diff --git a/数据处理.ipynb b/数据处理.ipynb index 01e08ef..b3e9a89 100644 --- a/数据处理.ipynb +++ b/数据处理.ipynb @@ -174,6 +174,13 @@ "## 穴位数据导入" ] }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 简单导出简介、内容" + ] + }, { "cell_type": "code", "execution_count": null, @@ -198,11 +205,476 @@ " dict1.setdefault(line1['title'][0],{})\n", " dict1[line1['title'][0]]['简介'] = line1['jj'] \n", " dict1[line1['title'][0]]['内容'] = dict2\n", - "filename = './file/穴位.json'\n", + "filename = './file/穴位1.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 数据导入文件中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "\n", + "dict1 = {}\n", + "filename = 'file/zhongyi/xuewei.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1.setdefault(line1['title'][0],{})\n", + " if len(list3) > 0:\n", + " \n", + " dict1[line1['title'][0]]['about'] = list3\n", + " dict1[line1['title'][0]].setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1[line1['title'][0]]['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + "filename = './file/穴位1.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl) " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 穴位数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"xuewei\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/xuewei.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " dict1['about'][item[0]] = item[1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 中医症状数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"zhongyizhengzhuang\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/zhongyizhengzhuang.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 疾病数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"jibing\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/jibing.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 术语数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"shuyu\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/shuyu.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " dict1['about'][item[0]] = item[1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:01:32.917184Z", + "iopub.status.busy": "2021-12-15T10:01:32.917184Z", + "iopub.status.idle": "2021-12-15T10:01:32.921185Z", + "shell.execute_reply": "2021-12-15T10:01:32.921185Z", + "shell.execute_reply.started": "2021-12-15T10:01:32.917184Z" + } + }, + "source": [ + "### 西医症状数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"xiyizhengzhuang\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/xiyizhengzhuang.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 药剂数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": 28, + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:13:01.097292Z", + "iopub.status.busy": "2021-12-15T10:13:01.097292Z", + "iopub.status.idle": "2021-12-15T10:13:04.127788Z", + "shell.execute_reply": "2021-12-15T10:13:04.127788Z", + "shell.execute_reply.started": "2021-12-15T10:13:01.097292Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"yaoji\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/yaoji.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj'] \n", + " dict2 = dict(zip(list1, list2)) \n", + " if len(list3) > 0: \n", + " jj = {}\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " jj[item[0]] = item[1] \n", + " dict1['name'] = jj['名称']\n", + " dict1['about'] = jj\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 药膳数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": 29, + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:15:38.181411Z", + "iopub.status.busy": "2021-12-15T10:15:38.181411Z", + "iopub.status.idle": "2021-12-15T10:15:38.468083Z", + "shell.execute_reply": "2021-12-15T10:15:38.467083Z", + "shell.execute_reply.started": "2021-12-15T10:15:38.181411Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"yaoshan\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/yaoshan.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " dict1['about'][item[0]] = item[1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 中草药数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": 30, + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:17:41.510430Z", + "iopub.status.busy": "2021-12-15T10:17:41.510430Z", + "iopub.status.idle": "2021-12-15T10:17:44.974610Z", + "shell.execute_reply": "2021-12-15T10:17:44.973618Z", + "shell.execute_reply.started": "2021-12-15T10:17:41.510430Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"zhongcaoyao\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/zhongcaoyao.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, { "cell_type": "markdown", "metadata": {}, @@ -355,15 +827,8 @@ }, { "cell_type": "code", - "execution_count": 78, + "execution_count": null, "metadata": { - "execution": { - "iopub.execute_input": "2021-11-24T11:15:33.859507Z", - "iopub.status.busy": "2021-11-24T11:15:33.858508Z", - "iopub.status.idle": "2021-11-24T11:15:33.864520Z", - "shell.execute_reply": "2021-11-24T11:15:33.864520Z", - "shell.execute_reply.started": "2021-11-24T11:15:33.859507Z" - }, "tags": [] }, "outputs": [],