From 1edc6efe08d50904f2a67079d3df2fa7dbec1714 Mon Sep 17 00:00:00 2001 From: 512song <512song@sina.com> Date: Mon, 20 Dec 2021 16:51:45 +0800 Subject: [PATCH] 20211220 --- .../数据处理-checkpoint-checkpoint.ipynb | 846 ++++++++++++++++++ 1 file changed, 846 insertions(+) create mode 100644 .ipynb_checkpoints/.ipynb_checkpoints/数据处理-checkpoint-checkpoint.ipynb diff --git a/.ipynb_checkpoints/.ipynb_checkpoints/数据处理-checkpoint-checkpoint.ipynb b/.ipynb_checkpoints/.ipynb_checkpoints/数据处理-checkpoint-checkpoint.ipynb new file mode 100644 index 0000000..9a19fdc --- /dev/null +++ b/.ipynb_checkpoints/.ipynb_checkpoints/数据处理-checkpoint-checkpoint.ipynb @@ -0,0 +1,846 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 正则表达式分割文本" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import re\n", + "\n", + "file_name = 'data/药食材性味归经.xlsx'\n", + "wb = openpyxl.load_workbook(file_name)\n", + "sheet = wb.active\n", + "#sheets = wb.sheetnames\n", + "\n", + "list1 = []\n", + "dict1 = {}\n", + "mo =r'[。入归].+经$'\n", + "mo1 = r'[二]'\n", + "mo2 = r'[,、;]'\n", + "for n in range(2,sheet.max_row):\n", + " name = sheet.cell(n,1).value\n", + " content = re.findall(mo,sheet.cell(n,2).value)\n", + " if len(content) > 0:\n", + " l = len(content[0])\n", + " gj = re.sub(mo1, '', content[0][1:l-1]) \n", + " list_gj = re.split(mo2,gj)\n", + " dict1[sheet.cell(n,1).value ] = list_gj\n", + " #dict1['guijing'] = list_gj\n", + " \n", + "filename = './data/药食材归经.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl,ensure_ascii=False) \n", + "\n", + "#print(list1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 药膳归经明细文件生成" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "filename = 'data/药膳210927.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "filename = 'data/药食材归经.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl)\n", + "k_zy = dict2.keys()\n", + "dict3 = {}\n", + "list1 = []\n", + "for item in dict1:\n", + " print(item['name'])\n", + " list1 = []\n", + " #print(i,item['zy'])\n", + " for m_zy in item['zy']:\n", + " if m_zy in k_zy:\n", + " dict4 = {}\n", + " print(m_zy,dict2[m_zy])\n", + " dict4[m_zy] = dict2[m_zy]\n", + " list1.append(dict4)\n", + " dict3[item['name']] = list1\n", + "filename = './data/药膳药食材.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict3, fl,ensure_ascii=False) \n", + " \n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 药膳归经权重生成" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import os,sys,shutil\n", + "\n", + "filename = './data/药膳药食材.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "list1 = []\n", + "for k, v in dict1.items():\n", + " if len(v) > 0 :\n", + " dict2 = {}\n", + " #print(k,v)\n", + " ys_name = k\n", + " dict2.setdefault(ys_name,{})\n", + " for item in v:\n", + " for k1, v1 in item.items():\n", + " for item1 in v1:\n", + " dict2[ys_name].setdefault(item1,0)\n", + " dict2[ys_name][item1] += 1\n", + " list1.append(dict2) \n", + " \n", + "dict_qz = {}\n", + "for item in list1:\n", + " for k, v in item.items():\n", + " qz = sorted(v.items(), key = lambda kv:(kv[1], kv[0]),reverse=True)\n", + " dict_qz[k] = qz\n", + "#print(dict_qz)\n", + "qz_key = dict_qz.keys()\n", + "filename = 'data/药膳210927.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "#print(dict1)\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet['A1'] = '药膳名称'\n", + "sheet['B1'] = '来源'\n", + "sheet['C1'] = '配方'\n", + "sheet['D1'] = '做法'\n", + "sheet['E1'] = '功效'\n", + "sheet['F1'] = '中药成分'\n", + "sheet['G1'] = '归经权重'\n", + "\n", + "i =2\n", + "for item in dict1:\n", + " sheet[f'A{i}'] = item['name']\n", + " sheet[f'B{i}'] = item['source']\n", + " sheet[f'C{i}'] = item['pf']\n", + " sheet[f'D{i}'] = item['zf']\n", + " sheet[f'E{i}'] = item['gx']\n", + " sheet[f'F{i}'] = ','.join(item['zy'])\n", + " if item['name'] in qz_key:\n", + " s = ''\n", + " for m_gj in dict_qz[item['name']]:\n", + " s = s+ m_gj[0] +'('+str(m_gj[1])+')'\n", + " sheet[f'G{i}'] = s\n", + " else:\n", + " sheet[f'G{i}'] = '暂无归经'\n", + " i += 1\n", + "\n", + "wb.save('data/test4.xlsx') \n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 穴位数据导入" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 简单导出简介、内容" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "dict1 = {}\n", + "filename = 'file/zhongyi/xuewei.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " dict2 = dict(zip(list1, list2))\n", + " #print(line1['title'],dict1)\n", + " #print(dict1.keys())\n", + " #print(line1)\n", + " dict1.setdefault(line1['title'][0],{})\n", + " dict1[line1['title'][0]]['简介'] = line1['jj'] \n", + " dict1[line1['title'][0]]['内容'] = dict2\n", + "filename = './file/穴位1.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl) " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 数据导入文件中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "\n", + "dict1 = {}\n", + "filename = 'file/zhongyi/xuewei.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1.setdefault(line1['title'][0],{})\n", + " if len(list3) > 0:\n", + " \n", + " dict1[line1['title'][0]]['about'] = list3\n", + " dict1[line1['title'][0]].setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1[line1['title'][0]]['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + "filename = './file/穴位1.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl) " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 穴位数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"xuewei\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/xuewei.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " dict1['about'][item[0]] = item[1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 中医症状数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"zhongyizhengzhuang\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/zhongyizhengzhuang.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 疾病数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"jibing\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/jibing.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 术语数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"shuyu\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/shuyu.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " dict1['about'][item[0]] = item[1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:01:32.917184Z", + "iopub.status.busy": "2021-12-15T10:01:32.917184Z", + "iopub.status.idle": "2021-12-15T10:01:32.921185Z", + "shell.execute_reply": "2021-12-15T10:01:32.921185Z", + "shell.execute_reply.started": "2021-12-15T10:01:32.917184Z" + } + }, + "source": [ + "### 西医症状数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"xiyizhengzhuang\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/xiyizhengzhuang.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 药剂数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": 28, + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:13:01.097292Z", + "iopub.status.busy": "2021-12-15T10:13:01.097292Z", + "iopub.status.idle": "2021-12-15T10:13:04.127788Z", + "shell.execute_reply": "2021-12-15T10:13:04.127788Z", + "shell.execute_reply.started": "2021-12-15T10:13:01.097292Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"yaoji\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/yaoji.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj'] \n", + " dict2 = dict(zip(list1, list2)) \n", + " if len(list3) > 0: \n", + " jj = {}\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " jj[item[0]] = item[1] \n", + " dict1['name'] = jj['名称']\n", + " dict1['about'] = jj\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 药膳数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": 29, + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:15:38.181411Z", + "iopub.status.busy": "2021-12-15T10:15:38.181411Z", + "iopub.status.idle": "2021-12-15T10:15:38.468083Z", + "shell.execute_reply": "2021-12-15T10:15:38.467083Z", + "shell.execute_reply.started": "2021-12-15T10:15:38.181411Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"yaoshan\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/yaoshan.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for s in list3:\n", + " item = s.strip().split(':')\n", + " dict1['about'][item[0]] = item[1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 中草药数据导入数据库中" + ] + }, + { + "cell_type": "code", + "execution_count": 30, + "metadata": { + "execution": { + "iopub.execute_input": "2021-12-15T10:17:41.510430Z", + "iopub.status.busy": "2021-12-15T10:17:41.510430Z", + "iopub.status.idle": "2021-12-15T10:17:44.974610Z", + "shell.execute_reply": "2021-12-15T10:17:44.973618Z", + "shell.execute_reply.started": "2021-12-15T10:17:41.510430Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import re\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "mydb = myclient['dayi']\n", + "mycol = mydb[\"zhongcaoyao\"]\n", + "db_list = []\n", + "filename = 'file/zhongyi/zhongcaoyao.json'\n", + "with open(filename,'r',encoding='utf-8') as fl:\n", + " for line in fl:\n", + " dict1 = {}\n", + " line1 = json.loads(re.sub(r'\\xa0','',line))\n", + " list1 = line1['tables']\n", + " list2 = line1['contents']\n", + " list3 = line1['jj']\n", + " dict2 = dict(zip(list1, list2))\n", + " dict1['name'] = line1['title'][0]\n", + " if len(list3) > 0:\n", + " dict1.setdefault('about',{})\n", + " for i in range(0,int(len(list3)/2)):\n", + " dict1['about'][list3[2*i]] = list3[2*i+1]\n", + " dict1.setdefault('content',{})\n", + " for k, v in dict2.items():\n", + " dict1['content'][k] = v \n", + " #dict1[line1['title'][0]]['内容'] = dict2\n", + " db_list.append(dict1)\n", + "x = mycol.insert_many(db_list)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 穴位数据导出Excel表" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "\n", + "filename = 'file/穴位.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "filename = 'file/穴位简要情况表.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet['A1'] = '穴位名称'\n", + "sheet['B1'] = '隶属'\n", + "sheet['C1'] = '位置'\n", + "sheet['D1'] = '主治'\n", + "sheet['E1'] = '功能'\n", + "sheet['F1'] = '操作'\n", + "sheet['G1'] = '主要配伍'\n", + "i = 2\n", + "for k, v in dict1.items():\n", + " sheet[f'A{i}'] = k\n", + " sheet[f'B{i}'] = v['简介'][0].split(':')[1]\n", + " sheet[f'C{i}'] = v['简介'][1].split(':')[1]\n", + " sheet[f'D{i}'] = v['简介'][2].split(':')[1]\n", + " sheet[f'E{i}'] = v['简介'][3].split(':')[1]\n", + " sheet[f'F{i}'] = v['简介'][4].split(':')[1]\n", + " sheet[f'G{i}'] = v['简介'][5].split(':')[1] \n", + " i += 1\n", + "wb.save(filename) \n", + "print('ok!')\n", + "\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 北海炼化体检数据提取" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import os,sys,shutil\n", + "\n", + "file_name = 'file/中石化北海炼化2020年团体体检报告.xlsx'\n", + "wb = openpyxl.load_workbook(file_name)\n", + "sheet = wb.active\n", + "dict1 = {}\n", + "for n in range(1,sheet.max_row+1):\n", + " bh = sheet.cell(n,2).value\n", + " name = sheet.cell(n,3).value\n", + " xb = sheet.cell(n,4).value\n", + " nl = sheet.cell(n,5).value\n", + " bz = sheet.cell(n,7).value\n", + " dict1.setdefault(bh,{})\n", + " dict1[bh]['姓名'] = name\n", + " dict1[bh]['性别'] = xb\n", + " dict1[bh]['年龄'] = nl\n", + " dict1[bh].setdefault('病症',[])\n", + " dict1[bh]['病症'].append(bz.strip())\n", + "wb.close()\n", + "#print(dict1)\n", + " \n", + " \n", + "filename = 'file/中石化北海炼化2020年体检人员情况表.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet['A1'] = '体检编号'\n", + "sheet['B1'] = '姓名'\n", + "sheet['C1'] = '性别'\n", + "sheet['D1'] = '年龄'\n", + "sheet['E1'] = '异常名称'\n", + "\n", + "i =2\n", + "for k, v in dict1.items():\n", + " sheet[f'A{i}'] = k\n", + " sheet[f'B{i}'] = v['姓名']\n", + " sheet[f'C{i}'] = v['性别']\n", + " sheet[f'D{i}'] = v['年龄']\n", + " sheet[f'E{i}'] = ','.join(v['病症']) \n", + " i += 1\n", + "\n", + "wb.save(filename) " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import openpyxl\n", + "import json\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://512song:songyi@192.168.3.100:27017/tice')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['tice']\n", + " \n", + "collist = mydb. list_collection_names()\n", + "print(collist)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import json\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://192.168.3.100:27017/')\n", + "#dblist = myclient.list_database_names()\n", + "mydb = myclient['tice']\n", + " \n", + "collist = mydb. list_collection_names()\n", + "print(collist)" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3 (ipykernel)", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.9.1" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +}