{ "cells": [ { "cell_type": "markdown", "metadata": {}, "source": [ "## 基础知识" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 正则表达式分割文本" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import re\n", "\n", "file_name = 'data/药食材性味归经.xlsx'\n", "wb = openpyxl.load_workbook(file_name)\n", "sheet = wb.active\n", "#sheets = wb.sheetnames\n", "\n", "list1 = []\n", "dict1 = {}\n", "mo =r'[。入归].+经$'\n", "mo1 = r'[二]'\n", "mo2 = r'[,、;]'\n", "for n in range(2,sheet.max_row):\n", " name = sheet.cell(n,1).value\n", " content = re.findall(mo,sheet.cell(n,2).value)\n", " if len(content) > 0:\n", " l = len(content[0])\n", " gj = re.sub(mo1, '', content[0][1:l-1]) \n", " list_gj = re.split(mo2,gj)\n", " dict1[sheet.cell(n,1).value ] = list_gj\n", " #dict1['guijing'] = list_gj\n", " \n", "filename = './data/药食材归经.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl,ensure_ascii=False) \n", "\n", "#print(list1)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import re\n", "\n", "s = '消化不良、胃下垂、急性胃炎、慢性胃炎、萎缩性胃炎、神经性呕吐、胆囊炎、胆石症、胆道蛔虫症、胸胁痛等'\n", "ss = '健中和胃,消食止呕,理气疏郁,清热利胆。'\n", "mo = '等$'\n", "mo2 = r'[,、;。]'\n", "s1 = re.sub(mo, '', s) \n", "list1 = re.split(mo2,s1)\n", "if '' in list1:\n", " list1.remove('')\n", "print(list1)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import re\n", "t = '5小时10分48秒'\n", "m = re.match(r'(.*)小时(.*)分(.*)秒', t)\n", "m.groups()" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import re\n", "t = '10分48秒'\n", "list1 = []\n", "if '小时' in t and '分' in t:\n", " m = re.match(r'(.*)小时(.*)分(.*)秒', t)\n", " list1 = [m[1],m[2],m[3]]\n", "elif '小时' in t:\n", " m = re.match(r'(.*)小时(.*)秒', t)\n", " list1 = [m[1],0,m[2]]\n", "elif '分' in t:\n", " m = re.match(r'(.*)分(.*)秒', t)\n", " list1 = [0,m[1],m[2]]\n", "print(list1)\n", "#m.group()" ] }, { "cell_type": "code", "execution_count": 3, "metadata": { "execution": { "iopub.execute_input": "2023-05-09T03:14:50.263302Z", "iopub.status.busy": "2023-05-09T03:14:50.262465Z", "iopub.status.idle": "2023-05-09T03:14:50.269294Z", "shell.execute_reply": "2023-05-09T03:14:50.268106Z", "shell.execute_reply.started": "2023-05-09T03:14:50.263258Z" }, "tags": [] }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "['气郁组']\n" ] } ], "source": [ "import re\n", "\n", "s = '2022-09-28-《气郁组》-视频学习详情_155229'\n", "m = re.findall(r'《(.+)》',s)\n", "print(m)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 日期计算" ] }, { "cell_type": "code", "execution_count": 12, "metadata": { "execution": { "iopub.execute_input": "2022-12-09T03:42:01.099247Z", "iopub.status.busy": "2022-12-09T03:42:01.098719Z", "iopub.status.idle": "2022-12-09T03:42:01.108853Z", "shell.execute_reply": "2022-12-09T03:42:01.106707Z", "shell.execute_reply.started": "2022-12-09T03:42:01.099199Z" }, "tags": [] }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "33\n" ] } ], "source": [ "import time\n", "\n", "birth = '1989-01-25'\n", "t_birth = time.strptime(birth,'%Y-%m-%d')\n", "days = (time.time() -time.mktime(t_birth))//(365*24*60*60)\n", "print(int(days))" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 字符串转换" ] }, { "cell_type": "code", "execution_count": 5, "metadata": { "execution": { "iopub.execute_input": "2023-05-12T04:10:03.168210Z", "iopub.status.busy": "2023-05-12T04:10:03.167782Z", "iopub.status.idle": "2023-05-12T04:10:03.176045Z", "shell.execute_reply": "2023-05-12T04:10:03.174828Z", "shell.execute_reply.started": "2023-05-12T04:10:03.168182Z" }, "tags": [] }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "bs b'\\xc0\\xee\\xb5t'\n", "decode-bs: 李祎\n", "gbcode: b'\\xc2\\xed\\xc1\\xa2\\xd1\\xc7'\n", "gbs: c2edc1a2d1c7\n" ] } ], "source": [ "import binascii\n", "\n", "gbs = 'C0EEB574'\n", "bs = binascii.a2b_hex(gbs)\n", "print('bs', bs)\n", "print('decode-bs:', bs.decode('gbk'))\n", "\n", "s = '马立亚'\n", "gbcode = s.encode('gbk') # 先转成 bytes格式\n", "print('gbcode:', gbcode)\n", "gbs = \"\".join([hex(ch)[2:] for ch in gbcode]) #\n", "print('gbs:', gbs)" ] }, { "cell_type": "markdown", "metadata": { "tags": [], "toc-hr-collapsed": true }, "source": [ "## 医药体测" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 药膳归经明细文件生成" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "\n", "filename = 'data/药膳210927.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "filename = 'data/药食材归经.json'\n", "with open(filename,'r') as fl:\n", " dict2 = json.load(fl)\n", "k_zy = dict2.keys()\n", "dict3 = {}\n", "list1 = []\n", "for item in dict1:\n", " print(item['name'])\n", " list1 = []\n", " #print(i,item['zy'])\n", " for m_zy in item['zy']:\n", " if m_zy in k_zy:\n", " dict4 = {}\n", " print(m_zy,dict2[m_zy])\n", " dict4[m_zy] = dict2[m_zy]\n", " list1.append(dict4)\n", " dict3[item['name']] = list1\n", "filename = './data/药膳药食材.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict3, fl,ensure_ascii=False) \n", " \n" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "### 药膳归经权重生成" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import os,sys,shutil\n", "\n", "filename = './data/药膳药食材.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list1 = []\n", "for k, v in dict1.items():\n", " if len(v) > 0 :\n", " dict2 = {}\n", " #print(k,v)\n", " ys_name = k\n", " dict2.setdefault(ys_name,{})\n", " for item in v:\n", " for k1, v1 in item.items():\n", " for item1 in v1:\n", " dict2[ys_name].setdefault(item1,0)\n", " dict2[ys_name][item1] += 1\n", " list1.append(dict2) \n", " \n", "dict_qz = {}\n", "for item in list1:\n", " for k, v in item.items():\n", " qz = sorted(v.items(), key = lambda kv:(kv[1], kv[0]),reverse=True)\n", " dict_qz[k] = qz\n", "#print(dict_qz)\n", "qz_key = dict_qz.keys()\n", "filename = 'data/药膳210927.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "#print(dict1)\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet['A1'] = '药膳名称'\n", "sheet['B1'] = '来源'\n", "sheet['C1'] = '配方'\n", "sheet['D1'] = '做法'\n", "sheet['E1'] = '功效'\n", "sheet['F1'] = '中药成分'\n", "sheet['G1'] = '归经权重'\n", "\n", "i =2\n", "for item in dict1:\n", " sheet[f'A{i}'] = item['name']\n", " sheet[f'B{i}'] = item['source']\n", " sheet[f'C{i}'] = item['pf']\n", " sheet[f'D{i}'] = item['zf']\n", " sheet[f'E{i}'] = item['gx']\n", " sheet[f'F{i}'] = ','.join(item['zy'])\n", " if item['name'] in qz_key:\n", " s = ''\n", " for m_gj in dict_qz[item['name']]:\n", " s = s+ m_gj[0] +'('+str(m_gj[1])+')'\n", " sheet[f'G{i}'] = s\n", " else:\n", " sheet[f'G{i}'] = '暂无归经'\n", " i += 1\n", "\n", "wb.save('data/test4.xlsx') \n", "\n" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "### 穴位数据导入" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "#### 简单导出简介、内容" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "dict1 = {}\n", "filename = 'file/zhongyi/xuewei.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " dict2 = dict(zip(list1, list2))\n", " #print(line1['title'],dict1)\n", " #print(dict1.keys())\n", " #print(line1)\n", " dict1.setdefault(line1['title'][0],{})\n", " dict1[line1['title'][0]]['简介'] = line1['jj'] \n", " dict1[line1['title'][0]]['内容'] = dict2\n", "filename = './file/穴位1.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "#### 数据导入文件中" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "\n", "dict1 = {}\n", "filename = 'file/zhongyi/xuewei.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1.setdefault(line1['title'][0],{})\n", " if len(list3) > 0:\n", " \n", " dict1[line1['title'][0]]['about'] = list3\n", " dict1[line1['title'][0]].setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1[line1['title'][0]]['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", "filename = './file/穴位1.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "#### 穴位数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "#dblist = myclient.list_database_names()\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"xuewei\"]\n", "db_list = []\n", "filename = 'file/zhongyi/xuewei.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for s in list3:\n", " item = s.strip().split(':')\n", " dict1['about'][item[0].strip()] = item[1].strip()\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "#### 中医症状数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"zhongyizhengzhuang\"]\n", "db_list = []\n", "filename = 'file/zhongyi/zhongyizhengzhuang.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for i in range(0,int(len(list3)/2)):\n", " dict1['about'][list3[2*i]] = list3[2*i+1]\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "#### 疾病数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"jibing\"]\n", "db_list = []\n", "filename = 'file/zhongyi/jibing.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for i in range(0,int(len(list3)/2)):\n", " dict1['about'][list3[2*i]] = list3[2*i+1]\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "#### 术语数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "#dblist = myclient.list_database_names()\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"shuyu\"]\n", "db_list = []\n", "filename = 'file/zhongyi/shuyu.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for s in list3:\n", " item = s.strip().split(':')\n", " dict1['about'][item[0]] = item[1]\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "execution": { "iopub.execute_input": "2021-12-15T10:01:32.917184Z", "iopub.status.busy": "2021-12-15T10:01:32.917184Z", "iopub.status.idle": "2021-12-15T10:01:32.921185Z", "shell.execute_reply": "2021-12-15T10:01:32.921185Z", "shell.execute_reply.started": "2021-12-15T10:01:32.917184Z" }, "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "#### 西医症状数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"xiyizhengzhuang\"]\n", "db_list = []\n", "filename = 'file/zhongyi/xiyizhengzhuang.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for i in range(0,int(len(list3)/2)):\n", " dict1['about'][list3[2*i]] = list3[2*i+1]\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "#### 药剂数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "#dblist = myclient.list_database_names()\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"yaoji\"]\n", "db_list = []\n", "filename = 'file/zhongyi/yaoji.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj'] \n", " dict2 = dict(zip(list1, list2)) \n", " if len(list3) > 0: \n", " jj = {}\n", " for s in list3:\n", " item = s.strip().split(':')\n", " jj[item[0]] = item[1] \n", " dict1['name'] = jj['名称']\n", " dict1['about'] = jj\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "#### 药膳数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "#dblist = myclient.list_database_names()\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"yaoshan\"]\n", "db_list = []\n", "filename = 'file/zhongyi/yaoshan.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for s in list3:\n", " item = s.strip().split(':')\n", " dict1['about'][item[0]] = item[1]\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "#### 中草药数据导入数据库中" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"zhongcaoyao\"]\n", "db_list = []\n", "filename = 'file/zhongyi/zhongcaoyao.json'\n", "with open(filename,'r',encoding='utf-8') as fl:\n", " for line in fl:\n", " dict1 = {}\n", " line1 = json.loads(re.sub(r'\\xa0','',line))\n", " list1 = line1['tables']\n", " list2 = line1['contents']\n", " list3 = line1['jj']\n", " dict2 = dict(zip(list1, list2))\n", " dict1['name'] = line1['title'][0]\n", " if len(list3) > 0:\n", " dict1.setdefault('about',{})\n", " for i in range(0,int(len(list3)/2)):\n", " dict1['about'][list3[2*i]] = list3[2*i+1]\n", " dict1.setdefault('content',{})\n", " for k, v in dict2.items():\n", " dict1['content'][k] = v \n", " #dict1[line1['title'][0]]['内容'] = dict2\n", " db_list.append(dict1)\n", "x = mycol.insert_many(db_list)\n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "### 穴位隶属整理" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"xuewei\"]\n", "dict1 = {}\n", "list1 = []\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about.隶属\":1 }):\n", " if 'about' in x.keys():\n", " m_ls = x['about']['隶属']\n", " dict1.setdefault(m_ls,[])\n", " dict1[m_ls].append(x['name'])\n", "'''\n", "filename = './file/穴位隶属.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) \n", "'''\n", "for item in dict1.keys():\n", " print(item)" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "### 穴位功能、主治统计" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"xuewei\"]\n", "\n", "dict1 = {}\n", "mo = '等$'\n", "mo2 = r'[,、;。]'\n", "\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n", " dict1.setdefault(x['name'],{})\n", " if '主治' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['主治']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['主治'] = list1\n", " if '功能' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['功能']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['功能'] = list1\n", " \n", "\n", "filename = './file/穴位主治功能统计.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) \n", "\n" ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "### 穴位数据导出Excel表" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "\n", "filename = 'file/穴位.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "filename = 'file/穴位简要情况表.xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet['A1'] = '穴位名称'\n", "sheet['B1'] = '隶属'\n", "sheet['C1'] = '位置'\n", "sheet['D1'] = '主治'\n", "sheet['E1'] = '功能'\n", "sheet['F1'] = '操作'\n", "sheet['G1'] = '主要配伍'\n", "i = 2\n", "for k, v in dict1.items():\n", " sheet[f'A{i}'] = k\n", " sheet[f'B{i}'] = v['简介'][0].split(':')[1]\n", " sheet[f'C{i}'] = v['简介'][1].split(':')[1]\n", " sheet[f'D{i}'] = v['简介'][2].split(':')[1]\n", " sheet[f'E{i}'] = v['简介'][3].split(':')[1]\n", " sheet[f'F{i}'] = v['简介'][4].split(':')[1]\n", " sheet[f'G{i}'] = v['简介'][5].split(':')[1] \n", " i += 1\n", "wb.save(filename) \n", "print('ok!')\n", "\n", "\n" ] }, { "cell_type": "markdown", "metadata": { "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "### 术语数据处理" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"shuyu\"]\n", "dict1 = {}\n", "list1 = []\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n", " if '类别' in x['about'].keys():\n", " m_lb = x['about']['类别'].replace(' ','')\n", " else:\n", " m_lb = '无类别'\n", " dict1.setdefault(re.sub('\\xa0+','',m_lb),[])\n", " dict1[m_lb].append(x['name'])\n", "filename = './file/术语类别.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) \n" ] }, { "cell_type": "markdown", "metadata": { "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ "### 疾病数据处理" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"jibing\"]\n", "dict1 = {}\n", "list1 = []\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n", " if '疾病分类' in x['about'].keys():\n", " m_lb = x['about']['疾病分类'].replace(' ','')\n", " else:\n", " m_lb = '无类别'\n", " dict1.setdefault(re.sub('\\xa0+','',m_lb),[])\n", " dict1[m_lb].append(x['name'])\n", "filename = './file/疾病类别.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 中草药数据处理" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"zhongcaoyao\"]\n", "\n", "dict1 = {}\n", "mo = '等$'\n", "mo2 = r'[,、;。]'\n", "\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n", " dict1.setdefault(x['name'],{})\n", " if '别名' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['别名']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['别名'] = list1\n", " if '功能' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['功能']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['功能'] = list1\n", " if '主治' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " dict1[x['name']]['主治'] = s = x['about']['主治']\n", " \n", "\n", "filename = './file/中草药主治功能统计.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 药剂数据处理" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"yaoji\"]\n", "\n", "dict1 = {}\n", "mo = '等$'\n", "mo2 = r'[,、;。]'\n", "\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n", " dict1.setdefault(x['name'],{})\n", " if '功用' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['功用']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['功用'] = list1\n", " if '主治' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " dict1[x['name']]['主治'] = s = x['about']['主治']\n", " \n", "\n", "filename = './file/药剂主治功用统计.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 药膳数据处理" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import re\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n", "mydb = myclient['dayi']\n", "mycol = mydb[\"yaoshan\"]\n", "\n", "dict1 = {}\n", "mo = '等$'\n", "mo2 = r'[,、;。]'\n", "\n", "for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n", " dict1.setdefault(x['name'],{})\n", " if '功效' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['功效']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['功效'] = list1\n", " if '相关疾病' in x['about'].keys():\n", " #dict1[x['name']].setdefault('主治',[])\n", " s = x['about']['相关疾病']\n", " s1 = re.sub(mo, '', s)\n", " list1 = re.split(mo2,s1)\n", " if '' in list1:\n", " list1.remove('')\n", " dict1[x['name']]['相关疾病'] = list1 \n", " \n", "\n", "filename = './file/药膳功能统计.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl) " ] }, { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "### 北海炼化体检数据提取" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import os,sys,shutil\n", "\n", "file_name = 'file/中石化北海炼化2020年团体体检报告.xlsx'\n", "wb = openpyxl.load_workbook(file_name)\n", "sheet = wb.active\n", "dict1 = {}\n", "for n in range(1,sheet.max_row+1):\n", " bh = sheet.cell(n,2).value\n", " name = sheet.cell(n,3).value\n", " xb = sheet.cell(n,4).value\n", " nl = sheet.cell(n,5).value\n", " bz = sheet.cell(n,7).value\n", " dict1.setdefault(bh,{})\n", " dict1[bh]['姓名'] = name\n", " dict1[bh]['性别'] = xb\n", " dict1[bh]['年龄'] = nl\n", " dict1[bh].setdefault('病症',[])\n", " dict1[bh]['病症'].append(bz.strip())\n", "wb.close()\n", "#print(dict1)\n", " \n", " \n", "filename = 'file/中石化北海炼化2020年体检人员情况表.xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet['A1'] = '体检编号'\n", "sheet['B1'] = '姓名'\n", "sheet['C1'] = '性别'\n", "sheet['D1'] = '年龄'\n", "sheet['E1'] = '异常名称'\n", "\n", "i =2\n", "for k, v in dict1.items():\n", " sheet[f'A{i}'] = k\n", " sheet[f'B{i}'] = v['姓名']\n", " sheet[f'C{i}'] = v['性别']\n", " sheet[f'D{i}'] = v['年龄']\n", " sheet[f'E{i}'] = ','.join(v['病症']) \n", " i += 1\n", "\n", "wb.save(filename) \n", "\n", "\n", "\n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## excel数据读取" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import openpyxl\n", "import json\n", "\n", "filename = 'data/长岭全成绩.xlsx'\n", "wb = openpyxl.load_workbook(filename)\n", "sheet = wb.active\n", "data1 =list(sheet.values)\n", "del data1[0]\n", "print(data1)\n", "#for data in data1:\n", "# print(data)\n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 图表生成" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 高考一分一段表生成" ] }, { "cell_type": "code", "execution_count": 14, "metadata": { "execution": { "iopub.execute_input": "2022-12-09T03:45:39.442887Z", "iopub.status.busy": "2022-12-09T03:45:39.442350Z", "iopub.status.idle": "2022-12-09T03:45:39.579904Z", "shell.execute_reply": "2022-12-09T03:45:39.578629Z", "shell.execute_reply.started": "2022-12-09T03:45:39.442840Z" }, "tags": [] }, "outputs": [], "source": [ "from pyecharts.globals import CurrentConfig, NotebookType\n", "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", "from pyecharts import options as opts\n", "from pyecharts.charts import Bar,Line\n", "import pyecharts.options as opts\n", "from pyecharts.faker import Faker\n", "import json\n", "\n", "list_x = []\n", "list_y = []\n", "filename = 'data/17-21年一分一段表.json'\n", "with open(filename,'r') as fl:\n", " m_xx = json.load(fl)\n", "dict1 =m_xx['2020']['z']\n", "for i in sorted(dict1,reverse=True): #降序\n", " list_x.append(i)\n", " list_y.append(dict1[i]['num_person'])\n", " #print(i,dict1[i]['num_person'])\n", "bar = (\n", " Bar()\n", " .add_xaxis(list_x)\n", " .add_yaxis(\"2020年一分一段表\", list_y, category_gap=0, color=Faker.rand_color())\n", " .set_series_opts(label_opts=opts.LabelOpts(is_show=False))\n", " .set_global_opts(title_opts=opts.TitleOpts(title=\"Bar-直方图\"))\n", " .render(\"bar_histogram2020.html\")\n", "# .set_global_opts(title_opts=opts.TitleOpts(title=\"运动步幅及步频\", subtitle=\"户外运动\"),)\n", ")\n", "#bar.load_javascript()" ] }, { "cell_type": "code", "execution_count": 16, "metadata": { "execution": { "iopub.execute_input": "2022-12-09T03:46:12.149945Z", "iopub.status.busy": "2022-12-09T03:46:12.149314Z", "iopub.status.idle": "2022-12-09T03:46:12.163591Z", "shell.execute_reply": "2022-12-09T03:46:12.161659Z", "shell.execute_reply.started": "2022-12-09T03:46:12.149895Z" }, "tags": [] }, "outputs": [ { "ename": "AttributeError", "evalue": "'str' object has no attribute 'render_notebook'", "output_type": "error", "traceback": [ "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m", "\u001b[0;31mAttributeError\u001b[0m Traceback (most recent call last)", "\u001b[0;32m\u001b[0m in \u001b[0;36m\u001b[0;34m\u001b[0m\n\u001b[0;32m----> 1\u001b[0;31m \u001b[0mbar\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mrender_notebook\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m", "\u001b[0;31mAttributeError\u001b[0m: 'str' object has no attribute 'render_notebook'" ] } ], "source": [ "bar.render_notebook()" ] }, { "cell_type": "code", "execution_count": 13, "metadata": { "execution": { "iopub.execute_input": "2022-12-09T03:45:26.505194Z", "iopub.status.busy": "2022-12-09T03:45:26.504668Z", "iopub.status.idle": "2022-12-09T03:45:26.519538Z", "shell.execute_reply": "2022-12-09T03:45:26.518053Z", "shell.execute_reply.started": "2022-12-09T03:45:26.505147Z" } }, "outputs": [ { "ename": "SyntaxError", "evalue": "unexpected EOF while parsing (, line 31)", "output_type": "error", "traceback": [ "\u001b[0;36m File \u001b[0;32m\"\"\u001b[0;36m, line \u001b[0;32m31\u001b[0m\n\u001b[0;31m .render(\"bar_histogram.html\")\u001b[0m\n\u001b[0m ^\u001b[0m\n\u001b[0;31mSyntaxError\u001b[0m\u001b[0;31m:\u001b[0m unexpected EOF while parsing\n" ] } ], "source": [ "from pyecharts.globals import CurrentConfig, NotebookType\n", "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", "from pyecharts import options as opts\n", "from pyecharts.charts import Bar,Line\n", "import pyecharts.options as opts\n", "from pyecharts.faker import Faker\n", "import json\n", "\n", "x_2021 = []\n", "y_2021 = []\n", "x_2020 = []\n", "y_2020 = []\n", "filename = 'data/17-21年一分一段表.json'\n", "with open(filename,'r') as fl:\n", " m_xx = json.load(fl)\n", "dict1 =m_xx['2021']['z']\n", "for i in sorted(dict1,reverse=True): #降序\n", " x_2021.append(i)\n", " y_2021.append(dict1[i]['num_person'])\n", "dict1 =m_xx['2020']['z']\n", "for i in sorted(dict1,reverse=True): #降序\n", " x_2021.append(i)\n", " y_2021.append(dict1[i]['num_person'])\n", " #print(i,dict1[i]['num_person'])\n", "bar = (\n", " Bar()\n", " .add_xaxis(list_x)\n", " .add_yaxis(\"人数\", list_y, category_gap=0, color=Faker.rand_color())\n", " .set_series_opts(label_opts=opts.LabelOpts(is_show=False))\n", " .set_global_opts(title_opts=opts.TitleOpts(title=\"Bar-直方图\"))\n", " .render(\"bar_histogram.html\")\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3 (ipykernel)", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.10.6" } }, "nbformat": 4, "nbformat_minor": 4 }