Files
jupyter/数据处理.ipynb
T
2024-07-25 21:53:21 +08:00

1846 lines
54 KiB
Plaintext
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 基础知识"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 正则表达式分割文本"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
"import re\n",
"\n",
"file_name = 'data/药食材性味归经.xlsx'\n",
"wb = openpyxl.load_workbook(file_name)\n",
"sheet = wb.active\n",
"#sheets = wb.sheetnames\n",
"\n",
"list1 = []\n",
"dict1 = {}\n",
"mo =r'[。入归].+经$'\n",
"mo1 = r'[二]'\n",
"mo2 = r'[,、;]'\n",
"for n in range(2,sheet.max_row):\n",
" name = sheet.cell(n,1).value\n",
" content = re.findall(mo,sheet.cell(n,2).value)\n",
" if len(content) > 0:\n",
" l = len(content[0])\n",
" gj = re.sub(mo1, '', content[0][1:l-1]) \n",
" list_gj = re.split(mo2,gj)\n",
" dict1[sheet.cell(n,1).value ] = list_gj\n",
" #dict1['guijing'] = list_gj\n",
" \n",
"filename = './data/药食材归经.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl,ensure_ascii=False) \n",
"\n",
"#print(list1)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"\n",
"s = '消化不良、胃下垂、急性胃炎、慢性胃炎、萎缩性胃炎、神经性呕吐、胆囊炎、胆石症、胆道蛔虫症、胸胁痛等'\n",
"ss = '健中和胃,消食止呕,理气疏郁,清热利胆。'\n",
"mo = '等$'\n",
"mo2 = r'[,、;。]'\n",
"s1 = re.sub(mo, '', s) \n",
"list1 = re.split(mo2,s1)\n",
"if '' in list1:\n",
" list1.remove('')\n",
"print(list1)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"t = '5小时10分48秒'\n",
"m = re.match(r'(.*)小时(.*)分(.*)秒', t)\n",
"m.groups()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"t = '10分48秒'\n",
"list1 = []\n",
"if '小时' in t and '分' in t:\n",
" m = re.match(r'(.*)小时(.*)分(.*)秒', t)\n",
" list1 = [m[1],m[2],m[3]]\n",
"elif '小时' in t:\n",
" m = re.match(r'(.*)小时(.*)秒', t)\n",
" list1 = [m[1],0,m[2]]\n",
"elif '分' in t:\n",
" m = re.match(r'(.*)分(.*)秒', t)\n",
" list1 = [0,m[1],m[2]]\n",
"print(list1)\n",
"#m.group()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"\n",
"s = '2022-09-28-《气郁组》-视频学习详情_155229'\n",
"m = re.findall(r'《(.+)》',s)\n",
"print(m)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 日期计算"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import time\n",
"\n",
"birth = '1989-01-25'\n",
"t_birth = time.strptime(birth,'%Y-%m-%d')\n",
"days = (time.time() -time.mktime(t_birth))//(365*24*60*60)\n",
"print(int(days))"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 字符串转换"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import binascii\n",
"\n",
"gbs = 'D4C0CDAF'\n",
"bs = binascii.a2b_hex(gbs)\n",
"print('bs', bs)\n",
"print('decode-bs:', bs.decode('gbk'))\n",
"\n",
"s = '马立亚'\n",
"gbcode = s.encode('gbk') # 先转成 bytes格式\n",
"print('gbcode:', gbcode)\n",
"gbs = \"\".join([hex(ch)[2:] for ch in gbcode]) #\n",
"print('gbs:', gbs)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 数据组合处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import itertools\n",
"\n",
"list1 = ['气虚','阳虚','阴虚','痰湿','湿热','血瘀','气郁','特禀']\n",
"list2 = [0,1,2,3,4,5,6,7]\n",
"\n",
"result=itertools.combinations(list1,7)\n",
"print(len(list(result)))\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 数字查询替换组合处理"
]
},
{
"cell_type": "code",
"execution_count": 21,
"metadata": {
"execution": {
"iopub.execute_input": "2024-05-18T09:23:04.791860Z",
"iopub.status.busy": "2024-05-18T09:23:04.791023Z",
"iopub.status.idle": "2024-05-18T09:23:04.798569Z",
"shell.execute_reply": "2024-05-18T09:23:04.798170Z",
"shell.execute_reply.started": "2024-05-18T09:23:04.791818Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Original Text: In 2021, we had a total of 456 sales.\n",
"Modified Text: In {{img_2021}}, we had a total of {{img_456}} sales.\n"
]
}
],
"source": [
"import re \n",
"def replace_numbers(text):\n",
" pattern = r'\\d+'\n",
" replacement = lambda x:'{{'+ f'img_{x.group()}'+'}}'\n",
" modified_text = re.sub(pattern, replacement, text)\n",
" return modified_text\n",
"\n",
"# Example usage of the function\n",
"example_text = \"In 2021, we had a total of 456 sales.\"\n",
"modified_text = replace_numbers(example_text)\n",
"print(\"Original Text:\", example_text)\n",
"print(\"Modified Text:\", modified_text)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 文本文件提取数据"
]
},
{
"cell_type": "code",
"execution_count": 18,
"metadata": {
"execution": {
"iopub.execute_input": "2024-07-25T05:15:01.475822Z",
"iopub.status.busy": "2024-07-25T05:15:01.475119Z",
"iopub.status.idle": "2024-07-25T05:15:01.484925Z",
"shell.execute_reply": "2024-07-25T05:15:01.483656Z",
"shell.execute_reply.started": "2024-07-25T05:15:01.475759Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"1刘宁宁;210105197906243163,;6226900714117146;中信银行上地支行,;13810477335\n",
"2李艳君;130102196606290626;6228480018747767178;中国农业银行股份有限公司北京北下关支行;13641398073\n",
"3何英;110108197202132727;4367420011031107855;中国建设银行北京北大南门支行;15101082137。\n",
"4王晓娜;11010819791230276X;6226900701079721;中信银行;18611172725\n",
"5彭翔吉;371328198801024031;6212260200195609850;中国工商银行崇文体育馆路支行;15210900779\n",
"6 章王楠;110108196201262795;4563510100879852469;中国银行北京安慧里支行;13911983185\n",
"1 薛文传;370921199601032418,;6217994630014645622;中国邮政储蓄银行宁阳县中心营业所;18728194757\n",
"2 张晓利;370921199601232428,;6223795315018773039;齐鲁银行领秀城支行;18953884309\n",
"3王荧铄;370921200311220088;6217002340044537993;中国建设银行宁阳支行;15610330246\n",
"4 符箐 ;622201198810291227;6217953400022774;浦发银行兰州市支行;19893189999\n",
"1杨建营;372431197201211319;6228480322626735417;农行杭州留下支行;13456947969\n",
"2张保(少林寺);342224197404010138;6217002430066673063;中国建设银行河南登封分行;(河南)15093080006(浙江)19884878916\n",
"3 杜洪宇;210404199503163010;6217002650002395485;中国建设银行湖北武当山支行;18671672127\n",
"4张业金;422129195010010036;6217002730004815712;中国建设银行武穴支行;13986516555\n",
"5刘敬儒;110104193607050830;6013820100008481833。;中国银行北京分行右安门支行;13611107631\n",
"6李剑方;132234195706078052;6217855000046012266;中国银行石家庄市中华大街支行;15710379999\n",
"7张长念;342222198006106013;6217230200007011051;中国工商银行海淀北太平庄支行;18911533236\n",
"8刘绥滨;510127196507040056;6227003811990408601;中国建设银行都江堰支行;13981805148\n",
"9王玉林;110108196409092721;6222030200028930042;工商银行清华大学支行;13521738338\n",
"10霍静虹;120111197709141024;6225880222824709;招商银行天津市南门外支行;13920559014\n",
"11姜周存;370102195012312939;6222081602006679608;山东济南市历下区中国工商银行经十东路支行;13964050787\n",
"12刘连俊;13092219610113001x;6228231735158089360;中国农业银行青县支行;13932726789\n",
"13任刚;510111195804064737;6222084402008897663;工商银行成都高新桐梓林南路支行;13908021945\n",
"14高宝东;142429194207131215;6214720508000053092;中国工商银行山西省晋中市太谷区支行;13834834395\n",
"15孙学孟;230102194808250416;6217001140006800145;中国建设银行;15604669719\n",
"16沙宗朝;372526197006130016;6212261611002728999;中国工商银行冠县支行;18606355008\n",
"17孙永田;110107194902050335;6217000010129556042;中国建设银行姓名:孙永田;13801392579\n",
"18陈 虎;622701198902081670;6230650004200973230;平凉农商行天门支行;19809330555\n",
"19王镖(8000元);622723197502051011;6217858500019105267;中国银行平凉分行;13993394577\n",
"20 杨 丽;110108195608191322;6222080200026479793;北京市海淀区红山口国防大学中国工商银行分行;13699112621\n",
"21张佑印;610124198111092419;6214860130311109;招商银行北京长安街支行;18801068122\n",
"22张永宏;612730198309030135;6217710726216872;中信银行北京上地支行;15960266950\n",
"23杨玉冰;110108196707206395;6217730719618595;中信银行北京市上地支行;19568701147\n"
]
}
],
"source": [
"fn = 'file/人员名单.txt'\n",
"\n",
"with open(fn, \"r\") as f:\n",
" data = f.readlines()\n",
"for i in range(0,int(len(data)/5)):\n",
" list1= []\n",
" for n in range(0,5):\n",
" s = data[5*i+n].replace('\\n','')\n",
" if n >0:\n",
" list1.append(s.split(':')[1].replace(' ',''))\n",
" else:\n",
" list1.append(s)\n",
" \n",
" print(';'.join(list1))\n",
" i+=1"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"## 医药体测"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 药膳归经明细文件生成"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"\n",
"filename = 'data/药膳210927.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"filename = 'data/药食材归经.json'\n",
"with open(filename,'r') as fl:\n",
" dict2 = json.load(fl)\n",
"k_zy = dict2.keys()\n",
"dict3 = {}\n",
"list1 = []\n",
"for item in dict1:\n",
" print(item['name'])\n",
" list1 = []\n",
" #print(i,item['zy'])\n",
" for m_zy in item['zy']:\n",
" if m_zy in k_zy:\n",
" dict4 = {}\n",
" print(m_zy,dict2[m_zy])\n",
" dict4[m_zy] = dict2[m_zy]\n",
" list1.append(dict4)\n",
" dict3[item['name']] = list1\n",
"filename = './data/药膳药食材.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict3, fl,ensure_ascii=False) \n",
" \n"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"### 药膳归经权重生成"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
"import os,sys,shutil\n",
"\n",
"filename = './data/药膳药食材.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"list1 = []\n",
"for k, v in dict1.items():\n",
" if len(v) > 0 :\n",
" dict2 = {}\n",
" #print(k,v)\n",
" ys_name = k\n",
" dict2.setdefault(ys_name,{})\n",
" for item in v:\n",
" for k1, v1 in item.items():\n",
" for item1 in v1:\n",
" dict2[ys_name].setdefault(item1,0)\n",
" dict2[ys_name][item1] += 1\n",
" list1.append(dict2) \n",
" \n",
"dict_qz = {}\n",
"for item in list1:\n",
" for k, v in item.items():\n",
" qz = sorted(v.items(), key = lambda kv:(kv[1], kv[0]),reverse=True)\n",
" dict_qz[k] = qz\n",
"#print(dict_qz)\n",
"qz_key = dict_qz.keys()\n",
"filename = 'data/药膳210927.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"#print(dict1)\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"sheet['A1'] = '药膳名称'\n",
"sheet['B1'] = '来源'\n",
"sheet['C1'] = '配方'\n",
"sheet['D1'] = '做法'\n",
"sheet['E1'] = '功效'\n",
"sheet['F1'] = '中药成分'\n",
"sheet['G1'] = '归经权重'\n",
"\n",
"i =2\n",
"for item in dict1:\n",
" sheet[f'A{i}'] = item['name']\n",
" sheet[f'B{i}'] = item['source']\n",
" sheet[f'C{i}'] = item['pf']\n",
" sheet[f'D{i}'] = item['zf']\n",
" sheet[f'E{i}'] = item['gx']\n",
" sheet[f'F{i}'] = ','.join(item['zy'])\n",
" if item['name'] in qz_key:\n",
" s = ''\n",
" for m_gj in dict_qz[item['name']]:\n",
" s = s+ m_gj[0] +'('+str(m_gj[1])+')'\n",
" sheet[f'G{i}'] = s\n",
" else:\n",
" sheet[f'G{i}'] = '暂无归经'\n",
" i += 1\n",
"\n",
"wb.save('data/test4.xlsx') \n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 药膳文件处理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### 药膳文件预处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"import json\n",
"\n",
"filename = 'data/286种药膳常用中药功能表.txt'\n",
"with open(filename, \"r\", encoding='utf-8') as f: \n",
" data = f.readlines()\n",
"dict1 = {}\n",
"for i in range(0,int(len(data)/7)) : \n",
" s = data[i*7].strip()\n",
" bh = s.split('.')[0]\n",
" name = s.split('.')[1].split('(')[0]\n",
" #print(bh,name)\n",
" id = str(i+1)\n",
" dict1.setdefault(id,{})\n",
" dict1[id]['name'] = name\n",
" for n in range(1,7):\n",
" ss = data[i*7+n].strip()\n",
" p = re.compile(r'【(.*?)】')\n",
" item = re.findall(p,ss)[0]\n",
" content = ss.split('】')[1]\n",
" dict1[id][item] = content\n",
"filename = 'data/286种药膳常用中药功能表.json'\n",
"with open(filename, 'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False)\n",
"print('ok')"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### 性味归经分解"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"import json\n",
"\n",
"filename = 'data/286种药膳常用中药功能表.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"p = re.compile(r'[入归](.*?)经')\n",
"for k, v in dict1.items():\n",
" if '性味归经' in v.keys():\n",
" s = v['性味归经']\n",
" item = re.search(p,s)\n",
" if item is not None:\n",
" ss = item.group()\n",
" sss = re.sub(ss,'',s)\n",
" \n",
" p1 = r'[;:、,。:]+。'\n",
" ssss = re.sub(p1,'。',sss)\n",
" #print(k,ss,ssss)\n",
" dict1[k]['归经'] = ss\n",
" dict1[k]['性味'] = ssss\n",
" else:\n",
" dict1[k]['归经'] = s\n",
" \n",
"filename = 'data/286种药膳常用中药功能表1.json'\n",
"with open(filename, 'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False)\n",
"print('ok') \n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"import json\n",
"\n",
"filename = 'data/286种药膳常用中药功能表.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"p = re.compile(r'[入归](.*?)经')\n",
"for k, v in dict1.items():\n",
" if '性味归经' in v.keys():\n",
" s = v['性味归经']\n",
" item = re.search(p,s)\n",
" #print(item)\n",
" if item is not None:\n",
" ss = item.group()\n",
" sss = re.sub(ss,'',s)\n",
" \n",
" p1 = r'[;:、,。:]+。'\n",
" ssss = re.sub(p1,'。',sss)\n",
" print(k,ssss)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 体质数据处理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### 体质对应数据导入"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"\n",
"filename = 'data/tijianbingzheng.txt'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"m_key = set()\n",
"for k, v in dict1.items():\n",
" for item in v.keys():\n",
" m_key.add(item)\n",
"print(dict1)"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"### 穴位数据导入"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"#### 简单导出简介、内容"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"dict1 = {}\n",
"filename = 'file/zhongyi/xuewei.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" dict2 = dict(zip(list1, list2))\n",
" #print(line1['title'],dict1)\n",
" #print(dict1.keys())\n",
" #print(line1)\n",
" dict1.setdefault(line1['title'][0],{})\n",
" dict1[line1['title'][0]]['简介'] = line1['jj'] \n",
" dict1[line1['title'][0]]['内容'] = dict2\n",
"filename = './file/穴位1.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) "
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"#### 数据导入文件中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"\n",
"dict1 = {}\n",
"filename = 'file/zhongyi/xuewei.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1.setdefault(line1['title'][0],{})\n",
" if len(list3) > 0:\n",
" \n",
" dict1[line1['title'][0]]['about'] = list3\n",
" dict1[line1['title'][0]].setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1[line1['title'][0]]['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
"filename = './file/穴位1.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) "
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"#### 穴位数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"#dblist = myclient.list_database_names()\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"xuewei\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/xuewei.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for s in list3:\n",
" item = s.strip().split(':')\n",
" dict1['about'][item[0].strip()] = item[1].strip()\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"#### 中医症状数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"zhongyizhengzhuang\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/zhongyizhengzhuang.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for i in range(0,int(len(list3)/2)):\n",
" dict1['about'][list3[2*i]] = list3[2*i+1]\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"#### 疾病数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"jibing\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/jibing.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for i in range(0,int(len(list3)/2)):\n",
" dict1['about'][list3[2*i]] = list3[2*i+1]\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"#### 术语数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"#dblist = myclient.list_database_names()\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"shuyu\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/shuyu.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for s in list3:\n",
" item = s.strip().split(':')\n",
" dict1['about'][item[0]] = item[1]\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"execution": {
"iopub.execute_input": "2021-12-15T10:01:32.917184Z",
"iopub.status.busy": "2021-12-15T10:01:32.917184Z",
"iopub.status.idle": "2021-12-15T10:01:32.921185Z",
"shell.execute_reply": "2021-12-15T10:01:32.921185Z",
"shell.execute_reply.started": "2021-12-15T10:01:32.917184Z"
},
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"#### 西医症状数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"xiyizhengzhuang\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/xiyizhengzhuang.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for i in range(0,int(len(list3)/2)):\n",
" dict1['about'][list3[2*i]] = list3[2*i+1]\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"#### 药剂数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"#dblist = myclient.list_database_names()\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"yaoji\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/yaoji.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj'] \n",
" dict2 = dict(zip(list1, list2)) \n",
" if len(list3) > 0: \n",
" jj = {}\n",
" for s in list3:\n",
" item = s.strip().split(':')\n",
" jj[item[0]] = item[1] \n",
" dict1['name'] = jj['名称']\n",
" dict1['about'] = jj\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"#### 药膳数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"#dblist = myclient.list_database_names()\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"yaoshan\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/yaoshan.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for s in list3:\n",
" item = s.strip().split(':')\n",
" dict1['about'][item[0]] = item[1]\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"#### 中草药数据导入数据库中"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"zhongcaoyao\"]\n",
"db_list = []\n",
"filename = 'file/zhongyi/zhongcaoyao.json'\n",
"with open(filename,'r',encoding='utf-8') as fl:\n",
" for line in fl:\n",
" dict1 = {}\n",
" line1 = json.loads(re.sub(r'\\xa0','',line))\n",
" list1 = line1['tables']\n",
" list2 = line1['contents']\n",
" list3 = line1['jj']\n",
" dict2 = dict(zip(list1, list2))\n",
" dict1['name'] = line1['title'][0]\n",
" if len(list3) > 0:\n",
" dict1.setdefault('about',{})\n",
" for i in range(0,int(len(list3)/2)):\n",
" dict1['about'][list3[2*i]] = list3[2*i+1]\n",
" dict1.setdefault('content',{})\n",
" for k, v in dict2.items():\n",
" dict1['content'][k] = v \n",
" #dict1[line1['title'][0]]['内容'] = dict2\n",
" db_list.append(dict1)\n",
"x = mycol.insert_many(db_list)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"### 穴位隶属整理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"xuewei\"]\n",
"dict1 = {}\n",
"list1 = []\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about.隶属\":1 }):\n",
" if 'about' in x.keys():\n",
" m_ls = x['about']['隶属']\n",
" dict1.setdefault(m_ls,[])\n",
" dict1[m_ls].append(x['name'])\n",
"'''\n",
"filename = './file/穴位隶属.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) \n",
"'''\n",
"for item in dict1.keys():\n",
" print(item)"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"### 穴位功能、主治统计"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"xuewei\"]\n",
"\n",
"dict1 = {}\n",
"mo = '等$'\n",
"mo2 = r'[,、;。]'\n",
"\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n",
" dict1.setdefault(x['name'],{})\n",
" if '主治' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['主治']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['主治'] = list1\n",
" if '功能' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['功能']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['功能'] = list1\n",
" \n",
"\n",
"filename = './file/穴位主治功能统计.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) \n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"### 穴位数据导出Excel表"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
"\n",
"filename = 'file/穴位.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"filename = 'file/穴位简要情况表.xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"sheet['A1'] = '穴位名称'\n",
"sheet['B1'] = '隶属'\n",
"sheet['C1'] = '位置'\n",
"sheet['D1'] = '主治'\n",
"sheet['E1'] = '功能'\n",
"sheet['F1'] = '操作'\n",
"sheet['G1'] = '主要配伍'\n",
"i = 2\n",
"for k, v in dict1.items():\n",
" sheet[f'A{i}'] = k\n",
" sheet[f'B{i}'] = v['简介'][0].split(':')[1]\n",
" sheet[f'C{i}'] = v['简介'][1].split(':')[1]\n",
" sheet[f'D{i}'] = v['简介'][2].split(':')[1]\n",
" sheet[f'E{i}'] = v['简介'][3].split(':')[1]\n",
" sheet[f'F{i}'] = v['简介'][4].split(':')[1]\n",
" sheet[f'G{i}'] = v['简介'][5].split(':')[1] \n",
" i += 1\n",
"wb.save(filename) \n",
"print('ok!')\n",
"\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"### 术语数据处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"shuyu\"]\n",
"dict1 = {}\n",
"list1 = []\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n",
" if '类别' in x['about'].keys():\n",
" m_lb = x['about']['类别'].replace(' ','')\n",
" else:\n",
" m_lb = '无类别'\n",
" dict1.setdefault(re.sub('\\xa0+','',m_lb),[])\n",
" dict1[m_lb].append(x['name'])\n",
"filename = './file/术语类别.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) \n"
]
},
{
"cell_type": "markdown",
"metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": []
},
"source": [
"### 疾病数据处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"jibing\"]\n",
"dict1 = {}\n",
"list1 = []\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n",
" if '疾病分类' in x['about'].keys():\n",
" m_lb = x['about']['疾病分类'].replace(' ','')\n",
" else:\n",
" m_lb = '无类别'\n",
" dict1.setdefault(re.sub('\\xa0+','',m_lb),[])\n",
" dict1[m_lb].append(x['name'])\n",
"filename = './file/疾病类别.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) "
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 中草药数据处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"zhongcaoyao\"]\n",
"\n",
"dict1 = {}\n",
"mo = '等$'\n",
"mo2 = r'[,、;。]'\n",
"\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n",
" dict1.setdefault(x['name'],{})\n",
" if '别名' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['别名']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['别名'] = list1\n",
" if '功能' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['功能']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['功能'] = list1\n",
" if '主治' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" dict1[x['name']]['主治'] = s = x['about']['主治']\n",
" \n",
"\n",
"filename = './file/中草药主治功能统计.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) "
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 药剂数据处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"yaoji\"]\n",
"\n",
"dict1 = {}\n",
"mo = '等$'\n",
"mo2 = r'[,、;。]'\n",
"\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n",
" dict1.setdefault(x['name'],{})\n",
" if '功用' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['功用']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['功用'] = list1\n",
" if '主治' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" dict1[x['name']]['主治'] = s = x['about']['主治']\n",
" \n",
"\n",
"filename = './file/药剂主治功用统计.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) "
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 药膳数据处理"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')\n",
"mydb = myclient['dayi']\n",
"mycol = mydb[\"yaoshan\"]\n",
"\n",
"dict1 = {}\n",
"mo = '等$'\n",
"mo2 = r'[,、;。]'\n",
"\n",
"for x in mycol.find({},{ \"_id\": 0,\"name\":1,\"about\":1 }):\n",
" dict1.setdefault(x['name'],{})\n",
" if '功效' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['功效']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['功效'] = list1\n",
" if '相关疾病' in x['about'].keys():\n",
" #dict1[x['name']].setdefault('主治',[])\n",
" s = x['about']['相关疾病']\n",
" s1 = re.sub(mo, '', s)\n",
" list1 = re.split(mo2,s1)\n",
" if '' in list1:\n",
" list1.remove('')\n",
" dict1[x['name']]['相关疾病'] = list1 \n",
" \n",
"\n",
"filename = './file/药膳功能统计.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl) "
]
},
{
"cell_type": "markdown",
"metadata": {
"tags": []
},
"source": [
"### 北海炼化体检数据提取"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
"import os,sys,shutil\n",
"\n",
"file_name = 'file/中石化北海炼化2020年团体体检报告.xlsx'\n",
"wb = openpyxl.load_workbook(file_name)\n",
"sheet = wb.active\n",
"dict1 = {}\n",
"for n in range(1,sheet.max_row+1):\n",
" bh = sheet.cell(n,2).value\n",
" name = sheet.cell(n,3).value\n",
" xb = sheet.cell(n,4).value\n",
" nl = sheet.cell(n,5).value\n",
" bz = sheet.cell(n,7).value\n",
" dict1.setdefault(bh,{})\n",
" dict1[bh]['姓名'] = name\n",
" dict1[bh]['性别'] = xb\n",
" dict1[bh]['年龄'] = nl\n",
" dict1[bh].setdefault('病症',[])\n",
" dict1[bh]['病症'].append(bz.strip())\n",
"wb.close()\n",
"#print(dict1)\n",
" \n",
" \n",
"filename = 'file/中石化北海炼化2020年体检人员情况表.xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"sheet['A1'] = '体检编号'\n",
"sheet['B1'] = '姓名'\n",
"sheet['C1'] = '性别'\n",
"sheet['D1'] = '年龄'\n",
"sheet['E1'] = '异常名称'\n",
"\n",
"i =2\n",
"for k, v in dict1.items():\n",
" sheet[f'A{i}'] = k\n",
" sheet[f'B{i}'] = v['姓名']\n",
" sheet[f'C{i}'] = v['性别']\n",
" sheet[f'D{i}'] = v['年龄']\n",
" sheet[f'E{i}'] = ','.join(v['病症']) \n",
" i += 1\n",
"wb.save(filename)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## excel数据读取"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import openpyxl\n",
"import json\n",
"\n",
"filename = 'data/长岭全成绩.xlsx'\n",
"wb = openpyxl.load_workbook(filename)\n",
"sheet = wb.active\n",
"data1 =list(sheet.values)\n",
"del data1[0]\n",
"print(data1)\n",
"#for data in data1:\n",
"# print(data)\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 图表生成"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 高考一分一段表生成"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from pyecharts.globals import CurrentConfig, NotebookType\n",
"CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n",
"from pyecharts import options as opts\n",
"from pyecharts.charts import Bar,Line\n",
"import pyecharts.options as opts\n",
"from pyecharts.faker import Faker\n",
"import json\n",
"\n",
"list_x = []\n",
"list_y = []\n",
"filename = 'data/17-21年一分一段表.json'\n",
"with open(filename,'r') as fl:\n",
" m_xx = json.load(fl)\n",
"dict1 =m_xx['2020']['z']\n",
"for i in sorted(dict1,reverse=True): #降序\n",
" list_x.append(i)\n",
" list_y.append(dict1[i]['num_person'])\n",
" #print(i,dict1[i]['num_person'])\n",
"bar = (\n",
" Bar()\n",
" .add_xaxis(list_x)\n",
" .add_yaxis(\"2020年一分一段表\", list_y, category_gap=0, color=Faker.rand_color())\n",
" .set_series_opts(label_opts=opts.LabelOpts(is_show=False))\n",
" .set_global_opts(title_opts=opts.TitleOpts(title=\"Bar-直方图\"))\n",
" .render(\"bar_histogram2020.html\")\n",
"# .set_global_opts(title_opts=opts.TitleOpts(title=\"运动步幅及步频\", subtitle=\"户外运动\"),)\n",
")\n",
"#bar.load_javascript()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"bar.render_notebook()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pyecharts.globals import CurrentConfig, NotebookType\n",
"CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n",
"from pyecharts import options as opts\n",
"from pyecharts.charts import Bar,Line\n",
"import pyecharts.options as opts\n",
"from pyecharts.faker import Faker\n",
"import json\n",
"\n",
"x_2021 = []\n",
"y_2021 = []\n",
"x_2020 = []\n",
"y_2020 = []\n",
"filename = 'data/17-21年一分一段表.json'\n",
"with open(filename,'r') as fl:\n",
" m_xx = json.load(fl)\n",
"dict1 =m_xx['2021']['z']\n",
"for i in sorted(dict1,reverse=True): #降序\n",
" x_2021.append(i)\n",
" y_2021.append(dict1[i]['num_person'])\n",
"dict1 =m_xx['2020']['z']\n",
"for i in sorted(dict1,reverse=True): #降序\n",
" x_2021.append(i)\n",
" y_2021.append(dict1[i]['num_person'])\n",
" #print(i,dict1[i]['num_person'])\n",
"bar = (\n",
" Bar()\n",
" .add_xaxis(list_x)\n",
" .add_yaxis(\"人数\", list_y, category_gap=0, color=Faker.rand_color())\n",
" .set_series_opts(label_opts=opts.LabelOpts(is_show=False))\n",
" .set_global_opts(title_opts=opts.TitleOpts(title=\"Bar-直方图\"))\n",
" .render(\"bar_histogram.html\")\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 生成雷达图"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from pyecharts.globals import CurrentConfig, NotebookType\n",
"import pyecharts.options as opts\n",
"CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n",
"from pyecharts.charts import Radar\n",
"\n",
"\n",
"v1 = [[90, 100, 80, 76, 88, 95]]\n",
"v2 = [[5000, 14000, 28000, 31000, 42000, 21000]]\n",
"\n",
"bar =(\n",
" Radar(init_opts=opts.InitOpts())\n",
" .add_schema(\n",
" schema=[\n",
" opts.RadarIndicatorItem(name=\"销售(sales)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"管理(Administration)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"信息技术(Information Technology)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"客服(Customer Support)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"研发(Development)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"市场(Marketing)\", max_=100),\n",
" ],\n",
" splitarea_opt=opts.SplitAreaOpts(\n",
" is_show=True, areastyle_opts=opts.AreaStyleOpts(opacity=1)\n",
" ),\n",
" textstyle_opts=opts.TextStyleOpts(color=\"#aaa\"),\n",
" )\n",
" .add(\n",
" series_name=\"预算分配(Allocated Budget)\",\n",
" data=v1,\n",
" linestyle_opts=opts.LineStyleOpts(color=\"#CD0000\"),\n",
" )\n",
" \n",
" .set_series_opts(label_opts=opts.LabelOpts(is_show=False))\n",
" .set_global_opts(\n",
" title_opts=opts.TitleOpts(title=\"基础雷达图\"), legend_opts=opts.LegendOpts()\n",
" )\n",
" #.render(\"basic_radar_chart.html\")\n",
" \n",
")\n",
"bar.load_javascript()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"bar.render_notebook()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from pyecharts.globals import CurrentConfig, NotebookType\n",
"import pyecharts.options as opts\n",
"CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n",
"from pyecharts.charts import Radar\n",
"from pyecharts.render import make_snapshot\n",
"from snapshot_phantomjs import snapshot\n",
"\n",
"\n",
"v1 = [[90, 100, 80, 76, 88, 95]]\n",
"v2 = [[5000, 14000, 28000, 31000, 42000, 21000]]\n",
"\n",
"bar =(\n",
" Radar(init_opts=opts.InitOpts())\n",
" .add_schema(\n",
" schema=[\n",
" opts.RadarIndicatorItem(name=\"销售(sales)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"管理(Administration)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"信息技术(Information Technology)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"客服(Customer Support)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"研发(Development)\", max_=100),\n",
" opts.RadarIndicatorItem(name=\"市场(Marketing)\", max_=100),\n",
" ],\n",
" splitarea_opt=opts.SplitAreaOpts(\n",
" is_show=True, areastyle_opts=opts.AreaStyleOpts(opacity=1)\n",
" ),\n",
" textstyle_opts=opts.TextStyleOpts(color=\"#aaa\"),\n",
" )\n",
" .add(\n",
" series_name=\"预算分配(Allocated Budget)\",\n",
" data=v1,\n",
" linestyle_opts=opts.LineStyleOpts(color=\"#CD0000\"),\n",
" )\n",
" \n",
" .set_series_opts(label_opts=opts.LabelOpts(is_show=False))\n",
" .set_global_opts(\n",
" title_opts=opts.TitleOpts(title=\"基础雷达图\"), legend_opts=opts.LegendOpts()\n",
" )\n",
" #.render(\"basic_radar_chart.html\")\n",
" \n",
")\n",
"make_snapshot(snapshot, bar.render(), \"bar0.png\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pygal \n",
"\n",
"radar_chart = pygal.Radar()\n",
"radar_chart.title = 'V8 benchmark results'\n",
"radar_chart.x_labels = ['Richards', 'DeltaBlue', 'Crypto', 'RayTrace', 'EarleyBoyer', 'RegExp', 'Splay', 'NavierStokes']\n",
"radar_chart.add('Chrome', [6395, 8212, 7520, 7218, 12464, 1660, 2123, 8607])\n",
"\n",
"radar_chart.render_to_png('chart.png')"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.12.3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}