Files
jupyter/数据处理.ipynb
T
2024-03-12 06:18:14 +00:00

59 KiB

基础知识

正则表达式分割文本

In [ ]:
import json
import openpyxl
import re

file_name = 'data/药食材性味归经.xlsx'
wb = openpyxl.load_workbook(file_name)
sheet = wb.active
#sheets = wb.sheetnames

list1 = []
dict1 = {}
mo =r'[。入归].+经$'
mo1 = r'[二]'
mo2 = r'[,、;]'
for n in range(2,sheet.max_row):
    name = sheet.cell(n,1).value
    content = re.findall(mo,sheet.cell(n,2).value)
    if len(content) > 0:
        l = len(content[0])
        gj = re.sub(mo1, '', content[0][1:l-1])        
        list_gj = re.split(mo2,gj)
        dict1[sheet.cell(n,1).value ] = list_gj
        #dict1['guijing'] = list_gj
        
filename = './data/药食材归经.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl,ensure_ascii=False) 

#print(list1)
In [ ]:
import re

s = '消化不良、胃下垂、急性胃炎、慢性胃炎、萎缩性胃炎、神经性呕吐、胆囊炎、胆石症、胆道蛔虫症、胸胁痛等'
ss = '健中和胃,消食止呕,理气疏郁,清热利胆。'
mo = '等$'
mo2 = r'[,、;。]'
s1 = re.sub(mo, '', s) 
list1 = re.split(mo2,s1)
if '' in list1:
    list1.remove('')
print(list1)
In [ ]:
import re
t = '5小时10分48秒'
m = re.match(r'(.*)小时(.*)分(.*)秒', t)
m.groups()
In [ ]:
import re
t = '10分48秒'
list1 = []
if '小时' in t and '分' in t:
    m = re.match(r'(.*)小时(.*)分(.*)秒', t)
    list1 = [m[1],m[2],m[3]]
elif '小时' in t:
    m = re.match(r'(.*)小时(.*)秒', t)
    list1 = [m[1],0,m[2]]
elif '分' in t:
    m = re.match(r'(.*)分(.*)秒', t)
    list1 = [0,m[1],m[2]]
print(list1)
#m.group()
In [ ]:
import re

s = '2022-09-28-《气郁组》-视频学习详情_155229'
m = re.findall(r'《(.+)》',s)
print(m)

日期计算

In [ ]:
import time

birth = '1989-01-25'
t_birth = time.strptime(birth,'%Y-%m-%d')
days = (time.time() -time.mktime(t_birth))//(365*24*60*60)
print(int(days))

字符串转换

In [ ]:
import binascii

gbs = 'D4C0CDAF'
bs = binascii.a2b_hex(gbs)
print('bs', bs)
print('decode-bs:', bs.decode('gbk'))

s = '马立亚'
gbcode = s.encode('gbk') # 先转成 bytes格式
print('gbcode:', gbcode)
gbs = "".join([hex(ch)[2:] for ch in gbcode])  #
print('gbs:', gbs)

数据组合处理

In [ ]:
import itertools

list1 = ['气虚','阳虚','阴虚','痰湿','湿热','血瘀','气郁','特禀']
list2 = [0,1,2,3,4,5,6,7]

result=itertools.combinations(list1,7)
print(len(list(result)))

医药体测

药膳归经明细文件生成

In [ ]:
import json

filename = 'data/药膳210927.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
filename = 'data/药食材归经.json'
with open(filename,'r') as fl:
    dict2 = json.load(fl)
k_zy = dict2.keys()
dict3 = {}
list1 = []
for item in dict1:
    print(item['name'])
    list1 = []
    #print(i,item['zy'])
    for m_zy in item['zy']:
        if m_zy in k_zy:
            dict4 = {}
            print(m_zy,dict2[m_zy])
            dict4[m_zy] = dict2[m_zy]
            list1.append(dict4)
    dict3[item['name']] = list1
filename = './data/药膳药食材.json'
with open(filename,'w') as fl:
    json.dump(dict3, fl,ensure_ascii=False)             
   

药膳归经权重生成

In [ ]:
import json
import openpyxl
import os,sys,shutil

filename = './data/药膳药食材.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
list1 = []
for k, v in dict1.items():
    if len(v) > 0 :
        dict2 = {}
        #print(k,v)
        ys_name = k
        dict2.setdefault(ys_name,{})
        for item in v:
            for k1, v1 in item.items():
                for item1 in v1:
                    dict2[ys_name].setdefault(item1,0)
                    dict2[ys_name][item1] += 1
        list1.append(dict2)                    
    
dict_qz = {}
for item in list1:
    for k, v in item.items():
        qz = sorted(v.items(), key = lambda kv:(kv[1], kv[0]),reverse=True)
        dict_qz[k] = qz
#print(dict_qz)
qz_key = dict_qz.keys()
filename = 'data/药膳210927.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
#print(dict1)
wb = openpyxl.Workbook()
sheet = wb.active
sheet['A1'] = '药膳名称'
sheet['B1'] = '来源'
sheet['C1'] = '配方'
sheet['D1'] = '做法'
sheet['E1'] = '功效'
sheet['F1'] = '中药成分'
sheet['G1'] = '归经权重'

i =2
for item in dict1:
    sheet[f'A{i}'] = item['name']
    sheet[f'B{i}'] = item['source']
    sheet[f'C{i}'] = item['pf']
    sheet[f'D{i}'] = item['zf']
    sheet[f'E{i}'] = item['gx']
    sheet[f'F{i}'] = ','.join(item['zy'])
    if item['name'] in qz_key:
        s = ''
        for m_gj in dict_qz[item['name']]:
            s = s+ m_gj[0] +'('+str(m_gj[1])+')'
        sheet[f'G{i}'] = s
    else:
        sheet[f'G{i}'] = '暂无归经'
    i += 1

wb.save('data/test4.xlsx')    

药膳文件处理

药膳文件预处理

In [ ]:
import re
import json

filename = 'data/286种药膳常用中药功能表.txt'
with open(filename, "r", encoding='utf-8') as f:  
    data = f.readlines()
dict1 = {}
for i in range(0,int(len(data)/7)) :    
    s = data[i*7].strip()
    bh = s.split('.')[0]
    name = s.split('.')[1].split('(')[0]
    #print(bh,name)
    id = str(i+1)
    dict1.setdefault(id,{})
    dict1[id]['name'] = name
    for n in range(1,7):
        ss = data[i*7+n].strip()
        p = re.compile(r'【(.*?)】')
        item = re.findall(p,ss)[0]
        content = ss.split('】')[1]
        dict1[id][item] = content
filename = 'data/286种药膳常用中药功能表.json'
with open(filename, 'w') as fl:
    json.dump(dict1, fl, ensure_ascii=False)
print('ok')

性味归经分解

In [164]:
import re
import json

filename = 'data/286种药膳常用中药功能表.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
p = re.compile(r'[入归](.*?)经')
for k, v in dict1.items():
    if '性味归经' in v.keys():
        s = v['性味归经']
        item = re.search(p,s)
        if item is not None:
            ss = item.group()
            sss = re.sub(ss,'',s)
            
            p1 = r'[;:、,。:]+。'
            ssss = re.sub(p1,'。',sss)
            #print(k,ss,ssss)
            dict1[k]['归经'] = ss
            dict1[k]['性味'] = ssss
        else:
            dict1[k]['归经'] = s
       
filename = 'data/286种药膳常用中药功能表1.json'
with open(filename, 'w') as fl:
    json.dump(dict1, fl, ensure_ascii=False)
print('ok')        
ok
In [163]:
import re
import json

filename = 'data/286种药膳常用中药功能表.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
p = re.compile(r'[入归](.*?)经')
for k, v in dict1.items():
    if '性味归经' in v.keys():
        s = v['性味归经']
        item = re.search(p,s)
        #print(item)
        if item is not None:
            ss = item.group()
            sss = re.sub(ss,'',s)
            
            p1 = r'[;:、,。:]+。'
            ssss = re.sub(p1,'。',sss)
            print(k,ssss)
1 性温,味甘、苦。
2 性平,味甘。
3 性凉,味甘、微苦。
4 性平,味甘、微苦。
5 性温,味甘。
6 性温,味苦、甘。
7 性平,味甘。
8 性温、味甘。
9 性平,味甘。
10 性寒、味苦。
11 味甘、苦,性寒。
12 性微温,味甘。
13 性温,味甘。
14 性平,味甘。
15 性温,味甘、辛。
16 性微温、味甘。
17 性凉,味苦、酸。
18 性温,味苦、甘、涩。
19 性平、味甘。
20 性温、味甘。
21 性凉,味甘、苦。
22 性寒,味甘、苦。
23 性寒,味甘、苦。
24 性平,味甘。
25 性平,味甘、微苦。
26 性平、味甘。
27 性平,味甘。
28 性平、味甘。
29 性微寒,味甘。
30 味甘、性平。
31 性寒、味甘。
32 性平,味苦、甘。
33 性寒,味甘、酸。
34 性寒,味甘。
35 性平,味咸。
36 性温,味甘、咸。
37 性平,味甘、咸。
38 性温,味甘。
39 性温、味甘。
40 性温,味辛、甘。
41 性温、味辛、有小毒。
42 性温,味辛。
43 性温、味辛。
44 味甘、咸,性温。
45 性温、味甘。
46 性温,味甘、咸。
47 性平、味咸。
48 性平,味辛、甘。
49 性温,味甘、微辛。
50 性微温,味苦、辛。
51 性温,味辛、甘。
52 味辛、苦,性温。
53 性温,味辛、甘。
54 性温、味辛。
55 性温、味辛。
56 性温、味辛。
57 性温、味辛。
58 味辛、性温。
59 性温,味辛、甘。
60 性温、味辛。
61 性温,味辛、苦。
62 性温,味甘、苦。
63 性温,味辛。
64 性温、味辛。
65 性温,味辛。
66 性凉、味辛。
67 性平,味辛。
68 性寒,味甘、咸。
69 性寒,味甘、苦。
70 性微寒,味甘、苦。
71 性微寒、味苦。
72 性凉,味甘、辛、微苦。
73 性凉,味甘。
74 性寒、味苦。
75 性寒,味辛。
76 性平,味苦。
77 性寒,味甘、辛。
78 性寒,味苦、甘。
79 性寒、味甘。
80 性凉,味甘、苦、酸。
81 性寒,味甘、淡。
82 性寒,味甘。
83 性寒、味苦。
84 性寒、味苦。
85 性寒,味苦、辛。
86 性凉,味甘、苦。
87 性凉,味苦。
88 性微寒,味苦。
89 性寒、味甘。
90 性寒,味苦。
91 性寒、味苦。
92 性寒,味苦、甘。
93 性微寒,味苦、辛。
94 性寒、味辛。
95 味甘、酸,性寒。
96 性平,味甘、淡。
97 性寒、味苦。
98 性寒,味苦。
99 性微寒,味苦,有小毒。
100 性平、味甘。
101 性寒,味辛、苦。
102 性寒,味甘、苦。
103 性寒,味辛。
104 性寒,味苦。
105 性寒、味苦。
106 性温,味酸涩、甘。
107 性凉,味甘、淡。
108 性寒、味苦。
109 性寒、味苦。
110 性寒、味苦。
111 味苦,性寒。
112 性寒、味苦。
113 性寒、微苦。
114 性寒,味苦、咸。
115 性寒,味苦、涩。
116 性寒,味甘、苦。
117 性微寒,味甜、微苦。
118 性凉,味辛、苦。
119 性微寒、味苦。
120 性寒,味甘、咸。
121 性寒、味苦。
122 性寒,味苦、微辛。
123 性寒,味苦、咸。
124 性寒、味甘。
125 性凉,味甘、苦。
126 性寒、味苦。
127 性温,味辛、苦。
128 苦,寒。
129 性平,味苦。
130 味苦、微辛,性平。
131 性寒,味苦、辛。
132 性温、味辛。
133 性平、味苦。
134 性温,苦、甘。
135 性温,味辛、咸。
136 性凉,味甘。
137 性温,味甘、咸。
138 味辛、苦,性微温。
139 性温,味苦、辛。
140 性平,味苦、辛。
141 性平,味苦、辛。
142 性温,味苦、辛。
143 性温,味辛。
144 性平,味辛。
145 性温,味辛、苦。
146 性温,味辛、苦。
147 性温,味辛。
148 性温、味辛。
149 性温、味辛。
150 性温,味辛。
151 性平,味甘、淡。
152 性平、味酸。
153 性寒、味甘。
154 性凉,味甘。
155 性平,味甘。
156 性凉、味甘。
157 性平、味甘。
158 性寒,味甘。
159 性寒,味苦。
160 性凉,味甘、淡。
161 性寒,味苦。
162 性微寒,味苦。
163 性寒,味辛、苦。
164 性寒,味甘。
165 性微寒,味甘、苦。
166 性寒,味甘、淡。
167 性平,味苦。
168 性微寒,味辛、苦。
169 性凉,味苦、辛。
170 性平,味苦。
171 性凉,味甘。
172 性热,味辛、甘。
173 性热,味辛、甘。
174 性热、味辛。
175 性温,味辛、苦。
176 性温、味辛。
177 味辛,性温。
178 性温、味辛。
179 性温,味苦、辛。
180 性微温,味苦、辛。
181 性寒、味苦。
182 性平,味辛、微苦、甘。
183 性温,味苦。
184 性温、味辛。
185 性温,味辛。
186 性寒、味苦。
187 性温,味辛。
188 性温,味甘、苦。
189 性温,味辛、苦。
190 性微温,味辛。
191 性平,味苦。
192 性温,味辛、苦。
193 性温,味辛、苦。
194 性温,味酸、甘。
195 性温,味甘、辛。
196 性微温、味甘。
197 性温,味甘。
198 性平,味辛、甘。
199 性平、味甘。
200 性温、味辛。
201 性微温、味苦。
202 性温、味辛。
203 性平,味苦、甘。
204 性凉,味辛、苦。
205 性平,味甘、苦。
206 性温,味苦、甘。
207 性温,味辛、苦。
208 性温,味苦、甘。
209 性凉,味辛、苦。
210 性温,味辛、苦。
211 性温,味辛、苦。
212 性平、味苦。
213 性温,味苦、辛。
214 性平,味苦。
215 性温,味甘。
216 性平,味甘、咸。
217 性温、味苦。
218 性温,味苦、辛。
219 性平,味咸、苦,有毒。
220 性凉,味咸。
221 性凉,味甘。
222 性微寒,味苦。
223 性寒、味甘。
224 性寒,味苦。
225 性温,味甘、微苦。
226 性凉,味苦、甜。
227 性平,味甘、涩。
228 性温,味苦、辛。
229 性热,味辛。
230 性温,味辛。
231 性温,味辛。
232 性微温,味苦、辛、咸。
233 性平,味苦、辛。
234 性凉,味苦、甘。
235 性寒,味甘、微苦。
236 性寒、味甘。
237 性寒,味苦、咸。
238 性寒,味咸。
239 性微寒,味苦、辛。
240 味辛,性温:、脾经。
241 性温、味辛。
242 性平,味甘、苦。
243 性寒,味甘。
244 性凉,味苦。
245 性微温,味甘、苦。
246 性凉、味甘。
247 性温,味淡、苦。
248 性平、味甘。
249 性平,味甘。
250 性微温,味苦、辛。
251 性平,味甘。
252 性平,味甘、苦。
253 性微寒,味甘:有毒。
254 性平,味甘、涩。
255 性寒,味咸。
256 性凉,味咸、湿。
257 性寒,味苦。
258 性平,味辛、苦。
259 性凉,味甘、苦。
260 性寒,味咸。
261 性凉,味甘。
262 性平、味甘。
263 性寒,味咸。
264 性平,味减、辛。
265 性平,味咸、辛。
266 性温,味辛。
267 性温,味辛、苦。
268 性微寒,味辛、苦。
269 性凉,味甘、咸。
270 性温,味酸。
271 性温,味酸。
272 性温,味苦、酸涩。
273 性温,味酸、涩。
274 性微温,味酸。
275 性平,味甘、酸。
276 性平,味酸、涩。
277 性平,味咸、甘。
278 性温,味减、涩。
279 性平,味甘、涩。
280 性平,味甘、涩。
281 性寒,味苦。
282 性大寒,味甘、苦。
283 性寒,味苦。
284 性平、味甘。
285 性平,味辛、苦、甘。
286 性平,味甘。
In [ ]:

体质数据处理

体质对应数据导入

In [ ]:
import json

filename = 'data/tijianbingzheng.txt'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
m_key = set()
for k, v in dict1.items():
    for item in v.keys():
        m_key.add(item)
print(dict1)

穴位数据导入

简单导出简介、内容

In [ ]:
import json
import re
dict1 = {}
filename = 'file/zhongyi/xuewei.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        dict2 = dict(zip(list1, list2))
        #print(line1['title'],dict1)
        #print(dict1.keys())
        #print(line1)
        dict1.setdefault(line1['title'][0],{})
        dict1[line1['title'][0]]['简介'] = line1['jj'] 
        dict1[line1['title'][0]]['内容'] = dict2
filename = './file/穴位1.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)        

数据导入文件中

In [ ]:
import json
import re

dict1 = {}
filename = 'file/zhongyi/xuewei.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1.setdefault(line1['title'][0],{})
        if len(list3) > 0:
            
            dict1[line1['title'][0]]['about'] = list3
        dict1[line1['title'][0]].setdefault('content',{})
        for k, v in dict2.items():
            dict1[line1['title'][0]]['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
filename = './file/穴位1.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)        

穴位数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
#dblist = myclient.list_database_names()
mydb = myclient['dayi']
mycol = mydb["xuewei"]
db_list = []
filename = 'file/zhongyi/xuewei.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for s in list3:
                item = s.strip().split(':')
                dict1['about'][item[0].strip()] = item[1].strip()
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

中医症状数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["zhongyizhengzhuang"]
db_list = []
filename = 'file/zhongyi/zhongyizhengzhuang.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for i in range(0,int(len(list3)/2)):
                dict1['about'][list3[2*i]] = list3[2*i+1]
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

疾病数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["jibing"]
db_list = []
filename = 'file/zhongyi/jibing.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for i in range(0,int(len(list3)/2)):
                dict1['about'][list3[2*i]] = list3[2*i+1]
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

术语数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
#dblist = myclient.list_database_names()
mydb = myclient['dayi']
mycol = mydb["shuyu"]
db_list = []
filename = 'file/zhongyi/shuyu.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for s in list3:
                item = s.strip().split(':')
                dict1['about'][item[0]] = item[1]
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

西医症状数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["xiyizhengzhuang"]
db_list = []
filename = 'file/zhongyi/xiyizhengzhuang.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for i in range(0,int(len(list3)/2)):
                dict1['about'][list3[2*i]] = list3[2*i+1]
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

药剂数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
#dblist = myclient.list_database_names()
mydb = myclient['dayi']
mycol = mydb["yaoji"]
db_list = []
filename = 'file/zhongyi/yaoji.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']        
        dict2 = dict(zip(list1, list2))        
        if len(list3) > 0: 
            jj = {}
            for s in list3:
                item = s.strip().split(':')
                jj[item[0]] = item[1]        
        dict1['name'] = jj['名称']
        dict1['about'] = jj
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

药膳数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
#dblist = myclient.list_database_names()
mydb = myclient['dayi']
mycol = mydb["yaoshan"]
db_list = []
filename = 'file/zhongyi/yaoshan.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for s in list3:
                item = s.strip().split(':')
                dict1['about'][item[0]] = item[1]
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

中草药数据导入数据库中

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["zhongcaoyao"]
db_list = []
filename = 'file/zhongyi/zhongcaoyao.json'
with open(filename,'r',encoding='utf-8') as fl:
    for line in fl:
        dict1 = {}
        line1 = json.loads(re.sub(r'\xa0','',line))
        list1 = line1['tables']
        list2 = line1['contents']
        list3 = line1['jj']
        dict2 = dict(zip(list1, list2))
        dict1['name'] = line1['title'][0]
        if len(list3) > 0:
            dict1.setdefault('about',{})
            for i in range(0,int(len(list3)/2)):
                dict1['about'][list3[2*i]] = list3[2*i+1]
        dict1.setdefault('content',{})
        for k, v in dict2.items():
            dict1['content'][k] = v 
        #dict1[line1['title'][0]]['内容'] = dict2
        db_list.append(dict1)
x = mycol.insert_many(db_list)
print('ok!')

穴位隶属整理

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["xuewei"]
dict1 = {}
list1 = []
for x in mycol.find({},{ "_id": 0,"name":1,"about.隶属":1 }):
    if 'about' in x.keys():
        m_ls = x['about']['隶属']
        dict1.setdefault(m_ls,[])
        dict1[m_ls].append(x['name'])
'''
filename = './file/穴位隶属.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)     
'''
for item in dict1.keys():
    print(item)

穴位功能、主治统计

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["xuewei"]

dict1 = {}
mo = '等$'
mo2 = r'[,、;。]'

for x in mycol.find({},{ "_id": 0,"name":1,"about":1 }):
    dict1.setdefault(x['name'],{})
    if '主治' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['主治']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['主治'] = list1
    if '功能' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['功能']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['功能'] = list1
        

filename = './file/穴位主治功能统计.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)     

穴位数据导出Excel表

In [ ]:
import json
import openpyxl

filename = 'file/穴位.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
filename = 'file/穴位简要情况表.xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
sheet['A1'] = '穴位名称'
sheet['B1'] = '隶属'
sheet['C1'] = '位置'
sheet['D1'] = '主治'
sheet['E1'] = '功能'
sheet['F1'] = '操作'
sheet['G1'] = '主要配伍'
i = 2
for k, v in dict1.items():
    sheet[f'A{i}'] = k
    sheet[f'B{i}'] = v['简介'][0].split(':')[1]
    sheet[f'C{i}'] = v['简介'][1].split(':')[1]
    sheet[f'D{i}'] = v['简介'][2].split(':')[1]
    sheet[f'E{i}'] = v['简介'][3].split(':')[1]
    sheet[f'F{i}'] = v['简介'][4].split(':')[1]
    sheet[f'G{i}'] = v['简介'][5].split(':')[1]      
    i += 1
wb.save(filename)   
print('ok!')


术语数据处理

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["shuyu"]
dict1 = {}
list1 = []
for x in mycol.find({},{ "_id": 0,"name":1,"about":1 }):
    if '类别' in x['about'].keys():
        m_lb = x['about']['类别'].replace(' ','')
    else:
        m_lb = '无类别'
    dict1.setdefault(re.sub('\xa0+','',m_lb),[])
    dict1[m_lb].append(x['name'])
filename = './file/术语类别.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)   

疾病数据处理

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["jibing"]
dict1 = {}
list1 = []
for x in mycol.find({},{ "_id": 0,"name":1,"about":1 }):
    if '疾病分类' in x['about'].keys():
        m_lb = x['about']['疾病分类'].replace(' ','')
    else:
        m_lb = '无类别'
    dict1.setdefault(re.sub('\xa0+','',m_lb),[])
    dict1[m_lb].append(x['name'])
filename = './file/疾病类别.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)   

中草药数据处理

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["zhongcaoyao"]

dict1 = {}
mo = '等$'
mo2 = r'[,、;。]'

for x in mycol.find({},{ "_id": 0,"name":1,"about":1 }):
    dict1.setdefault(x['name'],{})
    if '别名' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['别名']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['别名'] = list1
    if '功能' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['功能']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['功能'] = list1
    if '主治' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        dict1[x['name']]['主治'] = s = x['about']['主治']
        

filename = './file/中草药主治功能统计.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)     

药剂数据处理

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["yaoji"]

dict1 = {}
mo = '等$'
mo2 = r'[,、;。]'

for x in mycol.find({},{ "_id": 0,"name":1,"about":1 }):
    dict1.setdefault(x['name'],{})
    if '功用' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['功用']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['功用'] = list1
    if '主治' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        dict1[x['name']]['主治'] = s = x['about']['主治']
        

filename = './file/药剂主治功用统计.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)     

药膳数据处理

In [ ]:
import json
import re
import pymongo

myclient = pymongo.MongoClient('mongodb://10.147.17.6:27017/')
mydb = myclient['dayi']
mycol = mydb["yaoshan"]

dict1 = {}
mo = '等$'
mo2 = r'[,、;。]'

for x in mycol.find({},{ "_id": 0,"name":1,"about":1 }):
    dict1.setdefault(x['name'],{})
    if '功效' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['功效']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['功效'] = list1
    if '相关疾病' in x['about'].keys():
        #dict1[x['name']].setdefault('主治',[])
        s = x['about']['相关疾病']
        s1 = re.sub(mo, '', s)
        list1 = re.split(mo2,s1)
        if '' in list1:
            list1.remove('')
        dict1[x['name']]['相关疾病'] = list1    
        

filename = './file/药膳功能统计.json'
with open(filename,'w') as fl:
    json.dump(dict1, fl)     

北海炼化体检数据提取

In [ ]:
import json
import openpyxl
import os,sys,shutil

file_name = 'file/中石化北海炼化2020年团体体检报告.xlsx'
wb = openpyxl.load_workbook(file_name)
sheet = wb.active
dict1 = {}
for n in range(1,sheet.max_row+1):
    bh = sheet.cell(n,2).value
    name = sheet.cell(n,3).value
    xb = sheet.cell(n,4).value
    nl = sheet.cell(n,5).value
    bz = sheet.cell(n,7).value
    dict1.setdefault(bh,{})
    dict1[bh]['姓名'] = name
    dict1[bh]['性别'] = xb
    dict1[bh]['年龄'] = nl
    dict1[bh].setdefault('病症',[])
    dict1[bh]['病症'].append(bz.strip())
wb.close()
#print(dict1)
    
        
filename = 'file/中石化北海炼化2020年体检人员情况表.xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
sheet['A1'] = '体检编号'
sheet['B1'] = '姓名'
sheet['C1'] = '性别'
sheet['D1'] = '年龄'
sheet['E1'] = '异常名称'

i =2
for k, v in dict1.items():
    sheet[f'A{i}'] = k
    sheet[f'B{i}'] = v['姓名']
    sheet[f'C{i}'] = v['性别']
    sheet[f'D{i}'] = v['年龄']
    sheet[f'E{i}'] = ','.join(v['病症'])    
    i += 1
wb.save(filename)

excel数据读取

In [ ]:
import openpyxl
import json

filename = 'data/长岭全成绩.xlsx'
wb = openpyxl.load_workbook(filename)
sheet = wb.active
data1 =list(sheet.values)
del data1[0]
print(data1)
#for data in data1:
#    print(data)

图表生成

高考一分一段表生成

In [ ]:
from pyecharts.globals import CurrentConfig, NotebookType
CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB
from pyecharts import options as opts
from pyecharts.charts import Bar,Line
import pyecharts.options as opts
from pyecharts.faker import Faker
import json

list_x = []
list_y = []
filename = 'data/17-21年一分一段表.json'
with open(filename,'r') as fl:
    m_xx = json.load(fl)
dict1 =m_xx['2020']['z']
for i in sorted(dict1,reverse=True): #降序
    list_x.append(i)
    list_y.append(dict1[i]['num_person'])
    #print(i,dict1[i]['num_person'])
bar = (
    Bar()
    .add_xaxis(list_x)
    .add_yaxis("2020年一分一段表", list_y, category_gap=0, color=Faker.rand_color())
    .set_series_opts(label_opts=opts.LabelOpts(is_show=False))
    .set_global_opts(title_opts=opts.TitleOpts(title="Bar-直方图"))
    .render("bar_histogram2020.html")
#    .set_global_opts(title_opts=opts.TitleOpts(title="运动步幅及步频", subtitle="户外运动"),)
)
#bar.load_javascript()
In [ ]:
bar.render_notebook()
In [ ]:
from pyecharts.globals import CurrentConfig, NotebookType
CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB
from pyecharts import options as opts
from pyecharts.charts import Bar,Line
import pyecharts.options as opts
from pyecharts.faker import Faker
import json

x_2021 = []
y_2021 = []
x_2020 = []
y_2020 = []
filename = 'data/17-21年一分一段表.json'
with open(filename,'r') as fl:
    m_xx = json.load(fl)
dict1 =m_xx['2021']['z']
for i in sorted(dict1,reverse=True): #降序
    x_2021.append(i)
    y_2021.append(dict1[i]['num_person'])
dict1 =m_xx['2020']['z']
for i in sorted(dict1,reverse=True): #降序
    x_2021.append(i)
    y_2021.append(dict1[i]['num_person'])
    #print(i,dict1[i]['num_person'])
bar = (
    Bar()
    .add_xaxis(list_x)
    .add_yaxis("人数", list_y, category_gap=0, color=Faker.rand_color())
    .set_series_opts(label_opts=opts.LabelOpts(is_show=False))
    .set_global_opts(title_opts=opts.TitleOpts(title="Bar-直方图"))
    .render("bar_histogram.html")

生成雷达图

In [ ]:
from pyecharts.globals import CurrentConfig, NotebookType
import pyecharts.options as opts
CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB
from pyecharts.charts import Radar


v1 = [[90, 100, 80, 76, 88, 95]]
v2 = [[5000, 14000, 28000, 31000, 42000, 21000]]

bar =(
    Radar(init_opts=opts.InitOpts())
    .add_schema(
        schema=[
            opts.RadarIndicatorItem(name="销售(sales)", max_=100),
            opts.RadarIndicatorItem(name="管理(Administration)", max_=100),
            opts.RadarIndicatorItem(name="信息技术(Information Technology)", max_=100),
            opts.RadarIndicatorItem(name="客服(Customer Support)", max_=100),
            opts.RadarIndicatorItem(name="研发(Development)", max_=100),
            opts.RadarIndicatorItem(name="市场(Marketing)", max_=100),
        ],
        splitarea_opt=opts.SplitAreaOpts(
            is_show=True, areastyle_opts=opts.AreaStyleOpts(opacity=1)
        ),
        textstyle_opts=opts.TextStyleOpts(color="#aaa"),
    )
    .add(
        series_name="预算分配(Allocated Budget)",
        data=v1,
        linestyle_opts=opts.LineStyleOpts(color="#CD0000"),
    )
    
    .set_series_opts(label_opts=opts.LabelOpts(is_show=False))
    .set_global_opts(
        title_opts=opts.TitleOpts(title="基础雷达图"), legend_opts=opts.LegendOpts()
    )
    #.render("basic_radar_chart.html")
    
)
bar.load_javascript()
In [ ]:
bar.render_notebook()
In [ ]:
from pyecharts.globals import CurrentConfig, NotebookType
import pyecharts.options as opts
CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB
from pyecharts.charts import Radar
from pyecharts.render import make_snapshot
from snapshot_phantomjs import snapshot


v1 = [[90, 100, 80, 76, 88, 95]]
v2 = [[5000, 14000, 28000, 31000, 42000, 21000]]

bar =(
    Radar(init_opts=opts.InitOpts())
    .add_schema(
        schema=[
            opts.RadarIndicatorItem(name="销售(sales)", max_=100),
            opts.RadarIndicatorItem(name="管理(Administration)", max_=100),
            opts.RadarIndicatorItem(name="信息技术(Information Technology)", max_=100),
            opts.RadarIndicatorItem(name="客服(Customer Support)", max_=100),
            opts.RadarIndicatorItem(name="研发(Development)", max_=100),
            opts.RadarIndicatorItem(name="市场(Marketing)", max_=100),
        ],
        splitarea_opt=opts.SplitAreaOpts(
            is_show=True, areastyle_opts=opts.AreaStyleOpts(opacity=1)
        ),
        textstyle_opts=opts.TextStyleOpts(color="#aaa"),
    )
    .add(
        series_name="预算分配(Allocated Budget)",
        data=v1,
        linestyle_opts=opts.LineStyleOpts(color="#CD0000"),
    )
    
    .set_series_opts(label_opts=opts.LabelOpts(is_show=False))
    .set_global_opts(
        title_opts=opts.TitleOpts(title="基础雷达图"), legend_opts=opts.LegendOpts()
    )
    #.render("basic_radar_chart.html")
    
)
make_snapshot(snapshot, bar.render(), "bar0.png")
In [ ]:
import pygal 

radar_chart = pygal.Radar()
radar_chart.title = 'V8 benchmark results'
radar_chart.x_labels = ['Richards', 'DeltaBlue', 'Crypto', 'RayTrace', 'EarleyBoyer', 'RegExp', 'Splay', 'NavierStokes']
radar_chart.add('Chrome', [6395, 8212, 7520, 7218, 12464, 1660, 2123, 8607])

radar_chart.render_to_png('chart.png')
In [ ]: