37 KiB
37 KiB
In [ ]:
import openpyxl
import json
wb = openpyxl.load_workbook('data/南化人员信息表.xlsx',data_only=True)
sheet = wb.active
# sheets = wb.sheetnames
person = {}
for n in range(2, sheet.max_row+1):
code = int(sheet.cell(n, 4).value)
person.setdefault(code, {})
dict1 = {}
dict1['name'] = sheet.cell(n, 3).value
dict1['sex'] = sheet.cell(n, 5).value
dict1['unit'] = sheet.cell(n, 2).value
dict1['birth'] = str(sheet.cell(n, 6).value).replace('/','-').split(' ')[0]
dict1['phone'] = sheet.cell(n, 12).value
person[code] = dict1
filename = 'data/南京化工人员.json'
with open(filename, 'w') as fl:
json.dump(person, fl, ensure_ascii=False)
print(len(person),'ok')In [ ]:
import json
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
list1 = []
for k, v in dict1.items():
dict2 = {}
#if dict1['sex'] =='男':
# sex = 1
dict2 = {'id':k,'name':v['name'],'gender':v['sex'],'birth':v['birth'],'unit':v['unit']}
list1.append(dict2)
json_data = json.dumps(list1,ensure_ascii=False, indent=4)
# 将 json 数据写入文件
with open("data/data_南京化工人员.json", "w",encoding = 'utf-8') as file:
file.write(json_data)
print('ok')In [ ]:
import json
import datetime
import csv
from datetime import date
import my_module as My
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
filename = 'data/marks_20250703.csv'
re_ta = My.get_result(filename,dict1)
filename = 'data/result_南京化工.json'
with open(filename,'w') as fl:
json.dump(re_ta, fl, ensure_ascii=False)
print(len(re_ta))In [ ]:
import json
import time
import my_module as My
#list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','height','weight']
list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']
filename = 'data/result_南京化工.json'
with open(filename,'r') as fl:
dict2 = json.load(fl)
for k, v in dict2.items():
#print(k)
if v['sex'] == '男':
sex = 'M'
else:
sex = 'F'
if 'bmi' in v.keys():
#bmi_data = v['height']['成绩'].split()[0]+','+ v['weight']['成绩'].split()[0]
bmi_data = v['bmi']['成绩']
data1 = {'code':k,'sex':sex,'age':v['age'],'item':'HeightWeight','result':bmi_data}
dict2[k]['bmi'] = {}
dict2[k]['bmi']['成绩'] = bmi_data
dict2[k]['bmi']['score'] = My.cal_bmi(data1)
for item_en in list_item:
if item_en in v.keys():
data1 = {'code':k,'sex':sex,'age':v['age'],'item':item_en,'result':float(v[item_en]['成绩'].split()[0])}
#print(k,v['name'])
dict2[k][item_en]['score'] = My.cal_score(data1)
#print(k,v[item_en]['成绩'],cal_score(data1))
filename = f'data/result_南京化工.json'
with open(filename,'w') as fl:
json.dump(dict2,fl , ensure_ascii=False)
print('ok!') In [ ]:
import json
import openpyxl
items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']
title = ['编号','姓名','性别','单位','部门','身高','体重','肺活量','握力','坐位体前屈','纵跳','俯卧撑','一分钟仰卧起坐','单脚站立','选择反应时','台阶指数']
filename = 'data/result_南京化工.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict2 = json.load(fl)
list1 = []
for k, v in dict1.items():
list2 = []
list2.append(str(k).rjust(5,'0'))
list2.append(v['name'])
list2.append(dict2[k]['sex'])
list2.append(dict2[k]['unit'])
if 'bmi' in v.keys():
height = v['bmi']['成绩'].split(',')[0]
weight = v['bmi']['成绩'].split(',')[1]
list2.append(height)
list2.append(weight)
else:
list2.append('')
list2.append('')
for item in items:
if item in v.keys():
list2.append(v[item]['成绩'])
elif item =='name':
list2.append(v[item])
else:
list2.append('')
list1.append(list2)
filename = 'data/南京化工测试情况明细表(截至20250630).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
sheet.append(title)
for row in list1:
sheet.append(row)
wb.save(filename)In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
phone1 = set()
phone2 = set()
for k,v in dict1.items():
phone1.add(v['phone'])
list1 = []
filename = 'data/survey_records_20250813.csv'
with open(filename,'r',newline='') as csv_file:
fl = csv.reader(csv_file,delimiter=',')
header = next(fl)
for line in fl:
list1.append(line)
i =1
list2 = []
for item in list1:
content = json.loads(item[4])
code = int(content['phone'])
for k, v in dict1.items():
list3 = []
if v['phone'] == code:
list3.append(k)
list3.append(v['name'])
list3.append(v['sex'])
list3.append(v['unit'])
list3.append(code)
list2.append(list3)
filename = 'data/南化问卷情况表(第二批).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
#sheet.append(title)
for row in list2:
sheet.append(row)
wb.save(filename) In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date
filename = 'data/survey_records_20250813.csv'
with open(filename,'r',newline='') as csv_file:
fl = csv.reader(csv_file,delimiter=',')
header = next(fl)
for line in fl:
list1.append(line)
i =1
list2 = []
for item in list1:
list3 = []
content = json.loads(item[4])
phone = int(content['phone'])
name = content['name']
sex = content['gender']
list3.append(name)
list3.append(sex)
list3.append(phone)
list2.append(list3)
filename = 'data/南化问卷情况表(第二批).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
#sheet.append(title)
for row in list2:
sheet.append(row)
wb.save(filename) In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date
dict1 = {}
filename = 'data/result_南京化工.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict3 = json.load(fl)
phone = {}
for k,v in dict3.items():
if 'phone' in v.keys():
phone[v['phone']] = k
list1 = []
filename = 'data/survey_records_20250812.csv'
with open(filename,'r',newline='') as csv_file:
fl = csv.reader(csv_file,delimiter=',')
header = next(fl)
for line in fl:
list1.append(line)
nn = 0
for item in list1:
if int(item[3]) in phone.keys():
tcm = []
code = phone[int(item[3])]
for i in range(0,60):
tcm.append(0)
content = json.loads(item[4])
if code not in dict1.keys():
dict1[code] = dict3[code]
rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])
dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))
#dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]
else:
rq=date.fromisoformat('2025-07-01')
dict1[code]['rq'] = '2025-07-01'
for k, v in content.items():
if 'tcm' in k:
i = int(k[3:])
tcm[i-1] = int(v)
if 'tcm' in item[4]:
dict1[code]['tcm'] = tcm
birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))
days = (rq-birth).days
dict1[code]['age'] = int(days/365)
dict1[code]['month'] = int(days/365*12)
#print(phone[item[2]])
nn+=1
filename = 'data/result_南京化工-2.json'
with open(filename,'w') as fl:
json.dump(dict1, fl, ensure_ascii=False)
In [19]:
import json
import csv
import openpyxl
import time
from datetime import date
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict3 = json.load(fl)
list1 = []
filename = 'data/survey_records_20250901.csv'
with open(filename,'r',newline='') as csv_file:
fl = csv.reader(csv_file,delimiter=',')
header = next(fl)
for line in fl:
list1.append(line)
dict1 = {}
nn = 0
for item in list1:
tcm = []
for i in range(0,60):
tcm.append(0)
content = json.loads(item[4])
phone = content['phone']
name = content['name']
for k,v in dict3.items():
if name == v['name']:
code = k
unit = v['unit']
sex = v['sex']
dict1.setdefault(code,{})
#rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])
#dict1[phone]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))
for k, v in content.items():
if 'tcm' in k:
i = int(k[3:])
tcm[i-1] = int(v)
if 'tcm' in item[4]:
dict1[code]['tcm'] = tcm
dict1[code]['name'] = content['name']
dict1[code]['unit'] = unit
dict1[code]['sex'] = sex
dict1[code]['weight'] = content['weight']
dict1[code]['tun'] = content['hip']
dict1[code]['yao'] = content['waist']
#print(phone[item[2]])
nn+=1
filename = 'data/result_南京化工-2.json'
with open(filename,'w') as fl:
json.dump(dict1, fl, ensure_ascii=False)
print(len(dict1))83
In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date
dict1 = {}
filename = 'data/result_南京化工-1.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
wb = openpyxl.load_workbook('data/南化腰臀数据.xlsx',data_only=True)
sheet = wb.active
# sheets = wb.sheetnames
person = {}
for n in range(2, sheet.max_row+1):
code = str(sheet.cell(n, 1).value)
if code in dict1.keys():
yao = str(sheet.cell(n, 2).value)
tun = str(sheet.cell(n, 3).value)
dict1[code]['腰臀比'] = yao+','+tun
filename = 'data/result_南京化工-1.json'
with open(filename,'w') as fl:
json.dump(dict1, fl, ensure_ascii=False)
In [ ]:
import openpyxl
import json
questions = [
[1],
[-1, 2],
[-1, 2],
[-1, 8],
[-1, 3],
[1],
[-1],
[-1, 7],
[2],
[2],
[2],
[2, 3],
[2],
[2],
[3],
[3],
[3],
[3],
[3],
[4],
[4],
[4],
[4],
[4],
[4],
[4],
[4],
[5],
[5],
[5],
[5],
[5],
[5],
[5],
[5],
[6],
[6],
[6],
[6],
[6],
[6],
[7],
[7],
[7],
[7],
[7],
[7],
[8],
[8],
[8],
[8],
[8],
[8],
[9],
[9],
[9],
[9],
[9],
[9],
[9]
]
kinds = [
'平和',
'气虚',
'阳虚',
'阴虚',
'痰湿',
'湿热',
'血瘀',
'气郁',
'特禀'
]
def tcm_calc(arr):
qa = [8, 8, 7, 8, 8, 6, 7, 7, 7]
# 成绩数组
s = [0] * 9
# 遍历五进制
for i in range(len(questions)):
m = arr[i] - 1
for v in questions[i]:
if v < 0:
s[-v - 1] += 4 - m
else:
s[v - 1] += m
return [int((v / qa[i]) * 25) for i, v in enumerate(s)]
def tcm_kind(score):
kind = 0
near = False
max_kind = 0
max_score = 0
for i in range(1, 9):
if score[i] > max_score:
max_kind = i
max_score = score[i]
if score[0] >= 60 and max_score < 40:
if max_score >= 30:
near = True
kind = max_kind
else:
kind = max_kind
return {
"kind": kind,
"near": near
}
filename = 'data/result_南京化工-2.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
i = 1
list2 = []
for k, v in dict1.items():
if 'tcm' in v.keys():
list1 = []
tcm =v['tcm']
for item in tcm:
list1.append(item)
score = tcm_calc(list1)
result = tcm_kind(score)
kind = result['kind']
near = result['near']
#print(i,k,kinds[kind], near, score)
#i+=1
list3 = []
list3.append(k)
list3.append(v['name'])
list3.append(v['sex'])
list3.append(v['weight'])
list3.append(v['yao'])
list3.append(v['tun'])
list3.append(kinds[kind])
list3.append(near)
for item in score:
list3.append(item)
list2.append(list3)
filename = 'data/南化第二次问卷明细表(截至20250831).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
for row in list2:
sheet.append(row)
wb.save(filename)
print('ok') In [ ]:
from pathlib import Path
import json
import shutil
target_directory = Path('./file/南化体重')
new_path = './file/南化体重/new'
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
# 遍历目标目录及其子目录获取所有文件
for fl in target_directory.rglob('*.pdf'):
if fl.is_file():
fl_name = fl.stem
name = fl_name[12:]
for k, v in dict1.items():
if name == v['name']:
n_name = Path(new_path,str(k)+'-'+name+'.pdf')
shutil.copyfile(fl,n_name)
print(n_name)
In [ ]:
from pathlib import Path
import json
import shutil
import pymupdf4llm
#md_text = pymupdf4llm.to_markdown("data/1782596-唐荣.pdf")
llama_reader = pymupdf4llm.LlamaMarkdownReader()
#llama_docs = llama_reader.load_data("data/1782596-唐荣.pdf")
target_directory = Path('./file/北海体检报告')
new_path = './file/北海体检报告/md'
for fl in target_directory.rglob('*.pdf'):
if fl.is_file():
fl_name = fl.stem
llama_lists = pymupdf4llm.to_markdown(fl,page_chunks=True)
list1 = []
for item in llama_lists:
list1.append(item['text'])
llama_docs = '\n'.join(list1)
Path(new_path,fl_name+'.md').write_bytes(llama_docs.encode())
In [ ]:
from pathlib import Path
import json
import shutil
import pymupdf4llm
#md_text = pymupdf4llm.to_markdown("data/1782596-唐荣.pdf")
llama_reader = pymupdf4llm.LlamaMarkdownReader()
llama_docs = llama_reader.load_data("data/1782596-唐荣.pdf",page_chunks=True)
print(llama_docs)
In [ ]:
import requests
import json
import openpyxl
headers = {
"Content-Type": "application/json; charset=UTF-8"
}
filename = 'data/result_南京化工-2.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
list1 = []
file_path ='./南京化工第二批问卷/'
list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']
i=0
list2 = []
for k, v in dict1.items():
list1 = []
mydata = {}
id = str(k).rjust(4,"0")
mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'
mydata['title'] = '南化公司'
mydata['subtitle'] = ''#v['unit']
mydata['id'] = id
mydata['name'] = v['name']
if v['sex'] == '男':
mydata['gender'] = 'male'
else:
mydata['gender'] = 'female'
mydata['month'] = v['month']
mydata['fits'] = {}
survey_list = ['tcm','psy57','spine']
for item in survey_list:
if item in v.keys():
mydata.setdefault('surveys',{})
mydata['surveys'][item] = v[item]
#mydata['fits'] = {}
for item in list_item:
if item in v.keys():
mydata.setdefault('fits',{})
if item in ['lung','pushup','step','situp']:
mark = v[item]['成绩'].split()[0].split('.')[0]
else:
mark = v[item]['成绩'].split()[0]
mydata['fits'][item] = {'mark':mark,'score':v[item]['score']}
if len(mydata['fits']) >2 or len(mydata['surveys']) >0:
#if len(mydata['fits']) >2 :
list1.append(mydata)
list2.append([k,v['name']])
i+=1
x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)
#print(id,v['name'],x.text)
#print(mydata)
#x.close()
print(i)In [20]:
import requests
import json
import openpyxl
headers = {
"Content-Type": "application/json; charset=UTF-8"
}
filename = 'data/result_南京化工-2.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
list1 = []
file_path ='./南京化工第二批问卷/'
list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']
i=0
list2 = []
for k, v in dict1.items():
list1 = []
mydata = {}
id = str(k)
mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'
mydata['title'] = '南化公司'
mydata['subtitle'] = v['unit']
mydata['id'] = id
mydata['name'] = v['name']
if v['sex'] == 'm':
mydata['gender'] = 'male'
else:
mydata['gender'] = 'female'
#mydata['month'] = v['month']
#mydata['fits'] = {}
survey_list = ['tcm','psy57','spine']
for item in survey_list:
if item in v.keys():
mydata.setdefault('surveys',{})
mydata['surveys'][item] = v[item]
#mydata['fits'] = {}
if len(mydata['surveys']) >0:
#if len(mydata['fits']) >2 :
list1.append(mydata)
list2.append([k,v['name']])
i+=1
x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)
#print(id,v['name'],x.text)
#print(mydata)
#x.close()
print(i)83
In [ ]:
from pathlib import Path
import json
import shutil
target_directory = Path('./data/json')
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
# 遍历目标目录及其子目录获取所有文件
dict2 = {}
list2 = ['总胆固醇','甘油三酯','尿微量白蛋白']
for fl in target_directory.glob('*.json'):
if fl.is_file():
code = fl.stem
dict2.setdefault(code,{})
dict2[code] = dict1[code]
with open(fl,'r') as fl1:
dict3 = json.load(fl1)
for k, v in dict3.items():
if k =='血压情况' and len(v)>0:
dict2[code].setdefault('血压',{})
list1 = []
for item in v:
dict2[code]['血压'][item['项目']] = item['结果']
if '状态' in item.keys():
list1.append(item['项目']+item['状态'])
if len(list1)>0:
dict2[code]['血压']['状态'] = ','.join(list1)
if k in list2:
dict2[code].setdefault(k,{})
dict2[code][k]['结果'] = v['结果']
dict2[code][k]['参考值'] = v['参考值']
if '状态' in v.keys():
dict2[code][k]['状态'] = v['状态']
if k in ['空腹血糖','糖化血红蛋白']:
dict2[code].setdefault(k,{})
if '结果' in v.keys():
dict2[code][k]['结果'] = v['结果']
dict2[code][k]['参考值'] = v['参考值']
if '状态' in v.keys():
dict2[code][k]['状态'] = v['状态']
if k in ['ALT、AST、GGT','TSH、FT3、FT4']:
for item in v:
xm = item['项目']
dict2[code].setdefault(xm,{})
if '结果' in item.keys():
dict2[code][xm]['结果'] = item['结果']
if '参考值' in item.keys():
dict2[code][xm]['参考值'] = item['参考值']
if '状态' in item.keys():
dict2[code][xm]['状态'] = item['状态']
if k =='肾功能与尿微量白蛋白':
for item in v['肾功能']:
xm = item['项目']
dict2[code].setdefault(xm,{})
if '结果' in item.keys():
dict2[code][xm]['结果'] = item['结果']
if '参考值' in item.keys():
dict2[code][xm]['参考值'] = item['参考值']
if '状态' in item.keys():
dict2[code][xm]['状态'] = item['状态']
dict2[code].setdefault('尿微量白蛋白',{})
xm = v['尿微量白蛋白']
if '结果' in xm.keys() and len(xm['结果'])>0:
dict2[code]['尿微量白蛋白']['结果'] = xm['结果']
if '参考值' in xm.keys() and len(xm['参考值'])>0:
dict2[code]['尿微量白蛋白']['参考值'] = xm['参考值']
if '状态' in xm.keys():
dict2[code]['尿微量白蛋白']['状态'] = xm['状态']
filename = 'data/南京化工体检情况.json'
with open(filename,'w') as fl:
json.dump(dict2, fl, ensure_ascii=False) In [ ]:
import json
import csv
import openpyxl
filename = 'data/南京化工体检情况.json'
with open(filename,'r') as fl:
dict1 = json.load(fl)
list1 = ["总胆固醇","甘油三酯","空腹血糖","糖化血红蛋白","谷丙转氨酶 (ALT)","谷草转氨酶 (AST)","γ- 谷氨酰转肽酶 (GGT)","促甲状腺激素 (TSH)","游离三碘甲状腺原氨酸 (FT3)","游离甲状腺素 (FT4)","肌酐","尿素氮","尿酸","尿微量白蛋白"]
title = ['编号','姓名','性别','血压','状态']
for item in list1:
title.append(item)
title.append('状态')
list3 = []
for k, v in dict1.items():
list2 = []
list2.append(k)
list2.append(v['name'])
list2.append(v['sex'])
if '血压' in v.keys():
xueya = v['血压']['舒张压']+'/'+v['血压']['收缩压']
if '状态' in v['血压'].keys():
zt = v['血压']['状态']
else:
zt = ''
else:
xueya = ''
zt = ''
list2.append(xueya)
list2.append(zt)
for item in list1:
if item in v.keys() and '结果' in v[item]:
list2.append(v[item]['结果'])
if '状态' in v[item]:
list2.append(v[item]['状态'])
else:
list2.append('')
else:
list2.append('')
list2.append('')
list3.append(list2)
filename = 'data/南京化工体检相关数据明细.xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
sheet.append(title)
for row in list3:
sheet.append(row)
wb.save(filename)In [ ]:
from spire.pdf.common import *
from spire.pdf import *
# 创建PdfDocument类的实例
pdf = PdfDocument()
# 加载PDF文档
pdf.LoadFromFile("file/北海体检报告/2405280074.pdf")
# 将PDF转换为Markdown文件
pdf.SaveToFile("PDF转Markdown.md", FileFormat.Markdown)
pdf.Close()
In [ ]: