Files
jupyter/体测单位/南京化工.ipynb
T
2025-09-16 13:14:59 +08:00

37 KiB

体测人员导入

In [ ]:
import openpyxl
import json


wb = openpyxl.load_workbook('data/南化人员信息表.xlsx',data_only=True)
sheet = wb.active
# sheets = wb.sheetnames
person = {}

for n in range(2, sheet.max_row+1):
    code = int(sheet.cell(n, 4).value)
    person.setdefault(code, {})
    dict1 = {}
    dict1['name'] = sheet.cell(n, 3).value
    dict1['sex'] = sheet.cell(n, 5).value
    dict1['unit'] = sheet.cell(n, 2).value    
    dict1['birth'] = str(sheet.cell(n, 6).value).replace('/','-').split(' ')[0]
    dict1['phone'] = sheet.cell(n, 12).value 
    person[code] = dict1
filename = 'data/南京化工人员.json'
with open(filename, 'w') as fl:
    json.dump(person, fl, ensure_ascii=False)
print(len(person),'ok')

生成读卡系统文件

In [ ]:
import json

filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
list1 = []
for k, v in dict1.items():
    dict2 = {}
    #if dict1['sex'] =='男':
    #    sex = 1
    
    dict2 = {'id':k,'name':v['name'],'gender':v['sex'],'birth':v['birth'],'unit':v['unit']}
    list1.append(dict2)
json_data = json.dumps(list1,ensure_ascii=False, indent=4)  

# 将 json 数据写入文件
with open("data/data_南京化工人员.json", "w",encoding = 'utf-8') as file:
    file.write(json_data)   
print('ok')

获取人员测试成绩

In [ ]:
import json
import datetime
import csv
from datetime import date
import my_module as My

filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl) 
filename = 'data/marks_20250703.csv'
re_ta = My.get_result(filename,dict1)


filename = 'data/result_南京化工.json'
with open(filename,'w') as fl:
    json.dump(re_ta, fl, ensure_ascii=False) 
print(len(re_ta))

生成测试得分

In [ ]:
import json
import time
import my_module as My

#list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','height','weight']
list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']
filename = 'data/result_南京化工.json'
with open(filename,'r') as fl:
    dict2 = json.load(fl) 
for k, v in dict2.items():
    #print(k)
    if v['sex'] == '男':
        sex = 'M'
    else:
        sex = 'F'            
    if 'bmi' in v.keys():
        #bmi_data = v['height']['成绩'].split()[0]+','+ v['weight']['成绩'].split()[0]
        bmi_data = v['bmi']['成绩']
        data1 = {'code':k,'sex':sex,'age':v['age'],'item':'HeightWeight','result':bmi_data}
        dict2[k]['bmi'] = {}
        dict2[k]['bmi']['成绩'] = bmi_data
        dict2[k]['bmi']['score'] = My.cal_bmi(data1)
    for item_en in list_item:
        if item_en in v.keys():            
            data1 = {'code':k,'sex':sex,'age':v['age'],'item':item_en,'result':float(v[item_en]['成绩'].split()[0])}
            #print(k,v['name'])
            dict2[k][item_en]['score'] = My.cal_score(data1)
            #print(k,v[item_en]['成绩'],cal_score(data1))

filename = f'data/result_南京化工.json'
with open(filename,'w') as fl:
    json.dump(dict2,fl , ensure_ascii=False) 
print('ok!')        

导出测试人员信息

In [ ]:
import json
import openpyxl

items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']
title = ['编号','姓名','性别','单位','部门','身高','体重','肺活量','握力','坐位体前屈','纵跳','俯卧撑','一分钟仰卧起坐','单脚站立','选择反应时','台阶指数']

filename = 'data/result_南京化工.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)

filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict2 = json.load(fl)
    
list1 = []
for k, v in dict1.items():
    list2 = []
    list2.append(str(k).rjust(5,'0'))
    list2.append(v['name'])    
    list2.append(dict2[k]['sex'])
    list2.append(dict2[k]['unit'])
    if 'bmi' in v.keys():
        height = v['bmi']['成绩'].split(',')[0]
        weight = v['bmi']['成绩'].split(',')[1]
        list2.append(height)
        list2.append(weight)
    else:
        list2.append('')
        list2.append('')
    for item in items:
        if item in v.keys():
            list2.append(v[item]['成绩']) 
        elif item =='name':
            list2.append(v[item])
        else:
            list2.append('')
    
    list1.append(list2)
filename = 'data/南京化工测试情况明细表(截至20250630).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
sheet.append(title)
for row in list1:
    sheet.append(row)
    
wb.save(filename)

统计问卷人员情况

In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date


filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)

phone1 =  set()
phone2 = set()
for k,v in dict1.items():
    phone1.add(v['phone'])
list1 = []
filename = 'data/survey_records_20250813.csv'
with open(filename,'r',newline='') as csv_file:
    fl = csv.reader(csv_file,delimiter=',')
    header = next(fl)    
    for line in fl:
        list1.append(line)

i =1
list2 = []
for item in list1:
    content = json.loads(item[4])
    code = int(content['phone'])
    for k, v in dict1.items():
        list3 = []
        if v['phone'] == code:            
            list3.append(k)
            list3.append(v['name'])
            list3.append(v['sex'])
            list3.append(v['unit'])
            list3.append(code)
            list2.append(list3)
filename = 'data/南化问卷情况表(第二批).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
#sheet.append(title)
for row in list2:
    sheet.append(row)
    
wb.save(filename)  
In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date



filename = 'data/survey_records_20250813.csv'
with open(filename,'r',newline='') as csv_file:
    fl = csv.reader(csv_file,delimiter=',')
    header = next(fl)    
    for line in fl:
        list1.append(line)

i =1
list2 = []
for item in list1:
    list3 = []
    content = json.loads(item[4])
    phone = int(content['phone'])
    name = content['name']
    sex = content['gender']
    list3.append(name)
    list3.append(sex)
    list3.append(phone)
    list2.append(list3)
filename = 'data/南化问卷情况表(第二批).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
#sheet.append(title)
for row in list2:
    sheet.append(row)
    
wb.save(filename)  

导入问卷信息

In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date

dict1 = {}

filename = 'data/result_南京化工.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict3 = json.load(fl)

phone =  {}
for k,v in dict3.items():
    if 'phone' in v.keys():
        phone[v['phone']] = k

list1 = []
filename = 'data/survey_records_20250812.csv'
with open(filename,'r',newline='') as csv_file:
    fl = csv.reader(csv_file,delimiter=',')
    header = next(fl)    
    for line in fl:
        list1.append(line)



nn = 0
for item in list1:
     if int(item[3]) in phone.keys():        
        tcm = []
        code = phone[int(item[3])]
        
        for i in range(0,60):
            tcm.append(0)
        
       
        content = json.loads(item[4])
        if code not in dict1.keys():
            dict1[code] = dict3[code]
            rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])
            dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))
            #dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]
        else:
            rq=date.fromisoformat('2025-07-01')
            dict1[code]['rq'] = '2025-07-01'
        for k, v in content.items():
           
            if 'tcm' in k:
                i = int(k[3:])
                tcm[i-1] = int(v)           
        
        if 'tcm' in item[4]:            
            dict1[code]['tcm'] = tcm
       
        birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))
        
        days =  (rq-birth).days          
        dict1[code]['age'] = int(days/365)
        dict1[code]['month'] = int(days/365*12)
        #print(phone[item[2]])
        nn+=1
filename = 'data/result_南京化工-2.json'

with open(filename,'w') as fl:
    json.dump(dict1, fl, ensure_ascii=False)

导入问卷信息(新)

In [19]:
import json
import csv
import openpyxl
import time
from datetime import date


filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict3 = json.load(fl)


list1 = []
filename = 'data/survey_records_20250901.csv'
with open(filename,'r',newline='') as csv_file:
    fl = csv.reader(csv_file,delimiter=',')
    header = next(fl)    
    for line in fl:
        list1.append(line)


dict1 = {}
nn = 0
for item in list1:             
    tcm = []    
    for i in range(0,60):
        tcm.append(0)
       
    content = json.loads(item[4])   
    phone = content['phone']
    name = content['name']
    for k,v in dict3.items():
        if name == v['name']:
            code = k
            unit = v['unit']
            sex = v['sex']
    dict1.setdefault(code,{})
    
    #rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])
    #dict1[phone]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))
     
    for k, v in content.items():       
        if 'tcm' in k:
            i = int(k[3:])
            tcm[i-1] = int(v)           
    
    if 'tcm' in item[4]:            
        dict1[code]['tcm'] = tcm
   
    dict1[code]['name'] = content['name']
    dict1[code]['unit'] = unit
    dict1[code]['sex'] = sex
    dict1[code]['weight'] = content['weight']
    dict1[code]['tun'] = content['hip']
    dict1[code]['yao'] = content['waist']
    #print(phone[item[2]])
    nn+=1
filename = 'data/result_南京化工-2.json'

with open(filename,'w') as fl:
    json.dump(dict1, fl, ensure_ascii=False)
print(len(dict1))
83

导入腰臀数据

In [ ]:
import json
import csv
import openpyxl
import time
from datetime import date

dict1 = {}

filename = 'data/result_南京化工-1.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
wb = openpyxl.load_workbook('data/南化腰臀数据.xlsx',data_only=True)
sheet = wb.active
# sheets = wb.sheetnames
person = {}

for n in range(2, sheet.max_row+1):
    code = str(sheet.cell(n, 1).value)
    if code in dict1.keys():
        yao = str(sheet.cell(n, 2).value)
        tun = str(sheet.cell(n, 3).value)
        dict1[code]['腰臀比'] = yao+','+tun
filename = 'data/result_南京化工-1.json'

with open(filename,'w') as fl:
    json.dump(dict1, fl, ensure_ascii=False)

计算中医体质并导出

In [ ]:
import openpyxl
import json

questions = [
  [1],
  [-1, 2],
  [-1, 2],
  [-1, 8],
  [-1, 3],
  [1],
  [-1],
  [-1, 7],
  [2],
  [2],
  [2],
  [2, 3],
  [2],
  [2],
  [3],
  [3],
  [3],
  [3],
  [3],
  [4],
  [4],
  [4],
  [4],
  [4],
  [4],
  [4],
  [4],
  [5],
  [5],
  [5],
  [5],
  [5],
  [5],
  [5],
  [5],
  [6],
  [6],
  [6],
  [6],
  [6],
  [6],
  [7],
  [7],
  [7],
  [7],
  [7],
  [7],
  [8],
  [8],
  [8],
  [8],
  [8],
  [8],
  [9],
  [9],
  [9],
  [9],
  [9],
  [9],
  [9]
]

kinds = [
  '平和',
  '气虚',
  '阳虚',
  '阴虚',
  '痰湿',
  '湿热',
  '血瘀',
  '气郁',
  '特禀'
]

def tcm_calc(arr):
  qa = [8, 8, 7, 8, 8, 6, 7, 7, 7]
  # 成绩数组
  s = [0] * 9
  # 遍历五进制
  for i in range(len(questions)):
    m = arr[i] - 1
    for v in questions[i]:
      if v < 0:
        s[-v - 1] += 4 - m
      else:
        s[v - 1] += m
  return [int((v / qa[i]) * 25) for i, v in enumerate(s)]

def tcm_kind(score):
  kind = 0
  near = False
  max_kind = 0
  max_score = 0
  for i in range(1, 9):
    if score[i] > max_score:
      max_kind = i
      max_score = score[i]
  if score[0] >= 60 and max_score < 40:
    if max_score >= 30:
      near = True
      kind = max_kind
  else:
    kind = max_kind
  return {
    "kind": kind,
    "near": near
  }


filename = 'data/result_南京化工-2.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl) 
i = 1
list2 = []
for k, v in dict1.items():
    if 'tcm' in v.keys():
        list1 = []
        tcm =v['tcm']
        for item in tcm:
            list1.append(item)
        score = tcm_calc(list1)

        result = tcm_kind(score)
        kind = result['kind']
        near = result['near']
        #print(i,k,kinds[kind], near, score)
        #i+=1
        list3 = []
        list3.append(k)
        list3.append(v['name'])
        list3.append(v['sex'])
        list3.append(v['weight'])
        list3.append(v['yao'])
        list3.append(v['tun'])
        list3.append(kinds[kind])
        list3.append(near)
        for item in score:
            list3.append(item)
        list2.append(list3)

filename = 'data/南化第二次问卷明细表(截至20250831).xlsx'
wb = openpyxl.Workbook()
sheet = wb.active

for row in list2:
    sheet.append(row)
    
wb.save(filename)
print('ok')       

体检报告汇总

In [ ]:
from pathlib import Path
import json
import shutil


target_directory = Path('./file/南化体重')
new_path = './file/南化体重/new'
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
# 遍历目标目录及其子目录获取所有文件

for fl in target_directory.rglob('*.pdf'):
    if fl.is_file():
        fl_name = fl.stem
        name = fl_name[12:]        
        for k, v in dict1.items():            
            if name == v['name']:
                n_name = Path(new_path,str(k)+'-'+name+'.pdf')
                shutil.copyfile(fl,n_name)
                print(n_name)
                
                
In [ ]:
from pathlib import Path
import json
import shutil
import pymupdf4llm
#md_text = pymupdf4llm.to_markdown("data/1782596-唐荣.pdf")
llama_reader = pymupdf4llm.LlamaMarkdownReader()
#llama_docs = llama_reader.load_data("data/1782596-唐荣.pdf")


target_directory = Path('./file/北海体检报告')
new_path = './file/北海体检报告/md'


for fl in target_directory.rglob('*.pdf'):
    if fl.is_file():
        fl_name = fl.stem
        llama_lists = pymupdf4llm.to_markdown(fl,page_chunks=True)
        list1 = []
        for item in llama_lists:
            list1.append(item['text'])
        llama_docs = '\n'.join(list1)
        Path(new_path,fl_name+'.md').write_bytes(llama_docs.encode())
        
In [ ]:
from pathlib import Path
import json
import shutil
import pymupdf4llm
#md_text = pymupdf4llm.to_markdown("data/1782596-唐荣.pdf")
llama_reader = pymupdf4llm.LlamaMarkdownReader()
llama_docs = llama_reader.load_data("data/1782596-唐荣.pdf",page_chunks=True)
print(llama_docs)

生成报告

In [ ]:
import requests
import json
import openpyxl


headers = {
    "Content-Type": "application/json; charset=UTF-8"
    }
filename = 'data/result_南京化工-2.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
list1 = []
file_path ='./南京化工第二批问卷/'
list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']
i=0
list2 = []
for  k, v in dict1.items():
    list1 = []
    mydata = {}
    
    id = str(k).rjust(4,"0")
    mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'
    mydata['title'] = '南化公司'
    mydata['subtitle'] = ''#v['unit']
    mydata['id'] = id
    mydata['name'] = v['name']
    if v['sex'] == '男':
        mydata['gender'] = 'male'
    else:
        mydata['gender'] = 'female'
    
    mydata['month'] = v['month']
    mydata['fits'] = {}
    survey_list = ['tcm','psy57','spine']
    for item in survey_list:
        if item in  v.keys():
            mydata.setdefault('surveys',{})
            mydata['surveys'][item] = v[item]
            
    
    #mydata['fits'] = {}
    for item in list_item:
        if item in v.keys():
            mydata.setdefault('fits',{})
            if item in ['lung','pushup','step','situp']:
                mark = v[item]['成绩'].split()[0].split('.')[0]
            else:
                mark = v[item]['成绩'].split()[0]
            mydata['fits'][item] = {'mark':mark,'score':v[item]['score']}
    if len(mydata['fits']) >2 or len(mydata['surveys']) >0:
    #if len(mydata['fits']) >2 :    
        list1.append(mydata)
        list2.append([k,v['name']])
        i+=1
        x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)
        #print(id,v['name'],x.text)
        #print(mydata)
        #x.close()
print(i)

生成报告(单问卷)

In [20]:
import requests
import json
import openpyxl


headers = {
    "Content-Type": "application/json; charset=UTF-8"
    }
filename = 'data/result_南京化工-2.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
list1 = []
file_path ='./南京化工第二批问卷/'
list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']
i=0
list2 = []
for  k, v in dict1.items():
    list1 = []
    mydata = {}
    
    id = str(k)
    mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'
    mydata['title'] = '南化公司'
    mydata['subtitle'] = v['unit']
    mydata['id'] = id
    mydata['name'] = v['name']
    if v['sex'] == 'm':
        mydata['gender'] = 'male'
    else:
        mydata['gender'] = 'female'
    
    #mydata['month'] = v['month']
    #mydata['fits'] = {}
    survey_list = ['tcm','psy57','spine']
    for item in survey_list:
        if item in  v.keys():
            mydata.setdefault('surveys',{})
            mydata['surveys'][item] = v[item]
            
    
    #mydata['fits'] = {}
    
    if len(mydata['surveys']) >0:
    #if len(mydata['fits']) >2 :    
        list1.append(mydata)
        list2.append([k,v['name']])
        i+=1
        x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)
        #print(id,v['name'],x.text)
        #print(mydata)
        #x.close()
print(i)
83

导入体检报告数据

In [ ]:
from pathlib import Path
import json
import shutil


target_directory = Path('./data/json')
filename = 'data/南京化工人员.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)
# 遍历目标目录及其子目录获取所有文件
dict2 = {}
list2 =  ['总胆固醇','甘油三酯','尿微量白蛋白']
for fl in target_directory.glob('*.json'):
    if fl.is_file():
        code = fl.stem
        dict2.setdefault(code,{})
        dict2[code] = dict1[code]
        with open(fl,'r') as fl1:
            dict3 = json.load(fl1)
        for k, v in dict3.items():
            if k =='血压情况' and len(v)>0:
                dict2[code].setdefault('血压',{})
                list1 = [] 
                for item in v:
                    
                    dict2[code]['血压'][item['项目']] = item['结果']
                    if '状态' in item.keys():
                        list1.append(item['项目']+item['状态'])
                if len(list1)>0:
                    dict2[code]['血压']['状态'] = ','.join(list1)
                
            if k in list2:
                dict2[code].setdefault(k,{})
                dict2[code][k]['结果'] = v['结果']
                dict2[code][k]['参考值'] = v['参考值']
                if '状态' in v.keys():
                    dict2[code][k]['状态'] = v['状态']
            if k in ['空腹血糖','糖化血红蛋白']:
                dict2[code].setdefault(k,{})
                if '结果' in v.keys():
                    dict2[code][k]['结果'] = v['结果']
                    dict2[code][k]['参考值'] = v['参考值']
                if '状态' in v.keys():
                    dict2[code][k]['状态'] = v['状态']
            if k in ['ALT、AST、GGT','TSH、FT3、FT4']:
                for item in v:
                    xm = item['项目']
                    dict2[code].setdefault(xm,{})
                    if '结果' in item.keys():
                        dict2[code][xm]['结果'] = item['结果']
                    if '参考值' in item.keys():
                        dict2[code][xm]['参考值'] = item['参考值']
                    if '状态' in item.keys():
                        dict2[code][xm]['状态'] = item['状态']
            if k =='肾功能与尿微量白蛋白':
                for item in v['肾功能']:
                    xm = item['项目']
                    dict2[code].setdefault(xm,{})
                    if '结果' in item.keys():
                        dict2[code][xm]['结果'] = item['结果']
                    if '参考值' in item.keys():
                        dict2[code][xm]['参考值'] = item['参考值']
                    if '状态' in item.keys():
                        dict2[code][xm]['状态'] = item['状态']                
                dict2[code].setdefault('尿微量白蛋白',{})
                xm = v['尿微量白蛋白']
                if '结果' in xm.keys() and len(xm['结果'])>0:
                    dict2[code]['尿微量白蛋白']['结果'] = xm['结果']
                if '参考值' in xm.keys() and len(xm['参考值'])>0:
                    dict2[code]['尿微量白蛋白']['参考值'] = xm['参考值']
                if '状态' in xm.keys():
                    dict2[code]['尿微量白蛋白']['状态'] = xm['状态']    
                
            

filename = 'data/南京化工体检情况.json'

with open(filename,'w') as fl:
    json.dump(dict2, fl, ensure_ascii=False)                

导出体检报告数据

In [ ]:
import json
import csv
import openpyxl

filename = 'data/南京化工体检情况.json'
with open(filename,'r') as fl:
    dict1 = json.load(fl)

list1 = ["总胆固醇","甘油三酯","空腹血糖","糖化血红蛋白","谷丙转氨酶 (ALT)","谷草转氨酶 (AST)","γ- 谷氨酰转肽酶 (GGT)","促甲状腺激素 (TSH)","游离三碘甲状腺原氨酸 (FT3)","游离甲状腺素 (FT4)","肌酐","尿素氮","尿酸","尿微量白蛋白"]
title = ['编号','姓名','性别','血压','状态']
for item in list1:
    title.append(item)
    title.append('状态')
list3 = []
for k, v in dict1.items():
    list2 = []
    list2.append(k)
    list2.append(v['name'])
    list2.append(v['sex'])
    if '血压' in v.keys():
        xueya = v['血压']['舒张压']+'/'+v['血压']['收缩压']
        if '状态' in v['血压'].keys():
            zt = v['血压']['状态']
        else:
            zt = ''
    else:
        xueya = ''
        zt = ''
    
    list2.append(xueya)
    list2.append(zt)
    for item in list1:
        if item in v.keys() and '结果' in v[item]:
            list2.append(v[item]['结果'])
            if '状态' in v[item]:
                list2.append(v[item]['状态'])
            else:
                list2.append('')
        else:
            list2.append('')
            list2.append('')  
    list3.append(list2)

filename = 'data/南京化工体检相关数据明细.xlsx'
wb = openpyxl.Workbook()
sheet = wb.active
sheet.append(title)
for row in list3:
    sheet.append(row)
    
wb.save(filename)
In [ ]:
from spire.pdf.common import *
from spire.pdf import *

# 创建PdfDocument类的实例
pdf = PdfDocument()

# 加载PDF文档
pdf.LoadFromFile("file/北海体检报告/2405280074.pdf")

# 将PDF转换为Markdown文件
pdf.SaveToFile("PDF转Markdown.md", FileFormat.Markdown)
pdf.Close()
In [ ]: