763 lines
23 KiB
Plaintext
763 lines
23 KiB
Plaintext
{
|
||
"cells": [
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "04524c85-988e-4dbf-86eb-939a9db7aa28",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 体测人员导入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "bbba6efc-73cd-4db6-bae7-014724fee731",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import openpyxl\n",
|
||
"import json\n",
|
||
"\n",
|
||
"\n",
|
||
"wb = openpyxl.load_workbook('data/南化人员信息表.xlsx',data_only=True)\n",
|
||
"sheet = wb.active\n",
|
||
"# sheets = wb.sheetnames\n",
|
||
"person = {}\n",
|
||
"\n",
|
||
"for n in range(2, sheet.max_row+1):\n",
|
||
" code = int(sheet.cell(n, 4).value)\n",
|
||
" person.setdefault(code, {})\n",
|
||
" dict1 = {}\n",
|
||
" dict1['name'] = sheet.cell(n, 3).value\n",
|
||
" dict1['sex'] = sheet.cell(n, 5).value\n",
|
||
" dict1['unit'] = sheet.cell(n, 2).value \n",
|
||
" dict1['birth'] = str(sheet.cell(n, 6).value).replace('/','-').split(' ')[0]\n",
|
||
" dict1['phone'] = sheet.cell(n, 12).value \n",
|
||
" person[code] = dict1\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename, 'w') as fl:\n",
|
||
" json.dump(person, fl, ensure_ascii=False)\n",
|
||
"print(len(person),'ok')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "048a95aa-1691-45f6-b933-6eaf95ae1d30",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 生成读卡系统文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "ce406710-6b7d-4c51-98ec-78883bd3ce5f",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"list1 = []\n",
|
||
"for k, v in dict1.items():\n",
|
||
" dict2 = {}\n",
|
||
" #if dict1['sex'] =='男':\n",
|
||
" # sex = 1\n",
|
||
" \n",
|
||
" dict2 = {'id':k,'name':v['name'],'gender':v['sex'],'birth':v['birth'],'unit':v['unit']}\n",
|
||
" list1.append(dict2)\n",
|
||
"json_data = json.dumps(list1,ensure_ascii=False, indent=4) \n",
|
||
"\n",
|
||
"# 将 json 数据写入文件\n",
|
||
"with open(\"data/data_南京化工人员.json\", \"w\",encoding = 'utf-8') as file:\n",
|
||
" file.write(json_data) \n",
|
||
"print('ok')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "6bcd45c2-10af-4d5f-9e0b-5cd1df4f7a7f",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 获取人员测试成绩"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "970e171e-1360-448f-a28c-520ccb8f314a",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import datetime\n",
|
||
"import csv\n",
|
||
"from datetime import date\n",
|
||
"import my_module as My\n",
|
||
"\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl) \n",
|
||
"filename = 'data/marks_20250703.csv'\n",
|
||
"re_ta = My.get_result(filename,dict1)\n",
|
||
"\n",
|
||
"\n",
|
||
"filename = 'data/result_南京化工.json'\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(re_ta, fl, ensure_ascii=False) \n",
|
||
"print(len(re_ta))"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "6d381859-6d21-45d2-8313-55ebbabf8bd4",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 生成测试得分"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "0905ad6a-f7e4-43ed-ae29-c4850deb946b",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import time\n",
|
||
"import my_module as My\n",
|
||
"\n",
|
||
"#list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','height','weight']\n",
|
||
"list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']\n",
|
||
"filename = 'data/result_南京化工.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict2 = json.load(fl) \n",
|
||
"for k, v in dict2.items():\n",
|
||
" #print(k)\n",
|
||
" if v['sex'] == '男':\n",
|
||
" sex = 'M'\n",
|
||
" else:\n",
|
||
" sex = 'F' \n",
|
||
" if 'bmi' in v.keys():\n",
|
||
" #bmi_data = v['height']['成绩'].split()[0]+','+ v['weight']['成绩'].split()[0]\n",
|
||
" bmi_data = v['bmi']['成绩']\n",
|
||
" data1 = {'code':k,'sex':sex,'age':v['age'],'item':'HeightWeight','result':bmi_data}\n",
|
||
" dict2[k]['bmi'] = {}\n",
|
||
" dict2[k]['bmi']['成绩'] = bmi_data\n",
|
||
" dict2[k]['bmi']['score'] = My.cal_bmi(data1)\n",
|
||
" for item_en in list_item:\n",
|
||
" if item_en in v.keys(): \n",
|
||
" data1 = {'code':k,'sex':sex,'age':v['age'],'item':item_en,'result':float(v[item_en]['成绩'].split()[0])}\n",
|
||
" #print(k,v['name'])\n",
|
||
" dict2[k][item_en]['score'] = My.cal_score(data1)\n",
|
||
" #print(k,v[item_en]['成绩'],cal_score(data1))\n",
|
||
"\n",
|
||
"filename = f'data/result_南京化工.json'\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(dict2,fl , ensure_ascii=False) \n",
|
||
"print('ok!') "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "d034a61d-1fbd-417b-99bd-277e43ebb678",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 导出测试人员信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "867ad6b0-9e9d-48bc-a10a-3b9cb6125890",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"\n",
|
||
"items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']\n",
|
||
"title = ['编号','姓名','性别','单位','部门','身高','体重','肺活量','握力','坐位体前屈','纵跳','俯卧撑','一分钟仰卧起坐','单脚站立','选择反应时','台阶指数']\n",
|
||
"\n",
|
||
"filename = 'data/result_南京化工.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict2 = json.load(fl)\n",
|
||
" \n",
|
||
"list1 = []\n",
|
||
"for k, v in dict1.items():\n",
|
||
" list2 = []\n",
|
||
" list2.append(str(k).rjust(5,'0'))\n",
|
||
" list2.append(v['name']) \n",
|
||
" list2.append(dict2[k]['sex'])\n",
|
||
" list2.append(dict2[k]['unit'])\n",
|
||
" if 'bmi' in v.keys():\n",
|
||
" height = v['bmi']['成绩'].split(',')[0]\n",
|
||
" weight = v['bmi']['成绩'].split(',')[1]\n",
|
||
" list2.append(height)\n",
|
||
" list2.append(weight)\n",
|
||
" else:\n",
|
||
" list2.append('')\n",
|
||
" list2.append('')\n",
|
||
" for item in items:\n",
|
||
" if item in v.keys():\n",
|
||
" list2.append(v[item]['成绩']) \n",
|
||
" elif item =='name':\n",
|
||
" list2.append(v[item])\n",
|
||
" else:\n",
|
||
" list2.append('')\n",
|
||
" \n",
|
||
" list1.append(list2)\n",
|
||
"filename = 'data/南京化工测试情况明细表(截至20250630).xlsx'\n",
|
||
"wb = openpyxl.Workbook()\n",
|
||
"sheet = wb.active\n",
|
||
"sheet.append(title)\n",
|
||
"for row in list1:\n",
|
||
" sheet.append(row)\n",
|
||
" \n",
|
||
"wb.save(filename)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "699c6a40-a6ae-4300-9646-708cb85aa5e8",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 统计问卷人员情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 22,
|
||
"id": "def3f47a-24ef-4815-b63b-2a16b79b4c15",
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2025-08-13T08:03:28.281779Z",
|
||
"iopub.status.busy": "2025-08-13T08:03:28.281182Z",
|
||
"iopub.status.idle": "2025-08-13T08:03:28.305144Z",
|
||
"shell.execute_reply": "2025-08-13T08:03:28.304322Z",
|
||
"shell.execute_reply.started": "2025-08-13T08:03:28.281726Z"
|
||
},
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import csv\n",
|
||
"import openpyxl\n",
|
||
"import time\n",
|
||
"from datetime import date\n",
|
||
"\n",
|
||
"\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"\n",
|
||
"phone1 = set()\n",
|
||
"phone2 = set()\n",
|
||
"for k,v in dict1.items():\n",
|
||
" phone1.add(v['phone'])\n",
|
||
"list1 = []\n",
|
||
"filename = 'data/survey_records_20250813.csv'\n",
|
||
"with open(filename,'r',newline='') as csv_file:\n",
|
||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||
" header = next(fl) \n",
|
||
" for line in fl:\n",
|
||
" list1.append(line)\n",
|
||
"\n",
|
||
"i =1\n",
|
||
"list2 = []\n",
|
||
"for item in list1:\n",
|
||
" content = json.loads(item[4])\n",
|
||
" code = int(content['phone'])\n",
|
||
" for k, v in dict1.items():\n",
|
||
" list3 = []\n",
|
||
" if v['phone'] == code: \n",
|
||
" list3.append(k)\n",
|
||
" list3.append(v['name'])\n",
|
||
" list3.append(v['sex'])\n",
|
||
" list3.append(v['unit'])\n",
|
||
" list3.append(code)\n",
|
||
" list2.append(list3)\n",
|
||
"filename = 'data/南化问卷情况表(第二批).xlsx'\n",
|
||
"wb = openpyxl.Workbook()\n",
|
||
"sheet = wb.active\n",
|
||
"#sheet.append(title)\n",
|
||
"for row in list2:\n",
|
||
" sheet.append(row)\n",
|
||
" \n",
|
||
"wb.save(filename) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 23,
|
||
"id": "8b7c0528-17b6-4469-b15f-3c4b794e286e",
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2025-08-13T08:04:18.256439Z",
|
||
"iopub.status.busy": "2025-08-13T08:04:18.255730Z",
|
||
"iopub.status.idle": "2025-08-13T08:04:18.280866Z",
|
||
"shell.execute_reply": "2025-08-13T08:04:18.280162Z",
|
||
"shell.execute_reply.started": "2025-08-13T08:04:18.256379Z"
|
||
}
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import csv\n",
|
||
"import openpyxl\n",
|
||
"import time\n",
|
||
"from datetime import date\n",
|
||
"\n",
|
||
"\n",
|
||
"\n",
|
||
"filename = 'data/survey_records_20250813.csv'\n",
|
||
"with open(filename,'r',newline='') as csv_file:\n",
|
||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||
" header = next(fl) \n",
|
||
" for line in fl:\n",
|
||
" list1.append(line)\n",
|
||
"\n",
|
||
"i =1\n",
|
||
"list2 = []\n",
|
||
"for item in list1:\n",
|
||
" list3 = []\n",
|
||
" content = json.loads(item[4])\n",
|
||
" phone = int(content['phone'])\n",
|
||
" name = content['name']\n",
|
||
" sex = content['gender']\n",
|
||
" list3.append(name)\n",
|
||
" list3.append(sex)\n",
|
||
" list3.append(phone)\n",
|
||
" list2.append(list3)\n",
|
||
"filename = 'data/南化问卷情况表(第二批).xlsx'\n",
|
||
"wb = openpyxl.Workbook()\n",
|
||
"sheet = wb.active\n",
|
||
"#sheet.append(title)\n",
|
||
"for row in list2:\n",
|
||
" sheet.append(row)\n",
|
||
" \n",
|
||
"wb.save(filename) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "797717d5-47b9-4d47-be6c-8d54bede67d4",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 导入问卷信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "9d9dd574-9fbf-4d0a-96fd-9b55d98021d2",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import csv\n",
|
||
"import openpyxl\n",
|
||
"import time\n",
|
||
"from datetime import date\n",
|
||
"\n",
|
||
"dict1 = {}\n",
|
||
"\n",
|
||
"filename = 'data/result_南京化工.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict3 = json.load(fl)\n",
|
||
"\n",
|
||
"phone = {}\n",
|
||
"for k,v in dict3.items():\n",
|
||
" if 'phone' in v.keys():\n",
|
||
" phone[v['phone']] = k\n",
|
||
"\n",
|
||
"list1 = []\n",
|
||
"filename = 'data/survey_records_20250812.csv'\n",
|
||
"with open(filename,'r',newline='') as csv_file:\n",
|
||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||
" header = next(fl) \n",
|
||
" for line in fl:\n",
|
||
" list1.append(line)\n",
|
||
"\n",
|
||
"\n",
|
||
"\n",
|
||
"nn = 0\n",
|
||
"for item in list1:\n",
|
||
" if int(item[3]) in phone.keys(): \n",
|
||
" tcm = []\n",
|
||
" code = phone[int(item[3])]\n",
|
||
" \n",
|
||
" for i in range(0,60):\n",
|
||
" tcm.append(0)\n",
|
||
" \n",
|
||
" \n",
|
||
" content = json.loads(item[4])\n",
|
||
" if code not in dict1.keys():\n",
|
||
" dict1[code] = dict3[code]\n",
|
||
" rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])\n",
|
||
" dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))\n",
|
||
" #dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]\n",
|
||
" else:\n",
|
||
" rq=date.fromisoformat('2025-07-01')\n",
|
||
" dict1[code]['rq'] = '2025-07-01'\n",
|
||
" for k, v in content.items():\n",
|
||
" \n",
|
||
" if 'tcm' in k:\n",
|
||
" i = int(k[3:])\n",
|
||
" tcm[i-1] = int(v) \n",
|
||
" \n",
|
||
" if 'tcm' in item[4]: \n",
|
||
" dict1[code]['tcm'] = tcm\n",
|
||
" \n",
|
||
" birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))\n",
|
||
" \n",
|
||
" days = (rq-birth).days \n",
|
||
" dict1[code]['age'] = int(days/365)\n",
|
||
" dict1[code]['month'] = int(days/365*12)\n",
|
||
" #print(phone[item[2]])\n",
|
||
" nn+=1\n",
|
||
"filename = 'data/result_南京化工-2.json'\n",
|
||
"\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(dict1, fl, ensure_ascii=False)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "9b801085-ca98-4958-8dcc-bcd9985fcd4b",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import csv\n",
|
||
"import openpyxl\n",
|
||
"import time\n",
|
||
"from datetime import date\n",
|
||
"\n",
|
||
"dict1 = {}\n",
|
||
"\n",
|
||
"filename = 'data/result_南京化工.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict3 = json.load(fl)\n",
|
||
"\n",
|
||
"phone = {}\n",
|
||
"for k,v in dict3.items():\n",
|
||
" if 'phone' in v.keys():\n",
|
||
" phone[v['phone']] = k\n",
|
||
"\n",
|
||
"list1 = []\n",
|
||
"filename = 'data/survey_records_20250812.csv'\n",
|
||
"with open(filename,'r',newline='') as csv_file:\n",
|
||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||
" header = next(fl) \n",
|
||
" for line in fl:\n",
|
||
" list1.append(line)\n",
|
||
"\n",
|
||
"\n",
|
||
"\n",
|
||
"nn = 0\n",
|
||
"for item in list1:\n",
|
||
" \n",
|
||
" tcm = []\n",
|
||
" \n",
|
||
" \n",
|
||
" for i in range(0,60):\n",
|
||
" tcm.append(0)\n",
|
||
" \n",
|
||
" content = json.loads(item[4])\n",
|
||
" if int(content['phone']) in phone.keys(): \n",
|
||
" code = phone[int(content['phone'])]\n",
|
||
" print(code)\n",
|
||
" if code not in dict1.keys():\n",
|
||
" dict1[code] = dict3[code]\n",
|
||
" rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])\n",
|
||
" dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))\n",
|
||
" #dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]\n",
|
||
" else:\n",
|
||
" rq=date.fromisoformat('2025-08-12')\n",
|
||
" dict1[code]['rq'] = '2025-08-12'\n",
|
||
" for k, v in content.items():\n",
|
||
" \n",
|
||
" if 'tcm' in k:\n",
|
||
" i = int(k[3:])\n",
|
||
" tcm[i-1] = int(v) \n",
|
||
" \n",
|
||
" if 'tcm' in item[4]: \n",
|
||
" dict1[code]['tcm'] = tcm\n",
|
||
" \n",
|
||
" birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))\n",
|
||
" \n",
|
||
" days = (rq-birth).days \n",
|
||
" dict1[code]['age'] = int(days/365)\n",
|
||
" dict1[code]['month'] = int(days/365*12)\n",
|
||
" #print(phone[item[2]])\n",
|
||
" nn+=1\n",
|
||
"filename = 'data/result_南京化工-2.json'\n",
|
||
"\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(dict1, fl, ensure_ascii=False)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "89689b86-3fae-402e-a456-a646f0c7201f",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 导入腰臀数据"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "4a0678c6-0c64-4314-bb03-24b10a3d695a",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import csv\n",
|
||
"import openpyxl\n",
|
||
"import time\n",
|
||
"from datetime import date\n",
|
||
"\n",
|
||
"dict1 = {}\n",
|
||
"\n",
|
||
"filename = 'data/result_南京化工-1.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"wb = openpyxl.load_workbook('data/南化腰臀数据.xlsx',data_only=True)\n",
|
||
"sheet = wb.active\n",
|
||
"# sheets = wb.sheetnames\n",
|
||
"person = {}\n",
|
||
"\n",
|
||
"for n in range(2, sheet.max_row+1):\n",
|
||
" code = str(sheet.cell(n, 1).value)\n",
|
||
" if code in dict1.keys():\n",
|
||
" yao = str(sheet.cell(n, 2).value)\n",
|
||
" tun = str(sheet.cell(n, 3).value)\n",
|
||
" dict1[code]['腰臀比'] = yao+','+tun\n",
|
||
"filename = 'data/result_南京化工-1.json'\n",
|
||
"\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(dict1, fl, ensure_ascii=False)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "caf77c95-3090-4fa1-bcec-c9e8a38d4ca9",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 体检报告汇总"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "b815a478-d178-4b87-9080-a779d397d945",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"from pathlib import Path\n",
|
||
"import json\n",
|
||
"import shutil\n",
|
||
"\n",
|
||
"\n",
|
||
"target_directory = Path('./file/南化体重')\n",
|
||
"new_path = './file/南化体重/new'\n",
|
||
"filename = 'data/南京化工人员.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"# 遍历目标目录及其子目录获取所有文件\n",
|
||
"\n",
|
||
"for fl in target_directory.rglob('*.pdf'):\n",
|
||
" if fl.is_file():\n",
|
||
" fl_name = fl.stem\n",
|
||
" name = fl_name[12:] \n",
|
||
" for k, v in dict1.items(): \n",
|
||
" if name == v['name']:\n",
|
||
" n_name = Path(new_path,str(k)+'-'+name+'.pdf')\n",
|
||
" shutil.copyfile(fl,n_name)\n",
|
||
" print(n_name)\n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "4f192b79-5dc7-4517-b7f3-409e30d60dad",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"from pathlib import Path\n",
|
||
"import json\n",
|
||
"import shutil\n",
|
||
"import pymupdf4llm\n",
|
||
"#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n",
|
||
"llama_reader = pymupdf4llm.LlamaMarkdownReader()\n",
|
||
"#llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\")\n",
|
||
"\n",
|
||
"\n",
|
||
"target_directory = Path('./file/南化体重/new')\n",
|
||
"new_path = './file/南化体重/md'\n",
|
||
"\n",
|
||
"\n",
|
||
"for fl in target_directory.rglob('*.pdf'):\n",
|
||
" if fl.is_file():\n",
|
||
" fl_name = fl.stem\n",
|
||
" llama_lists = pymupdf4llm.to_markdown(fl,page_chunks=True)\n",
|
||
" list1 = []\n",
|
||
" for item in llama_lists:\n",
|
||
" list1.append(item['text'])\n",
|
||
" llama_docs = '\\n'.join(list1)\n",
|
||
" Path(new_path,fl_name+'.md').write_bytes(llama_docs.encode())\n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "e5863760-0d11-45b0-acb3-1a38c78d4fc7",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"from pathlib import Path\n",
|
||
"import json\n",
|
||
"import shutil\n",
|
||
"import pymupdf4llm\n",
|
||
"#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n",
|
||
"llama_reader = pymupdf4llm.LlamaMarkdownReader()\n",
|
||
"llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\",page_chunks=True)\n",
|
||
"print(llama_docs)\n",
|
||
"\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "8d5f4103-0d1e-4711-b324-ece360f8dcd3",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 生成报告"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "23ffd115-72a3-4f90-9e64-ffa6420df8a4",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"\n",
|
||
"\n",
|
||
"headers = {\n",
|
||
" \"Content-Type\": \"application/json; charset=UTF-8\"\n",
|
||
" }\n",
|
||
"filename = 'data/result_南京化工-1.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl)\n",
|
||
"list1 = []\n",
|
||
"file_path ='./南京化工/'\n",
|
||
"list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']\n",
|
||
"i=0\n",
|
||
"list2 = []\n",
|
||
"for k, v in dict1.items():\n",
|
||
" list1 = []\n",
|
||
" mydata = {}\n",
|
||
" \n",
|
||
" id = str(k).rjust(4,\"0\")\n",
|
||
" mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'\n",
|
||
" mydata['title'] = '南化公司'\n",
|
||
" mydata['subtitle'] = v['unit']\n",
|
||
" mydata['id'] = id\n",
|
||
" mydata['name'] = v['name']\n",
|
||
" if v['sex'] == '男':\n",
|
||
" mydata['gender'] = 'male'\n",
|
||
" else:\n",
|
||
" mydata['gender'] = 'female'\n",
|
||
" \n",
|
||
" mydata['month'] = v['month']\n",
|
||
" mydata['fits'] = {}\n",
|
||
" survey_list = ['tcm','psy57','spine']\n",
|
||
" for item in survey_list:\n",
|
||
" if item in v.keys():\n",
|
||
" mydata.setdefault('surveys',{})\n",
|
||
" mydata['surveys'][item] = v[item]\n",
|
||
" \n",
|
||
" \n",
|
||
" #mydata['fits'] = {}\n",
|
||
" for item in list_item:\n",
|
||
" if item in v.keys():\n",
|
||
" mydata.setdefault('fits',{})\n",
|
||
" if item in ['lung','pushup','step','situp']:\n",
|
||
" mark = v[item]['成绩'].split()[0].split('.')[0]\n",
|
||
" else:\n",
|
||
" mark = v[item]['成绩'].split()[0]\n",
|
||
" mydata['fits'][item] = {'mark':mark,'score':v[item]['score']}\n",
|
||
" if len(mydata['fits']) >2 or len(mydata['surveys']) >0:\n",
|
||
" #if len(mydata['fits']) >2 : \n",
|
||
" list1.append(mydata)\n",
|
||
" list2.append([k,v['name']])\n",
|
||
" i+=1\n",
|
||
" x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)\n",
|
||
" #print(id,v['name'],x.text)\n",
|
||
" #print(mydata)\n",
|
||
" #x.close()\n",
|
||
"print(i)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "87556725-f68a-484b-827c-f6735155e940",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
}
|
||
],
|
||
"metadata": {
|
||
"kernelspec": {
|
||
"display_name": "Python 3 (ipykernel)",
|
||
"language": "python",
|
||
"name": "python3"
|
||
},
|
||
"language_info": {
|
||
"codemirror_mode": {
|
||
"name": "ipython",
|
||
"version": 3
|
||
},
|
||
"file_extension": ".py",
|
||
"mimetype": "text/x-python",
|
||
"name": "python",
|
||
"nbconvert_exporter": "python",
|
||
"pygments_lexer": "ipython3",
|
||
"version": "3.12.3"
|
||
}
|
||
},
|
||
"nbformat": 4,
|
||
"nbformat_minor": 5
|
||
}
|