This commit is contained in:
512song committed 2025-08-13 20:36:47 +08:00
1 parent c0db3d691e
commit dcbfb8f1eb
5 files changed
+592 -384

No files matched your search

+175 -21
View File
@@ -241,9 +241,16 @@
},
{
"cell_type": "code",
"execution_count": null,
"execution_count": 22,
"id": "def3f47a-24ef-4815-b63b-2a16b79b4c15",
"metadata": {
"execution": {
"iopub.execute_input": "2025-08-13T08:03:28.281779Z",
"iopub.status.busy": "2025-08-13T08:03:28.281182Z",
"iopub.status.idle": "2025-08-13T08:03:28.305144Z",
"shell.execute_reply": "2025-08-13T08:03:28.304322Z",
"shell.execute_reply.started": "2025-08-13T08:03:28.281726Z"
},
"tags": []
},
"outputs": [],
@@ -264,7 +271,7 @@
"for k,v in dict1.items():\n",
" phone1.add(v['phone'])\n",
"list1 = []\n",
"filename = 'data/survey_records_20250711.csv'\n",
"filename = 'data/survey_records_20250813.csv'\n",
"with open(filename,'r',newline='') as csv_file:\n",
" fl = csv.reader(csv_file,delimiter=',')\n",
" header = next(fl) \n",
@@ -274,7 +281,8 @@
"i =1\n",
"list2 = []\n",
"for item in list1:\n",
" code = int(item[3])\n",
" content = json.loads(item[4])\n",
" code = int(content['phone'])\n",
" for k, v in dict1.items():\n",
" list3 = []\n",
" if v['phone'] == code: \n",
@@ -284,7 +292,7 @@
" list3.append(v['unit'])\n",
" list3.append(code)\n",
" list2.append(list3)\n",
"filename = 'data/南化问卷情况表.xlsx'\n",
"filename = 'data/南化问卷情况表(第二批).xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"#sheet.append(title)\n",
@@ -296,9 +304,17 @@
},
{
"cell_type": "code",
"execution_count": null,
"id": "9cb0172e-5f6c-4b47-8431-83dc5cc4601c",
"metadata": {},
"execution_count": 23,
"id": "8b7c0528-17b6-4469-b15f-3c4b794e286e",
"metadata": {
"execution": {
"iopub.execute_input": "2025-08-13T08:04:18.256439Z",
"iopub.status.busy": "2025-08-13T08:04:18.255730Z",
"iopub.status.idle": "2025-08-13T08:04:18.280866Z",
"shell.execute_reply": "2025-08-13T08:04:18.280162Z",
"shell.execute_reply.started": "2025-08-13T08:04:18.256379Z"
}
},
"outputs": [],
"source": [
"import json\n",
@@ -308,26 +324,34 @@
"from datetime import date\n",
"\n",
"\n",
"filename = 'data/南京化工人员.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"\n",
"phone1 = set()\n",
"phone2 = set()\n",
"for k,v in dict1.items():\n",
" phone1.add(v['phone'])\n",
"list1 = []\n",
"filename = 'data/survey_records_20250711.csv'\n",
"filename = 'data/survey_records_20250813.csv'\n",
"with open(filename,'r',newline='') as csv_file:\n",
" fl = csv.reader(csv_file,delimiter=',')\n",
" header = next(fl) \n",
" for line in fl:\n",
" list1.append(line)\n",
"\n",
"i =1\n",
"list2 = []\n",
"for item in list1:\n",
" code = int(item[3]) \n",
" if code not in phone1: \n",
" print(code)"
" list3 = []\n",
" content = json.loads(item[4])\n",
" phone = int(content['phone'])\n",
" name = content['name']\n",
" sex = content['gender']\n",
" list3.append(name)\n",
" list3.append(sex)\n",
" list3.append(phone)\n",
" list2.append(list3)\n",
"filename = 'data/南化问卷情况表(第二批).xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"#sheet.append(title)\n",
"for row in list2:\n",
" sheet.append(row)\n",
" \n",
"wb.save(filename) "
]
},
{
@@ -366,7 +390,7 @@
" phone[v['phone']] = k\n",
"\n",
"list1 = []\n",
"filename = 'data/survey_records_20250711.csv'\n",
"filename = 'data/survey_records_20250812.csv'\n",
"with open(filename,'r',newline='') as csv_file:\n",
" fl = csv.reader(csv_file,delimiter=',')\n",
" header = next(fl) \n",
@@ -410,12 +434,92 @@
" dict1[code]['month'] = int(days/365*12)\n",
" #print(phone[item[2]])\n",
" nn+=1\n",
"filename = 'data/result_南京化工-1.json'\n",
"filename = 'data/result_南京化工-2.json'\n",
"\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "9b801085-ca98-4958-8dcc-bcd9985fcd4b",
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import csv\n",
"import openpyxl\n",
"import time\n",
"from datetime import date\n",
"\n",
"dict1 = {}\n",
"\n",
"filename = 'data/result_南京化工.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"filename = 'data/南京化工人员.json'\n",
"with open(filename,'r') as fl:\n",
" dict3 = json.load(fl)\n",
"\n",
"phone = {}\n",
"for k,v in dict3.items():\n",
" if 'phone' in v.keys():\n",
" phone[v['phone']] = k\n",
"\n",
"list1 = []\n",
"filename = 'data/survey_records_20250812.csv'\n",
"with open(filename,'r',newline='') as csv_file:\n",
" fl = csv.reader(csv_file,delimiter=',')\n",
" header = next(fl) \n",
" for line in fl:\n",
" list1.append(line)\n",
"\n",
"\n",
"\n",
"nn = 0\n",
"for item in list1:\n",
" \n",
" tcm = []\n",
" \n",
" \n",
" for i in range(0,60):\n",
" tcm.append(0)\n",
" \n",
" content = json.loads(item[4])\n",
" if int(content['phone']) in phone.keys(): \n",
" code = phone[int(content['phone'])]\n",
" print(code)\n",
" if code not in dict1.keys():\n",
" dict1[code] = dict3[code]\n",
" rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])\n",
" dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))\n",
" #dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]\n",
" else:\n",
" rq=date.fromisoformat('2025-08-12')\n",
" dict1[code]['rq'] = '2025-08-12'\n",
" for k, v in content.items():\n",
" \n",
" if 'tcm' in k:\n",
" i = int(k[3:])\n",
" tcm[i-1] = int(v) \n",
" \n",
" if 'tcm' in item[4]: \n",
" dict1[code]['tcm'] = tcm\n",
" \n",
" birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))\n",
" \n",
" days = (rq-birth).days \n",
" dict1[code]['age'] = int(days/365)\n",
" dict1[code]['month'] = int(days/365*12)\n",
" #print(phone[item[2]])\n",
" nn+=1\n",
"filename = 'data/result_南京化工-2.json'\n",
"\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False)"
]
},
{
"cell_type": "markdown",
"id": "89689b86-3fae-402e-a456-a646f0c7201f",
@@ -499,6 +603,56 @@
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "4f192b79-5dc7-4517-b7f3-409e30d60dad",
"metadata": {},
"outputs": [],
"source": [
"from pathlib import Path\n",
"import json\n",
"import shutil\n",
"import pymupdf4llm\n",
"#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n",
"llama_reader = pymupdf4llm.LlamaMarkdownReader()\n",
"#llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\")\n",
"\n",
"\n",
"target_directory = Path('./file/南化体重/new')\n",
"new_path = './file/南化体重/md'\n",
"\n",
"\n",
"for fl in target_directory.rglob('*.pdf'):\n",
" if fl.is_file():\n",
" fl_name = fl.stem\n",
" llama_lists = pymupdf4llm.to_markdown(fl,page_chunks=True)\n",
" list1 = []\n",
" for item in llama_lists:\n",
" list1.append(item['text'])\n",
" llama_docs = '\\n'.join(list1)\n",
" Path(new_path,fl_name+'.md').write_bytes(llama_docs.encode())\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e5863760-0d11-45b0-acb3-1a38c78d4fc7",
"metadata": {},
"outputs": [],
"source": [
"from pathlib import Path\n",
"import json\n",
"import shutil\n",
"import pymupdf4llm\n",
"#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n",
"llama_reader = pymupdf4llm.LlamaMarkdownReader()\n",
"llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\",page_chunks=True)\n",
"print(llama_docs)\n",
"\n"
]
},
{
"cell_type": "markdown",
"id": "8d5f4103-0d1e-4711-b324-ece360f8dcd3",