20250813
This commit is contained in:
1 parent
c0db3d691e
commit
dcbfb8f1eb
5 files changed
+592
-384
No files matched your search
+175
-21
@@ -241,9 +241,16 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"execution_count": 22,
|
||||
"id": "def3f47a-24ef-4815-b63b-2a16b79b4c15",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2025-08-13T08:03:28.281779Z",
|
||||
"iopub.status.busy": "2025-08-13T08:03:28.281182Z",
|
||||
"iopub.status.idle": "2025-08-13T08:03:28.305144Z",
|
||||
"shell.execute_reply": "2025-08-13T08:03:28.304322Z",
|
||||
"shell.execute_reply.started": "2025-08-13T08:03:28.281726Z"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [],
|
||||
@@ -264,7 +271,7 @@
|
||||
"for k,v in dict1.items():\n",
|
||||
" phone1.add(v['phone'])\n",
|
||||
"list1 = []\n",
|
||||
"filename = 'data/survey_records_20250711.csv'\n",
|
||||
"filename = 'data/survey_records_20250813.csv'\n",
|
||||
"with open(filename,'r',newline='') as csv_file:\n",
|
||||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||||
" header = next(fl) \n",
|
||||
@@ -274,7 +281,8 @@
|
||||
"i =1\n",
|
||||
"list2 = []\n",
|
||||
"for item in list1:\n",
|
||||
" code = int(item[3])\n",
|
||||
" content = json.loads(item[4])\n",
|
||||
" code = int(content['phone'])\n",
|
||||
" for k, v in dict1.items():\n",
|
||||
" list3 = []\n",
|
||||
" if v['phone'] == code: \n",
|
||||
@@ -284,7 +292,7 @@
|
||||
" list3.append(v['unit'])\n",
|
||||
" list3.append(code)\n",
|
||||
" list2.append(list3)\n",
|
||||
"filename = 'data/南化问卷情况表.xlsx'\n",
|
||||
"filename = 'data/南化问卷情况表(第二批).xlsx'\n",
|
||||
"wb = openpyxl.Workbook()\n",
|
||||
"sheet = wb.active\n",
|
||||
"#sheet.append(title)\n",
|
||||
@@ -296,9 +304,17 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "9cb0172e-5f6c-4b47-8431-83dc5cc4601c",
|
||||
"metadata": {},
|
||||
"execution_count": 23,
|
||||
"id": "8b7c0528-17b6-4469-b15f-3c4b794e286e",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2025-08-13T08:04:18.256439Z",
|
||||
"iopub.status.busy": "2025-08-13T08:04:18.255730Z",
|
||||
"iopub.status.idle": "2025-08-13T08:04:18.280866Z",
|
||||
"shell.execute_reply": "2025-08-13T08:04:18.280162Z",
|
||||
"shell.execute_reply.started": "2025-08-13T08:04:18.256379Z"
|
||||
}
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
@@ -308,26 +324,34 @@
|
||||
"from datetime import date\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"filename = 'data/南京化工人员.json'\n",
|
||||
"with open(filename,'r') as fl:\n",
|
||||
" dict1 = json.load(fl)\n",
|
||||
"\n",
|
||||
"phone1 = set()\n",
|
||||
"phone2 = set()\n",
|
||||
"for k,v in dict1.items():\n",
|
||||
" phone1.add(v['phone'])\n",
|
||||
"list1 = []\n",
|
||||
"filename = 'data/survey_records_20250711.csv'\n",
|
||||
"filename = 'data/survey_records_20250813.csv'\n",
|
||||
"with open(filename,'r',newline='') as csv_file:\n",
|
||||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||||
" header = next(fl) \n",
|
||||
" for line in fl:\n",
|
||||
" list1.append(line)\n",
|
||||
"\n",
|
||||
"i =1\n",
|
||||
"list2 = []\n",
|
||||
"for item in list1:\n",
|
||||
" code = int(item[3]) \n",
|
||||
" if code not in phone1: \n",
|
||||
" print(code)"
|
||||
" list3 = []\n",
|
||||
" content = json.loads(item[4])\n",
|
||||
" phone = int(content['phone'])\n",
|
||||
" name = content['name']\n",
|
||||
" sex = content['gender']\n",
|
||||
" list3.append(name)\n",
|
||||
" list3.append(sex)\n",
|
||||
" list3.append(phone)\n",
|
||||
" list2.append(list3)\n",
|
||||
"filename = 'data/南化问卷情况表(第二批).xlsx'\n",
|
||||
"wb = openpyxl.Workbook()\n",
|
||||
"sheet = wb.active\n",
|
||||
"#sheet.append(title)\n",
|
||||
"for row in list2:\n",
|
||||
" sheet.append(row)\n",
|
||||
" \n",
|
||||
"wb.save(filename) "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -366,7 +390,7 @@
|
||||
" phone[v['phone']] = k\n",
|
||||
"\n",
|
||||
"list1 = []\n",
|
||||
"filename = 'data/survey_records_20250711.csv'\n",
|
||||
"filename = 'data/survey_records_20250812.csv'\n",
|
||||
"with open(filename,'r',newline='') as csv_file:\n",
|
||||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||||
" header = next(fl) \n",
|
||||
@@ -410,12 +434,92 @@
|
||||
" dict1[code]['month'] = int(days/365*12)\n",
|
||||
" #print(phone[item[2]])\n",
|
||||
" nn+=1\n",
|
||||
"filename = 'data/result_南京化工-1.json'\n",
|
||||
"filename = 'data/result_南京化工-2.json'\n",
|
||||
"\n",
|
||||
"with open(filename,'w') as fl:\n",
|
||||
" json.dump(dict1, fl, ensure_ascii=False)\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "9b801085-ca98-4958-8dcc-bcd9985fcd4b",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"import csv\n",
|
||||
"import openpyxl\n",
|
||||
"import time\n",
|
||||
"from datetime import date\n",
|
||||
"\n",
|
||||
"dict1 = {}\n",
|
||||
"\n",
|
||||
"filename = 'data/result_南京化工.json'\n",
|
||||
"with open(filename,'r') as fl:\n",
|
||||
" dict1 = json.load(fl)\n",
|
||||
"filename = 'data/南京化工人员.json'\n",
|
||||
"with open(filename,'r') as fl:\n",
|
||||
" dict3 = json.load(fl)\n",
|
||||
"\n",
|
||||
"phone = {}\n",
|
||||
"for k,v in dict3.items():\n",
|
||||
" if 'phone' in v.keys():\n",
|
||||
" phone[v['phone']] = k\n",
|
||||
"\n",
|
||||
"list1 = []\n",
|
||||
"filename = 'data/survey_records_20250812.csv'\n",
|
||||
"with open(filename,'r',newline='') as csv_file:\n",
|
||||
" fl = csv.reader(csv_file,delimiter=',')\n",
|
||||
" header = next(fl) \n",
|
||||
" for line in fl:\n",
|
||||
" list1.append(line)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"nn = 0\n",
|
||||
"for item in list1:\n",
|
||||
" \n",
|
||||
" tcm = []\n",
|
||||
" \n",
|
||||
" \n",
|
||||
" for i in range(0,60):\n",
|
||||
" tcm.append(0)\n",
|
||||
" \n",
|
||||
" content = json.loads(item[4])\n",
|
||||
" if int(content['phone']) in phone.keys(): \n",
|
||||
" code = phone[int(content['phone'])]\n",
|
||||
" print(code)\n",
|
||||
" if code not in dict1.keys():\n",
|
||||
" dict1[code] = dict3[code]\n",
|
||||
" rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])\n",
|
||||
" dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))\n",
|
||||
" #dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]\n",
|
||||
" else:\n",
|
||||
" rq=date.fromisoformat('2025-08-12')\n",
|
||||
" dict1[code]['rq'] = '2025-08-12'\n",
|
||||
" for k, v in content.items():\n",
|
||||
" \n",
|
||||
" if 'tcm' in k:\n",
|
||||
" i = int(k[3:])\n",
|
||||
" tcm[i-1] = int(v) \n",
|
||||
" \n",
|
||||
" if 'tcm' in item[4]: \n",
|
||||
" dict1[code]['tcm'] = tcm\n",
|
||||
" \n",
|
||||
" birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))\n",
|
||||
" \n",
|
||||
" days = (rq-birth).days \n",
|
||||
" dict1[code]['age'] = int(days/365)\n",
|
||||
" dict1[code]['month'] = int(days/365*12)\n",
|
||||
" #print(phone[item[2]])\n",
|
||||
" nn+=1\n",
|
||||
"filename = 'data/result_南京化工-2.json'\n",
|
||||
"\n",
|
||||
"with open(filename,'w') as fl:\n",
|
||||
" json.dump(dict1, fl, ensure_ascii=False)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "89689b86-3fae-402e-a456-a646f0c7201f",
|
||||
@@ -499,6 +603,56 @@
|
||||
" "
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "4f192b79-5dc7-4517-b7f3-409e30d60dad",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from pathlib import Path\n",
|
||||
"import json\n",
|
||||
"import shutil\n",
|
||||
"import pymupdf4llm\n",
|
||||
"#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n",
|
||||
"llama_reader = pymupdf4llm.LlamaMarkdownReader()\n",
|
||||
"#llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"target_directory = Path('./file/南化体重/new')\n",
|
||||
"new_path = './file/南化体重/md'\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"for fl in target_directory.rglob('*.pdf'):\n",
|
||||
" if fl.is_file():\n",
|
||||
" fl_name = fl.stem\n",
|
||||
" llama_lists = pymupdf4llm.to_markdown(fl,page_chunks=True)\n",
|
||||
" list1 = []\n",
|
||||
" for item in llama_lists:\n",
|
||||
" list1.append(item['text'])\n",
|
||||
" llama_docs = '\\n'.join(list1)\n",
|
||||
" Path(new_path,fl_name+'.md').write_bytes(llama_docs.encode())\n",
|
||||
" "
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "e5863760-0d11-45b0-acb3-1a38c78d4fc7",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from pathlib import Path\n",
|
||||
"import json\n",
|
||||
"import shutil\n",
|
||||
"import pymupdf4llm\n",
|
||||
"#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n",
|
||||
"llama_reader = pymupdf4llm.LlamaMarkdownReader()\n",
|
||||
"llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\",page_chunks=True)\n",
|
||||
"print(llama_docs)\n",
|
||||
"\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "8d5f4103-0d1e-4711-b324-ece360f8dcd3",
|
||||
|
||||
Reference in new issue
Block a user