diff --git a/体测单位/体质检测数据处理.ipynb b/体测单位/体质检测数据处理.ipynb index eb832bf..c47f845 100644 --- a/体测单位/体质检测数据处理.ipynb +++ b/体测单位/体质检测数据处理.ipynb @@ -2376,15 +2376,15 @@ }, { "cell_type": "code", - "execution_count": 35, + "execution_count": 1, "id": "6f9768ce-2ef5-4253-9fc6-49dbdcf00a19", "metadata": { "execution": { - "iopub.execute_input": "2025-06-30T06:04:34.269580Z", - "iopub.status.busy": "2025-06-30T06:04:34.268925Z", - "iopub.status.idle": "2025-06-30T06:04:34.282224Z", - "shell.execute_reply": "2025-06-30T06:04:34.281280Z", - "shell.execute_reply.started": "2025-06-30T06:04:34.269514Z" + "iopub.execute_input": "2025-07-27T15:45:05.548263Z", + "iopub.status.busy": "2025-07-27T15:45:05.547435Z", + "iopub.status.idle": "2025-07-27T15:45:05.559268Z", + "shell.execute_reply": "2025-07-27T15:45:05.558385Z", + "shell.execute_reply.started": "2025-07-27T15:45:05.548183Z" } }, "outputs": [], @@ -2515,16 +2515,32 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 2, "id": "1acaf159-637e-4101-9881-f6bb79400136", "metadata": { + "execution": { + "iopub.execute_input": "2025-07-27T15:45:48.440829Z", + "iopub.status.busy": "2025-07-27T15:45:48.440078Z", + "iopub.status.idle": "2025-07-27T15:45:48.629122Z", + "shell.execute_reply": "2025-07-27T15:45:48.628652Z", + "shell.execute_reply.started": "2025-07-27T15:45:48.440759Z" + }, "tags": [] }, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[['48', '气虚', False, 65, 53, 28, 15, 0, 0, 10, 3, 0], ['24', '痰湿', False, 59, 43, 25, 37, 53, 50, 17, 32, 25], ['30', '平和', False, 78, 15, 7, 6, 6, 0, 7, 3, 7], ['49', '气虚', False, 62, 43, 21, 31, 31, 25, 3, 10, 0], ['61', '平和', False, 84, 9, 28, 21, 25, 29, 28, 21, 3], ['17', '阳虚', False, 62, 31, 50, 40, 25, 25, 21, 7, 14], ['50', '气虚', False, 62, 40, 25, 25, 34, 29, 32, 39, 25], ['51', '平和', False, 96, 9, 0, 0, 12, 8, 3, 0, 3], ['59', '气虚', False, 43, 43, 32, 31, 31, 16, 17, 39, 35], ['52', '阳虚', False, 37, 50, 64, 28, 31, 50, 35, 28, 10], ['60', '湿热', False, 43, 65, 53, 43, 68, 70, 53, 46, 53], ['19', '阴虚', True, 93, 18, 28, 37, 21, 37, 28, 10, 17], ['29', '平和', False, 68, 12, 7, 0, 12, 12, 10, 14, 3], ['18', '气虚', False, 65, 50, 32, 28, 21, 12, 32, 7, 46], ['16', '气虚', True, 75, 34, 21, 25, 25, 25, 25, 25, 21], ['14', '平和', False, 93, 3, 0, 0, 0, 0, 0, 0, 0], ['71', '阳虚', False, 59, 40, 75, 50, 56, 62, 39, 32, 21], ['70', '平和', False, 78, 18, 0, 0, 0, 8, 0, 0, 21], ['73', '平和', False, 68, 28, 0, 6, 15, 16, 14, 17, 17], ['36', '痰湿', False, 62, 37, 35, 31, 43, 37, 25, 35, 21], ['13', '血瘀', False, 40, 65, 64, 62, 68, 75, 82, 60, 57], ['53', '痰湿', True, 84, 18, 3, 28, 31, 16, 14, 3, 0], ['57', '气虚', False, 53, 43, 25, 9, 34, 25, 10, 21, 21], ['56', '气虚', False, 43, 59, 57, 28, 43, 37, 35, 32, 32], ['58', '阳虚', False, 43, 53, 57, 53, 46, 45, 42, 46, 42], ['10', '平和', False, 100, 9, 0, 0, 0, 4, 0, 0, 7], ['26', '阴虚', False, 62, 34, 21, 46, 34, 33, 35, 42, 46], ['8', '湿热', False, 62, 43, 25, 31, 37, 54, 28, 35, 32], ['55', '阳虚', False, 56, 37, 50, 50, 50, 45, 42, 42, 46], ['54', '平和', False, 71, 9, 14, 18, 28, 29, 14, 25, 10], ['3', '特禀', False, 68, 37, 28, 28, 43, 41, 14, 39, 46], ['4', '湿热', False, 56, 40, 53, 37, 34, 54, 14, 7, 14], ['38', '湿热', False, 84, 15, 0, 0, 18, 41, 7, 14, 0], ['39', '痰湿', False, 75, 34, 39, 43, 56, 45, 17, 28, 7], ['40', '湿热', False, 50, 21, 0, 12, 18, 29, 7, 21, 17], ['35', '平和', False, 90, 15, 0, 6, 15, 12, 3, 0, 3], ['33', '湿热', True, 71, 28, 28, 28, 34, 37, 32, 25, 14], ['41', '痰湿', False, 78, 21, 14, 31, 53, 33, 17, 32, 50], ['34', '平和', False, 93, 18, 21, 15, 12, 4, 0, 0, 0], ['22', '气虚', False, 71, 43, 25, 18, 9, 33, 25, 28, 17], ['23', '气虚', False, 50, 46, 32, 31, 34, 16, 32, 21, 3], ['32', '平和', False, 75, 12, 7, 25, 15, 8, 7, 10, 0], ['72', '气虚', False, 50, 46, 46, 31, 40, 41, 35, 39, 28], ['21', '气虚', True, 68, 31, 17, 21, 18, 16, 17, 0, 0], ['20', '特禀', False, 71, 31, 14, 40, 37, 41, 28, 21, 42], ['74', '平和', False, 78, 25, 7, 28, 28, 25, 10, 21, 0], ['31', '湿热', False, 71, 46, 7, 31, 46, 50, 46, 7, 17], ['43', '气郁', False, 37, 43, 35, 21, 53, 45, 21, 71, 32], ['44', '阳虚', False, 75, 37, 60, 34, 50, 50, 10, 14, 46], ['45', '阴虚', False, 84, 25, 14, 56, 43, 50, 3, 17, 3], ['64', '特禀', False, 65, 40, 32, 34, 34, 41, 32, 25, 42], ['47', '湿热', False, 59, 37, 35, 28, 40, 54, 28, 14, 14], ['28', '湿热', False, 62, 50, 35, 46, 50, 54, 42, 28, 42], ['12', '阳虚', False, 53, 46, 50, 50, 43, 50, 35, 42, 32]]\n", + "ok\n" + ] + } + ], "source": [ "import openpyxl\n", "\n", - "filename = 'data/result_党建出版社2025-1.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院-1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", "i = 1\n", @@ -2550,7 +2566,7 @@ " list3.append(item)\n", " list2.append(list3)\n", "print(list2)\n", - "filename = 'data/党建出版社中医情况明细表2024.xlsx'\n", + "filename = 'data/新疆油田采油工艺研究院中医情况明细表.xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "\n", @@ -2640,7 +2656,7 @@ "source": [ "list4 = ['颈椎','胸椎','腰椎','骶尾椎']\n", "list5 = [[0,10,10],[10,17,7],[17,24,6],[24,26,2]]\n", - "filename = 'data/result_北海炼化2024.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院-1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list6 = []\n", @@ -2656,7 +2672,7 @@ " #print(k,list4[i],int(score/list5[i][2]*100)-100)\n", " list7.append(int(score/list5[i][2]*100)-100)\n", " list6.append(list7)\n", - "filename = 'data/北海炼化脊柱情况明细表2024.xlsx'\n", + "filename = 'data/新疆油田采油工艺研究院脊柱情况明细表.xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "\n", @@ -2726,15 +2742,15 @@ }, { "cell_type": "code", - "execution_count": 36, + "execution_count": 3, "id": "9ecd3ad8-c8af-4f3e-90e5-1618fea905a4", "metadata": { "execution": { - "iopub.execute_input": "2025-06-30T06:04:41.412312Z", - "iopub.status.busy": "2025-06-30T06:04:41.411683Z", - "iopub.status.idle": "2025-06-30T06:04:41.442284Z", - "shell.execute_reply": "2025-06-30T06:04:41.441732Z", - "shell.execute_reply.started": "2025-06-30T06:04:41.412252Z" + "iopub.execute_input": "2025-07-27T15:47:28.099497Z", + "iopub.status.busy": "2025-07-27T15:47:28.098682Z", + "iopub.status.idle": "2025-07-27T15:47:28.154741Z", + "shell.execute_reply": "2025-07-27T15:47:28.154263Z", + "shell.execute_reply.started": "2025-07-27T15:47:28.099431Z" } }, "outputs": [], @@ -2742,7 +2758,7 @@ "import openpyxl\n", "\n", "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']\n", - "filename = 'data/result_海淀区老干部大学第二期-1.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院-1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", "data_list = []\n", @@ -2824,7 +2840,7 @@ " list6.append('')\n", " i+=1\n", " data_list.append(list6)\n", - "filename = 'data/海淀区老干部大学第二期测试情况表.xlsx'\n", + "filename = 'data/新疆油田采油工艺研究院测试情况表.xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "#sheet.append(title)\n", diff --git a/体测单位/南京化工.ipynb b/体测单位/南京化工.ipynb index 3ceaa14..85bb539 100644 --- a/体测单位/南京化工.ipynb +++ b/体测单位/南京化工.ipynb @@ -241,9 +241,16 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 22, "id": "def3f47a-24ef-4815-b63b-2a16b79b4c15", "metadata": { + "execution": { + "iopub.execute_input": "2025-08-13T08:03:28.281779Z", + "iopub.status.busy": "2025-08-13T08:03:28.281182Z", + "iopub.status.idle": "2025-08-13T08:03:28.305144Z", + "shell.execute_reply": "2025-08-13T08:03:28.304322Z", + "shell.execute_reply.started": "2025-08-13T08:03:28.281726Z" + }, "tags": [] }, "outputs": [], @@ -264,7 +271,7 @@ "for k,v in dict1.items():\n", " phone1.add(v['phone'])\n", "list1 = []\n", - "filename = 'data/survey_records_20250711.csv'\n", + "filename = 'data/survey_records_20250813.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -274,7 +281,8 @@ "i =1\n", "list2 = []\n", "for item in list1:\n", - " code = int(item[3])\n", + " content = json.loads(item[4])\n", + " code = int(content['phone'])\n", " for k, v in dict1.items():\n", " list3 = []\n", " if v['phone'] == code: \n", @@ -284,7 +292,7 @@ " list3.append(v['unit'])\n", " list3.append(code)\n", " list2.append(list3)\n", - "filename = 'data/南化问卷情况表.xlsx'\n", + "filename = 'data/南化问卷情况表(第二批).xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "#sheet.append(title)\n", @@ -296,9 +304,17 @@ }, { "cell_type": "code", - "execution_count": null, - "id": "9cb0172e-5f6c-4b47-8431-83dc5cc4601c", - "metadata": {}, + "execution_count": 23, + "id": "8b7c0528-17b6-4469-b15f-3c4b794e286e", + "metadata": { + "execution": { + "iopub.execute_input": "2025-08-13T08:04:18.256439Z", + "iopub.status.busy": "2025-08-13T08:04:18.255730Z", + "iopub.status.idle": "2025-08-13T08:04:18.280866Z", + "shell.execute_reply": "2025-08-13T08:04:18.280162Z", + "shell.execute_reply.started": "2025-08-13T08:04:18.256379Z" + } + }, "outputs": [], "source": [ "import json\n", @@ -308,26 +324,34 @@ "from datetime import date\n", "\n", "\n", - "filename = 'data/南京化工人员.json'\n", - "with open(filename,'r') as fl:\n", - " dict1 = json.load(fl)\n", "\n", - "phone1 = set()\n", - "phone2 = set()\n", - "for k,v in dict1.items():\n", - " phone1.add(v['phone'])\n", - "list1 = []\n", - "filename = 'data/survey_records_20250711.csv'\n", + "filename = 'data/survey_records_20250813.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", " for line in fl:\n", " list1.append(line)\n", "\n", + "i =1\n", + "list2 = []\n", "for item in list1:\n", - " code = int(item[3]) \n", - " if code not in phone1: \n", - " print(code)" + " list3 = []\n", + " content = json.loads(item[4])\n", + " phone = int(content['phone'])\n", + " name = content['name']\n", + " sex = content['gender']\n", + " list3.append(name)\n", + " list3.append(sex)\n", + " list3.append(phone)\n", + " list2.append(list3)\n", + "filename = 'data/南化问卷情况表(第二批).xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "#sheet.append(title)\n", + "for row in list2:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename) " ] }, { @@ -366,7 +390,7 @@ " phone[v['phone']] = k\n", "\n", "list1 = []\n", - "filename = 'data/survey_records_20250711.csv'\n", + "filename = 'data/survey_records_20250812.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -410,12 +434,92 @@ " dict1[code]['month'] = int(days/365*12)\n", " #print(phone[item[2]])\n", " nn+=1\n", - "filename = 'data/result_南京化工-1.json'\n", + "filename = 'data/result_南京化工-2.json'\n", "\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl, ensure_ascii=False)\n" ] }, + { + "cell_type": "code", + "execution_count": null, + "id": "9b801085-ca98-4958-8dcc-bcd9985fcd4b", + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import csv\n", + "import openpyxl\n", + "import time\n", + "from datetime import date\n", + "\n", + "dict1 = {}\n", + "\n", + "filename = 'data/result_南京化工.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "filename = 'data/南京化工人员.json'\n", + "with open(filename,'r') as fl:\n", + " dict3 = json.load(fl)\n", + "\n", + "phone = {}\n", + "for k,v in dict3.items():\n", + " if 'phone' in v.keys():\n", + " phone[v['phone']] = k\n", + "\n", + "list1 = []\n", + "filename = 'data/survey_records_20250812.csv'\n", + "with open(filename,'r',newline='') as csv_file:\n", + " fl = csv.reader(csv_file,delimiter=',')\n", + " header = next(fl) \n", + " for line in fl:\n", + " list1.append(line)\n", + "\n", + "\n", + "\n", + "nn = 0\n", + "for item in list1:\n", + " \n", + " tcm = []\n", + " \n", + " \n", + " for i in range(0,60):\n", + " tcm.append(0)\n", + " \n", + " content = json.loads(item[4])\n", + " if int(content['phone']) in phone.keys(): \n", + " code = phone[int(content['phone'])]\n", + " print(code)\n", + " if code not in dict1.keys():\n", + " dict1[code] = dict3[code]\n", + " rq = date.fromisoformat(item[5].replace('/','-').split(' ')[0])\n", + " dict1[code]['rq'] = str(date.fromisoformat(item[5].replace('/','-').split(' ')[0]))\n", + " #dict1[code]['rq'] = item[5].replace('/','-').split(' ')[0]\n", + " else:\n", + " rq=date.fromisoformat('2025-08-12')\n", + " dict1[code]['rq'] = '2025-08-12'\n", + " for k, v in content.items():\n", + " \n", + " if 'tcm' in k:\n", + " i = int(k[3:])\n", + " tcm[i-1] = int(v) \n", + " \n", + " if 'tcm' in item[4]: \n", + " dict1[code]['tcm'] = tcm\n", + " \n", + " birth = date.fromisoformat(dict3[code]['birth'].replace('/','-'))\n", + " \n", + " days = (rq-birth).days \n", + " dict1[code]['age'] = int(days/365)\n", + " dict1[code]['month'] = int(days/365*12)\n", + " #print(phone[item[2]])\n", + " nn+=1\n", + "filename = 'data/result_南京化工-2.json'\n", + "\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl, ensure_ascii=False)" + ] + }, { "cell_type": "markdown", "id": "89689b86-3fae-402e-a456-a646f0c7201f", @@ -499,6 +603,56 @@ " " ] }, + { + "cell_type": "code", + "execution_count": null, + "id": "4f192b79-5dc7-4517-b7f3-409e30d60dad", + "metadata": {}, + "outputs": [], + "source": [ + "from pathlib import Path\n", + "import json\n", + "import shutil\n", + "import pymupdf4llm\n", + "#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n", + "llama_reader = pymupdf4llm.LlamaMarkdownReader()\n", + "#llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\")\n", + "\n", + "\n", + "target_directory = Path('./file/南化体重/new')\n", + "new_path = './file/南化体重/md'\n", + "\n", + "\n", + "for fl in target_directory.rglob('*.pdf'):\n", + " if fl.is_file():\n", + " fl_name = fl.stem\n", + " llama_lists = pymupdf4llm.to_markdown(fl,page_chunks=True)\n", + " list1 = []\n", + " for item in llama_lists:\n", + " list1.append(item['text'])\n", + " llama_docs = '\\n'.join(list1)\n", + " Path(new_path,fl_name+'.md').write_bytes(llama_docs.encode())\n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "e5863760-0d11-45b0-acb3-1a38c78d4fc7", + "metadata": {}, + "outputs": [], + "source": [ + "from pathlib import Path\n", + "import json\n", + "import shutil\n", + "import pymupdf4llm\n", + "#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n", + "llama_reader = pymupdf4llm.LlamaMarkdownReader()\n", + "llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\",page_chunks=True)\n", + "print(llama_docs)\n", + "\n" + ] + }, { "cell_type": "markdown", "id": "8d5f4103-0d1e-4711-b324-ece360f8dcd3", diff --git a/体测单位/宁夏能化.ipynb b/体测单位/宁夏能化.ipynb index abe0b45..e91c189 100644 --- a/体测单位/宁夏能化.ipynb +++ b/体测单位/宁夏能化.ipynb @@ -10,27 +10,12 @@ }, { "cell_type": "code", - "execution_count": 13, + "execution_count": null, "id": "bbba6efc-73cd-4db6-bae7-014724fee731", "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T06:47:52.647999Z", - "iopub.status.busy": "2025-06-24T06:47:52.647468Z", - "iopub.status.idle": "2025-06-24T06:47:52.837761Z", - "shell.execute_reply": "2025-06-24T06:47:52.837173Z", - "shell.execute_reply.started": "2025-06-24T06:47:52.647952Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "1698 ok\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -65,26 +50,10 @@ }, { "cell_type": "code", - "execution_count": 36, + "execution_count": null, "id": "ddddbac1-ab71-436a-ab64-2a15600f5b1b", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:47:30.639088Z", - "iopub.status.busy": "2025-06-24T07:47:30.638546Z", - "iopub.status.idle": "2025-06-24T07:47:30.944089Z", - "shell.execute_reply": "2025-06-24T07:47:30.943556Z", - "shell.execute_reply.started": "2025-06-24T07:47:30.639038Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "1953 ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -165,27 +134,12 @@ }, { "cell_type": "code", - "execution_count": 37, + "execution_count": null, "id": "970e171e-1360-448f-a28c-520ccb8f314a", "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:47:35.746440Z", - "iopub.status.busy": "2025-06-24T07:47:35.746188Z", - "iopub.status.idle": "2025-06-24T07:47:35.874766Z", - "shell.execute_reply": "2025-06-24T07:47:35.874285Z", - "shell.execute_reply.started": "2025-06-24T07:47:35.746417Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "1699\n" - ] - } - ], + "outputs": [], "source": [ "import json\n", "import datetime\n", @@ -234,26 +188,10 @@ }, { "cell_type": "code", - "execution_count": 22, + "execution_count": null, "id": "2082c096-9f36-430c-9823-f09d8b6f9b42", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:24:22.507391Z", - "iopub.status.busy": "2025-06-24T07:24:22.506635Z", - "iopub.status.idle": "2025-06-24T07:24:22.548344Z", - "shell.execute_reply": "2025-06-24T07:24:22.547831Z", - "shell.execute_reply.started": "2025-06-24T07:24:22.507322Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "['2018652', '2019665', '2019160', '2019512', '3427477', '2019423', '2019494', '2019901']\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "filename = 'data/result_宁夏能化人员.json'\n", "with open(filename,'r') as fl:\n", @@ -281,26 +219,10 @@ }, { "cell_type": "code", - "execution_count": 38, + "execution_count": null, "id": "0905ad6a-f7e4-43ed-ae29-c4850deb946b", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:47:41.560238Z", - "iopub.status.busy": "2025-06-24T07:47:41.559637Z", - "iopub.status.idle": "2025-06-24T07:47:41.657989Z", - "shell.execute_reply": "2025-06-24T07:47:41.657434Z", - "shell.execute_reply.started": "2025-06-24T07:47:41.560182Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok!\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import time\n", @@ -347,16 +269,9 @@ }, { "cell_type": "code", - "execution_count": 42, + "execution_count": null, "id": "867ad6b0-9e9d-48bc-a10a-3b9cb6125890", "metadata": { - "execution": { - "iopub.execute_input": "2025-06-26T01:35:41.139581Z", - "iopub.status.busy": "2025-06-26T01:35:41.139145Z", - "iopub.status.idle": "2025-06-26T01:35:41.663585Z", - "shell.execute_reply": "2025-06-26T01:35:41.663054Z", - "shell.execute_reply.started": "2025-06-26T01:35:41.139557Z" - }, "tags": [] }, "outputs": [], @@ -578,26 +493,10 @@ }, { "cell_type": "code", - "execution_count": 44, + "execution_count": null, "id": "d31020be-c9c0-404f-8e0b-f8f507385f48", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-27T04:14:46.467739Z", - "iopub.status.busy": "2025-06-27T04:14:46.466974Z", - "iopub.status.idle": "2025-06-27T04:14:47.053693Z", - "shell.execute_reply": "2025-06-27T04:14:47.053104Z", - "shell.execute_reply.started": "2025-06-27T04:14:46.467668Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import openpyxl\n", @@ -730,26 +629,10 @@ }, { "cell_type": "code", - "execution_count": 39, + "execution_count": null, "id": "9ffec868-3851-42f6-8431-c5c5d3d15424", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:47:57.736323Z", - "iopub.status.busy": "2025-06-24T07:47:57.736061Z", - "iopub.status.idle": "2025-06-24T07:48:02.653903Z", - "shell.execute_reply": "2025-06-24T07:48:02.653030Z", - "shell.execute_reply.started": "2025-06-24T07:47:57.736300Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "8\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import requests\n", "import json\n", @@ -823,26 +706,10 @@ }, { "cell_type": "code", - "execution_count": 40, + "execution_count": null, "id": "59f69be9-ab27-4308-8641-927dbcfda87e", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:48:39.717546Z", - "iopub.status.busy": "2025-06-24T07:48:39.716729Z", - "iopub.status.idle": "2025-06-24T07:48:40.238267Z", - "shell.execute_reply": "2025-06-24T07:48:40.237638Z", - "shell.execute_reply.started": "2025-06-24T07:48:39.717462Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "1699\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import os,sys,shutil\n", "import json\n", @@ -880,26 +747,10 @@ }, { "cell_type": "code", - "execution_count": 41, + "execution_count": null, "id": "896955ac-351a-4914-94a6-56e6d1ae9e6e", - "metadata": { - "execution": { - "iopub.execute_input": "2025-06-24T07:48:42.641359Z", - "iopub.status.busy": "2025-06-24T07:48:42.641096Z", - "iopub.status.idle": "2025-06-24T07:48:42.812702Z", - "shell.execute_reply": "2025-06-24T07:48:42.812095Z", - "shell.execute_reply.started": "2025-06-24T07:48:42.641334Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import os,sys,shutil\n", "import json\n", @@ -936,10 +787,147 @@ "print('ok')" ] }, + { + "cell_type": "markdown", + "id": "3e5363cb-233c-4050-9c3a-18207d7f597a", + "metadata": {}, + "source": [ + "## 统计体质检测等级明细表" + ] + }, + { + "cell_type": "code", + "execution_count": 8, + "id": "360831b7-e204-4801-9f72-879e338ed962", + "metadata": { + "execution": { + "iopub.execute_input": "2025-08-13T11:53:04.589116Z", + "iopub.status.busy": "2025-08-13T11:53:04.588495Z", + "iopub.status.idle": "2025-08-13T11:53:04.762457Z", + "shell.execute_reply": "2025-08-13T11:53:04.761877Z", + "shell.execute_reply.started": "2025-08-13T11:53:04.589057Z" + } + }, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "\n", + "\n", + "filename = 'data/data_宁夏能化人员.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "list1 = []\n", + "for k, v in dict1.items():\n", + " list1.append([k,v['name'],v['unit'],v['sex'],v['age'],v['level'],v['score']])\n", + " \n", + "filename = 'data/宁夏能化等级情况表.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "#sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename) \n", + " " + ] + }, + { + "cell_type": "markdown", + "id": "d9705453-5c8c-40d2-83c5-7c2ddfc5f094", + "metadata": {}, + "source": [ + "## 导出体检报告阳性数据" + ] + }, { "cell_type": "code", "execution_count": null, - "id": "a1907b6f-3ba7-4773-8f6e-6559af18131e", + "id": "e9f6d0ce-7be5-4bb9-8a64-9281a99eb897", + "metadata": {}, + "outputs": [], + "source": [ + "import openpyxl\n", + "import json\n", + "\n", + "\n", + "wb = openpyxl.load_workbook('data/2025年中国石化长城能源化工(宁夏)有限公司体检数据.xlsx',data_only=True)\n", + "sheet = wb.active\n", + "# sheets = wb.sheetnames\n", + "person = {}\n", + "\n", + "for n in range(2, sheet.max_row+1):\n", + " code = int(sheet.cell(n, 1).value)\n", + " person.setdefault(code, {})\n", + " dict1 = {}\n", + " dict1['name'] = sheet.cell(n, 3).value\n", + " dict1['sex'] = sheet.cell(n, 4).value\n", + " dict1['unit'] = sheet.cell(n, 2).value\n", + " dict1['age'] = sheet.cell(n, 5).value\n", + " content = sheet.cell(n, 6).value\n", + " dict1['signs'] = content.split(',')\n", + " person[code] = dict1\n", + "filename = 'data/宁夏能化体检数据2025.json'\n", + "with open(filename, 'w') as fl:\n", + " json.dump(person, fl, ensure_ascii=False)\n", + "print(len(person),'ok')" + ] + }, + { + "cell_type": "markdown", + "id": "f315bd79-122f-4dd6-9666-d4945fe216cb", + "metadata": {}, + "source": [ + "## 筛选体重相关异常数据人员" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "e3117237-b921-41fe-9b65-4918c5b2a165", + "metadata": { + "execution": { + "iopub.execute_input": "2025-08-13T08:52:05.672800Z", + "iopub.status.busy": "2025-08-13T08:52:05.672283Z", + "iopub.status.idle": "2025-08-13T08:52:05.820901Z", + "shell.execute_reply": "2025-08-13T08:52:05.820357Z", + "shell.execute_reply.started": "2025-08-13T08:52:05.672751Z" + } + }, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "\n", + "filename = 'data/宁夏能化体检数据2025.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "list1 = []\n", + "list2 = ['肥胖','超重','脂肪肝','血脂','高脂血症','动脉粥样硬化']\n", + "for k, v in dict1.items():\n", + " zz = set()\n", + " for item in v['signs']:\n", + " for xm in list2:\n", + " if xm in item:\n", + " zz.add(item)\n", + " if len(zz) > 0:\n", + " list1.append([k,v['name'],v['unit'],v['sex'],v['age'],','.join(zz)])\n", + "\n", + "filename = 'data/宁夏能化体检体重相关人员.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "#sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename) \n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6133f0d5-75aa-41c0-a624-b354d89b5ba3", "metadata": {}, "outputs": [], "source": [] diff --git a/体测单位/新疆油田采油工艺研究院.ipynb b/体测单位/新疆油田采油工艺研究院.ipynb index 2cc8466..03f3f1f 100644 --- a/体测单位/新疆油田采油工艺研究院.ipynb +++ b/体测单位/新疆油田采油工艺研究院.ipynb @@ -10,27 +10,12 @@ }, { "cell_type": "code", - "execution_count": 30, + "execution_count": null, "id": "bbba6efc-73cd-4db6-bae7-014724fee731", "metadata": { - "execution": { - "iopub.execute_input": "2025-07-26T00:04:43.702270Z", - "iopub.status.busy": "2025-07-26T00:04:43.701709Z", - "iopub.status.idle": "2025-07-26T00:04:43.722751Z", - "shell.execute_reply": "2025-07-26T00:04:43.722227Z", - "shell.execute_reply.started": "2025-07-26T00:04:43.702216Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "71 ok\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -60,6 +45,41 @@ "print(len(person),'ok')" ] }, + { + "cell_type": "code", + "execution_count": null, + "id": "de3144e7-544a-4430-bf69-3d20479d50c9", + "metadata": {}, + "outputs": [], + "source": [ + "import openpyxl\n", + "import json\n", + "\n", + "\n", + "wb = openpyxl.load_workbook('data/新疆油田采油工艺研究院-1.xlsx',data_only=True)\n", + "sheet = wb.active\n", + "# sheets = wb.sheetnames\n", + "person = {}\n", + "\n", + "for n in range(2, sheet.max_row+1):\n", + " code = int(sheet.cell(n, 1).value)\n", + " person.setdefault(code, {})\n", + " dict1 = {}\n", + " dict1['name'] = sheet.cell(n, 2).value\n", + " dict1['sex'] = sheet.cell(n, 3).value\n", + " if sheet.cell(n,4).value is not None:\n", + " dict1['unit'] = sheet.cell(n, 4).value\n", + " else:\n", + " dict1['unit'] = ''\n", + " dict1['birth'] = str(sheet.cell(n, 5).value).replace('/','-').split(' ')[0]\n", + " dict1['phone'] = sheet.cell(n, 6).value \n", + " person[code] = dict1\n", + "filename = 'data/新疆油田采油工艺研究院-1.json'\n", + "with open(filename, 'w') as fl:\n", + " json.dump(person, fl, ensure_ascii=False)\n", + "print(len(person),'ok')" + ] + }, { "cell_type": "markdown", "id": "6bcd45c2-10af-4d5f-9e0b-5cd1df4f7a7f", @@ -70,27 +90,38 @@ }, { "cell_type": "code", - "execution_count": 31, + "execution_count": null, "id": "970e171e-1360-448f-a28c-520ccb8f314a", "metadata": { - "execution": { - "iopub.execute_input": "2025-07-26T00:04:50.800932Z", - "iopub.status.busy": "2025-07-26T00:04:50.800184Z", - "iopub.status.idle": "2025-07-26T00:04:50.821256Z", - "shell.execute_reply": "2025-07-26T00:04:50.820361Z", - "shell.execute_reply.started": "2025-07-26T00:04:50.800866Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "71\n" - ] - } - ], + "outputs": [], + "source": [ + "import json\n", + "import datetime\n", + "import csv\n", + "from datetime import date\n", + "import my_module as My\n", + "\n", + "filename = 'data/新疆油田采油工艺研究院-1.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl) \n", + "filename = 'data/marks_20250813.csv'\n", + "re_ta = My.get_result(filename,dict1)\n", + "\n", + "\n", + "filename = 'data/result_新疆油田采油工艺研究院1.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(re_ta, fl, ensure_ascii=False) \n", + "print(len(re_ta))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "d1270459-e27a-46cc-b436-ec7e82b47be2", + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import datetime\n", @@ -101,7 +132,7 @@ "filename = 'data/新疆油田采油工艺研究院.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", - "filename = 'data/marks_20250726.csv'\n", + "filename = 'data/marks_20250727.csv'\n", "re_ta = My.get_result(filename,dict1)\n", "\n", "\n", @@ -121,26 +152,10 @@ }, { "cell_type": "code", - "execution_count": 32, + "execution_count": null, "id": "0905ad6a-f7e4-43ed-ae29-c4850deb946b", - "metadata": { - "execution": { - "iopub.execute_input": "2025-07-26T00:04:58.023332Z", - "iopub.status.busy": "2025-07-26T00:04:58.022777Z", - "iopub.status.idle": "2025-07-26T00:04:58.041468Z", - "shell.execute_reply": "2025-07-26T00:04:58.040883Z", - "shell.execute_reply.started": "2025-07-26T00:04:58.023283Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok!\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import time\n", @@ -148,7 +163,7 @@ "\n", "#list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','height','weight']\n", "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']\n", - "filename = 'data/result_新疆油田采油工艺研究院.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院1.json'\n", "with open(filename,'r') as fl:\n", " dict2 = json.load(fl) \n", "for k, v in dict2.items():\n", @@ -171,7 +186,7 @@ " dict2[k][item_en]['score'] = My.cal_score(data1)\n", " #print(k,v[item_en]['成绩'],cal_score(data1))\n", "\n", - "filename = f'data/result_新疆油田采油工艺研究院.json'\n", + "filename = f'data/result_新疆油田采油工艺研究院1.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict2,fl , ensure_ascii=False) \n", "print('ok!') " @@ -187,15 +202,15 @@ }, { "cell_type": "code", - "execution_count": 33, + "execution_count": 7, "id": "867ad6b0-9e9d-48bc-a10a-3b9cb6125890", "metadata": { "execution": { - "iopub.execute_input": "2025-07-26T00:05:04.440594Z", - "iopub.status.busy": "2025-07-26T00:05:04.439940Z", - "iopub.status.idle": "2025-07-26T00:05:04.474727Z", - "shell.execute_reply": "2025-07-26T00:05:04.474128Z", - "shell.execute_reply.started": "2025-07-26T00:05:04.440537Z" + "iopub.execute_input": "2025-08-13T02:46:29.876079Z", + "iopub.status.busy": "2025-08-13T02:46:29.875587Z", + "iopub.status.idle": "2025-08-13T02:46:29.904595Z", + "shell.execute_reply": "2025-08-13T02:46:29.904071Z", + "shell.execute_reply.started": "2025-08-13T02:46:29.876014Z" }, "tags": [] }, @@ -207,11 +222,11 @@ "items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']\n", "title = ['编号','姓名','性别','单位','部门','身高','体重','肺活量','握力','坐位体前屈','纵跳','俯卧撑','一分钟仰卧起坐','单脚站立','选择反应时','台阶指数']\n", "\n", - "filename = 'data/result_新疆油田采油工艺研究院.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "\n", - "filename = 'data/新疆油田采油工艺研究院.json'\n", + "filename = 'data/新疆油田采油工艺研究院-1.json'\n", "with open(filename,'r') as fl:\n", " dict2 = json.load(fl)\n", " \n", @@ -239,7 +254,59 @@ " list2.append('')\n", " \n", " list1.append(list2)\n", - "filename = 'data/新疆油田采油工艺研究院测试情况明细表(截至20250725).xlsx'\n", + "filename = 'data/新疆油田采油工艺研究院测试情况明细表(第二批).xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename)" + ] + }, + { + "cell_type": "markdown", + "id": "f44edca7-a1bd-4937-a290-bf80ea91dd5b", + "metadata": {}, + "source": [ + "## 导出未体测人员" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "bf6b900f-1fcc-49c8-9b03-7723f9ca2429", + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "\n", + "items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']\n", + "title = ['编号','姓名','性别','单位','部门']\n", + "\n", + "filename = 'data/result_新疆油田采油工艺研究院.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "\n", + "filename = 'data/新疆油田采油工艺研究院.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl)\n", + "print(dict1.keys()) \n", + "list1 = []\n", + "for k, v in dict2.items():\n", + " \n", + " if k not in dict1.keys():\n", + " \n", + " list2 = []\n", + " list2.append(str(k).rjust(5,'0'))\n", + " list2.append(v['name']) \n", + " list2.append(dict2[k]['sex'])\n", + " list2.append(dict2[k]['unit'])\n", + " \n", + " list1.append(list2)\n", + "print(list1)\n", + "filename = 'data/新疆油田采油工艺研究院未体测人员(截至20250725).xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet.append(title)\n", @@ -259,16 +326,9 @@ }, { "cell_type": "code", - "execution_count": 9, + "execution_count": null, "id": "def3f47a-24ef-4815-b63b-2a16b79b4c15", "metadata": { - "execution": { - "iopub.execute_input": "2025-07-25T13:18:57.344342Z", - "iopub.status.busy": "2025-07-25T13:18:57.344014Z", - "iopub.status.idle": "2025-07-25T13:18:57.363693Z", - "shell.execute_reply": "2025-07-25T13:18:57.363137Z", - "shell.execute_reply.started": "2025-07-25T13:18:57.344308Z" - }, "tags": [] }, "outputs": [], @@ -321,40 +381,10 @@ }, { "cell_type": "code", - "execution_count": 29, + "execution_count": null, "id": "9cb0172e-5f6c-4b47-8431-83dc5cc4601c", - "metadata": { - "execution": { - "iopub.execute_input": "2025-07-25T13:55:37.064250Z", - "iopub.status.busy": "2025-07-25T13:55:37.063559Z", - "iopub.status.idle": "2025-07-25T13:55:37.077165Z", - "shell.execute_reply": "2025-07-25T13:55:37.076333Z", - "shell.execute_reply.started": "2025-07-25T13:55:37.064186Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "18610346996 杨霖坤\n", - "13121563838 高洁\n", - "15349912576 韩晓强\n", - "13899592125 粟刘\n", - "15509909309 关子越\n", - "13579524663 王金龙\n", - "13121073480 罗腾\n", - "13899586957 袁鹏\n", - "13999311222 闫鹏\n", - "15609905966 李阳\n", - "18909908078 吕长智\n", - "18040802080 王立坤\n", - "17709901414 侯丽娜\n", - "17399375199 史瑶\n", - "18699008113 荣玉健\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import csv\n", @@ -397,26 +427,10 @@ }, { "cell_type": "code", - "execution_count": 34, + "execution_count": null, "id": "2fa85089-8dc7-42f7-9ff3-c37cbe0899ed", - "metadata": { - "execution": { - "iopub.execute_input": "2025-07-26T00:06:01.368354Z", - "iopub.status.busy": "2025-07-26T00:06:01.367653Z", - "iopub.status.idle": "2025-07-26T00:06:01.401313Z", - "shell.execute_reply": "2025-07-26T00:06:01.400714Z", - "shell.execute_reply.started": "2025-07-26T00:06:01.368289Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "54\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import csv\n", @@ -426,10 +440,10 @@ "\n", "dict1 = {}\n", "\n", - "filename = 'data/result_新疆油田采油工艺研究院.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", - "filename = 'data/新疆油田采油工艺研究院.json'\n", + "filename = 'data/新疆油田采油工艺研究院-1.json'\n", "with open(filename,'r') as fl:\n", " dict3 = json.load(fl)\n", "\n", @@ -439,7 +453,7 @@ " phone[v['phone']] = k\n", "\n", "list1 = []\n", - "filename = 'data/survey_records_20250725.csv'\n", + "filename = 'data/survey_records_20250813.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -489,7 +503,7 @@ " dict1[code]['month'] = int(days/365*12)\n", " #print(phone[item[2]])\n", " nn+=1\n", - "filename = 'data/result_新疆油田采油工艺研究院-1.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院1-1.json'\n", "\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl, ensure_ascii=False)\n", @@ -506,15 +520,15 @@ }, { "cell_type": "code", - "execution_count": 35, + "execution_count": 6, "id": "23ffd115-72a3-4f90-9e64-ffa6420df8a4", "metadata": { "execution": { - "iopub.execute_input": "2025-07-26T00:06:07.337672Z", - "iopub.status.busy": "2025-07-26T00:06:07.336904Z", - "iopub.status.idle": "2025-07-26T00:06:50.437010Z", - "shell.execute_reply": "2025-07-26T00:06:50.436483Z", - "shell.execute_reply.started": "2025-07-26T00:06:07.337571Z" + "iopub.execute_input": "2025-08-13T02:43:03.163356Z", + "iopub.status.busy": "2025-08-13T02:43:03.162804Z", + "iopub.status.idle": "2025-08-13T02:43:17.799933Z", + "shell.execute_reply": "2025-08-13T02:43:17.799273Z", + "shell.execute_reply.started": "2025-08-13T02:43:03.163305Z" } }, "outputs": [ @@ -522,7 +536,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "71\n" + "24\n" ] } ], @@ -535,11 +549,11 @@ "headers = {\n", " \"Content-Type\": \"application/json; charset=UTF-8\"\n", " }\n", - "filename = 'data/result_新疆油田采油工艺研究院-1.json'\n", + "filename = 'data/result_新疆油田采油工艺研究院1-1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list1 = []\n", - "file_path ='./新疆油田采油工艺研究院/'\n", + "file_path ='./新疆油田采油工艺研究院1/'\n", "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']\n", "i=0\n", "list2 = []\n", diff --git a/文件操作.ipynb b/文件操作.ipynb index 0b3d51d..d2bc8fc 100644 --- a/文件操作.ipynb +++ b/文件操作.ipynb @@ -414,25 +414,9 @@ }, { "cell_type": "code", - "execution_count": 33, - "metadata": { - "execution": { - "iopub.execute_input": "2025-03-09T02:47:43.965042Z", - "iopub.status.busy": "2025-03-09T02:47:43.964342Z", - "iopub.status.idle": "2025-03-09T02:47:44.398962Z", - "shell.execute_reply": "2025-03-09T02:47:44.398531Z", - "shell.execute_reply.started": "2025-03-09T02:47:43.964980Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "\n" - ] - } - ], + "execution_count": null, + "metadata": {}, + "outputs": [], "source": [ "from bs4 import BeautifulSoup\n", "import re\n", @@ -482,16 +466,8 @@ }, { "cell_type": "code", - "execution_count": 29, - "metadata": { - "execution": { - "iopub.execute_input": "2025-03-09T02:36:40.207084Z", - "iopub.status.busy": "2025-03-09T02:36:40.206317Z", - "iopub.status.idle": "2025-03-09T02:36:40.223987Z", - "shell.execute_reply": "2025-03-09T02:36:40.223140Z", - "shell.execute_reply.started": "2025-03-09T02:36:40.207017Z" - } - }, + "execution_count": null, + "metadata": {}, "outputs": [], "source": [ "from bs4 import BeautifulSoup\n", @@ -1600,6 +1576,66 @@ "# print(err)" ] }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymupdf4llm\n", + "#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n", + "llama_reader = pymupdf4llm.LlamaMarkdownReader()\n", + "llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\")\n", + "\n", + "print(llama_docs)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymupdf4llm\n", + "md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n", + "#llama_reader = pymupdf4llm.LlamaMarkdownReader()\n", + "#llama_docs = llama_reader.load_data(\"data/1782596-唐荣.pdf\")\n", + "\n", + "print(md_text)" + ] + }, + { + "cell_type": "code", + "execution_count": 9, + "metadata": { + "execution": { + "iopub.execute_input": "2025-08-05T04:13:25.191718Z", + "iopub.status.busy": "2025-08-05T04:13:25.191013Z", + "iopub.status.idle": "2025-08-05T04:13:26.589724Z", + "shell.execute_reply": "2025-08-05T04:13:26.589155Z", + "shell.execute_reply.started": "2025-08-05T04:13:25.191654Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[{'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 1}, 'toc_items': [], 'tables': [], 'images': [{'number': 4, 'bbox': Rect(233.94000244140625, 713.0, 375.66998291015625, 854.72998046875), 'transform': (141.72999572753906, 0.0, -0.0, 141.72999572753906, 233.94000244140625, 713.0), 'width': 190, 'height': 190, 'colorspace': 3, 'cs-name': 'DeviceRGB', 'xres': 96, 'yres': 96, 'bpc': 8, 'size': 15730, 'has-mask': False}, {'number': 11, 'bbox': Rect(251.69000244140625, 279.6000061035156, 353.739990234375, 387.32000732421875), 'transform': (102.05000305175781, 0.0, -0.0, 107.72000122070312, 251.69000244140625, 279.6000061035156), 'width': 137, 'height': 145, 'colorspace': 3, 'cs-name': 'DeviceRGB', 'xres': 96, 'yres': 96, 'bpc': 8, 'size': 7001, 'has-mask': False}], 'graphics': [], 'text': '```\\n 医保卡号:\\n\\n 体检号:325031000089\\n 检查类型:在岗期间\\n\\n# **`南京江北医院`** **`职业健康检查表`**\\n\\n```\\n\\n#### `体检编号` `姓 名`\\n\\n\\n#### `325031000089` `唐荣`\\n\\n\\n#### `身 份 证 32011219771117****`\\n\\n\\n#### `用人单位`\\n\\n\\n```\\n南化公司苯化工部\\n\\n```\\n\\n#### `联系电话 13382785142`\\n\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 2}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '## **`职业健康检查表说明`**\\n\\n#### `一、本单位保证职业健康检查的科学性、公正性和准确性。` `二、本单位职业健康检查活动依据国家《职业健康检查管理办法》、《` `职业健康监护技术规范》等规定进行。` `三、本单位《江苏省职业健康检查机构备案回执》的编号:苏卫职检` `备字[2021]BG第016号` `四、本检查表涂改、增删无效,未加盖单位印章无效。` `五、未经本单位同意,不得部分复制本检查表。` `六、用人单位和劳动者应确保一般项目、职业史、接触的职业病危害` `因素、既往病史等项目的真实性。` `七、发现健康损害或者疑似职业病病人时,用人单位应根据本单位的` `主检意见及国家法律法规要求安排复查或医学观察或进行职业病` `诊断。` `八、对检查结果若有异议,可直接向本单位进行咨询。` `地址(Address):南京市化工园区葛关路552号` `邮政编码(Post Coad):210048` `电话(Tel):025-57067000`\\n\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 3}, 'toc_items': [], 'tables': [{'bbox': (57.689998626708984, 206.18099975585938, 562.260009765625, 297.70001220703125), 'rows': 2, 'columns': 6}, {'bbox': (57.689998626708984, 315.7010192871094, 562.260009765625, 367.9800109863281), 'rows': 2, 'columns': 5}], 'images': [], 'graphics': [], 'text': '```\\n 中华人民共和国预防性健康检查用表\\n\\n### **`有害作业人员健康检查表`**\\n\\n```\\n\\n```\\n单位名称:\\n\\n```\\n\\n```\\n南化公司苯化工部\\n\\n```\\n\\n```\\n1977年11月17日\\n\\n```\\n\\n```\\n姓 名: 唐荣 性 别:男\\n\\n```\\n\\n```\\n出生年月:\\n\\n```\\n\\n```\\n国 籍:中华人民共和国\\n\\n```\\n\\n```\\n民 族:汉族 婚 否: /\\n\\n```\\n\\n```\\n0 月 接害工龄:20 年 0 月\\n\\n```\\n\\n```\\n体检日期:\\n\\n```\\n\\n```\\n总 工 龄: 29 年 0 月 接害工龄:\\n\\n```\\n\\n```\\n年 0 月 接害工龄:20 年\\n\\n```\\n\\n```\\n2025年03月10日\\n\\n```\\n\\n```\\n29 0 20\\n\\n```\\n\\n```\\n接触有害因素:\\n\\n```\\n\\n```\\n苯,苯胺,硝基苯,硫酸及三氧化硫,硝酸,氮氧化物,氢氧化钠,氢气\\n\\n```\\n\\n\\n\\n\\n\\n\\n\\n\\n\\n|一、职业史|Col2|Col3|Col4|Col5|Col6|\\n|---|---|---|---|---|---|\\n|`起止日期`|`工作单位`|`部门`|`工种`|` 有害因素种类、`
`名称`|`防护措施`|\\n|`2005-03-10-`
`至今`|`南化公司苯化工部`|`苯胺作业区`|`有机合成工`|`苯,苯胺,硝基苯,`
`硫酸及三氧化硫,`
`硝酸,氮氧化物,氢`
`氧化钠,氢气`|`口罩,手套,耳塞`|\\n\\n\\n\\n|二、既往病史|Col2|Col3|Col4|Col5|\\n|---|---|---|---|---|\\n|`疾病名称`|`诊断日期`|`诊断单位`|`治疗经过`|`转归`|\\n|`无`|`/`|`/`|`/`|`/`|\\n\\n```\\n三、月经史 初潮: / 岁 经期: / 天 周期: / 天 停经年龄: / 岁\\n\\n 是否经期: /\\n\\n四、生育史 现有子女 / 人、流产 / 次、早产/ 次、死产/ 次、异常胎/ 次\\n\\n五、烟酒史 现在每天吸 10 支/天、共 29 年;\\n\\n 不饮酒 / ml/日、共 / 年。\\n\\n六、其他 无特殊情况\\n\\n七、自觉症状\\n\\n症状 症状 症状 症状\\n\\n目前无不适症状\\n\\n问诊者: 录入日期:2025年03月10日\\n\\n 第1页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 4}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第4页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n八、临床\\n\\n体重视力检查\\n\\n项目 结果 项目 结果\\n\\n身高 173cm 体重 86kg\\n\\n体重指数(BMI) 28.73↑ 裸眼视力左 4.5\\n\\n裸眼视力右 4.5 矫正视力(左) \\n矫正视力(右) \\n检查者: 体检日期:2025年03月10日\\n\\n内科检查\\n\\n项目 结果 项目 结果\\n\\n肺 双肺呼吸音清晰,未闻及干湿性啰音 心律 正常\\n\\n杂音 正常 腹部压痛 正常\\n\\n心率 96次/分 内科其它 \\\\\\n\\n医生: 体检日期:2025年03月10日\\n\\n血压\\n\\n项目 结果 项目 结果\\n\\n收缩压 155mmHg↑ 脉搏 96次/分\\n\\n舒张压 99mmHg↑\\n\\n医生: 体检日期:2025年03月10日\\n\\n耳鼻喉科\\n\\n项目 结果 项目 结果\\n\\n扁桃体 扁桃体Ⅱ度 咽喉 慢性咽炎\\n\\n鼻腔 正常 听力(右) 正常\\n\\n听力(左) 正常 耳疾 无\\n\\n嗅觉 正常\\n\\n医生: 体检日期:2025年03月10日\\n\\n眼科常规检查\\n\\n项目 结果 项目 结果\\n\\n辨色力 正常 眼结膜 正常\\n\\n眼部病变 \\\\\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第2页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 5}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第5页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n医生: 体检日期:2025年03月10日\\n\\n外科检查\\n\\n项目 结果 项目 结果\\n\\n皮肤 未见明显异常 淋巴结 未见明显异常\\n\\n甲状腺 未见明显异常 四肢 未见明显异常\\n\\n脊椎 未见明显异常 腋下 未见明显异常\\n\\n指甲 未见明显异常 骨关节 未见明显异常\\n\\n外科其它 /\\n\\n医生: 体检日期:2025年03月10日\\n\\n口腔检查\\n\\n项目 结果 项目 结果\\n\\n口腔粘膜 正常 唇腭 正常\\n\\n舌体 正常 牙周 正常\\n\\n牙齿 龋齿、残根 口腔科其它 /\\n\\n医生: 体检日期:2025年03月10日\\n\\n神经科\\n\\n项目 结果 项目 结果\\n\\n膝腱反射 正常反射 病理反射 未引出\\n\\n肌力 Ⅴ级 三颤 阴性\\n\\n皮肤划痕症 阴性 跟腱反射 正常反射\\n\\n前庭功能检查 阴性 肌张力 阴性\\n\\n共济运动 未见明显异常 感觉异常 未见明显异常\\n\\n运动功能 未见明显异常\\n\\n医生: 体检日期:2025年03月10日\\n\\n尿常规\\n\\n项目 结果 项目 结果\\n\\n结晶 0.00/μl 尿白细胞脂酶(LEU) 阴性Cell/uL\\n\\n医生: 体检日期:2025年03月10日\\n\\n九、器械类检查\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第3页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 6}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第6页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n彩超\\n\\n项目 结果\\n\\n彩超(双肾) 左肾强光点多个\\n\\n彩超(甲状腺) 甲状腺右叶囊性结节一个,大小约3.0*1.8*2.3毫米,TI-RADS 2类\\n\\n彩超(前列腺) 前列腺大小约44*33毫米,内见部分增强光斑(前列腺稍大伴部分钙化)\\n\\n肝 轻-中度脂肪肝;肝内低回声区一个,大小约10*10毫米,考虑低脂区可能\\n\\n胆囊 未见明显异常\\n\\n胰 未见明显异常\\n\\n脾 未见明显异常\\n\\n```\\n\\n```\\n小结:\\n\\n```\\n\\n```\\n彩超(双肾):左肾强光点多个;彩超(甲状腺):甲状腺右叶囊性结节一个,大小约3.0*1.8*2.3毫米,T\\nI-RADS\\n2类;彩超(前列腺):前列腺大小约44*33毫米,内见部分增强光斑(前列腺稍大伴部分钙化);肝\\n:轻-中度脂肪肝;肝内低回声区一个,大小约10*10毫米,考虑低脂区可能。\\n\\n```\\n\\n```\\n医生: 体检日期:2025年03月10日\\n\\n心电图\\n\\n项目 结果\\n\\n心电图 窦性心律;\\n\\n 小结: 窦性心律\\n\\n医生: 体检日期:2025年03月10日\\n\\n肺功能\\n\\n项目 结果 参考值\\n\\nFVC% 80.1 % ≥80\\n\\nFEV1% 81.7 % ≥70\\n\\nFEV1/FVC% 90.07 % 70-100\\n\\n 小结: 无明显异常\\n\\n检查者: 体检日期:2025年03月10日\\n\\n胸部平扫CT\\n\\n项目 结果\\n\\n胸部平扫CT 胸廓对称,纵隔气管居中。两侧肋骨走行自然,胸壁肌间隙清晰,胸壁软组织形态如常,未\\n 见异常密度影。右肺中叶外侧段(IM35)、左肺上叶前段(IM18、IM20)、左肺上叶下舌段\\n (IM41)见多发实性小结节,较大者位于右肺中叶外侧段(IM35),大小约为5mm×4mm。余\\n 肺未见异常密度影。两侧肺门未见增大。气管、支气管通畅。胸腺区呈脂肪密度影,未见增\\n 宽。两侧肺门及纵隔内未见明显肿大淋巴结影。两侧胸腔未见积液影,两侧胸膜未见增厚。\\n 心脏形态、大小如常,主动脉壁见钙化影。所及肝实质密度稍减低。\\n\\n 小结:\\n\\n医生: 体检日期:2025年03月10日\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第4页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 7}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第7页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n十、化验室项目\\n\\n血常规\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n\\n白细胞计数 7.1 3.5-9.5 10^9/L 红细胞计数 4.70 4.3-5.8 10^12/L\\n\\n血红蛋白 139.00 130-175 g/L 红细胞压积 42.1 40-50 %\\n\\n红细胞平均体积 89.50 82-100 fL 平均血红蛋白量 29.70 27.00-34.00 pg\\n\\n平均血红蛋白浓度 330.00 316-354 g/L 血小板计数 340.00 125-350 10^9/L\\n\\n淋巴细胞百分数 34.40 20-50 % 淋巴细胞绝对值 2.44 1.1-3.2 10^9/L\\n\\n中性粒细胞绝对值 4.11 1.8-6.3 10^9/L 中性粒细胞百分数 57.90 40-75 %\\n\\n血小板比积 0.33↑ 0.11-0.28 % 红细胞分布宽度(SD) 44.10 39-46 fl\\n\\n平均血小板体积 9.80 6.00-14.00 fL 血小板分布宽度 16.40 9.00-18.00 fl\\n\\n单核细胞绝对值 0.38 0.1-0.6 10^9/L 单核细胞百分数 5.40 3.0-10.0 %\\n\\n嗜酸性粒细胞 0.15 0.02-0.52 10^9/L 嗜碱性粒细胞 0.01 0-0.06 10^9/L\\n\\n嗜酸性粒细胞百分数 2.10 0.5-5 % 嗜碱性粒细胞百分数 0.20 0-1 %\\n\\n检验员: 复核员: 报告日期: 2025年03月10日\\n\\n尿常规\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n\\n尿维生素C +3 阴性 mmol/L 尿比重(SG) 1.030↑ 1.003-1.03\\n\\n尿胆原(URO) 正常 正常 umol/L 尿胆红素 阴性 阴性 umol/L\\n\\n尿酮体(KET) 阴性 阴性 mmol/L 尿隐血(BLD) +- 阴性 Cell/ul\\n\\n尿蛋白(PRO) +1 0.03-0.14 g/L 亚硝酸盐 阴性 阴性\\n\\n尿葡萄糖(GLU) 阴性 阴性 mmol/L 尿酸碱度测定 5.00 4.60-8.00\\n\\n微白蛋白 150.00↑ 0.00-20.00 mg/L 尿钙 1.00↓ 2.50-7.50 mmol/L\\n\\n尿肌酐 17.60 2.00-22.00 mmol/L 尿红细胞 2.50 0-10 /μl\\n\\n透明度 透明 清晰透明 鳞状上皮 0.00 0-23 /μl\\n\\n非鳞状上皮细胞 0.00 0-3 /uL 草酸钙结晶 0.00 0-10 /ul\\n\\n颜色 淡黄色 淡黄色 管型 0.00 0-1 /ul\\n\\n透明管型 0.00 0-1 /μl 细菌 0.00 0-33 /ul\\n\\n酵母菌 0.00 0-0 粘液丝 +- 0-118 /ul\\n微量蛋白/肌酐(ACR +1 正常 尿酸结晶 0.00 0-3 /ul\\n)\\n\\n尿白细胞 阴性 阴性 Cell/uL\\n\\n检验员: 复核员: 报告日期: 2025年03月10日\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第5页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 8}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第8页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n生化\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n\\n谷丙转氨酶 24 9-50 U/L 谷草转氨酶 17 15-40 U/L\\n\\n总胆红素 3.1 0-26 μmol/L 总蛋白 80.9 65-85 g/L\\n\\n白蛋白 48.4 40-55 g/L 球蛋白 32.5 20-45 g/L\\n\\nγ-谷氨酰转肽酶 48 10-60 U/L 肌酐 75 57-97 μmol/L\\n\\n尿素氮 5.2 2.76-8.07 mmol/L 尿酸 384.0 202.3-416.5 μmol/L\\n\\n总胆固醇 5.49↑ 0-5.2 mmol/L 甘油三酯 5.32↑ 0-1.7 mmol/L\\n\\n低密度脂蛋白 3.07 0-4.12 mmol/L 高密度脂蛋白 0.88↓ 1-2.0 mmol/L\\n\\n葡萄糖 5.70 3.89-6.1 mmol/L\\n\\n检验员: 复核员: 报告日期: 2025年03月10日\\n\\n甲状腺功能\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n游离三碘甲状腺原氨 5.79 3.1-6.8 pmol/L 游离甲状腺素 14.91 12-22 pmol/L\\n酸\\n\\n促甲状腺激素 4.95↑ 0.27-4.2 uIU/ml\\n\\n检验员: 复核员: 报告日期: 2025年03月10日\\n\\n肿瘤标志物\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n\\n癌胚抗原 3.47 0-4.7 ng/ml 甲胎蛋白 3.60 0-7.0 ng/ml\\n\\n检验员: 复核员: 报告日期: 2025年03月10日\\n\\n糖化血红蛋白(HbAIC)\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n\\n糖化血红蛋白测定(H\\n 6.05↑ 4-6.0 %\\nbAIC)\\n\\n检验员: 复核员: 报告日期: 2025年03月10日\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第6页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 9}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第9页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n幽门螺杆菌抗体\\n\\n项目 结果 参考值 计量单位 项目 结果 参考值 计量单位\\n\\n```\\n\\n```\\n幽门螺杆菌尿素酶抗\\n体检测\\n\\n```\\n\\n```\\n弱阳性(±\\n)\\n\\n```\\n\\n```\\n检验员: 复核员: 报告日期: 2025年03月11日\\n\\n十一、检查结论\\n\\n 检查结果:\\n (1)[体重视力检查]:体重指数偏高:28.73;裸眼视力左:4.5;裸眼视力右:4.5;\\n (2)[血压]:收缩压偏高:155mmHg;舒张压偏高:99mmHg;\\n (3)[耳鼻喉科]:扁桃体:扁桃体Ⅱ度;咽喉:慢性咽炎;\\n (4)[口腔检查]:牙齿:龋齿、残根;\\n (5)[彩超]:彩超(双肾):左肾强光点多个;彩超(甲状腺):甲状腺右叶囊性结节一个,大小约3.0*1.8*2.3\\n 毫米,TI-RADS\\n 2类;彩超(前列腺):前列腺大小约44*33毫米,内见部分增强光斑(前列腺稍大伴部分钙化);肝:轻 中度脂肪肝;肝内低回声区一个,大小约10*10毫米,考虑低脂区可能;\\n (6)[心电图]:窦性心律;\\n (7)[尿常规]:尿隐血(BLD):+-;尿蛋白(PRO):+1;微白蛋白偏高:150.00mg/L;尿钙偏低:1.00mmo\\n l/L;微量蛋白/肌酐(ACR):+1;\\n (8)[生化]:总胆固醇偏高:5.49mmol/L;甘油三酯偏高:5.32mmol/L;高密度脂蛋白偏低:0.88mmol/L;\\n (9)[甲状腺功能]:促甲状腺激素偏高:4.95uIU/ml;\\n (10)[胸部平扫CT]:胸部CT平扫:\\n 1、两肺多发微小结节,建议年度随访复查;\\n 2、主动脉硬化。\\n 附见:轻度脂肪肝;\\n (11)[糖化血红蛋白(HbAIC)]:糖化血红蛋白测定(HbAIC)偏高:6.05%;\\n (12)[幽门螺杆菌抗体]:幽门螺杆菌尿素酶抗体检测:弱阳性(±);\\n 其余所检项目未见明显异常。\\n\\n 主检结论:\\n 本次职业健康检查发现除目标疾病以外的其他疾病或某项指标异常。\\n 建议定期复查。\\n\\n 健康建议:\\n 1、糖化血红蛋白测定(HbAIC)偏高:\\n 糖化血红蛋白是血糖检查最常用的监测指标,能反应近2-3个月的平均血糖水平,因此,也是监测糖尿病血\\n 糖控制水平的指标\\n\\n 2、微白蛋白偏高:150.00:\\n 建议注意休息,多饮水,必要时肾内科就诊\\n\\n 3、尿蛋白(PRO):+1:\\n 尿内出现蛋白称为蛋白尿。引起蛋白尿的原因较多:剧烈运动、发热、寒冷、精神紧张时可出现暂时性蛋\\n 白尿,属于生理性蛋白尿。病理性蛋白尿多由急性肾炎、糖尿病肾病、系统性红斑狼疮、肾小管受炎症或\\n 药物刺激引起,膀胱炎、尿道炎、尿道出血及尿液内混入阴道分泌物也可出现假性蛋白尿。建议您肾内科\\n 专科就诊。\\n\\n 4、甲状腺右叶囊性结节:\\n 建议门诊内分泌科或甲乳外科诊疗。\\n\\n 5、两肺多发微小结节:\\n 建议定期复查\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第7页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 10}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '```\\n 第10页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n 6、促甲状腺激素偏高:\\n 建议内分泌科诊疗。\\n\\n 7、高密度脂蛋白偏低:\\n 当发生HDL降低时,动脉血管壁胆固醇沉积的机会增加,导致血管硬化,诱发冠心病,定期复查。\\n\\n 8、轻-中度脂肪肝:\\n 肝脏脂肪含量高于5%称为脂肪肝,引起脂肪肝的主要原因有:营养失调(营养过量)、饮酒、糖尿病、高\\n 脂血症、中毒和药物等。脂肪肝是可逆性的,合理饮食、运动及治疗后可恢复。\\n 建议:①低脂饮食,少吃动物内脏,多进食蔬菜、水果 ②严格限酒,适当运动\\n ③定期复查彩超及血脂,必要时肝病科诊疗。\\n\\n 9、甘油三酯偏高:\\n 甘油三酯测定是脂类代谢的重要指标之一。采血前2-3天尽可能少食含脂质类的食物,空腹12小时抽血,以\\n 排除和减少饮食的影响。血清甘油三脂水平受年龄、性别和饮食的影响。甘油三酯的增高与冠心病的发生\\n 有着重要的相关性。增高见于家族性高甘油三酯血症、饮食大量甘油三酯和继发于某些疾病如:糖尿病、\\n 甲状腺功能减退、肾病综合征、动脉硬化病等。\\n 建议:低脂饮食,多进食蔬菜、水果,增加体育锻炼;明显增高者,在医师指导下使用降脂药物。心内科门\\n 诊定期复查。\\n\\n 10、总胆固醇偏高:\\n 总胆固醇测定主要反映体内胆固醇的代谢情况,血清胆固醇受遗传、饮食、季节、生活方式、性别、年龄\\n 和环境诸多因素的影响,需多次检测并动态观察。\\n 建议\\n 1.胆固醇增高:见于脂肪肝、肝脏肿瘤、动脉粥样硬化、糖尿病、肾病综合症、甲状腺功能减退、阻塞性\\n 黄疸、原发性高胆固醇血症及高脂血症等。\\n 2.明显增高者,在医师指导下使用降脂药物治疗。\\n 3.低胆固醇饮食,少食动物内脏、蛋黄、蟹黄等食物,定期复查。\\n\\n 11、肝内低回声区:\\n 建议消化内科诊疗\\n\\n 12、体重指数偏高:\\n 体重指数(BMI)--即体重(公斤)除以身高(米)的平方,BMI正常是18.5~24.9。24.9~29.9\\n 视为超重;>30 视为肥胖。\\n 建议:①合理控制饮食量,合理搭配饮食,低盐、低脂和低糖饮食。②长期坚持体育锻炼,坚持自测体重\\n\\n 。\\n\\n 13、收缩压偏高:\\n 建议心血管内科专科诊疗\\n\\n 14、左肾强光点:\\n 建议:①多饮水排尿,多吃新鲜蔬菜水果,少吃油腻辛辣食物,多做下蹲动作。②定期复查彩超。\\n\\n 15、舒张压偏高:\\n 建议心血管内科专科诊疗\\n\\n 16、主动脉硬化:\\n 建议心内科诊疗\\n\\n 17、前列腺稍大:\\n 1.多吃新鲜蔬菜、水果,忌辛辣刺激性食物,忌酒。2.尽可能少骑自行车,忌长时间憋尿。3.保持心情舒\\n 畅,避免过度劳累,适度进行体育活动。建议定期复查彩超。\\n\\n 主检医师(签字):\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第8页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 11}, 'toc_items': [], 'tables': [], 'images': [{'number': 4, 'bbox': Rect(377.6499938964844, 142.2300262451172, 519.3800048828125, 283.96002197265625), 'transform': (141.72999572753906, 0.0, -0.0, 141.72999572753906, 377.6499938964844, 142.2300262451172), 'width': 190, 'height': 190, 'colorspace': 3, 'cs-name': 'DeviceRGB', 'xres': 96, 'yres': 96, 'bpc': 8, 'size': 15730, 'has-mask': False}, {'number': 3, 'bbox': Rect(355.3299865722656, 96.88002014160156, 463.04998779296875, 139.4000244140625), 'transform': (107.72000122070312, 0.0, -0.0, 42.52000045776367, 355.3299865722656, 96.88002014160156), 'width': 145, 'height': 58, 'colorspace': 3, 'cs-name': 'DeviceRGB', 'xres': 96, 'yres': 96, 'bpc': 8, 'size': 6109, 'has-mask': False}], 'graphics': [], 'text': '```\\n 第11页 共12页\\n\\n 姓名: 唐荣 性别: 男 年龄:47 体检编号:325031000089 体检日期: 2025-03-10\\n\\n备注:请仔细阅读体检总检和体检报告!!!\\n\\n 第9页\\n\\n```\\n\\n', 'words': []}, {'metadata': {'format': 'PDF 1.5', 'title': '', 'author': 'FastReport', 'subject': 'FastReport PDF export', 'keywords': '', 'creator': '', 'producer': '', 'creationDate': 'D:20250628093605', 'modDate': 'D:20250628093605', 'trapped': '', 'encryption': None, 'file_path': 'data/1782596-唐荣.pdf', 'page_count': 12, 'page': 12}, 'toc_items': [], 'tables': [], 'images': [], 'graphics': [], 'text': '', 'words': []}]\n" + ] + } + ], + "source": [ + "from pathlib import Path\n", + "import json\n", + "import shutil\n", + "import pymupdf4llm\n", + "#md_text = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\")\n", + "#llama_reader = pymupdf4llm.LlamaMarkdownReader()\n", + "llama_docs = pymupdf4llm.to_markdown(\"data/1782596-唐荣.pdf\",page_chunks=True)\n", + "print(llama_docs)" + ] + }, { "cell_type": "code", "execution_count": null,