diff --git a/体测单位/my_module1.py b/体测单位/my_module1.py index 0ef8587..e9ea826 100644 --- a/体测单位/my_module1.py +++ b/体测单位/my_module1.py @@ -163,4 +163,38 @@ def get_result(filename,dict1): re_ta[user].setdefault(item_name,{}) score = result[4] re_ta[user][item_name]['成绩'] = score + return(re_ta) +def get_result_2023(filename,dict1): + re_ta = {} + list1 = [] + with open(filename,'r',newline='') as csv_file: + fl = csv.reader(csv_file,delimiter=',') + header = next(fl) + for line in fl: + #line = re.sub('[\r\n\f ]{1,}', '', line) + list1.append(line) + + for result in list1: + user = str(result[2]).lower() + rq = date.fromisoformat(result[5].replace('/','-')) + if user in dict1.keys(): + l_xm = [] + m_item = str(result[3]) + re_ta.setdefault(user,{}) + re_ta[user]['name'] = dict1[user]['name'] + re_ta[user]['sex'] = dict1[user]['sex'] + if 'phone' in dict1[user].keys(): + re_ta[user]['phone'] = dict1[user]['phone'] + l_xm = ['bmi','lung','grip','flexion','jump','pushup','balance','reaction','step','situp'] + re_ta[user]['unit'] = dict1[user]['unit'] + birth = date.fromisoformat(dict1[user]['birth'].replace('/','-')) + item_name = result[3] + if item_name in l_xm: + days = (rq-birth).days + re_ta[user]['age'] = int(days/365) + re_ta[user]['month'] = int(days/365*12) + re_ta[user]['rq'] = result[5] + re_ta[user].setdefault(item_name,{}) + score = result[4] + re_ta[user][item_name]['成绩'] = score return(re_ta) \ No newline at end of file diff --git a/体测单位/体质检测数据处理.ipynb b/体测单位/体质检测数据处理.ipynb index 0aff157..16b22c5 100644 --- a/体测单位/体质检测数据处理.ipynb +++ b/体测单位/体质检测数据处理.ipynb @@ -391,16 +391,32 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 3, "id": "d6498431-2103-4d27-bba9-f1669b710361", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T14:03:11.519074Z", + "iopub.status.busy": "2025-12-03T14:03:11.518752Z", + "iopub.status.idle": "2025-12-03T14:03:11.537309Z", + "shell.execute_reply": "2025-12-03T14:03:11.536828Z", + "shell.execute_reply.started": "2025-12-03T14:03:11.519047Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "('769530','185','bmi','165.0,67.0','2025-12-03'),('769530','186','bmi','178.0,81.0','2025-12-03'),('769530','187','bmi','171.0,68.0','2025-12-03'),('769530','188','bmi','168.0,65.0','2025-12-03'),('769530','189','bmi','167.0,69.0','2025-12-03'),('769530','191','bmi','183.0,75.0','2025-12-03'),('769530','192','bmi','170.0,78.0','2025-12-03'),('769530','193','bmi','178.0,78.0','2025-12-03'),('769530','194','bmi','153.0,52.0','2025-12-03'),('769530','195','bmi','166.0,69.0','2025-12-03'),('769530','196','bmi','155.0,59.0','2025-12-03'),('769530','198','bmi','162.0,59.0','2025-12-03'),('769530','185','grip','36.0','2025-12-03'),('769530','186','grip','41.4','2025-12-03'),('769530','187','grip','58.0','2025-12-03'),('769530','188','grip','26.7','2025-12-03'),('769530','189','grip','31.0','2025-12-03'),('769530','190','grip','49.0','2025-12-03'),('769530','191','grip','39.0','2025-12-03'),('769530','192','grip','42.0','2025-12-03'),('769530','193','grip','43.7','2025-12-03'),('769530','194','grip','22.5','2025-12-03'),('769530','195','grip','29.0','2025-12-03'),('769530','196','grip','31.2','2025-12-03'),('769530','197','grip','30.0','2025-12-03'),('769530','198','grip','39.7','2025-12-03'),('769530','185','balance','7.47','2025-12-03'),('769530','186','balance','82','2025-12-03'),('769530','187','balance','30','2025-12-03'),('769530','188','balance','9.51','2025-12-03'),('769530','189','balance','21','2025-12-03'),('769530','190','balance','24','2025-12-03'),('769530','192','balance','35','2025-12-03'),('769530','193','balance','25','2025-12-03'),('769530','194','balance','20.55','2025-12-03'),('769530','195','balance','24','2025-12-03'),('769530','196','balance','15','2025-12-03'),('769530','197','balance','34','2025-12-03'),('769530','198','balance','28','2025-12-03'),('769530','185','lung','3130','2025-12-03'),('769530','186','lung','4035','2025-12-03'),('769530','187','lung','4040','2025-12-03'),('769530','188','lung','2252','2025-12-03'),('769530','189','lung','3417','2025-12-03'),('769530','190','lung','4693','2025-12-03'),('769530','191','lung','4254','2025-12-03'),('769530','192','lung','4031','2025-12-03'),('769530','193','lung','4849','2025-12-03'),('769530','194','lung','2259','2025-12-03'),('769530','195','lung','4106','2025-12-03'),('769530','196','lung','2399','2025-12-03'),('769530','197','lung','3100','2025-12-03'),('769530','198','lung','3310','2025-12-03'),('769530','187','flexion','10.0','2025-12-03'),('769530','188','flexion','10.1','2025-12-03'),('769530','189','flexion','19.8','2025-12-03'),('769530','190','flexion','-0.4','2025-12-03'),('769530','191','flexion','0.4','2025-12-03'),('769530','192','flexion','-12.4','2025-12-03'),('769530','193','flexion','3.8','2025-12-03'),('769530','194','flexion','0.6','2025-12-03'),('769530','195','flexion','6.8','2025-12-03'),('769530','196','flexion','7.8','2025-12-03'),('769530','197','flexion','-1.7','2025-12-03'),('769530','198','flexion','7.2','2025-12-03'),('769530','188','reaction','0.431','2025-12-03'),('769530','189','reaction','0.411','2025-12-03'),('769530','190','reaction','0.416','2025-12-03'),('769530','191','reaction','0.461','2025-12-03'),('769530','192','reaction','0.426','2025-12-03'),('769530','193','reaction','0.454','2025-12-03'),('769530','194','reaction','0.414','2025-12-03'),('769530','195','reaction','0.414','2025-12-03'),('769530','196','reaction','0.463','2025-12-03'),('769530','197','reaction','0.450','2025-12-03'),('769530','198','reaction','0.457','2025-12-03')\n" + ] + } + ], "source": [ "import openpyxl\n", "import json\n", "\n", "\n", - "wb = openpyxl.load_workbook('data/镇海手工数据.xlsx',data_only=True)\n", + "wb = openpyxl.load_workbook('data/采油工艺研究院手工录入.xlsx',data_only=True)\n", "sheet = wb.active\n", "s = ''\n", "list1 = []\n", diff --git a/体测单位/天津石化.ipynb b/体测单位/天津石化.ipynb index 5892037..94c8feb 100644 --- a/体测单位/天津石化.ipynb +++ b/体测单位/天津石化.ipynb @@ -2793,9 +2793,7 @@ { "cell_type": "markdown", "id": "08bcbf58-4356-4f30-850f-0d0ff0e2528b", - "metadata": { - "jp-MarkdownHeadingCollapsed": true - }, + "metadata": {}, "source": [ "# 第三次体测(2024年10月)" ] @@ -3205,26 +3203,10 @@ }, { "cell_type": "code", - "execution_count": 2, + "execution_count": null, "id": "98d2da23-cfd4-461a-890a-4533f0409712", - "metadata": { - "execution": { - "iopub.execute_input": "2024-11-12T03:00:45.476616Z", - "iopub.status.busy": "2024-11-12T03:00:45.475867Z", - "iopub.status.idle": "2024-11-12T03:00:48.668118Z", - "shell.execute_reply": "2024-11-12T03:00:48.667607Z", - "shell.execute_reply.started": "2024-11-12T03:00:45.476543Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import os,sys,shutil\n", "import json\n", @@ -3265,12 +3247,27 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 48, "id": "9dcbc866-2481-4cf6-a563-df4bb6596dfd", "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T13:04:55.617841Z", + "iopub.status.busy": "2025-12-03T13:04:55.617249Z", + "iopub.status.idle": "2025-12-03T13:04:56.054475Z", + "shell.execute_reply": "2025-12-03T13:04:56.053478Z", + "shell.execute_reply.started": "2025-12-03T13:04:55.617812Z" + }, "tags": [] }, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok\n" + ] + } + ], "source": [ "import os,sys,shutil\n", "import json\n", @@ -3318,26 +3315,10 @@ }, { "cell_type": "code", - "execution_count": 1, + "execution_count": null, "id": "e92d76a8-6841-4a1b-8529-caa5e37d624a", - "metadata": { - "execution": { - "iopub.execute_input": "2024-11-22T02:08:10.643448Z", - "iopub.status.busy": "2024-11-22T02:08:10.642748Z", - "iopub.status.idle": "2024-11-22T02:08:12.178978Z", - "shell.execute_reply": "2024-11-22T02:08:12.178427Z", - "shell.execute_reply.started": "2024-11-22T02:08:10.643386Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import openpyxl\n", @@ -3415,26 +3396,10 @@ }, { "cell_type": "code", - "execution_count": 24, + "execution_count": null, "id": "c101eccb-a2f3-4404-9583-682cec049160", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-21T13:09:40.491935Z", - "iopub.status.busy": "2025-11-21T13:09:40.491689Z", - "iopub.status.idle": "2025-11-21T13:09:41.068111Z", - "shell.execute_reply": "2025-11-21T13:09:41.067547Z", - "shell.execute_reply.started": "2025-11-21T13:09:40.491915Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "5057 ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -3471,26 +3436,10 @@ }, { "cell_type": "code", - "execution_count": 7, + "execution_count": null, "id": "85bae94e-f480-4c0b-a120-e516e6a5c14d", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-10T10:34:12.756830Z", - "iopub.status.busy": "2025-11-10T10:34:12.756579Z", - "iopub.status.idle": "2025-11-10T10:34:12.798482Z", - "shell.execute_reply": "2025-11-10T10:34:12.797837Z", - "shell.execute_reply.started": "2025-11-10T10:34:12.756807Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "5053 ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "\n", @@ -3523,28 +3472,10 @@ }, { "cell_type": "code", - "execution_count": 1, + "execution_count": null, "id": "0bebc944-ffa0-4bc9-8ae5-5643fe1015dd", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-25T09:08:42.753636Z", - "iopub.status.busy": "2025-11-25T09:08:42.752981Z", - "iopub.status.idle": "2025-11-25T09:08:42.813419Z", - "shell.execute_reply": "2025-11-25T09:08:42.812811Z", - "shell.execute_reply.started": "2025-11-25T09:08:42.753578Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "03603055 赵文远 电仪部-电气作业一区\n", - "03603051 李英豪 电仪部-电气作业一区\n", - "5060 ok\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "\n", @@ -3584,26 +3515,10 @@ }, { "cell_type": "code", - "execution_count": 1, + "execution_count": null, "id": "52de3ba0-f430-444b-87e9-1f13c3e2dcc2", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-29T12:48:41.369858Z", - "iopub.status.busy": "2025-11-29T12:48:41.369285Z", - "iopub.status.idle": "2025-11-29T12:48:41.616095Z", - "shell.execute_reply": "2025-11-29T12:48:41.615602Z", - "shell.execute_reply.started": "2025-11-29T12:48:41.369808Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "3749\n" - ] - } - ], + "metadata": {}, + "outputs": [], "source": [ "import json\n", "import datetime\n", @@ -3615,7 +3530,7 @@ "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", "filename = 'data/marks_20251129.csv'\n", - "re_ta = My.get_result(filename,dict1)\n", + "re_ta = My.get_result_2023(filename,dict1)\n", "\n", "\n", "filename = 'data/result_天津石化2025.json'\n", @@ -3634,17 +3549,9 @@ }, { "cell_type": "code", - "execution_count": 6, + "execution_count": null, "id": "04990b9f-a8f9-4442-935e-d223fde24e72", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-30T09:11:15.004533Z", - "iopub.status.busy": "2025-11-30T09:11:15.003858Z", - "iopub.status.idle": "2025-11-30T09:11:15.536091Z", - "shell.execute_reply": "2025-11-30T09:11:15.535622Z", - "shell.execute_reply.started": "2025-11-30T09:11:15.004455Z" - } - }, + "metadata": {}, "outputs": [], "source": [ "import json\n", @@ -3686,23 +3593,15 @@ }, { "cell_type": "code", - "execution_count": 7, + "execution_count": null, "id": "fcb3c816-5260-4807-9536-b8203680ed1c", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-26T12:46:57.881794Z", - "iopub.status.busy": "2025-11-26T12:46:57.881481Z", - "iopub.status.idle": "2025-11-26T12:46:57.922877Z", - "shell.execute_reply": "2025-11-26T12:46:57.922323Z", - "shell.execute_reply.started": "2025-11-26T12:46:57.881771Z" - } - }, + "metadata": {}, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "\n", - "filename = 'data/result_天津石化2025.json'\n", + "filename = 'data/result_天津石化2025-2.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "\n", @@ -3733,7 +3632,7 @@ " list2 = [k,v['人数'],v['体测人数']]\n", " list1.append(list2)\n", "\n", - "filename = 'data/天津石化部门测试情况(截至20241126).xlsx'\n", + "filename = 'data/天津石化部门测试情况(截至20241130).xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet.append(title)\n", @@ -3753,17 +3652,9 @@ }, { "cell_type": "code", - "execution_count": 8, + "execution_count": null, "id": "df8b9405-a2d9-413d-bcf4-205704b0243a", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-26T12:47:05.652854Z", - "iopub.status.busy": "2025-11-26T12:47:05.652259Z", - "iopub.status.idle": "2025-11-26T12:47:05.810426Z", - "shell.execute_reply": "2025-11-26T12:47:05.809906Z", - "shell.execute_reply.started": "2025-11-26T12:47:05.652799Z" - } - }, + "metadata": {}, "outputs": [], "source": [ "import json\n", @@ -3806,17 +3697,9 @@ }, { "cell_type": "code", - "execution_count": 5, + "execution_count": null, "id": "06979464-2285-4862-94de-84e772f44bbf", - "metadata": { - "execution": { - "iopub.execute_input": "2025-11-29T13:28:01.451927Z", - "iopub.status.busy": "2025-11-29T13:28:01.451274Z", - "iopub.status.idle": "2025-11-29T13:28:02.297890Z", - "shell.execute_reply": "2025-11-29T13:28:02.297407Z", - "shell.execute_reply.started": "2025-11-29T13:28:01.451868Z" - } - }, + "metadata": {}, "outputs": [], "source": [ "import openpyxl\n", @@ -3856,6 +3739,442 @@ "wb.save(filename)" ] }, + { + "cell_type": "markdown", + "id": "d4bd8a8f-6326-4568-8417-abb62e35fe00", + "metadata": {}, + "source": [ + "## 导入手工数据" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "97150a5f-9fcd-4b0a-aed9-61b10b26f2da", + "metadata": {}, + "outputs": [], + "source": [ + "import openpyxl\n", + "import json\n", + "from datetime import date\n", + "\n", + "wb = openpyxl.load_workbook('data/天津石化手工数据2025.xlsx')\n", + "sheet = wb.active\n", + "filename = 'data/result_天津石化2025.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl) \n", + "\n", + "for n in range(2, sheet.max_row+1):\n", + " code = str(sheet.cell(n, 1).value)\n", + " item = sheet.cell(n, 2).value\n", + " mark = str(sheet.cell(n, 3).value)\n", + " rq = str(sheet.cell(n, 3).value)\n", + " if code in dict2.keys():\n", + " dict2[code].setdefault(item,{})\n", + " dict2[code][item]['成绩'] = mark\n", + " else:\n", + " print(code)\n", + " \n", + "filename = 'data/result_天津石化2025-1.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict2, fl, ensure_ascii=False) \n", + "print(len(re_ta)) " + ] + }, + { + "cell_type": "markdown", + "id": "58336a12-561b-4af7-be00-71c6d0a8ef75", + "metadata": {}, + "source": [ + "## 导入腰臀数据" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f50b8f39-0311-4537-8eae-9a84af01702d", + "metadata": {}, + "outputs": [], + "source": [ + "import openpyxl\n", + "import json\n", + "from datetime import date\n", + "\n", + "wb = openpyxl.load_workbook('data/天津石化腰围臀围数据.xlsx')\n", + "sheet = wb.active\n", + "filename = 'data/result_天津石化2025-1.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl) \n", + "\n", + "for n in range(2, sheet.max_row+1):\n", + " code = str(sheet.cell(n, 1).value)\n", + " item = 'wh'\n", + " mark = str(sheet.cell(n, 2).value)+','+str(sheet.cell(n, 3).value)\n", + " if code in dict2.keys():\n", + " dict2[code].setdefault(item,{})\n", + " dict2[code][item]['成绩'] = mark\n", + " else:\n", + " print(code)\n", + " \n", + "filename = 'data/result_天津石化2025-2.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict2, fl, ensure_ascii=False) \n", + "print(len(dict2)) " + ] + }, + { + "cell_type": "markdown", + "id": "43d2d7dd-b747-4177-80df-5ca36bea8541", + "metadata": {}, + "source": [ + "## 统计体测人员项目数" + ] + }, + { + "cell_type": "code", + "execution_count": 40, + "id": "227b2b23-1eb4-4da9-896f-5db070228f5b", + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T11:08:04.377758Z", + "iopub.status.busy": "2025-12-03T11:08:04.377114Z", + "iopub.status.idle": "2025-12-03T11:08:04.744096Z", + "shell.execute_reply": "2025-12-03T11:08:04.743518Z", + "shell.execute_reply.started": "2025-12-03T11:08:04.377699Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "01738573 曹志玲 研究院 2\n", + "01730621 张敬东 电仪部 2\n", + "01733160 王金江 烯烃部 2\n", + "01736847 章洪 热电部 2\n", + "01736378 杨桂强 热电部 1\n", + "01737731 徐欣 运输销售部 1\n", + "01736804 李振江 热电部 2\n", + "01737858 于学宁 运输销售部 2\n", + "01737542 刘呈健 水务部 1\n" + ] + } + ], + "source": [ + "import json\n", + "\n", + "filename = 'data/result_天津石化2025-2.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "xm = ['bmi','lung','grip','flexion','jump','pushup','balance','reaction','step','situp']\n", + "list3 = []\n", + "for k,v in dict1.items():\n", + " list1 = []\n", + " for item in v.keys():\n", + " if item in xm:\n", + " list1.append(item)\n", + " if len(list1) <3:\n", + " print(k,v['name'],v['unit'],len(list1))\n", + " list3.append(k)\n", + "filename = 'data/天津石化人员2025.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl)\n", + "\n", + "list1 = []\n", + "\n", + "i = 1\n", + "for k, v in dict1.items():\n", + " if k not in list3:\n", + " list2 = [i,k,v['name'],v['unit'],dict2[k]['sub_unit'],v['rq'],]\n", + " i+=1\n", + " list1.append(list2)\n", + "#print(list1)\n", + "filename = f'data/天津石化体测人员名单(可出报告).xlsx'\n", + "title = ['序号','员工编号','姓名','部门','车间','体测日期']\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row) \n", + "wb.save(filename)" + ] + }, + { + "cell_type": "markdown", + "id": "620b0be5-0651-4f62-9b4d-8cc590168a61", + "metadata": {}, + "source": [ + "## 导出测试人员信息" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "d2900f9b-573e-41b1-aeae-8cec37a52826", + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "\n", + "items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']\n", + "title = ['编号','姓名','性别','单位','部门','身高','体重','肺活量','握力','坐位体前屈','纵跳','俯卧撑','一分钟仰卧起坐','单脚站立','选择反应时','台阶指数']\n", + "\n", + "filename = 'data/result_天津石化2025-1.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "\n", + "filename = 'data/天津石化人员2025.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl)\n", + " \n", + "list1 = []\n", + "for k, v in dict1.items():\n", + " list2 = []\n", + " list2.append(str(k).rjust(5,'0'))\n", + " list2.append(v['name']) \n", + " list2.append(dict2[k]['sex'])\n", + " list2.append(dict2[k]['unit'])\n", + " list2.append(dict2[k]['birth'])\n", + " if 'phone' in v.keys():\n", + " list2.append(dict2[k]['phone'])\n", + " else:\n", + " list2.append('')\n", + " \n", + " if 'bmi' in v.keys():\n", + " height = v['bmi']['成绩'].split(',')[0]\n", + " weight = v['bmi']['成绩'].split(',')[1]\n", + " list2.append(height)\n", + " list2.append(weight)\n", + " else:\n", + " list2.append('')\n", + " list2.append('')\n", + " for item in items:\n", + " if item in v.keys():\n", + " list2.append(v[item]['成绩']) \n", + " elif item =='name':\n", + " list2.append(v[item])\n", + " else:\n", + " list2.append('')\n", + " \n", + " list1.append(list2)\n", + "filename = 'data/天津石化体测情况表(2025年).xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename)" + ] + }, + { + "cell_type": "markdown", + "id": "3df27e2f-55c0-446a-af01-074e85461bab", + "metadata": {}, + "source": [ + "### 生成报告(按照编号)" + ] + }, + { + "cell_type": "code", + "execution_count": 44, + "id": "d0368e0f-270b-47fd-aec0-cae3cadd68b2", + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T11:31:43.458591Z", + "iopub.status.busy": "2025-12-03T11:31:43.458312Z", + "iopub.status.idle": "2025-12-03T11:31:44.427029Z", + "shell.execute_reply": "2025-12-03T11:31:44.426528Z", + "shell.execute_reply.started": "2025-12-03T11:31:43.458563Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "1\n" + ] + } + ], + "source": [ + "import requests\n", + "import json\n", + "import openpyxl\n", + "\n", + "\n", + "headers = {\n", + " \"Content-Type\": \"application/json; charset=UTF-8\"\n", + " }\n", + "filename = 'data/result_天津石化2025-2.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "filename = 'data/天津石化人员2025.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl)\n", + "list1 = []\n", + "file_path ='./天津石化2025/'\n", + "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi','wh']\n", + "i=0\n", + "list2 = []\n", + "person = ['03603316']\n", + "bumen =['南港乙烯项目管理部']\n", + "for k, v in dict1.items():\n", + " list1 = []\n", + " mydata = {}\n", + " \n", + " #id = str(k).rjust(8,\"0\")\n", + " id = str(k)\n", + " #if v['unit'] in bumen:\n", + " if id in person:\n", + " mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'\n", + " mydata['title'] = '中石化(天津)石油化工\\n有限公司'\n", + " mydata['subtitle'] = v['unit'] + ' '+dict2[k]['sub_unit']\n", + " mydata['id'] = id\n", + " mydata['name'] = v['name']\n", + " if v['sex'] == '男':\n", + " mydata['gender'] = 'male'\n", + " else:\n", + " mydata['gender'] = 'female'\n", + " \n", + " mydata['month'] = v['month']\n", + " mydata['fits'] = {}\n", + " \n", + " for item in list_item:\n", + " if item in v.keys():\n", + " mydata.setdefault('fits',{})\n", + " if item in ['lung','pushup','step','situp']:\n", + " mark = v[item]['成绩'].split()[0].split('.')[0]\n", + " else:\n", + " mark = v[item]['成绩'].split()[0]\n", + " if item == 'bmi':\n", + " mydata['fits']['heightWeight'] = {'mark':mark,'score':-1}\n", + " else:\n", + " mydata['fits'][item] = {'mark':mark,'score':-1}\n", + " #if len(mydata['fits']) >2 or len(mydata['surveys']) >0:\n", + " if len(mydata['fits']) >2 : \n", + " list1.append(mydata)\n", + " list2.append([k,v['name']])\n", + " i+=1\n", + " x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)\n", + " #print(id,v['name'],x.text)\n", + " #print(mydata)\n", + " x.close()\n", + "print(i)" + ] + }, + { + "cell_type": "markdown", + "id": "428ff4de-5c44-4f13-bfd0-3ccca2d2aff4", + "metadata": {}, + "source": [ + "## 核对报告人员" + ] + }, + { + "cell_type": "code", + "execution_count": 41, + "id": "8467b539-c5ca-4db3-8231-f634b70dff88", + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T11:13:39.490227Z", + "iopub.status.busy": "2025-12-03T11:13:39.489101Z", + "iopub.status.idle": "2025-12-03T11:13:39.648115Z", + "shell.execute_reply": "2025-12-03T11:13:39.647529Z", + "shell.execute_reply.started": "2025-12-03T11:13:39.490158Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "03603316 马新淼 炼油部\n", + "01738573 曹志玲 研究院\n", + "01730621 张敬东 电仪部\n", + "01733160 王金江 烯烃部\n", + "01736847 章洪 热电部\n", + "01736378 杨桂强 热电部\n", + "01737731 徐欣 运输销售部\n", + "01736804 李振江 热电部\n", + "01737858 于学宁 运输销售部\n", + "01737542 刘呈健 水务部\n" + ] + } + ], + "source": [ + "import os,sys,shutil\n", + "import json\n", + "import glob\n", + "from pathlib import Path\n", + "import openpyxl\n", + "\n", + "filename = 'data/result_天津石化2025-2.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "\n", + "fi_path = '/home/songyi/pdf-typescript-fit2023/天津石化2025'\n", + "\n", + "fls = glob.glob(f'{fi_path}/*.pdf')\n", + "list1 = []\n", + "title = ['测试编号','姓名','性别','部门']\n", + "for fn in fls:\n", + " list2 = []\n", + " fi_name =Path(fn).stem.split('-')[0] \n", + " list1.append(fi_name)\n", + "for k, v in dict1.items():\n", + " if k not in list1:\n", + " print(k,v['name'],v['unit'])" + ] + }, + { + "cell_type": "markdown", + "id": "1620d417-e80c-473a-81f8-f4e705ae6982", + "metadata": {}, + "source": [ + "## 体测报告按部门、车间分类" + ] + }, + { + "cell_type": "code", + "execution_count": 46, + "id": "b9f88d36-e72a-4f26-8357-62c7b0e2128d", + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T12:03:30.161578Z", + "iopub.status.busy": "2025-12-03T12:03:30.161039Z", + "iopub.status.idle": "2025-12-03T12:03:31.205217Z", + "shell.execute_reply": "2025-12-03T12:03:31.204754Z", + "shell.execute_reply.started": "2025-12-03T12:03:30.161525Z" + } + }, + "outputs": [], + "source": [ + "import os,sys,shutil\n", + "import json\n", + "import glob\n", + "from pathlib import Path\n", + "\n", + "fi_path = '/home/songyi/pdf-typescript-fit2023/天津石化2025'\n", + "new_path = 'file/天津石化2025'\n", + "old = []\n", + "dict2 = {}\n", + "\n", + "filename = 'data/天津石化人员2025.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "fls = glob.glob(f'{fi_path}/*.pdf')\n", + "for fn in fls:\n", + " fi_name =Path(fn).stem.split('-')[0]\n", + " code = fi_name\n", + " unit_path = Path(new_path,dict1[str(code)]['unit'],dict1[str(code)]['sub_unit'])\n", + " unit_path.mkdir(parents = True, exist_ok = True)\n", + " n_name = Path(unit_path,Path(fn).stem+'.pdf')\n", + " if not os.path.exists(n_name):\n", + " shutil.copyfile(fn,n_name)" + ] + }, { "cell_type": "markdown", "id": "85a03b00-9624-4ae9-aa6d-386798d457df", diff --git a/体测单位/宁夏能化.ipynb b/体测单位/宁夏能化.ipynb index 81cda03..f8f0640 100644 --- a/体测单位/宁夏能化.ipynb +++ b/体测单位/宁夏能化.ipynb @@ -1131,9 +1131,17 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 1, "id": "22be8f39-afbb-448a-8ae6-c825c7d38757", - "metadata": {}, + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-01T08:20:03.713528Z", + "iopub.status.busy": "2025-12-01T08:20:03.713011Z", + "iopub.status.idle": "2025-12-01T08:20:04.212086Z", + "shell.execute_reply": "2025-12-01T08:20:04.211619Z", + "shell.execute_reply.started": "2025-12-01T08:20:03.713468Z" + } + }, "outputs": [], "source": [ "import json\n", @@ -1142,7 +1150,7 @@ "\n", "title = []\n", "\n", - "filename = 'data/surveys_records_2025-11-24.json'\n", + "filename = 'data/surveys_records_2025-11-30.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list1 = []\n", @@ -1167,7 +1175,7 @@ " \n", " list1.append(list2)\n", "all_title = ['id', 'date_created','name', 'gender', 'birth', 'code', 'unit', 'height', 'weight', 'next_weight', 'waist', 'hip', 'level4', 'level3', 'level2', 'recipe', 'level1', 'last_level4', 'last_level3', 'last_level2', 'last_level1', 'last_recipe', 'last_lose', 'last_sport', 'sport_type', 'sport_duration', 'last_sport_time', 'last_lose-Comment', 'last_recipe-Comment', 'last_lose_weight', 'last_sport-Comment', 'sport_type-Comment']\n", - "filename = 'data/宁夏能化干预人员问卷(20251123).xlsx'\n", + "filename = 'data/宁夏能化干预人员问卷(20251130).xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet.append(all_title)\n", @@ -1219,7 +1227,7 @@ " password=\"songyi\"\n", ")\n", "cur = conn.cursor()\n", - "wb = openpyxl.load_workbook('data/宁夏能化干预人员问卷(20251123).xlsx',data_only=True)\n", + "wb = openpyxl.load_workbook('data/宁夏能化干预人员问卷(20251130).xlsx',data_only=True)\n", "sheet = wb.active\n", "# sheets = wb.sheetnames\n", "person = {}\n", @@ -1262,10 +1270,36 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 2, "id": "91c53866-3725-4ceb-b0eb-23fb14e20561", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-01T08:22:23.093768Z", + "iopub.status.busy": "2025-12-01T08:22:23.093084Z", + "iopub.status.idle": "2025-12-01T08:22:23.315461Z", + "shell.execute_reply": "2025-12-01T08:22:23.314813Z", + "shell.execute_reply.started": "2025-12-01T08:22:23.093711Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "('李涛', '男', '企业管理部(法律事务部)', '1984-10-16', '02020309', 1)\n", + "('马宏伟', '男', 'BDO运行部', '1998-08-26', '03497946', 1)\n", + "('吴婷', '女', '电气仪表中心', '1989-06-12', '02020010', 1)\n", + "('罗继发', '男', '安全环保部', '1986-09-15', '02019647', 1)\n", + "('苟小锐', '男', '安全环保部', '1974-04-08', '02018416', 1)\n", + "('王振华', '男', '设备工程部', '1981-10-31', '02018482', 1)\n", + "('莫文宁', '男', '物资采购中心', '1972-09-02', '02020208', 1)\n", + "('张保华', '男', '质检中心', '1991-12-11', '02018721', 1)\n", + "('邓蓉芳', '女', '质检中心', '1980-11-12', '02018677', 1)\n", + "('冯波', '男', '乙炔运行部', '1984-07-01', '02019339', 1)\n", + "('尹小俊', '男', '乙炔运行部', '1977-06-05', '2019493', 1)\n" + ] + } + ], "source": [ "import json\n", "import openpyxl\n", @@ -1278,7 +1312,7 @@ " password=\"songyi\"\n", ")\n", "cur = conn.cursor()\n", - "wb = openpyxl.load_workbook('data/宁夏能化干预人员问卷(20251123).xlsx',data_only=True)\n", + "wb = openpyxl.load_workbook('data/宁夏能化干预人员问卷(20251130).xlsx',data_only=True)\n", "sheet = wb.active\n", "# sheets = wb.sheetnames\n", "org_id = 1\n", @@ -1404,15 +1438,15 @@ }, { "cell_type": "code", - "execution_count": 106, + "execution_count": 4, "id": "b7133988-213c-4b52-8d6b-17b56b8ea310", "metadata": { "execution": { - "iopub.execute_input": "2025-11-24T05:18:02.990193Z", - "iopub.status.busy": "2025-11-24T05:18:02.989559Z", - "iopub.status.idle": "2025-11-24T05:18:03.293675Z", - "shell.execute_reply": "2025-11-24T05:18:03.293166Z", - "shell.execute_reply.started": "2025-11-24T05:18:02.990132Z" + "iopub.execute_input": "2025-12-01T08:24:08.522567Z", + "iopub.status.busy": "2025-12-01T08:24:08.522021Z", + "iopub.status.idle": "2025-12-01T08:24:08.789832Z", + "shell.execute_reply": "2025-12-01T08:24:08.789290Z", + "shell.execute_reply.started": "2025-12-01T08:24:08.522517Z" } }, "outputs": [ @@ -1420,7 +1454,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "483 ok\n" + "361 ok\n" ] } ], @@ -1428,7 +1462,7 @@ "import json\n", "import openpyxl\n", "\n", - "wb = openpyxl.load_workbook('data/宁夏能化干预人员问卷(20251123).xlsx',data_only=True)\n", + "wb = openpyxl.load_workbook('data/宁夏能化干预人员问卷(20251130).xlsx',data_only=True)\n", "sheet = wb.active\n", "# sheets = wb.sheetnames\n", "person = {}\n", @@ -1447,7 +1481,7 @@ " person[code] = dict1\n", "#print(person)\n", "\n", - "filename = 'data/surveys_records_2025-11-24.json'\n", + "filename = 'data/surveys_records_2025-11-30.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list1 = []\n", @@ -1505,7 +1539,7 @@ " dict2['last_lose-Comment'] = data['last_lose-Comment']\n", " if code in person.keys():\n", " person[code][rq] = dict2\n", - "filename = 'data/宁夏能化干预人员问卷情况(20251124).json'\n", + "filename = 'data/宁夏能化干预人员问卷情况(20251130).json'\n", "\n", "with open(filename,'w') as fl:\n", " json.dump(person, fl, ensure_ascii=False)\n", @@ -1609,15 +1643,15 @@ }, { "cell_type": "code", - "execution_count": 107, + "execution_count": 5, "id": "55b20059-ae36-493e-b0c5-c7e35473efc7", "metadata": { "execution": { - "iopub.execute_input": "2025-11-24T05:21:09.966817Z", - "iopub.status.busy": "2025-11-24T05:21:09.966532Z", - "iopub.status.idle": "2025-11-24T05:21:11.798693Z", - "shell.execute_reply": "2025-11-24T05:21:11.798041Z", - "shell.execute_reply.started": "2025-11-24T05:21:09.966796Z" + "iopub.execute_input": "2025-12-01T08:24:33.067615Z", + "iopub.status.busy": "2025-12-01T08:24:33.067012Z", + "iopub.status.idle": "2025-12-01T08:24:34.348805Z", + "shell.execute_reply": "2025-12-01T08:24:34.348172Z", + "shell.execute_reply.started": "2025-12-01T08:24:33.067560Z" } }, "outputs": [], @@ -1636,7 +1670,7 @@ "cur = conn.cursor()\n", "event_id = 1\n", "org_id = 1\n", - "filename = 'data/宁夏能化干预人员问卷情况(20251124).json'\n", + "filename = 'data/宁夏能化干预人员问卷情况(20251130).json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list1 = ['name','sex','birth','unit']\n", diff --git a/体测单位/新疆油田采油工艺研究院.ipynb b/体测单位/新疆油田采油工艺研究院.ipynb index 364b426..67543d8 100644 --- a/体测单位/新疆油田采油工艺研究院.ipynb +++ b/体测单位/新疆油田采油工艺研究院.ipynb @@ -90,15 +90,15 @@ }, { "cell_type": "code", - "execution_count": 1, + "execution_count": 25, "id": "cac3736a-56da-403b-9e62-5caaf989a6d4", "metadata": { "execution": { - "iopub.execute_input": "2025-11-29T08:30:03.181743Z", - "iopub.status.busy": "2025-11-29T08:30:03.181050Z", - "iopub.status.idle": "2025-11-29T08:30:06.219308Z", - "shell.execute_reply": "2025-11-29T08:30:06.218748Z", - "shell.execute_reply.started": "2025-11-29T08:30:03.181684Z" + "iopub.execute_input": "2025-12-03T13:46:10.295376Z", + "iopub.status.busy": "2025-12-03T13:46:10.294744Z", + "iopub.status.idle": "2025-12-03T13:46:13.043043Z", + "shell.execute_reply": "2025-12-03T13:46:13.042530Z", + "shell.execute_reply.started": "2025-12-03T13:46:10.295324Z" } }, "outputs": [ @@ -106,7 +106,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "122 ok\n" + "190 ok\n" ] } ], @@ -211,15 +211,15 @@ }, { "cell_type": "code", - "execution_count": 2, + "execution_count": 30, "id": "925f7152-b365-4df8-8d98-73569aef68f8", "metadata": { "execution": { - "iopub.execute_input": "2025-11-29T08:34:02.133638Z", - "iopub.status.busy": "2025-11-29T08:34:02.132876Z", - "iopub.status.idle": "2025-11-29T08:34:02.154487Z", - "shell.execute_reply": "2025-11-29T08:34:02.153682Z", - "shell.execute_reply.started": "2025-11-29T08:34:02.133574Z" + "iopub.execute_input": "2025-12-03T14:07:05.330354Z", + "iopub.status.busy": "2025-12-03T14:07:05.329571Z", + "iopub.status.idle": "2025-12-03T14:07:05.346870Z", + "shell.execute_reply": "2025-12-03T14:07:05.345950Z", + "shell.execute_reply.started": "2025-12-03T14:07:05.330293Z" } }, "outputs": [ @@ -227,7 +227,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "61\n" + "95\n" ] } ], @@ -241,7 +241,7 @@ "filename = 'data/新疆油田采油工艺研究院-202511.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", - "filename = 'data/marks_20251129.csv'\n", + "filename = 'data/marks_20251203.csv'\n", "re_ta = My.get_result(filename,dict1)\n", "\n", "\n", @@ -261,15 +261,15 @@ }, { "cell_type": "code", - "execution_count": 9, + "execution_count": 28, "id": "0905ad6a-f7e4-43ed-ae29-c4850deb946b", "metadata": { "execution": { - "iopub.execute_input": "2025-11-21T02:51:14.528104Z", - "iopub.status.busy": "2025-11-21T02:51:14.527417Z", - "iopub.status.idle": "2025-11-21T02:51:14.544192Z", - "shell.execute_reply": "2025-11-21T02:51:14.543168Z", - "shell.execute_reply.started": "2025-11-21T02:51:14.528044Z" + "iopub.execute_input": "2025-12-03T13:55:35.038083Z", + "iopub.status.busy": "2025-12-03T13:55:35.036944Z", + "iopub.status.idle": "2025-12-03T13:55:35.052613Z", + "shell.execute_reply": "2025-12-03T13:55:35.051549Z", + "shell.execute_reply.started": "2025-12-03T13:55:35.038026Z" } }, "outputs": [ @@ -319,15 +319,15 @@ }, { "cell_type": "code", - "execution_count": 3, + "execution_count": 31, "id": "0e71f6b8-e17e-49f8-a577-7a2eb6f5e994", "metadata": { "execution": { - "iopub.execute_input": "2025-11-29T08:34:10.450721Z", - "iopub.status.busy": "2025-11-29T08:34:10.450172Z", - "iopub.status.idle": "2025-11-29T08:34:10.467986Z", - "shell.execute_reply": "2025-11-29T08:34:10.466883Z", - "shell.execute_reply.started": "2025-11-29T08:34:10.450670Z" + "iopub.execute_input": "2025-12-03T14:07:15.401160Z", + "iopub.status.busy": "2025-12-03T14:07:15.400541Z", + "iopub.status.idle": "2025-12-03T14:07:15.419441Z", + "shell.execute_reply": "2025-12-03T14:07:15.418453Z", + "shell.execute_reply.started": "2025-12-03T14:07:15.401101Z" } }, "outputs": [ @@ -696,15 +696,15 @@ }, { "cell_type": "code", - "execution_count": 4, + "execution_count": 32, "id": "83872322-129c-410d-a9a3-ef6f4fe8dcde", "metadata": { "execution": { - "iopub.execute_input": "2025-11-29T08:35:10.179405Z", - "iopub.status.busy": "2025-11-29T08:35:10.178792Z", - "iopub.status.idle": "2025-11-29T08:35:10.198448Z", - "shell.execute_reply": "2025-11-29T08:35:10.197970Z", - "shell.execute_reply.started": "2025-11-29T08:35:10.179353Z" + "iopub.execute_input": "2025-12-03T14:07:28.969613Z", + "iopub.status.busy": "2025-12-03T14:07:28.969043Z", + "iopub.status.idle": "2025-12-03T14:07:28.992695Z", + "shell.execute_reply": "2025-12-03T14:07:28.991843Z", + "shell.execute_reply.started": "2025-12-03T14:07:28.969559Z" } }, "outputs": [ @@ -712,7 +712,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "61\n" + "95\n" ] } ], @@ -724,7 +724,7 @@ "from datetime import date\n", "\n", "list1 = []\n", - "filename = 'data/sql_20251129.csv'\n", + "filename = 'data/sql_20251203.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -939,15 +939,15 @@ }, { "cell_type": "code", - "execution_count": 5, + "execution_count": 33, "id": "28790176-7666-4b2d-83a5-9207cb8ad757", "metadata": { "execution": { - "iopub.execute_input": "2025-11-29T08:38:54.922892Z", - "iopub.status.busy": "2025-11-29T08:38:54.922316Z", - "iopub.status.idle": "2025-11-29T08:38:57.059105Z", - "shell.execute_reply": "2025-11-29T08:38:57.057160Z", - "shell.execute_reply.started": "2025-11-29T08:38:54.922836Z" + "iopub.execute_input": "2025-12-03T14:08:59.269168Z", + "iopub.status.busy": "2025-12-03T14:08:59.268598Z", + "iopub.status.idle": "2025-12-03T14:09:06.148725Z", + "shell.execute_reply": "2025-12-03T14:09:06.148173Z", + "shell.execute_reply.started": "2025-12-03T14:08:59.269114Z" } }, "outputs": [ @@ -955,7 +955,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "2\n" + "14\n" ] } ], @@ -975,7 +975,7 @@ "file_path ='./新疆油田采油工艺研究院2511/'\n", "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']\n", "i=0\n", - "list2 = ['154','49']\n", + "list2 = ['185','186','187','188','189','190','191','192','193','194','195','196','197','198']\n", "for k, v in dict1.items():\n", " if k in list2:\n", " list1 = []\n", diff --git a/文件管理.ipynb b/文件管理.ipynb index b6cec13..358d485 100644 --- a/文件管理.ipynb +++ b/文件管理.ipynb @@ -1270,18 +1270,55 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 5, "id": "2f9f474d-265c-4beb-bd0e-53372ae9e57e", "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T04:26:40.331997Z", + "iopub.status.busy": "2025-12-03T04:26:40.331341Z", + "iopub.status.idle": "2025-12-03T04:26:41.149586Z", + "shell.execute_reply": "2025-12-03T04:26:41.149140Z", + "shell.execute_reply.started": "2025-12-03T04:26:40.331947Z" + }, "tags": [] }, - "outputs": [], + "outputs": [ + { + "name": "stderr", + "output_type": "stream", + "text": [ + "<>:51: SyntaxWarning: invalid escape sequence '\\s'\n", + "<>:51: SyntaxWarning: invalid escape sequence '\\s'\n", + "/tmp/ipykernel_1730998/3799497539.py:51: SyntaxWarning: invalid escape sequence '\\s'\n", + " data =[re.sub('\\s+', '', cell) if cell is not None else None for cell in row]\n" + ] + }, + { + "name": "stdout", + "output_type": "stream", + "text": [ + "['姓名', '邵希强']\n", + "['性别', '男']\n", + "['测试标准', '国民体质测定标准']\n", + "['身高体重指数', '26.02', '60分']\n", + "['握力', '33.5千克', '50分']\n", + "['肺活量', '3050毫升', '70分']\n", + "['选择反应时', '0.572秒', '75分']\n", + "['指标', '您的结果', '亚洲男性平均值']\n", + "['臀围', '100', '88.82']\n", + "['身高腰围指数', '53.19', '42.79']\n", + "['建议类别', '建议项']\n", + "['增加摄入', '豆类、水果、蔬菜、坚果类、蛋类、菌类、乳制品、肉类']\n", + "['生活习惯', '提高睡眠质量、保持心情舒畅、换季时避开感染源、减少用眼']\n" + ] + } + ], "source": [ "import pdfplumber\n", "import json\n", "import re\n", "\n", - "file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n", + "file = \"file/01730823-邵希强.pdf\"\n", "pdf = pdfplumber.open(file)\n", "list1 = []\n", "dict1 = {}\n", @@ -1347,58 +1384,98 @@ }, { "cell_type": "code", - "execution_count": null, - "id": "71e1da77-6744-4d86-a4bc-88feeb3023e8", - "metadata": { - "tags": [] - }, - "outputs": [], - "source": [ - "import pdfplumber\n", - "import json\n", - "import re\n", - "\n", - "file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n", - "pdf = pdfplumber.open(file)\n", - "list1 = []\n", - "dict1 = {}\n", - "n = 1\n", - "for page in pdf.pages:\n", - " for pdf_table in page.extract_tables():\n", - " list2 = []\n", - " table = []\n", - " cells = []\n", - " for row in pdf_table:\n", - " print(row)\n", - " print('******')\n", - " print('--------')\n", - " " - ] - }, - { - "cell_type": "code", - "execution_count": null, + "execution_count": 13, "id": "b0764366-12cf-4880-8e0f-9ae18b46d392", "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T05:52:34.403674Z", + "iopub.status.busy": "2025-12-03T05:52:34.402297Z", + "iopub.status.idle": "2025-12-03T05:52:34.532929Z", + "shell.execute_reply": "2025-12-03T05:52:34.532427Z", + "shell.execute_reply.started": "2025-12-03T05:52:34.403605Z" + }, "tags": [] }, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[['姓名', '邵希强']]\n", + "[['性别', '男']]\n", + "[['测试标准', '国民体质测定标准']]\n", + "[['身高体重指数', '26.02', '60分']]\n", + "[['握力', '33.5 千克', '50分']]\n", + "[['肺活量', '3050 毫升', '70分']]\n", + "[['选择反应时', '0.572 秒', '75分']]\n" + ] + } + ], "source": [ - "import camelot\n", - "import json\n", - "import re\n", + "import pdfplumber\n", + "name = 'file/01730823-邵希强.pdf'\n", "\n", - "file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n", - "tables = camelot.read_pdf(file, pages='3',flavor='stream')\n", - "# 2.导出pdf所有的表格为csv文件\n", - "tables.export('foo.json', f='json')\n", - "print('ok!')" + "pdf = pdfplumber.open(name)\n", + "tables =pdf.pages[1].extract_tables()\n", + "df1 = tables\n", + "for item in df1:\n", + " print(item)" + ] + }, + { + "cell_type": "code", + "execution_count": 30, + "id": "04dd6436-8374-4096-a2a3-d79f4d93a360", + "metadata": { + "execution": { + "iopub.execute_input": "2025-12-03T06:25:24.215544Z", + "iopub.status.busy": "2025-12-03T06:25:24.214765Z", + "iopub.status.idle": "2025-12-03T06:25:24.280529Z", + "shell.execute_reply": "2025-12-03T06:25:24.280065Z", + "shell.execute_reply.started": "2025-12-03T06:25:24.215449Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "['北体乾元健康管理中心', '姓名 邵希强', '中石化(天津)石油化工 编号 01730823', '有限公司 性别 男', '年龄 53', '国民体质检测结果与健康处方', '肺活量', '握力 身高体重指数', '坐位体前屈 选择反应时', '纵跳 闭眼单脚站立', '俯卧撑', '测试标准 国民体质测定标准', '闭眼单脚站立 2.8 秒 10分', '身高体重指数 26.02 60分', '坐位体前屈 13.5 厘米 90分', '握力 33.5 千克 50分', '纵跳 18.0 厘米 30分', '肺活量 3050 毫升 70分', '俯卧撑 5 次 50分', '选择反应时 0.572 秒 75分', '腰臀比 0.90 正常', '请注意:以上测试项目及格线为3分。列表中红色项目需重点关注,建议在专家指导下进行科学锻炼,', '减少潜在的运动风险。', '感谢您完成测试,以《国民体质测定标准》综合评级,您的总分为58分,等级为四级(不合格)。', '(其中项目纵跳因超出年龄范围得分仅供参考,不计入总分)', '北体乾元体质监测报告']\n", + "11 21\n", + "闭眼单脚站立 2.8 秒 10分\n", + "身高体重指数 26.02 60分\n", + "坐位体前屈 13.5 厘米 90分\n", + "握力 33.5 千克 50分\n", + "纵跳 18.0 厘米 30分\n", + "肺活量 3050 毫升 70分\n", + "俯卧撑 5 次 50分\n", + "选择反应时 0.572 秒 75分\n", + "腰臀比 0.90 正常\n" + ] + } + ], + "source": [ + "import pdfplumber\n", + "name = 'file/01730823-邵希强.pdf'\n", + "pdf = pdfplumber.open(name)\n", + "text = pdf.pages[1].extract_text()#######页码从0开始计数\n", + "#print(text)\n", + "list1 = text.split('\\n')\n", + "print(list1)\n", + "for item in list1:\n", + " if '测试标准 国民体质测定标准' in item:\n", + " list_min = list1.index(item)\n", + " if '请注意:以上测试项目' in item:\n", + " list_max = list1.index(item)\n", + "print(list_min,list_max)\n", + "for i in range(list_min+1,list_max):\n", + " print(list1[i])\n" ] }, { "cell_type": "code", "execution_count": null, - "id": "04dd6436-8374-4096-a2a3-d79f4d93a360", + "id": "ac034bf9-cba9-4a72-87ba-d77721cc7097", "metadata": {}, "outputs": [], "source": []