diff --git a/体测单位/北海炼化.ipynb b/体测单位/北海炼化.ipynb index a7b8b09..2569320 100644 --- a/体测单位/北海炼化.ipynb +++ b/体测单位/北海炼化.ipynb @@ -4,6 +4,7 @@ "cell_type": "markdown", "id": "1c72f0cf-165b-4025-af9d-6771e4a0a5d1", "metadata": { + "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ @@ -635,6 +636,7 @@ "cell_type": "markdown", "id": "c103301b-d540-4bf5-861a-0eb2e313e26e", "metadata": { + "jp-MarkdownHeadingCollapsed": true, "tags": [] }, "source": [ @@ -1996,27 +1998,12 @@ }, { "cell_type": "code", - "execution_count": 149, + "execution_count": null, "id": "7410324c-92c2-4a20-9715-f003d4915c8a", "metadata": { - "execution": { - "iopub.execute_input": "2024-04-10T11:51:24.039781Z", - "iopub.status.busy": "2024-04-10T11:51:24.039544Z", - "iopub.status.idle": "2024-04-10T11:51:24.665413Z", - "shell.execute_reply": "2024-04-10T11:51:24.664830Z", - "shell.execute_reply.started": "2024-04-10T11:51:24.039764Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -2074,109 +2061,19 @@ }, { "cell_type": "code", - "execution_count": 148, + "execution_count": 158, "id": "28825d80-9123-4ec5-b107-9ad133f11a82", "metadata": { "execution": { - "iopub.execute_input": "2024-04-10T11:47:45.024742Z", - "iopub.status.busy": "2024-04-10T11:47:45.024528Z", - "iopub.status.idle": "2024-04-10T11:47:45.386052Z", - "shell.execute_reply": "2024-04-10T11:47:45.385453Z", - "shell.execute_reply.started": "2024-04-10T11:47:45.024726Z" + "iopub.execute_input": "2024-04-11T03:17:59.468194Z", + "iopub.status.busy": "2024-04-11T03:17:59.467979Z", + "iopub.status.idle": "2024-04-11T03:17:59.981566Z", + "shell.execute_reply": "2024-04-11T03:17:59.981079Z", + "shell.execute_reply.started": "2024-04-11T03:17:59.468179Z" }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "畸胎瘤\n", - "细胞角蛋白19片段增高\n", - "癌胚抗原(定量)(CEA)增高\n", - "尿糖阳性\n", - "脊柱侧弯\n", - "甲状腺结节\n", - "游离三碘甲状腺原氨酸(FT3)增高\n", - "2级高血压(中度)\n", - "肾错构瘤\n", - "胆囊结石\n", - "天门冬氨酸氨基转移酶(AST)增高\n", - "颈椎退行性变\n", - "血脂异常\n", - "听力下降\n", - "肺少许慢性炎症\n", - "卵巢囊肿\n", - "色弱\n", - "肥胖\n", - "窦性心动过缓\n", - "预激综合征\n", - "甲胎蛋白(AFP)增高\n", - "体重偏低\n", - "糖链抗原(CA-125)偏高\n", - "前列腺特异抗原(PSA)增高\n", - "肾钙乳症\n", - "血压高\n", - "轻度ST段改变\n", - "肺大疱\n", - "心影增大\n", - "肝囊肿\n", - "乳腺囊肿\n", - "胆管扩张\n", - "C-14呼气试验阳性\n", - "肾结石\n", - "胆囊壁稍毛糙\n", - "尿蛋白阳性\n", - "血清γ-谷氨酰基转移酶(GGT)增高\n", - "咽部充血\n", - "糖链抗原(CA-242)偏高\n", - "脂肪肝\n", - "前列腺囊肿\n", - "动脉血管弹性下降\n", - "肝脂肪度为中度\n", - "胆囊息肉\n", - "白内障\n", - "前列腺稍大并钙化灶\n", - "肝纤维化为轻度\n", - "肝血管瘤\n", - "血压低\n", - "空腹血糖增高\n", - "肝脂肪度为轻度\n", - "前列腺增大\n", - "肝纤维化为中度\n", - "人乳头瘤病毒基因分型(HPV)高危型阳\n", - "前列腺钙化灶\n", - "EB病毒抗体(EB-Ab)阳性\n", - "颈椎骨质增生\n", - "肝脂肪度为重度\n", - "胆囊息肉样病变\n", - "轻度ST-T改变\n", - "胆囊泥沙样结石\n", - "肝实质光点密集\n", - "肾囊肿\n", - "血尿酸增高\n", - "尿潜血阳性\n", - "心电图提示:V1呈QS型\n", - "血肌酐增高\n", - "室性早搏\n", - "宫颈息肉\n", - "肺结节\n", - "胆固醇结晶\n", - "神经元特异性烯醇化酶(NSE)偏高\n", - "心房纤颤\n", - "肝纤维化为重度\n", - "糖化血红蛋白(HbA1c)增高\n", - "甲状腺囊肿\n", - "肺少许炎症\n", - "脾囊肿\n", - "肝低回声区\n", - "屈光不正\n", - "血清丙氨酸氨基转移酶(ALT)增高\n", - "肝实质光点稍密集\n", - "动脉血管弹性稍下降\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -2184,15 +2081,25 @@ "wb = openpyxl.load_workbook('data/北海体检报告数据整理2024.xlsx')\n", "#sheet = wb.active\n", "# sheets = wb.sheetnames\n", - "sheet = wb['Sheet3']\n", + "sheet = wb['原始']\n", "xm = set()\n", + "list1 = []\n", "for n in range(2, sheet.max_row+1):\n", " if sheet.cell(n,2).value is None:\n", " break\n", " else: \n", - " xm.add(sheet.cell(n, 7).value) \n", - "for item in xm:\n", - " print(item)" + " xm.add(sheet.cell(n, 6).value) \n", + "for i, item in enumerate(xm):\n", + " list1.append([i+1,item])\n", + "filename = 'data/中国石化北海炼化有限责任公司2023年体检指标统计表1.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "title = ['序号','项目名称']\n", + "sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename) " ] }, { diff --git a/文件管理1.ipynb b/文件管理1.ipynb index 722a884..9441db2 100755 --- a/文件管理1.ipynb +++ b/文件管理1.ipynb @@ -638,24 +638,23 @@ "for i, table in enumerate(doc.tables):\n", " print(f\"Table {i}:\")\n", " for row in table.rows:\n", - " print(type(row))\n", + " #print(type(row))\n", " for cell in row.cells:\n", - " if cell.text =='序号':\n", - " print(cell.text, end=\" | \")\n", + " print(cell.text, end=\" | \")\n", " print() # 每一行结束后换行\n" ] }, { "cell_type": "code", - "execution_count": 18, - "id": "c3ccfa25-f31c-46d4-bfd0-903832cc620d", + "execution_count": 36, + "id": "fe615899-447e-489a-9f0d-90a239031e7c", "metadata": { "execution": { - "iopub.execute_input": "2024-04-10T11:08:13.063231Z", - "iopub.status.busy": "2024-04-10T11:08:13.063012Z", - "iopub.status.idle": "2024-04-10T11:08:25.173599Z", - "shell.execute_reply": "2024-04-10T11:08:25.173048Z", - "shell.execute_reply.started": "2024-04-10T11:08:13.063213Z" + "iopub.execute_input": "2024-04-11T03:04:05.540001Z", + "iopub.status.busy": "2024-04-11T03:04:05.539789Z", + "iopub.status.idle": "2024-04-11T03:04:19.212186Z", + "shell.execute_reply": "2024-04-11T03:04:19.211594Z", + "shell.execute_reply.started": "2024-04-11T03:04:05.539987Z" }, "tags": [] }, @@ -675,14 +674,52 @@ "doc = docx.Document(r\"data/中国石化北海炼化有限责任公司2023年职业体检团体报告.docx\")\n", "list2 = []\n", "for i, table in enumerate(doc.tables):\n", + " list1= []\n", + " for row in table.rows:\n", + " list3 = []\n", + " for cell in row.cells:\n", + " if cell.text.strip()!='':\n", + " list3.append(cell.text)\n", + " list1.append(list3)\n", + " if list1[0][0] =='序号':\n", + " for n in range(1,len(list1)):\n", + " list2.append(list1[n])\n", + "filename = '北海职业体测报告情况表2024.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "\n", + "for row in list2:\n", + " sheet.append(row)\n", " \n", - " if table.cell(0, 0).text == '序号':\n", + "wb.save(filename)\n", + "print('ok!')\n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "c3ccfa25-f31c-46d4-bfd0-903832cc620d", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import docx\n", + "import json\n", + "import openpyxl\n", + "doc = docx.Document(r\"data/中国石化北海炼化有限责任公司2023年健康体检团体报告.docx\")\n", + "list2 = []\n", + "for i, table in enumerate(doc.tables):\n", + " \n", + " if table.cell(0, 0).text == '序号' or table.cell(0, 1).text == '序号':\n", " #print(f\"Table {i}:\")\n", " \n", " for row in table.rows:\n", " list1= []\n", " for cell in row.cells:\n", - " list1.append(cell.text)\n", + " if cell.text.strip()!='':\n", + " list1.append(cell.text)\n", " if list1[0] !='序号':\n", " list2.append(list1)\n", "filename = '北海体测报告情况表2024.xlsx'\n",