This commit is contained in:
512song committed 2024-04-11 11:35:29 +08:00
1 parent 55437a8820
commit 0d4e4454b4
2 files changed
+72 -128

No files matched your search

+25 -118
View File
@@ -4,6 +4,7 @@
"cell_type": "markdown", "cell_type": "markdown",
"id": "1c72f0cf-165b-4025-af9d-6771e4a0a5d1", "id": "1c72f0cf-165b-4025-af9d-6771e4a0a5d1",
"metadata": { "metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": [] "tags": []
}, },
"source": [ "source": [
@@ -635,6 +636,7 @@
"cell_type": "markdown", "cell_type": "markdown",
"id": "c103301b-d540-4bf5-861a-0eb2e313e26e", "id": "c103301b-d540-4bf5-861a-0eb2e313e26e",
"metadata": { "metadata": {
"jp-MarkdownHeadingCollapsed": true,
"tags": [] "tags": []
}, },
"source": [ "source": [
@@ -1996,27 +1998,12 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": 149, "execution_count": null,
"id": "7410324c-92c2-4a20-9715-f003d4915c8a", "id": "7410324c-92c2-4a20-9715-f003d4915c8a",
"metadata": { "metadata": {
"execution": {
"iopub.execute_input": "2024-04-10T11:51:24.039781Z",
"iopub.status.busy": "2024-04-10T11:51:24.039544Z",
"iopub.status.idle": "2024-04-10T11:51:24.665413Z",
"shell.execute_reply": "2024-04-10T11:51:24.664830Z",
"shell.execute_reply.started": "2024-04-10T11:51:24.039764Z"
},
"tags": [] "tags": []
}, },
"outputs": [ "outputs": [],
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [ "source": [
"import openpyxl\n", "import openpyxl\n",
"import json\n", "import json\n",
@@ -2074,109 +2061,19 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": 148, "execution_count": 158,
"id": "28825d80-9123-4ec5-b107-9ad133f11a82", "id": "28825d80-9123-4ec5-b107-9ad133f11a82",
"metadata": { "metadata": {
"execution": { "execution": {
"iopub.execute_input": "2024-04-10T11:47:45.024742Z", "iopub.execute_input": "2024-04-11T03:17:59.468194Z",
"iopub.status.busy": "2024-04-10T11:47:45.024528Z", "iopub.status.busy": "2024-04-11T03:17:59.467979Z",
"iopub.status.idle": "2024-04-10T11:47:45.386052Z", "iopub.status.idle": "2024-04-11T03:17:59.981566Z",
"shell.execute_reply": "2024-04-10T11:47:45.385453Z", "shell.execute_reply": "2024-04-11T03:17:59.981079Z",
"shell.execute_reply.started": "2024-04-10T11:47:45.024726Z" "shell.execute_reply.started": "2024-04-11T03:17:59.468179Z"
}, },
"tags": [] "tags": []
}, },
"outputs": [ "outputs": [],
{
"name": "stdout",
"output_type": "stream",
"text": [
"畸胎瘤\n",
"细胞角蛋白19片段增高\n",
"癌胚抗原(定量)(CEA)增高\n",
"尿糖阳性\n",
"脊柱侧弯\n",
"甲状腺结节\n",
"游离三碘甲状腺原氨酸(FT3)增高\n",
"2级高血压(中度)\n",
"肾错构瘤\n",
"胆囊结石\n",
"天门冬氨酸氨基转移酶(AST)增高\n",
"颈椎退行性变\n",
"血脂异常\n",
"听力下降\n",
"肺少许慢性炎症\n",
"卵巢囊肿\n",
"色弱\n",
"肥胖\n",
"窦性心动过缓\n",
"预激综合征\n",
"甲胎蛋白(AFP)增高\n",
"体重偏低\n",
"糖链抗原(CA-125)偏高\n",
"前列腺特异抗原(PSA)增高\n",
"肾钙乳症\n",
"血压高\n",
"轻度ST段改变\n",
"肺大疱\n",
"心影增大\n",
"肝囊肿\n",
"乳腺囊肿\n",
"胆管扩张\n",
"C-14呼气试验阳性\n",
"肾结石\n",
"胆囊壁稍毛糙\n",
"尿蛋白阳性\n",
"血清γ-谷氨酰基转移酶(GGT)增高\n",
"咽部充血\n",
"糖链抗原(CA-242)偏高\n",
"脂肪肝\n",
"前列腺囊肿\n",
"动脉血管弹性下降\n",
"肝脂肪度为中度\n",
"胆囊息肉\n",
"白内障\n",
"前列腺稍大并钙化灶\n",
"肝纤维化为轻度\n",
"肝血管瘤\n",
"血压低\n",
"空腹血糖增高\n",
"肝脂肪度为轻度\n",
"前列腺增大\n",
"肝纤维化为中度\n",
"人乳头瘤病毒基因分型(HPV)高危型阳\n",
"前列腺钙化灶\n",
"EB病毒抗体(EB-Ab)阳性\n",
"颈椎骨质增生\n",
"肝脂肪度为重度\n",
"胆囊息肉样病变\n",
"轻度ST-T改变\n",
"胆囊泥沙样结石\n",
"肝实质光点密集\n",
"肾囊肿\n",
"血尿酸增高\n",
"尿潜血阳性\n",
"心电图提示:V1呈QS型\n",
"血肌酐增高\n",
"室性早搏\n",
"宫颈息肉\n",
"肺结节\n",
"胆固醇结晶\n",
"神经元特异性烯醇化酶(NSE)偏高\n",
"心房纤颤\n",
"肝纤维化为重度\n",
"糖化血红蛋白(HbA1c)增高\n",
"甲状腺囊肿\n",
"肺少许炎症\n",
"脾囊肿\n",
"肝低回声区\n",
"屈光不正\n",
"血清丙氨酸氨基转移酶(ALT)增高\n",
"肝实质光点稍密集\n",
"动脉血管弹性稍下降\n"
]
}
],
"source": [ "source": [
"import openpyxl\n", "import openpyxl\n",
"import json\n", "import json\n",
@@ -2184,15 +2081,25 @@
"wb = openpyxl.load_workbook('data/北海体检报告数据整理2024.xlsx')\n", "wb = openpyxl.load_workbook('data/北海体检报告数据整理2024.xlsx')\n",
"#sheet = wb.active\n", "#sheet = wb.active\n",
"# sheets = wb.sheetnames\n", "# sheets = wb.sheetnames\n",
"sheet = wb['Sheet3']\n", "sheet = wb['原始']\n",
"xm = set()\n", "xm = set()\n",
"list1 = []\n",
"for n in range(2, sheet.max_row+1):\n", "for n in range(2, sheet.max_row+1):\n",
" if sheet.cell(n,2).value is None:\n", " if sheet.cell(n,2).value is None:\n",
" break\n", " break\n",
" else: \n", " else: \n",
" xm.add(sheet.cell(n, 7).value) \n", " xm.add(sheet.cell(n, 6).value) \n",
"for item in xm:\n", "for i, item in enumerate(xm):\n",
" print(item)" " list1.append([i+1,item])\n",
"filename = 'data/中国石化北海炼化有限责任公司2023年体检指标统计表1.xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"title = ['序号','项目名称']\n",
"sheet.append(title)\n",
"for row in list1:\n",
" sheet.append(row)\n",
" \n",
"wb.save(filename) "
] ]
}, },
{ {
+47 -10
View File
@@ -638,24 +638,23 @@
"for i, table in enumerate(doc.tables):\n", "for i, table in enumerate(doc.tables):\n",
" print(f\"Table {i}:\")\n", " print(f\"Table {i}:\")\n",
" for row in table.rows:\n", " for row in table.rows:\n",
" print(type(row))\n", " #print(type(row))\n",
" for cell in row.cells:\n", " for cell in row.cells:\n",
" if cell.text =='序号':\n",
" print(cell.text, end=\" | \")\n", " print(cell.text, end=\" | \")\n",
" print() # 每一行结束后换行\n" " print() # 每一行结束后换行\n"
] ]
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": 18, "execution_count": 36,
"id": "c3ccfa25-f31c-46d4-bfd0-903832cc620d", "id": "fe615899-447e-489a-9f0d-90a239031e7c",
"metadata": { "metadata": {
"execution": { "execution": {
"iopub.execute_input": "2024-04-10T11:08:13.063231Z", "iopub.execute_input": "2024-04-11T03:04:05.540001Z",
"iopub.status.busy": "2024-04-10T11:08:13.063012Z", "iopub.status.busy": "2024-04-11T03:04:05.539789Z",
"iopub.status.idle": "2024-04-10T11:08:25.173599Z", "iopub.status.idle": "2024-04-11T03:04:19.212186Z",
"shell.execute_reply": "2024-04-10T11:08:25.173048Z", "shell.execute_reply": "2024-04-11T03:04:19.211594Z",
"shell.execute_reply.started": "2024-04-10T11:08:13.063213Z" "shell.execute_reply.started": "2024-04-11T03:04:05.539987Z"
}, },
"tags": [] "tags": []
}, },
@@ -675,13 +674,51 @@
"doc = docx.Document(r\"data/中国石化北海炼化有限责任公司2023年职业体检团体报告.docx\")\n", "doc = docx.Document(r\"data/中国石化北海炼化有限责任公司2023年职业体检团体报告.docx\")\n",
"list2 = []\n", "list2 = []\n",
"for i, table in enumerate(doc.tables):\n", "for i, table in enumerate(doc.tables):\n",
" list1= []\n",
" for row in table.rows:\n",
" list3 = []\n",
" for cell in row.cells:\n",
" if cell.text.strip()!='':\n",
" list3.append(cell.text)\n",
" list1.append(list3)\n",
" if list1[0][0] =='序号':\n",
" for n in range(1,len(list1)):\n",
" list2.append(list1[n])\n",
"filename = '北海职业体测报告情况表2024.xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"\n", "\n",
" if table.cell(0, 0).text == '序号':\n", "for row in list2:\n",
" sheet.append(row)\n",
" \n",
"wb.save(filename)\n",
"print('ok!')\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c3ccfa25-f31c-46d4-bfd0-903832cc620d",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import docx\n",
"import json\n",
"import openpyxl\n",
"doc = docx.Document(r\"data/中国石化北海炼化有限责任公司2023年健康体检团体报告.docx\")\n",
"list2 = []\n",
"for i, table in enumerate(doc.tables):\n",
" \n",
" if table.cell(0, 0).text == '序号' or table.cell(0, 1).text == '序号':\n",
" #print(f\"Table {i}:\")\n", " #print(f\"Table {i}:\")\n",
" \n", " \n",
" for row in table.rows:\n", " for row in table.rows:\n",
" list1= []\n", " list1= []\n",
" for cell in row.cells:\n", " for cell in row.cells:\n",
" if cell.text.strip()!='':\n",
" list1.append(cell.text)\n", " list1.append(cell.text)\n",
" if list1[0] !='序号':\n", " if list1[0] !='序号':\n",
" list2.append(list1)\n", " list2.append(list1)\n",