diff --git a/数据处理.ipynb b/数据处理.ipynb index 4340191..01e08ef 100644 --- a/数据处理.ipynb +++ b/数据处理.ipynb @@ -212,30 +212,11 @@ }, { "cell_type": "code", - "execution_count": 71, + "execution_count": null, "metadata": { - "execution": { - "iopub.execute_input": "2021-11-24T03:41:43.078279Z", - "iopub.status.busy": "2021-11-24T03:41:43.077279Z", - "iopub.status.idle": "2021-11-24T03:41:43.162298Z", - "shell.execute_reply": "2021-11-24T03:41:43.161298Z", - "shell.execute_reply.started": "2021-11-24T03:41:43.078279Z" - }, "tags": [] }, - "outputs": [ - { - "ename": "IndexError", - "evalue": "list index out of range", - "output_type": "error", - "traceback": [ - "\u001b[1;31m---------------------------------------------------------------------------\u001b[0m", - "\u001b[1;31mIndexError\u001b[0m Traceback (most recent call last)", - "\u001b[1;32m~\\AppData\\Local\\Temp/ipykernel_166520/2691368177.py\u001b[0m in \u001b[0;36m\u001b[1;34m\u001b[0m\n\u001b[0;32m 23\u001b[0m \u001b[0msheet\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34mf'E{i}'\u001b[0m\u001b[1;33m]\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mv\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34m'简介'\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m3\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m':'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m1\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 24\u001b[0m \u001b[0msheet\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34mf'F{i}'\u001b[0m\u001b[1;33m]\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mv\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34m'简介'\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m4\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m':'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m1\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[1;32m---> 25\u001b[1;33m \u001b[0msheet\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34mf'G{i}'\u001b[0m\u001b[1;33m]\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mv\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34m'简介'\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m5\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m':'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m1\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 26\u001b[0m \u001b[0mi\u001b[0m \u001b[1;33m+=\u001b[0m \u001b[1;36m1\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 27\u001b[0m \u001b[0mwb\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msave\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mfilename\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n", - "\u001b[1;31mIndexError\u001b[0m: list index out of range" - ] - } - ], + "outputs": [], "source": [ "import json\n", "import openpyxl\n", @@ -278,15 +259,8 @@ }, { "cell_type": "code", - "execution_count": 68, + "execution_count": null, "metadata": { - "execution": { - "iopub.execute_input": "2021-11-24T03:23:57.451845Z", - "iopub.status.busy": "2021-11-24T03:23:57.451845Z", - "iopub.status.idle": "2021-11-24T03:23:57.624934Z", - "shell.execute_reply": "2021-11-24T03:23:57.624934Z", - "shell.execute_reply.started": "2021-11-24T03:23:57.451845Z" - }, "tags": [] }, "outputs": [], @@ -379,6 +353,26 @@ "print(collist)" ] }, + { + "cell_type": "code", + "execution_count": 78, + "metadata": { + "execution": { + "iopub.execute_input": "2021-11-24T11:15:33.859507Z", + "iopub.status.busy": "2021-11-24T11:15:33.858508Z", + "iopub.status.idle": "2021-11-24T11:15:33.864520Z", + "shell.execute_reply": "2021-11-24T11:15:33.864520Z", + "shell.execute_reply.started": "2021-11-24T11:15:33.859507Z" + }, + "tags": [] + }, + "outputs": [], + "source": [ + "list1 = []\n", + "if not list1:\n", + " print('true')" + ] + }, { "cell_type": "code", "execution_count": null, diff --git a/文件管理1.ipynb b/文件管理1.ipynb index b73796b..797d093 100755 --- a/文件管理1.ipynb +++ b/文件管理1.ipynb @@ -596,10 +596,145 @@ "doc2.save(\"first-and-last-10.pdf\")\n" ] }, + { + "cell_type": "markdown", + "id": "917dfa96-aaea-4c40-8b89-0fe093c65c33", + "metadata": {}, + "source": [ + "## PDF文件读取表格" + ] + }, { "cell_type": "code", "execution_count": null, - "id": "ed609077-c179-4041-8ad2-200f02c4ea83", + "id": "2f9f474d-265c-4beb-bd0e-53372ae9e57e", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import pdfplumber\n", + "import json\n", + "import re\n", + "\n", + "file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n", + "pdf = pdfplumber.open(file)\n", + "list1 = []\n", + "dict1 = {}\n", + "n = 1\n", + "for page in pdf.pages:\n", + " # print(page.extract_text())\n", + " \n", + " \n", + " for pdf_table in page.extract_tables():\n", + " list2 = []\n", + " table = []\n", + " cells = []\n", + " for row in pdf_table:\n", + " if not any(row):\n", + " # 如果一行全为空,则视为一条记录结束\n", + " if any(cells):\n", + " table.append(cells)\n", + " cells = []\n", + " elif all(row):\n", + " # 如果一行全不为空,则本条为新行,上一条结束\n", + " if any(cells):\n", + " table.append(cells)\n", + " cells = []\n", + " table.append(row)\n", + " else:\n", + " if len(cells) == 0:\n", + " cells = row\n", + " else:\n", + " for i in range(len(row)):\n", + " if row[i] is not None and not cells[i]:\n", + " cells[i] = row[i]\n", + " elif row[i] is not None and cells[i]: \n", + " table.append(cells)\n", + " cells = row\n", + " #cells[i] = row[i]\n", + " break\n", + " elif row[i] is None and cells[i]:\n", + " continue\n", + " elif row[i] is None and not cells[i]:\n", + " #cells[i] = row[i]\n", + " continue \n", + " # cells[i] = row[i] if cells[i] is None else cells[i] + row[i]\n", + " \n", + " list2 = [] \n", + " for row in table:\n", + " data =[re.sub('\\s+', '', cell) if cell is not None else None for cell in row]\n", + " print(data)\n", + " \n", + " #print(json.dumps(data_list, indent=2, ensure_ascii=False))\n", + " #with open('Test1.json','a',encoding=\"utf-8\") as file: # json文件的存放位置\n", + " # json.dump(data_list, file, ensure_ascii=False)\n", + " list2.append(data)\n", + " if len(list2) > 0:\n", + " dict1[n] = list2\n", + " #print(dict1)\n", + " #list1.append(list2)\n", + " n +=1\n", + "with open('Test1.json','w',encoding=\"utf-8\") as file:\n", + " json.dump(dict1, file, ensure_ascii=False)\n", + " \n", + "pdf.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "71e1da77-6744-4d86-a4bc-88feeb3023e8", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import pdfplumber\n", + "import json\n", + "import re\n", + "\n", + "file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n", + "pdf = pdfplumber.open(file)\n", + "list1 = []\n", + "dict1 = {}\n", + "n = 1\n", + "for page in pdf.pages:\n", + " for pdf_table in page.extract_tables():\n", + " list2 = []\n", + " table = []\n", + " cells = []\n", + " for row in pdf_table:\n", + " print(row)\n", + " print('******')\n", + " print('--------')\n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "b0764366-12cf-4880-8e0f-9ae18b46d392", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import camelot\n", + "import json\n", + "import re\n", + "\n", + "file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n", + "tables = camelot.read_pdf(file, pages='3',flavor='stream')\n", + "# 2.导出pdf所有的表格为csv文件\n", + "tables.export('foo.json', f='json')\n", + "print('ok!')" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "04dd6436-8374-4096-a2a3-d79f4d93a360", "metadata": {}, "outputs": [], "source": []