This commit is contained in:
512song committed 2021-11-25 17:36:29 +08:00
1 parent 265dbb8d66
commit 46a6e6e350
2 files changed
+159 -30

No files matched your search

+23 -29
View File
@@ -212,30 +212,11 @@
},
{
"cell_type": "code",
"execution_count": 71,
"execution_count": null,
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-24T03:41:43.078279Z",
"iopub.status.busy": "2021-11-24T03:41:43.077279Z",
"iopub.status.idle": "2021-11-24T03:41:43.162298Z",
"shell.execute_reply": "2021-11-24T03:41:43.161298Z",
"shell.execute_reply.started": "2021-11-24T03:41:43.078279Z"
},
"tags": []
},
"outputs": [
{
"ename": "IndexError",
"evalue": "list index out of range",
"output_type": "error",
"traceback": [
"\u001b[1;31m---------------------------------------------------------------------------\u001b[0m",
"\u001b[1;31mIndexError\u001b[0m Traceback (most recent call last)",
"\u001b[1;32m~\\AppData\\Local\\Temp/ipykernel_166520/2691368177.py\u001b[0m in \u001b[0;36m<module>\u001b[1;34m\u001b[0m\n\u001b[0;32m 23\u001b[0m \u001b[0msheet\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34mf'E{i}'\u001b[0m\u001b[1;33m]\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mv\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34m'简介'\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m3\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m':'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m1\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 24\u001b[0m \u001b[0msheet\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34mf'F{i}'\u001b[0m\u001b[1;33m]\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mv\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34m'简介'\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m4\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m':'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m1\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[1;32m---> 25\u001b[1;33m \u001b[0msheet\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34mf'G{i}'\u001b[0m\u001b[1;33m]\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mv\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;34m'简介'\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m5\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m':'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m1\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 26\u001b[0m \u001b[0mi\u001b[0m \u001b[1;33m+=\u001b[0m \u001b[1;36m1\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 27\u001b[0m \u001b[0mwb\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msave\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mfilename\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n",
"\u001b[1;31mIndexError\u001b[0m: list index out of range"
]
}
],
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
@@ -278,15 +259,8 @@
},
{
"cell_type": "code",
"execution_count": 68,
"execution_count": null,
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-24T03:23:57.451845Z",
"iopub.status.busy": "2021-11-24T03:23:57.451845Z",
"iopub.status.idle": "2021-11-24T03:23:57.624934Z",
"shell.execute_reply": "2021-11-24T03:23:57.624934Z",
"shell.execute_reply.started": "2021-11-24T03:23:57.451845Z"
},
"tags": []
},
"outputs": [],
@@ -379,6 +353,26 @@
"print(collist)"
]
},
{
"cell_type": "code",
"execution_count": 78,
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-24T11:15:33.859507Z",
"iopub.status.busy": "2021-11-24T11:15:33.858508Z",
"iopub.status.idle": "2021-11-24T11:15:33.864520Z",
"shell.execute_reply": "2021-11-24T11:15:33.864520Z",
"shell.execute_reply.started": "2021-11-24T11:15:33.859507Z"
},
"tags": []
},
"outputs": [],
"source": [
"list1 = []\n",
"if not list1:\n",
" print('true')"
]
},
{
"cell_type": "code",
"execution_count": null,
+136 -1
View File
@@ -596,10 +596,145 @@
"doc2.save(\"first-and-last-10.pdf\")\n"
]
},
{
"cell_type": "markdown",
"id": "917dfa96-aaea-4c40-8b89-0fe093c65c33",
"metadata": {},
"source": [
"## PDF文件读取表格"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ed609077-c179-4041-8ad2-200f02c4ea83",
"id": "2f9f474d-265c-4beb-bd0e-53372ae9e57e",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pdfplumber\n",
"import json\n",
"import re\n",
"\n",
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
"pdf = pdfplumber.open(file)\n",
"list1 = []\n",
"dict1 = {}\n",
"n = 1\n",
"for page in pdf.pages:\n",
" # print(page.extract_text())\n",
" \n",
" \n",
" for pdf_table in page.extract_tables():\n",
" list2 = []\n",
" table = []\n",
" cells = []\n",
" for row in pdf_table:\n",
" if not any(row):\n",
" # 如果一行全为空,则视为一条记录结束\n",
" if any(cells):\n",
" table.append(cells)\n",
" cells = []\n",
" elif all(row):\n",
" # 如果一行全不为空,则本条为新行,上一条结束\n",
" if any(cells):\n",
" table.append(cells)\n",
" cells = []\n",
" table.append(row)\n",
" else:\n",
" if len(cells) == 0:\n",
" cells = row\n",
" else:\n",
" for i in range(len(row)):\n",
" if row[i] is not None and not cells[i]:\n",
" cells[i] = row[i]\n",
" elif row[i] is not None and cells[i]: \n",
" table.append(cells)\n",
" cells = row\n",
" #cells[i] = row[i]\n",
" break\n",
" elif row[i] is None and cells[i]:\n",
" continue\n",
" elif row[i] is None and not cells[i]:\n",
" #cells[i] = row[i]\n",
" continue \n",
" # cells[i] = row[i] if cells[i] is None else cells[i] + row[i]\n",
" \n",
" list2 = [] \n",
" for row in table:\n",
" data =[re.sub('\\s+', '', cell) if cell is not None else None for cell in row]\n",
" print(data)\n",
" \n",
" #print(json.dumps(data_list, indent=2, ensure_ascii=False))\n",
" #with open('Test1.json','a',encoding=\"utf-8\") as file: # json文件的存放位置\n",
" # json.dump(data_list, file, ensure_ascii=False)\n",
" list2.append(data)\n",
" if len(list2) > 0:\n",
" dict1[n] = list2\n",
" #print(dict1)\n",
" #list1.append(list2)\n",
" n +=1\n",
"with open('Test1.json','w',encoding=\"utf-8\") as file:\n",
" json.dump(dict1, file, ensure_ascii=False)\n",
" \n",
"pdf.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "71e1da77-6744-4d86-a4bc-88feeb3023e8",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pdfplumber\n",
"import json\n",
"import re\n",
"\n",
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
"pdf = pdfplumber.open(file)\n",
"list1 = []\n",
"dict1 = {}\n",
"n = 1\n",
"for page in pdf.pages:\n",
" for pdf_table in page.extract_tables():\n",
" list2 = []\n",
" table = []\n",
" cells = []\n",
" for row in pdf_table:\n",
" print(row)\n",
" print('******')\n",
" print('--------')\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b0764366-12cf-4880-8e0f-9ae18b46d392",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import camelot\n",
"import json\n",
"import re\n",
"\n",
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
"tables = camelot.read_pdf(file, pages='3',flavor='stream')\n",
"# 2.导出pdf所有的表格为csv文件\n",
"tables.export('foo.json', f='json')\n",
"print('ok!')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "04dd6436-8374-4096-a2a3-d79f4d93a360",
"metadata": {},
"outputs": [],
"source": []