20211125
This commit is contained in:
1 parent
265dbb8d66
commit
46a6e6e350
2 files changed
+159
-30
No files matched your search
+136
-1
@@ -596,10 +596,145 @@
|
||||
"doc2.save(\"first-and-last-10.pdf\")\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "917dfa96-aaea-4c40-8b89-0fe093c65c33",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## PDF文件读取表格"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "ed609077-c179-4041-8ad2-200f02c4ea83",
|
||||
"id": "2f9f474d-265c-4beb-bd0e-53372ae9e57e",
|
||||
"metadata": {
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pdfplumber\n",
|
||||
"import json\n",
|
||||
"import re\n",
|
||||
"\n",
|
||||
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
|
||||
"pdf = pdfplumber.open(file)\n",
|
||||
"list1 = []\n",
|
||||
"dict1 = {}\n",
|
||||
"n = 1\n",
|
||||
"for page in pdf.pages:\n",
|
||||
" # print(page.extract_text())\n",
|
||||
" \n",
|
||||
" \n",
|
||||
" for pdf_table in page.extract_tables():\n",
|
||||
" list2 = []\n",
|
||||
" table = []\n",
|
||||
" cells = []\n",
|
||||
" for row in pdf_table:\n",
|
||||
" if not any(row):\n",
|
||||
" # 如果一行全为空,则视为一条记录结束\n",
|
||||
" if any(cells):\n",
|
||||
" table.append(cells)\n",
|
||||
" cells = []\n",
|
||||
" elif all(row):\n",
|
||||
" # 如果一行全不为空,则本条为新行,上一条结束\n",
|
||||
" if any(cells):\n",
|
||||
" table.append(cells)\n",
|
||||
" cells = []\n",
|
||||
" table.append(row)\n",
|
||||
" else:\n",
|
||||
" if len(cells) == 0:\n",
|
||||
" cells = row\n",
|
||||
" else:\n",
|
||||
" for i in range(len(row)):\n",
|
||||
" if row[i] is not None and not cells[i]:\n",
|
||||
" cells[i] = row[i]\n",
|
||||
" elif row[i] is not None and cells[i]: \n",
|
||||
" table.append(cells)\n",
|
||||
" cells = row\n",
|
||||
" #cells[i] = row[i]\n",
|
||||
" break\n",
|
||||
" elif row[i] is None and cells[i]:\n",
|
||||
" continue\n",
|
||||
" elif row[i] is None and not cells[i]:\n",
|
||||
" #cells[i] = row[i]\n",
|
||||
" continue \n",
|
||||
" # cells[i] = row[i] if cells[i] is None else cells[i] + row[i]\n",
|
||||
" \n",
|
||||
" list2 = [] \n",
|
||||
" for row in table:\n",
|
||||
" data =[re.sub('\\s+', '', cell) if cell is not None else None for cell in row]\n",
|
||||
" print(data)\n",
|
||||
" \n",
|
||||
" #print(json.dumps(data_list, indent=2, ensure_ascii=False))\n",
|
||||
" #with open('Test1.json','a',encoding=\"utf-8\") as file: # json文件的存放位置\n",
|
||||
" # json.dump(data_list, file, ensure_ascii=False)\n",
|
||||
" list2.append(data)\n",
|
||||
" if len(list2) > 0:\n",
|
||||
" dict1[n] = list2\n",
|
||||
" #print(dict1)\n",
|
||||
" #list1.append(list2)\n",
|
||||
" n +=1\n",
|
||||
"with open('Test1.json','w',encoding=\"utf-8\") as file:\n",
|
||||
" json.dump(dict1, file, ensure_ascii=False)\n",
|
||||
" \n",
|
||||
"pdf.close()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "71e1da77-6744-4d86-a4bc-88feeb3023e8",
|
||||
"metadata": {
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pdfplumber\n",
|
||||
"import json\n",
|
||||
"import re\n",
|
||||
"\n",
|
||||
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
|
||||
"pdf = pdfplumber.open(file)\n",
|
||||
"list1 = []\n",
|
||||
"dict1 = {}\n",
|
||||
"n = 1\n",
|
||||
"for page in pdf.pages:\n",
|
||||
" for pdf_table in page.extract_tables():\n",
|
||||
" list2 = []\n",
|
||||
" table = []\n",
|
||||
" cells = []\n",
|
||||
" for row in pdf_table:\n",
|
||||
" print(row)\n",
|
||||
" print('******')\n",
|
||||
" print('--------')\n",
|
||||
" "
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "b0764366-12cf-4880-8e0f-9ae18b46d392",
|
||||
"metadata": {
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import camelot\n",
|
||||
"import json\n",
|
||||
"import re\n",
|
||||
"\n",
|
||||
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
|
||||
"tables = camelot.read_pdf(file, pages='3',flavor='stream')\n",
|
||||
"# 2.导出pdf所有的表格为csv文件\n",
|
||||
"tables.export('foo.json', f='json')\n",
|
||||
"print('ok!')"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "04dd6436-8374-4096-a2a3-d79f4d93a360",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": []
|
||||
|
||||
Reference in new issue
Block a user