20211108
This commit is contained in:
1 parent
b208898522
commit
a707437a2b
2 files changed
+217
-87
No files matched your search
+140
-79
@@ -3,7 +3,12 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "4925cb3a-bea3-4a4d-9590-6b91b62ca5b4",
|
||||
"metadata": {},
|
||||
"metadata": {
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"source": [
|
||||
"# 体质检测管理"
|
||||
]
|
||||
@@ -419,15 +424,15 @@
|
||||
"import pdfplumber\n",
|
||||
"\n",
|
||||
"# 读取 PDF 文档\n",
|
||||
"pdf = pdfplumber.open(\"5.pdf\")\n",
|
||||
"pdf = pdfplumber.open(\"file/211027/114.pdf\")\n",
|
||||
"\n",
|
||||
"# 获取页数\n",
|
||||
"print(\"总页数:\",len(pdf.pages))\n",
|
||||
"print(\"-----------------------------------------\")\n",
|
||||
"\n",
|
||||
"# 读取第 4 页;索引从 1 开始\n",
|
||||
"page = pdf.pages[1] \n",
|
||||
"print(\"本页:\",page.page_number + 1)\n",
|
||||
"n = len(pdf.pages)\n",
|
||||
"# 读取第 4 页;索引从 0开始\n",
|
||||
"page = pdf.pages[n-2] \n",
|
||||
"print(\"本页:\",page.page_number)\n",
|
||||
"print(\"-----------------------------------------\")\n",
|
||||
"text = page.extract_text()\n",
|
||||
"#for s in text:\n",
|
||||
@@ -577,8 +582,8 @@
|
||||
" PDFSyntaxError\n",
|
||||
")\n",
|
||||
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
|
||||
"fl_path = 'file/210924'\n",
|
||||
"new_path ='file/new'\n",
|
||||
"fl_path = 'file/split'\n",
|
||||
"new_path ='file/new_split'\n",
|
||||
"old = glob.glob(f'{fl_path}/*.pdf')\n",
|
||||
"old.sort()\n",
|
||||
"for old_file in old:\n",
|
||||
@@ -750,27 +755,12 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 2,
|
||||
"execution_count": null,
|
||||
"id": "9e33d1bc-1b71-47f7-928a-16ec435449d6",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2021-11-01T01:09:32.605273Z",
|
||||
"iopub.status.busy": "2021-11-01T01:09:32.604346Z",
|
||||
"iopub.status.idle": "2021-11-01T01:09:32.942228Z",
|
||||
"shell.execute_reply": "2021-11-01T01:09:32.940432Z",
|
||||
"shell.execute_reply.started": "2021-11-01T01:09:32.605167Z"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"ok\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import openpyxl\n",
|
||||
"import json\n",
|
||||
@@ -813,27 +803,12 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 13,
|
||||
"execution_count": null,
|
||||
"id": "2475a1c3-c2e7-45d6-9685-32e28e606df5",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2021-11-01T02:03:59.201350Z",
|
||||
"iopub.status.busy": "2021-11-01T02:03:59.200377Z",
|
||||
"iopub.status.idle": "2021-11-01T02:04:00.932234Z",
|
||||
"shell.execute_reply": "2021-11-01T02:04:00.930325Z",
|
||||
"shell.execute_reply.started": "2021-11-01T02:03:59.201245Z"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"ok\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import openpyxl\n",
|
||||
"import json\n",
|
||||
@@ -868,29 +843,12 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 11,
|
||||
"execution_count": null,
|
||||
"id": "61887c54-66be-4158-9721-7a995daf1745",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2021-11-01T02:03:47.177819Z",
|
||||
"iopub.status.busy": "2021-11-01T02:03:47.176858Z",
|
||||
"iopub.status.idle": "2021-11-01T02:03:48.719338Z",
|
||||
"shell.execute_reply": "2021-11-01T02:03:48.717050Z",
|
||||
"shell.execute_reply.started": "2021-11-01T02:03:47.177713Z"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"355-479已存在!\n",
|
||||
"401-379已存在!\n",
|
||||
"493-765已存在!\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import openpyxl\n",
|
||||
"\n",
|
||||
@@ -919,27 +877,12 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 12,
|
||||
"execution_count": null,
|
||||
"id": "499a41ef-1b1f-4973-b733-8c322ac1ce09",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2021-11-01T02:03:52.339731Z",
|
||||
"iopub.status.busy": "2021-11-01T02:03:52.339110Z",
|
||||
"iopub.status.idle": "2021-11-01T02:03:52.366118Z",
|
||||
"shell.execute_reply": "2021-11-01T02:03:52.364000Z",
|
||||
"shell.execute_reply.started": "2021-11-01T02:03:52.339665Z"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"538\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"filename = '北海炼化公司职工心理健康量表统计.json'\n",
|
||||
@@ -950,10 +893,128 @@
|
||||
"print(len(dict1))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "2d5d65e6-0b14-46a8-8636-ba4755dabafe",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 拆分调查报告PDF页面"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 1,
|
||||
"id": "d1b557c1-e34c-4b15-8a5b-fcd5e4b62a64",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2021-11-08T12:32:32.248333Z",
|
||||
"iopub.status.busy": "2021-11-08T12:32:32.246928Z",
|
||||
"iopub.status.idle": "2021-11-08T12:33:45.304866Z",
|
||||
"shell.execute_reply": "2021-11-08T12:33:45.302666Z",
|
||||
"shell.execute_reply.started": "2021-11-08T12:32:32.248068Z"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pdfplumber\n",
|
||||
"import glob\n",
|
||||
"import os,sys\n",
|
||||
"import json\n",
|
||||
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"filename = '北海炼化公司职工心理健康量表统计.json'\n",
|
||||
"list1 = []\n",
|
||||
"with open(filename,'r') as fl:\n",
|
||||
" dict1 = json.load(fl)\n",
|
||||
"'''\n",
|
||||
"for k, v in dict1.items():\n",
|
||||
" if None in v.values():\n",
|
||||
" list1.append(k)\n",
|
||||
"'''\n",
|
||||
"\n",
|
||||
"list1 = dict1.keys()\n",
|
||||
"\n",
|
||||
"fl_path = 'file/211027'\n",
|
||||
"new_path ='file/split/'\n",
|
||||
"old = glob.glob(f'{fl_path}/*.pdf')\n",
|
||||
"old.sort()\n",
|
||||
"\n",
|
||||
"for old_file in old:\n",
|
||||
" name = os.path.basename(old_file).split('.')[0]\n",
|
||||
" if name in list1:\n",
|
||||
" pdfWriter = PdfFileWriter()\n",
|
||||
" pdf_reader = PdfFileReader(old_file)\n",
|
||||
" n = pdf_reader.numPages\n",
|
||||
" pdfWriter.addPage(pdf_reader.getPage(n-2))\n",
|
||||
" split_file = f'{new_path}{name}.pdf'\n",
|
||||
" with open(split_file, 'wb') as pdfOutputFile:\n",
|
||||
" pdfWriter.write(pdfOutputFile)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "7b6d3d04-2844-4bf4-986c-503d62a39e84",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 拆分PD指定F页"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 2,
|
||||
"id": "664c3aca-74e3-4a07-a141-a1dde3ef96e2",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2021-11-08T12:51:27.262359Z",
|
||||
"iopub.status.busy": "2021-11-08T12:51:27.261120Z",
|
||||
"iopub.status.idle": "2021-11-08T12:52:40.868282Z",
|
||||
"shell.execute_reply": "2021-11-08T12:52:40.866331Z",
|
||||
"shell.execute_reply.started": "2021-11-08T12:51:27.262250Z"
|
||||
}
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pdfplumber\n",
|
||||
"import glob\n",
|
||||
"import os,sys\n",
|
||||
"import json\n",
|
||||
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"filename = '北海炼化公司职工心理健康量表统计.json'\n",
|
||||
"list1 = []\n",
|
||||
"with open(filename,'r') as fl:\n",
|
||||
" dict1 = json.load(fl)\n",
|
||||
"'''\n",
|
||||
"for k, v in dict1.items():\n",
|
||||
" if None in v.values():\n",
|
||||
" list1.append(k)\n",
|
||||
"'''\n",
|
||||
"\n",
|
||||
"list1 = dict1.keys()\n",
|
||||
"\n",
|
||||
"fl_path = 'file/211027'\n",
|
||||
"new_path ='file/split/'\n",
|
||||
"old = glob.glob(f'{fl_path}/*.pdf')\n",
|
||||
"old.sort()\n",
|
||||
"\n",
|
||||
"for old_file in old:\n",
|
||||
" name = os.path.basename(old_file).split('.')[0]\n",
|
||||
" pdfWriter = PdfFileWriter()\n",
|
||||
" pdf_reader = PdfFileReader(old_file)\n",
|
||||
" n = pdf_reader.numPages\n",
|
||||
" pdfWriter.addPage(pdf_reader.getPage(n-1))\n",
|
||||
" split_file = f'{new_path}{name}.pdf'\n",
|
||||
" with open(split_file, 'wb') as pdfOutputFile:\n",
|
||||
" pdfWriter.write(pdfOutputFile)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "8ddd87ea-75f9-4296-8334-c28e6359fedd",
|
||||
"id": "4756d26b-5f4e-4a5c-abcd-a7c179ce5f1e",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": []
|
||||
|
||||
Reference in new issue
Block a user