This commit is contained in:
512song committed 2021-11-08 22:13:52 +08:00
1 parent b208898522
commit a707437a2b
2 files changed
+217 -87

No files matched your search

+140 -79
View File
@@ -3,7 +3,12 @@
{
"cell_type": "markdown",
"id": "4925cb3a-bea3-4a4d-9590-6b91b62ca5b4",
"metadata": {},
"metadata": {
"jupyter": {
"source_hidden": true
},
"tags": []
},
"source": [
"# 体质检测管理"
]
@@ -419,15 +424,15 @@
"import pdfplumber\n",
"\n",
"# 读取 PDF 文档\n",
"pdf = pdfplumber.open(\"5.pdf\")\n",
"pdf = pdfplumber.open(\"file/211027/114.pdf\")\n",
"\n",
"# 获取页数\n",
"print(\"总页数:\",len(pdf.pages))\n",
"print(\"-----------------------------------------\")\n",
"\n",
"# 读取第 4 页;索引从 1 开始\n",
"page = pdf.pages[1] \n",
"print(\"本页:\",page.page_number + 1)\n",
"n = len(pdf.pages)\n",
"# 读取第 4 页;索引从 0开始\n",
"page = pdf.pages[n-2] \n",
"print(\"本页:\",page.page_number)\n",
"print(\"-----------------------------------------\")\n",
"text = page.extract_text()\n",
"#for s in text:\n",
@@ -577,8 +582,8 @@
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"fl_path = 'file/210924'\n",
"new_path ='file/new'\n",
"fl_path = 'file/split'\n",
"new_path ='file/new_split'\n",
"old = glob.glob(f'{fl_path}/*.pdf')\n",
"old.sort()\n",
"for old_file in old:\n",
@@ -750,27 +755,12 @@
},
{
"cell_type": "code",
"execution_count": 2,
"execution_count": null,
"id": "9e33d1bc-1b71-47f7-928a-16ec435449d6",
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-01T01:09:32.605273Z",
"iopub.status.busy": "2021-11-01T01:09:32.604346Z",
"iopub.status.idle": "2021-11-01T01:09:32.942228Z",
"shell.execute_reply": "2021-11-01T01:09:32.940432Z",
"shell.execute_reply.started": "2021-11-01T01:09:32.605167Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"outputs": [],
"source": [
"import openpyxl\n",
"import json\n",
@@ -813,27 +803,12 @@
},
{
"cell_type": "code",
"execution_count": 13,
"execution_count": null,
"id": "2475a1c3-c2e7-45d6-9685-32e28e606df5",
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-01T02:03:59.201350Z",
"iopub.status.busy": "2021-11-01T02:03:59.200377Z",
"iopub.status.idle": "2021-11-01T02:04:00.932234Z",
"shell.execute_reply": "2021-11-01T02:04:00.930325Z",
"shell.execute_reply.started": "2021-11-01T02:03:59.201245Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"outputs": [],
"source": [
"import openpyxl\n",
"import json\n",
@@ -868,29 +843,12 @@
},
{
"cell_type": "code",
"execution_count": 11,
"execution_count": null,
"id": "61887c54-66be-4158-9721-7a995daf1745",
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-01T02:03:47.177819Z",
"iopub.status.busy": "2021-11-01T02:03:47.176858Z",
"iopub.status.idle": "2021-11-01T02:03:48.719338Z",
"shell.execute_reply": "2021-11-01T02:03:48.717050Z",
"shell.execute_reply.started": "2021-11-01T02:03:47.177713Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"355-479已存在!\n",
"401-379已存在!\n",
"493-765已存在!\n"
]
}
],
"outputs": [],
"source": [
"import openpyxl\n",
"\n",
@@ -919,27 +877,12 @@
},
{
"cell_type": "code",
"execution_count": 12,
"execution_count": null,
"id": "499a41ef-1b1f-4973-b733-8c322ac1ce09",
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-01T02:03:52.339731Z",
"iopub.status.busy": "2021-11-01T02:03:52.339110Z",
"iopub.status.idle": "2021-11-01T02:03:52.366118Z",
"shell.execute_reply": "2021-11-01T02:03:52.364000Z",
"shell.execute_reply.started": "2021-11-01T02:03:52.339665Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"538\n"
]
}
],
"outputs": [],
"source": [
"import json\n",
"filename = '北海炼化公司职工心理健康量表统计.json'\n",
@@ -950,10 +893,128 @@
"print(len(dict1))"
]
},
{
"cell_type": "markdown",
"id": "2d5d65e6-0b14-46a8-8636-ba4755dabafe",
"metadata": {},
"source": [
"## 拆分调查报告PDF页面"
]
},
{
"cell_type": "code",
"execution_count": 1,
"id": "d1b557c1-e34c-4b15-8a5b-fcd5e4b62a64",
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-08T12:32:32.248333Z",
"iopub.status.busy": "2021-11-08T12:32:32.246928Z",
"iopub.status.idle": "2021-11-08T12:33:45.304866Z",
"shell.execute_reply": "2021-11-08T12:33:45.302666Z",
"shell.execute_reply.started": "2021-11-08T12:32:32.248068Z"
},
"tags": []
},
"outputs": [],
"source": [
"import pdfplumber\n",
"import glob\n",
"import os,sys\n",
"import json\n",
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
"\n",
"\n",
"filename = '北海炼化公司职工心理健康量表统计.json'\n",
"list1 = []\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"'''\n",
"for k, v in dict1.items():\n",
" if None in v.values():\n",
" list1.append(k)\n",
"'''\n",
"\n",
"list1 = dict1.keys()\n",
"\n",
"fl_path = 'file/211027'\n",
"new_path ='file/split/'\n",
"old = glob.glob(f'{fl_path}/*.pdf')\n",
"old.sort()\n",
"\n",
"for old_file in old:\n",
" name = os.path.basename(old_file).split('.')[0]\n",
" if name in list1:\n",
" pdfWriter = PdfFileWriter()\n",
" pdf_reader = PdfFileReader(old_file)\n",
" n = pdf_reader.numPages\n",
" pdfWriter.addPage(pdf_reader.getPage(n-2))\n",
" split_file = f'{new_path}{name}.pdf'\n",
" with open(split_file, 'wb') as pdfOutputFile:\n",
" pdfWriter.write(pdfOutputFile)"
]
},
{
"cell_type": "markdown",
"id": "7b6d3d04-2844-4bf4-986c-503d62a39e84",
"metadata": {},
"source": [
"## 拆分PD指定F页"
]
},
{
"cell_type": "code",
"execution_count": 2,
"id": "664c3aca-74e3-4a07-a141-a1dde3ef96e2",
"metadata": {
"execution": {
"iopub.execute_input": "2021-11-08T12:51:27.262359Z",
"iopub.status.busy": "2021-11-08T12:51:27.261120Z",
"iopub.status.idle": "2021-11-08T12:52:40.868282Z",
"shell.execute_reply": "2021-11-08T12:52:40.866331Z",
"shell.execute_reply.started": "2021-11-08T12:51:27.262250Z"
}
},
"outputs": [],
"source": [
"import pdfplumber\n",
"import glob\n",
"import os,sys\n",
"import json\n",
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
"\n",
"\n",
"filename = '北海炼化公司职工心理健康量表统计.json'\n",
"list1 = []\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"'''\n",
"for k, v in dict1.items():\n",
" if None in v.values():\n",
" list1.append(k)\n",
"'''\n",
"\n",
"list1 = dict1.keys()\n",
"\n",
"fl_path = 'file/211027'\n",
"new_path ='file/split/'\n",
"old = glob.glob(f'{fl_path}/*.pdf')\n",
"old.sort()\n",
"\n",
"for old_file in old:\n",
" name = os.path.basename(old_file).split('.')[0]\n",
" pdfWriter = PdfFileWriter()\n",
" pdf_reader = PdfFileReader(old_file)\n",
" n = pdf_reader.numPages\n",
" pdfWriter.addPage(pdf_reader.getPage(n-1))\n",
" split_file = f'{new_path}{name}.pdf'\n",
" with open(split_file, 'wb') as pdfOutputFile:\n",
" pdfWriter.write(pdfOutputFile)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8ddd87ea-75f9-4296-8334-c28e6359fedd",
"id": "4756d26b-5f4e-4a5c-abcd-a7c179ce5f1e",
"metadata": {},
"outputs": [],
"source": []
+77 -8
View File
@@ -29,8 +29,8 @@
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd() + '/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n",
"fi_path = os.getcwd() + '/file/210924'\n",
"fi_xls = os.getcwd() + '/北海炼化名单表(202110).xlsx'\n",
"fi_path = os.getcwd() + '/file/211028'\n",
"old = []\n",
"dict1 = {}\n",
"\n",
@@ -38,14 +38,14 @@
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1): \n",
" m_name = sheet.cell(n,1).value.strip()\n",
" m_name = str(sheet.cell(n,1).value)\n",
" m_depart = sheet.cell(n,5).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,5).value)\n",
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
" \n",
"# 创建部门办公室 \n",
"m_path = os.getcwd()+'/file/210924/new'\n",
"m_path = os.getcwd()+'/file/211028/new'\n",
"for pn in depart:\n",
" if not os.path.exists(m_path + '/' + pn):\n",
" os.mkdir(m_path + '/' + pn)\n",
@@ -402,7 +402,9 @@
"cell_type": "code",
"execution_count": null,
"id": "0b0d621c-f2be-4dcb-a1f0-4b16196d25ee",
"metadata": {},
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from newpdfminer.pdfparser import PDFParser, PDFDocument\n",
@@ -411,7 +413,7 @@
"from newpdfminer.layout import LAParams, LTTextBox\n",
"from newpdfminer.pdfinterp import PDFTextExtractionNotAllowed\n",
"\n",
"path = \"5.pdf\"\n",
"path = \"file/211027/114.pdf\"\n",
"\n",
"# 用文件对象来创建一个pdf文档分析器\n",
"praser = PDFParser(open(path, 'rb'))\n",
@@ -504,9 +506,11 @@
},
{
"cell_type": "code",
"execution_count": 38,
"execution_count": null,
"id": "5ff3d5dd-b8dd-4707-8e29-26ba6cf97070",
"metadata": {},
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
@@ -534,6 +538,71 @@
"\n",
"pdf2image1(file_path, dir_path)"
]
},
{
"cell_type": "markdown",
"id": "7e86692e-43fe-4aaa-a72f-2fe25fc3597e",
"metadata": {},
"source": [
"### Pdf文件拆分"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c4931fc0-7f0b-4f78-8c8f-151dce930a6b",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os, sys\n",
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
"\n",
"file = \"file/211027/114.pdf\"\n",
"pdf_reader = PdfFileReader(file)\n",
"n = pdf_reader.numPages\n",
"print(n)\n",
"pdfWriter = PdfFileWriter()\n",
"pageObj = pdf_reader.getPage(n-2)\n",
"pageObj.scaleBy(1)\n",
"pdfWriter.addPage(pageObj)\n",
"with open('split_114.pdf', 'wb') as pdfOutputFile:\n",
" pdfWriter.write(pdfOutputFile)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2382429a-7b73-42bc-bef6-67a0b69a449d",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"\n",
"file = \"file/211027/114.pdf\"\n",
"doc = fitz.open(file)\n",
"n = doc.page_count\n",
"print(n)\n",
"doc2 = fitz.open()\n",
"page = doc[n - 2]\n",
"#doc2.insert_pdf(doc, to_page = n-2) # first 10 pages\n",
"doc2.insert_pdf(doc, from_page = n-2,to_page = n-2)\n",
"#print(type(page))\n",
"#doc2.insert_pdf(page) # last 10 pages\n",
"doc2.save(\"first-and-last-10.pdf\")\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ed609077-c179-4041-8ad2-200f02c4ea83",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {