From a707437a2bef2c50c494390c255c0e692598338f Mon Sep 17 00:00:00 2001 From: 512song Date: Mon, 8 Nov 2021 22:13:52 +0800 Subject: [PATCH] 20211108 --- 体质检测管理.ipynb | 219 +++++++++++++++++++++++++++++---------------- 文件管理1.ipynb | 85 ++++++++++++++++-- 2 files changed, 217 insertions(+), 87 deletions(-) diff --git a/体质检测管理.ipynb b/体质检测管理.ipynb index aa3d8f2..a511ff8 100644 --- a/体质检测管理.ipynb +++ b/体质检测管理.ipynb @@ -3,7 +3,12 @@ { "cell_type": "markdown", "id": "4925cb3a-bea3-4a4d-9590-6b91b62ca5b4", - "metadata": {}, + "metadata": { + "jupyter": { + "source_hidden": true + }, + "tags": [] + }, "source": [ "# 体质检测管理" ] @@ -419,15 +424,15 @@ "import pdfplumber\n", "\n", "# 读取 PDF 文档\n", - "pdf = pdfplumber.open(\"5.pdf\")\n", + "pdf = pdfplumber.open(\"file/211027/114.pdf\")\n", "\n", "# 获取页数\n", "print(\"总页数:\",len(pdf.pages))\n", "print(\"-----------------------------------------\")\n", - "\n", - "# 读取第 4 页;索引从 1 开始\n", - "page = pdf.pages[1] \n", - "print(\"本页:\",page.page_number + 1)\n", + "n = len(pdf.pages)\n", + "# 读取第 4 页;索引从 0开始\n", + "page = pdf.pages[n-2] \n", + "print(\"本页:\",page.page_number)\n", "print(\"-----------------------------------------\")\n", "text = page.extract_text()\n", "#for s in text:\n", @@ -577,8 +582,8 @@ " PDFSyntaxError\n", ")\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", - "fl_path = 'file/210924'\n", - "new_path ='file/new'\n", + "fl_path = 'file/split'\n", + "new_path ='file/new_split'\n", "old = glob.glob(f'{fl_path}/*.pdf')\n", "old.sort()\n", "for old_file in old:\n", @@ -750,27 +755,12 @@ }, { "cell_type": "code", - "execution_count": 2, + "execution_count": null, "id": "9e33d1bc-1b71-47f7-928a-16ec435449d6", "metadata": { - "execution": { - "iopub.execute_input": "2021-11-01T01:09:32.605273Z", - "iopub.status.busy": "2021-11-01T01:09:32.604346Z", - "iopub.status.idle": "2021-11-01T01:09:32.942228Z", - "shell.execute_reply": "2021-11-01T01:09:32.940432Z", - "shell.execute_reply.started": "2021-11-01T01:09:32.605167Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -813,27 +803,12 @@ }, { "cell_type": "code", - "execution_count": 13, + "execution_count": null, "id": "2475a1c3-c2e7-45d6-9685-32e28e606df5", "metadata": { - "execution": { - "iopub.execute_input": "2021-11-01T02:03:59.201350Z", - "iopub.status.busy": "2021-11-01T02:03:59.200377Z", - "iopub.status.idle": "2021-11-01T02:04:00.932234Z", - "shell.execute_reply": "2021-11-01T02:04:00.930325Z", - "shell.execute_reply.started": "2021-11-01T02:03:59.201245Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "ok\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "import json\n", @@ -868,29 +843,12 @@ }, { "cell_type": "code", - "execution_count": 11, + "execution_count": null, "id": "61887c54-66be-4158-9721-7a995daf1745", "metadata": { - "execution": { - "iopub.execute_input": "2021-11-01T02:03:47.177819Z", - "iopub.status.busy": "2021-11-01T02:03:47.176858Z", - "iopub.status.idle": "2021-11-01T02:03:48.719338Z", - "shell.execute_reply": "2021-11-01T02:03:48.717050Z", - "shell.execute_reply.started": "2021-11-01T02:03:47.177713Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "355-479已存在!\n", - "401-379已存在!\n", - "493-765已存在!\n" - ] - } - ], + "outputs": [], "source": [ "import openpyxl\n", "\n", @@ -919,27 +877,12 @@ }, { "cell_type": "code", - "execution_count": 12, + "execution_count": null, "id": "499a41ef-1b1f-4973-b733-8c322ac1ce09", "metadata": { - "execution": { - "iopub.execute_input": "2021-11-01T02:03:52.339731Z", - "iopub.status.busy": "2021-11-01T02:03:52.339110Z", - "iopub.status.idle": "2021-11-01T02:03:52.366118Z", - "shell.execute_reply": "2021-11-01T02:03:52.364000Z", - "shell.execute_reply.started": "2021-11-01T02:03:52.339665Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "538\n" - ] - } - ], + "outputs": [], "source": [ "import json\n", "filename = '北海炼化公司职工心理健康量表统计.json'\n", @@ -950,10 +893,128 @@ "print(len(dict1))" ] }, + { + "cell_type": "markdown", + "id": "2d5d65e6-0b14-46a8-8636-ba4755dabafe", + "metadata": {}, + "source": [ + "## 拆分调查报告PDF页面" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "id": "d1b557c1-e34c-4b15-8a5b-fcd5e4b62a64", + "metadata": { + "execution": { + "iopub.execute_input": "2021-11-08T12:32:32.248333Z", + "iopub.status.busy": "2021-11-08T12:32:32.246928Z", + "iopub.status.idle": "2021-11-08T12:33:45.304866Z", + "shell.execute_reply": "2021-11-08T12:33:45.302666Z", + "shell.execute_reply.started": "2021-11-08T12:32:32.248068Z" + }, + "tags": [] + }, + "outputs": [], + "source": [ + "import pdfplumber\n", + "import glob\n", + "import os,sys\n", + "import json\n", + "from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n", + "\n", + "\n", + "filename = '北海炼化公司职工心理健康量表统计.json'\n", + "list1 = []\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "'''\n", + "for k, v in dict1.items():\n", + " if None in v.values():\n", + " list1.append(k)\n", + "'''\n", + "\n", + "list1 = dict1.keys()\n", + "\n", + "fl_path = 'file/211027'\n", + "new_path ='file/split/'\n", + "old = glob.glob(f'{fl_path}/*.pdf')\n", + "old.sort()\n", + "\n", + "for old_file in old:\n", + " name = os.path.basename(old_file).split('.')[0]\n", + " if name in list1:\n", + " pdfWriter = PdfFileWriter()\n", + " pdf_reader = PdfFileReader(old_file)\n", + " n = pdf_reader.numPages\n", + " pdfWriter.addPage(pdf_reader.getPage(n-2))\n", + " split_file = f'{new_path}{name}.pdf'\n", + " with open(split_file, 'wb') as pdfOutputFile:\n", + " pdfWriter.write(pdfOutputFile)" + ] + }, + { + "cell_type": "markdown", + "id": "7b6d3d04-2844-4bf4-986c-503d62a39e84", + "metadata": {}, + "source": [ + "## 拆分PD指定F页" + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "664c3aca-74e3-4a07-a141-a1dde3ef96e2", + "metadata": { + "execution": { + "iopub.execute_input": "2021-11-08T12:51:27.262359Z", + "iopub.status.busy": "2021-11-08T12:51:27.261120Z", + "iopub.status.idle": "2021-11-08T12:52:40.868282Z", + "shell.execute_reply": "2021-11-08T12:52:40.866331Z", + "shell.execute_reply.started": "2021-11-08T12:51:27.262250Z" + } + }, + "outputs": [], + "source": [ + "import pdfplumber\n", + "import glob\n", + "import os,sys\n", + "import json\n", + "from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n", + "\n", + "\n", + "filename = '北海炼化公司职工心理健康量表统计.json'\n", + "list1 = []\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "'''\n", + "for k, v in dict1.items():\n", + " if None in v.values():\n", + " list1.append(k)\n", + "'''\n", + "\n", + "list1 = dict1.keys()\n", + "\n", + "fl_path = 'file/211027'\n", + "new_path ='file/split/'\n", + "old = glob.glob(f'{fl_path}/*.pdf')\n", + "old.sort()\n", + "\n", + "for old_file in old:\n", + " name = os.path.basename(old_file).split('.')[0]\n", + " pdfWriter = PdfFileWriter()\n", + " pdf_reader = PdfFileReader(old_file)\n", + " n = pdf_reader.numPages\n", + " pdfWriter.addPage(pdf_reader.getPage(n-1))\n", + " split_file = f'{new_path}{name}.pdf'\n", + " with open(split_file, 'wb') as pdfOutputFile:\n", + " pdfWriter.write(pdfOutputFile)" + ] + }, { "cell_type": "code", "execution_count": null, - "id": "8ddd87ea-75f9-4296-8334-c28e6359fedd", + "id": "4756d26b-5f4e-4a5c-abcd-a7c179ce5f1e", "metadata": {}, "outputs": [], "source": [] diff --git a/文件管理1.ipynb b/文件管理1.ipynb index 5175762..c14b4c6 100755 --- a/文件管理1.ipynb +++ b/文件管理1.ipynb @@ -29,8 +29,8 @@ "import openpyxl\n", "import math\n", "\n", - "fi_xls = os.getcwd() + '/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n", - "fi_path = os.getcwd() + '/file/210924'\n", + "fi_xls = os.getcwd() + '/北海炼化名单表(202110).xlsx'\n", + "fi_path = os.getcwd() + '/file/211028'\n", "old = []\n", "dict1 = {}\n", "\n", @@ -38,14 +38,14 @@ "sheet = wb.active\n", "depart = []\n", "for n in range(2,sheet.max_row+1): \n", - " m_name = sheet.cell(n,1).value.strip()\n", + " m_name = str(sheet.cell(n,1).value)\n", " m_depart = sheet.cell(n,5).value\n", " if m_depart not in depart:\n", " depart.append(sheet.cell(n,5).value)\n", " dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n", " \n", "# 创建部门办公室 \n", - "m_path = os.getcwd()+'/file/210924/new'\n", + "m_path = os.getcwd()+'/file/211028/new'\n", "for pn in depart:\n", " if not os.path.exists(m_path + '/' + pn):\n", " os.mkdir(m_path + '/' + pn)\n", @@ -402,7 +402,9 @@ "cell_type": "code", "execution_count": null, "id": "0b0d621c-f2be-4dcb-a1f0-4b16196d25ee", - "metadata": {}, + "metadata": { + "tags": [] + }, "outputs": [], "source": [ "from newpdfminer.pdfparser import PDFParser, PDFDocument\n", @@ -411,7 +413,7 @@ "from newpdfminer.layout import LAParams, LTTextBox\n", "from newpdfminer.pdfinterp import PDFTextExtractionNotAllowed\n", "\n", - "path = \"5.pdf\"\n", + "path = \"file/211027/114.pdf\"\n", "\n", "# 用文件对象来创建一个pdf文档分析器\n", "praser = PDFParser(open(path, 'rb'))\n", @@ -504,9 +506,11 @@ }, { "cell_type": "code", - "execution_count": 38, + "execution_count": null, "id": "5ff3d5dd-b8dd-4707-8e29-26ba6cf97070", - "metadata": {}, + "metadata": { + "tags": [] + }, "outputs": [], "source": [ "import fitz\n", @@ -534,6 +538,71 @@ "\n", "pdf2image1(file_path, dir_path)" ] + }, + { + "cell_type": "markdown", + "id": "7e86692e-43fe-4aaa-a72f-2fe25fc3597e", + "metadata": {}, + "source": [ + "### Pdf文件拆分" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "c4931fc0-7f0b-4f78-8c8f-151dce930a6b", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import os, sys\n", + "from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n", + "\n", + "file = \"file/211027/114.pdf\"\n", + "pdf_reader = PdfFileReader(file)\n", + "n = pdf_reader.numPages\n", + "print(n)\n", + "pdfWriter = PdfFileWriter()\n", + "pageObj = pdf_reader.getPage(n-2)\n", + "pageObj.scaleBy(1)\n", + "pdfWriter.addPage(pageObj)\n", + "with open('split_114.pdf', 'wb') as pdfOutputFile:\n", + " pdfWriter.write(pdfOutputFile)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "2382429a-7b73-42bc-bef6-67a0b69a449d", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import fitz\n", + "import os\n", + "\n", + "file = \"file/211027/114.pdf\"\n", + "doc = fitz.open(file)\n", + "n = doc.page_count\n", + "print(n)\n", + "doc2 = fitz.open()\n", + "page = doc[n - 2]\n", + "#doc2.insert_pdf(doc, to_page = n-2) # first 10 pages\n", + "doc2.insert_pdf(doc, from_page = n-2,to_page = n-2)\n", + "#print(type(page))\n", + "#doc2.insert_pdf(page) # last 10 pages\n", + "doc2.save(\"first-and-last-10.pdf\")\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "ed609077-c179-4041-8ad2-200f02c4ea83", + "metadata": {}, + "outputs": [], + "source": [] } ], "metadata": {