This commit is contained in:
512song committed 2026-01-22 22:08:19 +08:00
1 parent 72a94b8de3
commit 262b081f94
6 files changed
+958 -650

No files matched your search

+221 -71
View File
@@ -64,6 +64,67 @@
"#print(dict1)"
]
},
{
"cell_type": "markdown",
"id": "bb7b6152-0fad-4917-912a-379ed49c06f9",
"metadata": {},
"source": [
"## 批量修改文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d8abca28-6e1f-4f0a-a715-c1697cdf3f6a",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import json\n",
"import glob\n",
"from pathlib import Path\n",
"\n",
"filepath = 'file/line/'\n",
"new = 'file/Line1/'\n",
"files = glob.glob(f'{filepath}*.SAC')\n",
"for fn in files:\n",
" fi_name =Path(fn).stem.split('_')[0]\n",
" n_name = Path(new,fi_name+'.SAC')\n",
" shutil.copyfile(fn,n_name)\n",
" \n",
" "
]
},
{
"cell_type": "markdown",
"id": "e22401f7-9c02-4c4b-a935-e844e56d099b",
"metadata": {},
"source": [
"## 汇总目录下所有文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "62ff063a-5380-4e08-b115-acd967441ca4",
"metadata": {},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"from pathlib import Path\n",
"\n",
"fi_path = Path('file/北海/2022年')\n",
"new_path = 'file/北海/new/2022年'\n",
"pdf_files = list(fi_path.glob('**/*.pdf'))\n",
"fls = list(fi_path.glob('**/*.pdf'))\n",
"for fn in fls:\n",
" fi_name =Path(fn).name\n",
" n_name = Path(new_path,fi_name)\n",
" shutil.copyfile(fn,n_name)"
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -530,6 +591,22 @@
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "35b46e24-cbcd-4975-bd69-b75b886e7347",
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "code",
"execution_count": null,
"id": "da9d533f-3b61-4628-8d1e-470aa0a493a8",
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "markdown",
"id": "d936ca99-269b-46f8-8582-fb02ed2cd3cc",
@@ -933,16 +1010,9 @@
},
{
"cell_type": "code",
"execution_count": 1,
"execution_count": null,
"id": "52db5faa-ed27-4ac1-a9e0-b260d8737eb6",
"metadata": {
"execution": {
"iopub.execute_input": "2022-10-05T01:52:52.306098Z",
"iopub.status.busy": "2022-10-05T01:52:52.305370Z",
"iopub.status.idle": "2022-10-05T01:52:53.053478Z",
"shell.execute_reply": "2022-10-05T01:52:53.052957Z",
"shell.execute_reply.started": "2022-10-05T01:52:52.305888Z"
},
"tags": []
},
"outputs": [],
@@ -960,10 +1030,7 @@
"for fn in fl:\n",
" doc = fitz.open(fn)\n",
" doc2.insert_pdf(doc, from_page = 1,to_page = 1)\n",
"doc2.save(\"检查.pdf\") \n",
" \n",
" \n",
"\n"
"doc2.save(\"检查.pdf\") \n"
]
},
{
@@ -1002,7 +1069,7 @@
" s = os.path.basename(fn).split('.')[0].rjust(5,'0') \n",
" lurl=f'pic/{s}.jpg'\n",
" pm.save(lurl) \n",
"fitz.\n",
"\n",
"fl=glob.glob('./pic/*.jpg')\n",
"fl.sort()\n",
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
@@ -1013,29 +1080,62 @@
"print('ok!')"
]
},
{
"cell_type": "markdown",
"id": "fcc0b4d6-9c41-4d8e-8362-36355f8c9354",
"metadata": {},
"source": [
"#### 多线程提取图片"
]
},
{
"cell_type": "code",
"execution_count": 2,
"id": "6e7c67f9-0bef-4860-9dc5-2c8fee885fd4",
"execution_count": null,
"id": "21e20dcb-1852-4357-a08c-9ee9eacfc96b",
"metadata": {
"execution": {
"iopub.execute_input": "2022-10-05T01:53:22.899805Z",
"iopub.status.busy": "2022-10-05T01:53:22.899116Z",
"iopub.status.idle": "2022-10-05T01:59:52.903430Z",
"shell.execute_reply": "2022-10-05T01:59:52.902838Z",
"shell.execute_reply.started": "2022-10-05T01:53:22.899730Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok!\n"
]
}
],
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"import glob\n",
"from multiprocessing.dummy import Pool\n",
"\n",
"def get_image(fn):\n",
" doc = fitz.open(fn)\n",
" zoom = 100\n",
" page = doc[1]\n",
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0)\n",
" pm = page.get_pixmap(matrix=trans)\n",
" s = os.path.basename(fn).split('.')[0].rjust(5,'0') \n",
" lurl=f'pic1/{s}.jpg'\n",
" pm.save(lurl)\n",
" print(f'{s}已成功生成!') \n",
"fi_path = 'file/2/'\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"pool = Pool(8)\n",
"pool.map(get_image,fl)\n",
"pool.close()\n",
"pool.join()"
]
},
{
"cell_type": "markdown",
"id": "264d3c99-a112-4ade-abda-7785bf8c0d1c",
"metadata": {},
"source": [
"#### 提取页面并分卷压缩"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6e7c67f9-0bef-4860-9dc5-2c8fee885fd4",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os\n",
"import glob\n",
@@ -1166,7 +1266,7 @@
"import json\n",
"import re\n",
"\n",
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
"file = \"file/01730823-邵希强.pdf\"\n",
"pdf = pdfplumber.open(file)\n",
"list1 = []\n",
"dict1 = {}\n",
@@ -1230,36 +1330,6 @@
"pdf.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "71e1da77-6744-4d86-a4bc-88feeb3023e8",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pdfplumber\n",
"import json\n",
"import re\n",
"\n",
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
"pdf = pdfplumber.open(file)\n",
"list1 = []\n",
"dict1 = {}\n",
"n = 1\n",
"for page in pdf.pages:\n",
" for pdf_table in page.extract_tables():\n",
" list2 = []\n",
" table = []\n",
" cells = []\n",
" for row in pdf_table:\n",
" print(row)\n",
" print('******')\n",
" print('--------')\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -1269,15 +1339,14 @@
},
"outputs": [],
"source": [
"import camelot\n",
"import json\n",
"import re\n",
"import pdfplumber\n",
"name = 'file/01730823-邵希强.pdf'\n",
"\n",
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
"tables = camelot.read_pdf(file, pages='3',flavor='stream')\n",
"# 2.导出pdf所有的表格为csv文件\n",
"tables.export('foo.json', f='json')\n",
"print('ok!')"
"pdf = pdfplumber.open(name)\n",
"tables =pdf.pages[1].extract_tables()\n",
"df1 = tables\n",
"for item in df1:\n",
" print(item)"
]
},
{
@@ -1286,6 +1355,87 @@
"id": "04dd6436-8374-4096-a2a3-d79f4d93a360",
"metadata": {},
"outputs": [],
"source": [
"import pdfplumber\n",
"name = 'file/03499391-宋文路.pdf'\n",
"pdf = pdfplumber.open(name)\n",
"text = pdf.pages[1].extract_text()#######页码从0开始计数\n",
"#print(text)\n",
"list1 = text.split('\\n')\n",
"print(list1)\n",
"for item in list1:\n",
" if '测试标准 国民体质测定标准' in item:\n",
" list_min = list1.index(item)\n",
" if '请注意:以上测试项目' in item:\n",
" list_max = list1.index(item)\n",
"print(list_min,list_max)\n",
"for i in range(list_min+1,list_max):\n",
" print(list1[i].split(' ')[0])\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ac034bf9-cba9-4a72-87ba-d77721cc7097",
"metadata": {},
"outputs": [],
"source": [
"import pdfplumber\n",
"name = 'file/03499391-宋文路.pdf'\n",
"pdf = pdfplumber.open(name)\n",
"text = pdf.pages[1].extract_text()#######页码从0开始计数\n",
"#print(text)\n",
"list1 = text.split('\\n')\n",
"print(list1)\n",
"for item in list1:\n",
" if '感谢您完成测试' in item:\n",
" i = list1.index(item)\n",
" ss = ''.join(list1[i:])\n",
" print(ss)"
]
},
{
"cell_type": "markdown",
"id": "26623725-92da-48d9-a476-78a1befd06ef",
"metadata": {},
"source": [
"## 合并PDF文件"
]
},
{
"cell_type": "code",
"execution_count": 6,
"id": "9b37b17e-66d3-4acc-8edb-760ea17123b4",
"metadata": {
"execution": {
"iopub.execute_input": "2026-01-17T11:00:13.781489Z",
"iopub.status.busy": "2026-01-17T11:00:13.780803Z",
"iopub.status.idle": "2026-01-17T11:00:14.805023Z",
"shell.execute_reply": "2026-01-17T11:00:14.804489Z",
"shell.execute_reply.started": "2026-01-17T11:00:13.781426Z"
}
},
"outputs": [],
"source": [
"from pypdf import PdfWriter\n",
"import glob\n",
"\n",
"fi_path = 'file/2025/'\n",
"fls = glob.glob(f'{fi_path}*.pdf')\n",
"fls.sort()\n",
"merger = PdfWriter()\n",
"for pdf in fls:\n",
" merger.append(pdf)\n",
"merger.write(\"file/2025年总账.pdf\")\n",
"merger.close()\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6d5c9dd3-bd09-49b3-8a22-07bb4440aace",
"metadata": {},
"outputs": [],
"source": []
}
],
@@ -1305,7 +1455,7 @@
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
"version": "3.12.3"
}
},
"nbformat": 4,