Files
jupyter/文件管理.ipynb
T
2026-01-22 22:08:19 +08:00

1464 lines
42 KiB
Plaintext
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"cells": [
{
"cell_type": "markdown",
"id": "333a5431-a697-4330-a587-455d6d18ebb0",
"metadata": {},
"source": [
"# 文件管理"
]
},
{
"cell_type": "markdown",
"id": "bcd48bb6-4756-4748-ac8b-854eeee92a3b",
"metadata": {},
"source": [
"## 数字文件名转换为文本文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "54b320e8-c3c1-4214-9a9e-db13c2558604",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd() + '/data/高新区.xlsx'\n",
"fi_path = os.getcwd() + '/file/220720'\n",
"old = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1): \n",
" m_name = str(sheet.cell(n,1).value)\n",
" m_depart = sheet.cell(n,5).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,5).value)\n",
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
" \n",
"# 创建部门办公室 \n",
"m_path = os.getcwd()+'/file/220427/new'\n",
"for pn in depart:\n",
" if not os.path.exists(m_path + '/' + pn):\n",
" os.mkdir(m_path + '/' + pn)\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn) \n",
"old.sort()\n",
"for n in old: \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(dict1)"
]
},
{
"cell_type": "markdown",
"id": "bb7b6152-0fad-4917-912a-379ed49c06f9",
"metadata": {},
"source": [
"## 批量修改文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d8abca28-6e1f-4f0a-a715-c1697cdf3f6a",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import json\n",
"import glob\n",
"from pathlib import Path\n",
"\n",
"filepath = 'file/line/'\n",
"new = 'file/Line1/'\n",
"files = glob.glob(f'{filepath}*.SAC')\n",
"for fn in files:\n",
" fi_name =Path(fn).stem.split('_')[0]\n",
" n_name = Path(new,fi_name+'.SAC')\n",
" shutil.copyfile(fn,n_name)\n",
" \n",
" "
]
},
{
"cell_type": "markdown",
"id": "e22401f7-9c02-4c4b-a935-e844e56d099b",
"metadata": {},
"source": [
"## 汇总目录下所有文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "62ff063a-5380-4e08-b115-acd967441ca4",
"metadata": {},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"from pathlib import Path\n",
"\n",
"fi_path = Path('file/北海/2022年')\n",
"new_path = 'file/北海/new/2022年'\n",
"pdf_files = list(fi_path.glob('**/*.pdf'))\n",
"fls = list(fi_path.glob('**/*.pdf'))\n",
"for fn in fls:\n",
" fi_name =Path(fn).name\n",
" n_name = Path(new_path,fi_name)\n",
" shutil.copyfile(fn,n_name)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "4a43668e-47ba-491f-a29c-e289885badb0",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd() + '/data/高新区.xlsx'\n",
"fi_path = os.getcwd() + '/file/220720'\n",
"old = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1): \n",
" m_name = str(sheet.cell(n,1).value)\n",
" m_depart = sheet.cell(n,5).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,5).value)\n",
" dict1[sheet.cell(n,4).value] = [sheet.cell(n,6).value,sheet.cell(n,3).value]\n",
"#print(dict1)\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn) \n",
"old.sort()\n",
"\n",
"\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"i = 0\n",
" \n",
"for item in old:\n",
" m_bh = item\n",
" m_name = dict1[item][0]\n",
" m_dep = dict1[item][1]\n",
" i += 1\n",
" sheet[f'A{i}'] = m_bh\n",
" sheet[f'B{i}'] = m_name\n",
" sheet[f'C{i}'] = m_dep\n",
" \n",
" \n",
"wb.save('data/高新区报告.xlsx') "
]
},
{
"cell_type": "markdown",
"id": "72866db8-89be-4498-b854-a3aefe4c897a",
"metadata": {},
"source": [
"## 低碳院人员信息转换"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "433c3f52-f4ae-4544-8d06-3636feae71ba",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd() + '/data/北京低碳监测.xlsx'\n",
"fi_path = os.getcwd() + '/file/220724'\n",
"old = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row): \n",
" m_name = sheet.cell(n,6).value\n",
" m_depart = sheet.cell(n,3).value\n",
" if sheet.cell(n,4).value is not None:\n",
" dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n",
"\n",
"m_path = os.getcwd()+'/file/220724/new'\n",
"\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn) \n",
"old.sort()\n",
"for n in old: \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(dict1)"
]
},
{
"cell_type": "markdown",
"id": "cda8fbed-c0f2-4bfc-a23f-6b412bbcd0ba",
"metadata": {},
"source": [
"## 低碳院人员信息核验"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "a34c9f6b-4ef3-4678-89bb-8b692e183968",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"import json\n",
"\n",
"fi_xls = os.getcwd() + '/data/北京低碳监测.xlsx'\n",
"fi_path = os.getcwd() + '/file/220724'\n",
"old = []\n",
"dict1 = {}\n",
"dict2 = {}\n",
"dict3 = {}\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1): \n",
" m_name = sheet.cell(n,6).value\n",
" m_depart = sheet.cell(n,3).value\n",
" if sheet.cell(n,4).value is not None:\n",
" dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n",
"\n",
"fi_xls = os.getcwd() + '/data/1658883945637.xlsx'\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"for n in range(2,sheet.max_row+1): \n",
" m_dm = sheet.cell(n,3).value\n",
" m_depart = sheet.cell(n,2).value\n",
" m_name = sheet.cell(n,4).value\n",
" if sheet.cell(n,3).value is not None:\n",
" dict2[sheet.cell(n,3).value] = [m_name,m_depart,m_dm]\n",
"#print(dict2)\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn) \n",
"old.sort()\n",
"for n in old:\n",
" m_name = dict1[n][0]\n",
" m_depart = dict1[n][1]\n",
" for item in dict2.values():\n",
" if m_name in item and m_depart in item:\n",
" dict3[n] = [m_name,m_depart,item[2]]\n",
" break\n",
"for k in old:\n",
" if k not in dict3.keys():\n",
" print(k)\n",
"with open('data/低碳院报告.json','w') as fl2:\n",
" json.dump(dict3,fl2,ensure_ascii=False) "
]
},
{
"cell_type": "markdown",
"id": "3244663e-890c-48de-ae6a-1f666cb1bb76",
"metadata": {},
"source": [
"## 低碳院人员附件核验"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8e20ead8-754a-4535-95ea-b103264e3783",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import os\n",
"\n",
"fl_name = 'data/低碳院报告.json'\n",
"\n",
"with open(fl_name,'r') as fl:\n",
" m_xx = json.load(fl)\n",
"for k, v in m_xx.items():\n",
" m_bh = str(k).rjust(5,\"0\")\n",
" fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n",
" if os.path.exists(fl_name):\n",
" print(fl_name)\n",
" else:\n",
" print(f'{v[0]}不存在')"
]
},
{
"cell_type": "markdown",
"id": "a7976e2e-23fa-4464-89f5-0e96f721bc95",
"metadata": {},
"source": [
"## 低碳院人员导出"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d77c3ff5-91fb-4449-80fb-57bc467c27bd",
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
" \n",
"\n",
"fl_name = 'data/低碳院报告.json'\n",
"\n",
"with open(fl_name,'r') as fl:\n",
" m_xx = json.load(fl)\n",
" \n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"i = 0\n",
" \n",
"for k, v in m_xx.items():\n",
" m_bh = str(k).rjust(5,\"0\")\n",
" m_name = v[0]\n",
" fl = f'{m_bh}-{v[0]}.pdf'\n",
" m_rec = v[2]+'@ceic.com'\n",
" i += 1\n",
" sheet[f'A{i}'] = i\n",
" sheet[f'B{i}'] = m_bh\n",
" sheet[f'C{i}'] = m_name\n",
" sheet[f'D{i}'] = m_rec \n",
" sheet[f'E{i}'] = fl\n",
" \n",
"wb.save('data/低碳院邮件.xlsx') \n",
" "
]
},
{
"cell_type": "markdown",
"id": "bbf0b425-c0ce-4608-ae52-85a78ddcd6c0",
"metadata": {},
"source": [
"## 批量更换扩展名"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2c2bd1d1-4d84-4639-ba20-c97c218ae275",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import glob\n",
"import os,shutil\n",
"\n",
"filepath = 'file/yimo/'\n",
"old = 'SGF'\n",
"new = 'sgf'\n",
"files = glob.glob(f'{filepath}*.{old}')\n",
"for f in files:\n",
" new_file =os.path.splitext(f)[0]+'.'+new\n",
" os.rename(f,new_file)"
]
},
{
"cell_type": "markdown",
"id": "69712868-6786-430b-8591-f6828d781588",
"metadata": {},
"source": [
"## PDF文件压缩"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8492dc14-739e-4c78-a302-f4b57ae571dd",
"metadata": {},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"\n",
"\n",
"def covert2pic(zoom):\n",
" if os.path.exists('.pdf'): # 临时文件,需为空\n",
" os.removedirs('.pdf')\n",
" os.mkdir('.pdf')\n",
" for pg in range(totaling):\n",
" page = doc[pg]\n",
" zoom = int(zoom) #值越大,分辨率越高,文件越清晰\n",
" rotate = int(0)\n",
" print(page)\n",
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0).preRotate(rotate)\n",
" pm = page.getPixmap(matrix=trans, alpha=False)\n",
" \n",
" lurl='.pdf/%s.jpg' % str(pg+1)\n",
" pm.writePNG(lurl)\n",
" doc.close()\n",
"\n",
"def pic2pdf(obj):\n",
" doc = fitz.open()\n",
" for pg in range(totaling):\n",
" img = '.pdf/%s.jpg' % str(pg+1)\n",
" imgdoc = fitz.open(img) # 打开图片\n",
" pdfbytes = imgdoc.convertToPDF() # 使用图片创建单页的 PDF\n",
" os.remove(img) \n",
" imgpdf = fitz.open(\"pdf\", pdfbytes)\n",
" doc.insertPDF(imgpdf) # 将当前页插入文档\n",
" if os.path.exists(obj): # 若文件存在先删除\n",
" os.remove(obj)\n",
" doc.save(obj) # 保存pdf文件\n",
" doc.close()\n",
"\n",
"\n",
"def pdfz(sor, obj, zoom): \n",
" covert2pic(zoom)\n",
" pic2pdf(obj)\n",
" \n",
"\n",
"\n",
"sor = \"5.pdf\" # 需要压缩的PDF文件\n",
"obj = \"new-\" + sor\n",
"doc = fitz.open(sor) \n",
"totaling = doc.pageCount\n",
"\n",
"zoom = 150 # 清晰度调节,缩放比率\n",
"pdfz(sor, obj, zoom)\n",
"os.removedirs('.pdf')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "88d84849-5b98-48cb-86b1-e96757a8615f",
"metadata": {},
"outputs": [],
"source": [
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path('检查.pdf', dpi=100,fmt='jpg', output_folder='./pic')\n",
"print(path)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "058a40ba-4948-4675-8400-e4c762e836ee",
"metadata": {},
"outputs": [],
"source": [
"import img2pdf \n",
"import os,sys\n",
"import glob\n",
"\n",
"fl=glob.glob('./pic/*.jpg')\n",
"fl.sort()\n",
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
"with open(\"name.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(fl,layout_fun=layout_fun))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e997ad75-b2d9-43bf-a1de-8833cfe0cf6a",
"metadata": {},
"outputs": [],
"source": [
"import glob\n",
"import fitz # 导入本模块需安装pymupdf库\n",
"import os,sys\n",
"\n",
"def pic2pdf_1(img_path, pdf_path, pdf_name):\n",
" doc = fitz.open()\n",
" fl=os.listdir(img_path)\n",
" fl.sort()\n",
" width, height = fitz.PaperSize(\"a4\")\n",
" #width, height = fitz.paper_size(\"a4\")\n",
" for img in fl:\n",
" fn = img_path+'/'+img\n",
" if os.path.isfile(fn):\n",
" imgdoc = fitz.open(img_path+'/'+img)\n",
" pdfbytes = imgdoc.convertToPDF()\n",
" imgpdf = fitz.open(\"pdf\", pdfbytes,width = width, height = height)\n",
" doc.insertPDF(imgpdf)\n",
" doc.save(pdf_path +'/'+ pdf_name)\n",
" doc.close()\n",
"\n",
"img_path = os.getcwd() +'/pic'\n",
"pdf_path = os.getcwd()\n",
"pic2pdf_1(img_path=img_path, pdf_path=pdf_path, pdf_name='1.pdf')"
]
},
{
"cell_type": "markdown",
"id": "1f38e4b6-ef20-4fef-801c-ff47951d2ad6",
"metadata": {},
"source": [
"## 图片文件扫描识别"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8c990b6a-7f8e-4527-a150-65f60bd131ac",
"metadata": {},
"outputs": [],
"source": [
"#将指定目录下图片文件进行文字识别\n",
"import os,sys\n",
"from aip import AipOcr\n",
"import glob\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"\n",
"fi_path = os.getcwd()+'/pic'\n",
"fl = glob.glob(f'{fi_path}/*.jpg')\n",
"fl.sort()\n",
"for fl1 in fl:\n",
" file_name = fl1\n",
" image = get_file_content(file_name)\n",
" result= client.basicGeneral(image, options)\n",
" if 'words_result' in result:\n",
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
" print('\\n')\n",
" \n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "35b46e24-cbcd-4975-bd69-b75b886e7347",
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "code",
"execution_count": null,
"id": "da9d533f-3b61-4628-8d1e-470aa0a493a8",
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "markdown",
"id": "d936ca99-269b-46f8-8582-fb02ed2cd3cc",
"metadata": {},
"source": [
"## word文档读取"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d8043425-be9e-4e9e-bbfc-3a2d8885a860",
"metadata": {},
"outputs": [],
"source": [
"import docx\n",
"import json\n",
"Doc = docx.Document(r\"症状对症处方选穴.docx\")\n",
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
"testList = []\n",
"text1 = []\n",
"dict1 = {}\n",
"for text in Doc.paragraphs:\n",
" testList.append(text)\n",
"i = 0\n",
"for pg in testList:\n",
" \n",
" if len(pg.text) >0:\n",
" s = ''.join(pg.text.split()) \n",
" if s[0:1] == '第':\n",
" i += 1\n",
" n = 0\n",
" \n",
" dict1.setdefault(i,{})\n",
" dict1[i]['title'] = pg.text.split()[1]\n",
" dict1[i]['sub'] = {}\n",
" \n",
" #dict1[i]['title'].setdefault(i,{})\n",
" elif s[0:1] == '(':\n",
" sub = s.split(')')[1]\n",
" dict2 = {}\n",
" dict1[i]['sub'].setdefault(sub,[])\n",
" else:\n",
" dict1[i]['sub'][sub].append(s)\n",
"filename = '症状对症处方选穴.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl,ensure_ascii=False) \n",
"print('ok') "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c440b599-9c80-4072-8543-d9c381214f1d",
"metadata": {},
"outputs": [],
"source": [
"import docx\n",
"import json\n",
"Doc = docx.Document(r\"常见疾病辨病处方选穴.docx\")\n",
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
"testList = []\n",
"text1 = []\n",
"dict1 = {}\n",
"for text in Doc.paragraphs:\n",
" testList.append(text)\n",
"i = 0\n",
"for pg in testList:\n",
" \n",
" if len(pg.text) >0:\n",
" s = ''.join(pg.text.split()) \n",
" if s[0:1] == '第':\n",
" i += 1\n",
" n = 0\n",
" \n",
" dict1.setdefault(i,{})\n",
" dict1[i]['title'] = pg.text.split('节')[1]\n",
" dict1[i]['sub'] = {}\n",
" \n",
" #dict1[i]['title'].setdefault(i,{})\n",
" elif s[0:1] == '(':\n",
" sub = s.split(')')[1]\n",
" dict2 = {}\n",
" dict1[i]['sub'].setdefault(sub,[])\n",
" else:\n",
" dict1[i]['sub'][sub].append(s)\n",
"filename = '常见疾病辨病处方选穴.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl,ensure_ascii=False)\n",
"print('ok') "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b5e500b5-14d4-4417-8b5e-46df71a6b6b8",
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import re\n",
"filename = '症状对症处方选穴.json'\n",
"pattern = r'[\\d\\.]'\n",
"mo =r'[\\u4e00-\\u9fa5]+'\n",
"dict2 = {}\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl) \n",
"for k,v in dict1.items():\n",
" m_title = v['title']\n",
" for k1,v1 in v['sub'].items():\n",
" sub_title = k1 \n",
" s = ''\n",
" for mx in v1: \n",
" s = s + mx\n",
" list1 = re.split(pattern, s)\n",
" list2 = []\n",
" for ss in list1:\n",
" if ss != '':\n",
" list3 = re.findall(mo,ss)\n",
" list2.append(list3)\n",
" \n",
" #print(m_title,sub_title,list2)\n",
" dict2.setdefault(m_title,{})\n",
" dict2[m_title][sub_title] = list2\n",
"filename = 'new_症状对症处方选穴.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict2, fl,ensure_ascii=False) \n",
"print('ok') \n",
" "
]
},
{
"cell_type": "markdown",
"id": "24d3fdb2-65ac-4a2b-b65a-f64a01a3fdf5",
"metadata": {},
"source": [
"## word使用模板生成文档"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "9957a90c-7735-4224-b5b5-4a4b1b7688c9",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import openpyxl\n",
"from docxtpl import DocxTemplate\n",
"\n",
"wb = openpyxl.load_workbook('file/火车餐厅人员信息表1.xlsx')\n",
"sheet = wb.active\n",
"# sheets = wb.sheetnames\n",
"list1 = []\n",
"context = {}\n",
"for n in range(2, sheet.max_row+1):\n",
" dict1 = {}\n",
" dict1['xh'] = sheet.cell(n,1).value\n",
" dict1['xm'] = sheet.cell(n,2).value\n",
" dict1['sfzhm'] = sheet.cell(n,3).value\n",
" dict1['dh'] = sheet.cell(n,4).value\n",
" dict1['cx'] = sheet.cell(n,5).value\n",
" dict1['xzdz'] = sheet.cell(n,6).value\n",
" list1.append(dict1)\n",
"context['info'] = list1\n",
"tpl = DocxTemplate(\"file/火车餐厅人员信息表.docx\")\n",
"tpl.render(context)\n",
"tpl.save('file/test1.docx')"
]
},
{
"cell_type": "markdown",
"id": "8a555d4f-8b2c-4f2c-af7e-fe8dfbd33f5f",
"metadata": {},
"source": [
"## PDF文件读取"
]
},
{
"cell_type": "markdown",
"id": "308511e7-0f2d-40bc-a151-68f26ee635a2",
"metadata": {},
"source": [
"### pdfminer读取PDF文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0b0d621c-f2be-4dcb-a1f0-4b16196d25ee",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from newpdfminer.pdfparser import PDFParser, PDFDocument\n",
"from newpdfminer.pdfinterp import PDFResourceManager, PDFPageInterpreter\n",
"from newpdfminer.converter import PDFPageAggregator\n",
"from newpdfminer.layout import LAParams, LTTextBox\n",
"from newpdfminer.pdfinterp import PDFTextExtractionNotAllowed\n",
"\n",
"path = \"file/211027/114.pdf\"\n",
"\n",
"# 用文件对象来创建一个pdf文档分析器\n",
"praser = PDFParser(open(path, 'rb'))\n",
"# 创建一个PDF文档\n",
"doc = PDFDocument()\n",
"# 连接分析器 与文档对象\n",
"praser.set_document(doc)\n",
"doc.set_parser(praser)\n",
"\n",
"# 提供初始化密码\n",
"# 如果没有密码 就创建一个空的字符串\n",
"doc.initialize()\n",
"\n",
"# 检测文档是否提供txt转换,不提供就忽略\n",
"if not doc.is_extractable:\n",
" raise PDFTextExtractionNotAllowed\n",
"else:\n",
" # 创建PDf 资源管理器 来管理共享资源\n",
" rsrcmgr = PDFResourceManager()\n",
" # 创建一个PDF设备对象\n",
" laparams = LAParams()\n",
" device = PDFPageAggregator(rsrcmgr, laparams=laparams)\n",
" # 创建一个PDF解释器对象\n",
" interpreter = PDFPageInterpreter(rsrcmgr, device)\n",
"\n",
" # 循环遍历列表,每次处理一个page的内容\n",
" for page in doc.get_pages():\n",
" interpreter.process_page(page) \n",
" # 接受该页面的LTPage对象\n",
" layout = device.get_result()\n",
" # 这里layout是一个LTPage对象,里面存放着这个 page 解析出的各种对象\n",
" # 包括 LTTextBox, LTFigure, LTImage, LTTextBoxHorizontal 等 \n",
" for x in layout:\n",
" if isinstance(x, LTTextBox):\n",
" print(x.get_text().strip())"
]
},
{
"cell_type": "markdown",
"id": "8fff7f1e-b52c-4666-8a2c-e9cde85a8e01",
"metadata": {},
"source": [
"### pdfplumber读取PDF文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c8685b70-55b3-439e-852d-c4fbb67b6326",
"metadata": {},
"outputs": [],
"source": [
"import pdfplumber\n",
"from PIL import Image\n",
"\n",
"# 读取 PDF 文档\n",
"pdf = pdfplumber.open(\"5.pdf\")\n",
"\n",
"# 获取页数\n",
"print(\"总页数:\",len(pdf.pages))\n",
"print(\"-----------------------------------------\")\n",
"\n",
"# 读取第 4 页;索引从 1 开始\n",
"page = pdf.pages[0] \n",
"print(\"本页:\",page.page_number + 1)\n",
"print(\"-----------------------------------------\")\n",
"# 单页文件存储为图片\n",
"#im = page.to_image()\n",
"#im.save('./5_1.png', format='PNG')\n",
"# 单页文件中原始图片读取,不能读取图表文件\n",
"#imgs = page.images\n",
"#for i, img in enumerate(imgs):\n",
"# size = img['width'], img['height']\n",
"# data = img['stream'].get_data()\n",
"# out_path = f'003pdf_images_{i}.png'\n",
"# with open(out_path, 'wb') as fimg_out:\n",
"# fimg_out.write(data)\n",
"# 导出第 4 页文本\n",
"text = page.extract_text()\n",
"print(text)\n"
]
},
{
"cell_type": "markdown",
"id": "59e8775b-e2d6-4963-9376-4a944108ec58",
"metadata": {},
"source": [
"### fitz1.8读取PDF文件中的图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5ff3d5dd-b8dd-4707-8e29-26ba6cf97070",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"import re\n",
"import os\n",
"\n",
"file_path = '5.pdf' # PDF 文件路径\n",
"dir_path = './data' # 存放图片的文件夹\n",
"\n",
"def pdf2image1(path, pic_path):\n",
" checkIM = r\"/Subtype(?= */Image)\"\n",
" pdf = fitz.open(path)\n",
" lenXREF = pdf.xref_length()\n",
" count = 1\n",
" for i in range(1, lenXREF):\n",
" text = pdf.xref_object(i)\n",
" isImage = re.search(checkIM, text)\n",
" if not isImage:\n",
" continue\n",
" pix = fitz.Pixmap(pdf, i)\n",
" new_name = f\"img_{count}.png\"\n",
" pix.writePNG(os.path.join(pic_path, new_name))\n",
" count += 1\n",
" pix = None\n",
"\n",
"pdf2image1(file_path, dir_path)"
]
},
{
"cell_type": "markdown",
"id": "7e86692e-43fe-4aaa-a72f-2fe25fc3597e",
"metadata": {},
"source": [
"### Pdf文件拆分"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c4931fc0-7f0b-4f78-8c8f-151dce930a6b",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os, sys\n",
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
"\n",
"file = \"file/211027/114.pdf\"\n",
"pdf_reader = PdfFileReader(file)\n",
"n = pdf_reader.numPages\n",
"print(n)\n",
"pdfWriter = PdfFileWriter()\n",
"pageObj = pdf_reader.getPage(n-2)\n",
"pageObj.scaleBy(1)\n",
"pdfWriter.addPage(pageObj)\n",
"with open('split_114.pdf', 'wb') as pdfOutputFile:\n",
" pdfWriter.write(pdfOutputFile)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2382429a-7b73-42bc-bef6-67a0b69a449d",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"\n",
"file = \"file/211027/114.pdf\"\n",
"doc = fitz.open(file)\n",
"n = doc.page_count\n",
"print(n)\n",
"doc2 = fitz.open()\n",
"page = doc[n - 2]\n",
"#doc2.insert_pdf(doc, to_page = n-2) # first 10 pages\n",
"doc2.insert_pdf(doc, from_page = n-2,to_page = n-2)\n",
"#print(type(page))\n",
"#doc2.insert_pdf(page) # last 10 pages\n",
"doc2.save(\"first-and-last-10.pdf\")\n"
]
},
{
"cell_type": "markdown",
"id": "18a469e6-8c49-425c-8610-7df4bc59248d",
"metadata": {},
"source": [
"### 提取Pdf页面合并文件"
]
},
{
"cell_type": "markdown",
"id": "21a588fe-cc3c-473b-b9d0-5fcdb0274c5c",
"metadata": {},
"source": [
"#### 简单提取"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "52db5faa-ed27-4ac1-a9e0-b260d8737eb6",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"import glob\n",
"\n",
"fi_path = 'file/2/'\n",
"dict1 = {}\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"list1 = []\n",
"i = 0\n",
"doc2 = fitz.open()\n",
"for fn in fl:\n",
" doc = fitz.open(fn)\n",
" doc2.insert_pdf(doc, from_page = 1,to_page = 1)\n",
"doc2.save(\"检查.pdf\") \n"
]
},
{
"cell_type": "markdown",
"id": "bc39eb1c-e321-4816-9bbd-04a639edc924",
"metadata": {},
"source": [
"#### 提取页面并压缩"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5b3482fa-ae72-4513-b76d-f4f059a73ff1",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"import glob\n",
"import img2pdf\n",
"\n",
"fi_path = 'file/2/'\n",
"zoom = 100\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"\n",
"i = 0\n",
"doc2 = fitz.open()\n",
"for fn in fl:\n",
" doc = fitz.open(fn)\n",
" page = doc[1]\n",
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0)\n",
" pm = page.get_pixmap(matrix=trans)\n",
" s = os.path.basename(fn).split('.')[0].rjust(5,'0') \n",
" lurl=f'pic/{s}.jpg'\n",
" pm.save(lurl) \n",
"\n",
"fl=glob.glob('./pic/*.jpg')\n",
"fl.sort()\n",
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
"with open(\"检查.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(fl,layout_fun=layout_fun))\n",
"\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"id": "fcc0b4d6-9c41-4d8e-8362-36355f8c9354",
"metadata": {},
"source": [
"#### 多线程提取图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "21e20dcb-1852-4357-a08c-9ee9eacfc96b",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"import glob\n",
"from multiprocessing.dummy import Pool\n",
"\n",
"def get_image(fn):\n",
" doc = fitz.open(fn)\n",
" zoom = 100\n",
" page = doc[1]\n",
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0)\n",
" pm = page.get_pixmap(matrix=trans)\n",
" s = os.path.basename(fn).split('.')[0].rjust(5,'0') \n",
" lurl=f'pic1/{s}.jpg'\n",
" pm.save(lurl)\n",
" print(f'{s}已成功生成!') \n",
"fi_path = 'file/2/'\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"pool = Pool(8)\n",
"pool.map(get_image,fl)\n",
"pool.close()\n",
"pool.join()"
]
},
{
"cell_type": "markdown",
"id": "264d3c99-a112-4ade-abda-7785bf8c0d1c",
"metadata": {},
"source": [
"#### 提取页面并分卷压缩"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6e7c67f9-0bef-4860-9dc5-2c8fee885fd4",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os\n",
"import glob\n",
"import img2pdf\n",
"import fitz\n",
"\n",
"fi_path = 'file/2022-09-30/'\n",
"zoom = 100\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"\n",
"for fn in fl:\n",
" doc = fitz.open(fn)\n",
" page = doc[1]\n",
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0)\n",
" pm = page.get_pixmap(matrix=trans)\n",
" s = os.path.basename(fn).split('.')[0].rjust(5,'0') \n",
" lurl=f'pic/{s}.jpg'\n",
" pm.save(lurl) \n",
"\n",
"\n",
"fl=glob.glob('./pic/*.jpg')\n",
"fl.sort()\n",
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
"i = 1\n",
"n = 1\n",
"list1 = []\n",
"for fl1 in fl:\n",
" list1.append(fl1)\n",
" if len(list1) == 200:\n",
" with open(f\"检查{str(i)}.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(list1,layout_fun=layout_fun))\n",
" i +=1\n",
" list1 = []\n",
"if len(list1)>0:\n",
" with open(f\"检查{str(i)}.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(list1,layout_fun=layout_fun))\n",
"\n",
"print('ok!')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f4e85968-7f5b-47f6-afd9-17847bc69b16",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os\n",
"import glob\n",
"import img2pdf\n",
"\n",
"fi_path = 'file/2/'\n",
"\n",
"fl=glob.glob('./pic/*.jpg')\n",
"fl.sort()\n",
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
"i = 1\n",
"n = 1\n",
"list1 = []\n",
"for fl1 in fl:\n",
" list1.append(fl1)\n",
" if len(list1) == 200:\n",
" with open(f\"检查{str(i)}.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(list1,layout_fun=layout_fun))\n",
" i +=1\n",
" list1 = []\n",
"if len(list1)>0:\n",
" with open(f\"检查{str(i)}.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(list1,layout_fun=layout_fun))\n",
"\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"id": "224fca0a-f585-424d-8fcb-59a571215fae",
"metadata": {},
"source": [
"#### 使用PyPdf2简单提取"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e4037936-790d-45f8-b50c-6b1300014f32",
"metadata": {},
"outputs": [],
"source": [
"import os, sys\n",
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
"import glob\n",
"\n",
"fi_path = 'file/2/'\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"pdfWriter = PdfFileWriter()\n",
"i = 0\n",
"doc2 = fitz.open()\n",
"for fn in fl:\n",
" pdf_reader = PdfFileReader(fn)\n",
" pageObj = pdf_reader.getPage(1)\n",
" pdfWriter.addPage(pageObj)\n",
"with open('split_114.pdf', 'wb') as pdfOutputFile: \n",
" pdfWriter.write(pdfOutputFile) \n"
]
},
{
"cell_type": "markdown",
"id": "917dfa96-aaea-4c40-8b89-0fe093c65c33",
"metadata": {},
"source": [
"## PDF文件读取表格"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2f9f474d-265c-4beb-bd0e-53372ae9e57e",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pdfplumber\n",
"import json\n",
"import re\n",
"\n",
"file = \"file/01730823-邵希强.pdf\"\n",
"pdf = pdfplumber.open(file)\n",
"list1 = []\n",
"dict1 = {}\n",
"n = 1\n",
"for page in pdf.pages:\n",
" # print(page.extract_text())\n",
" \n",
" \n",
" for pdf_table in page.extract_tables():\n",
" list2 = []\n",
" table = []\n",
" cells = []\n",
" for row in pdf_table:\n",
" if not any(row):\n",
" # 如果一行全为空,则视为一条记录结束\n",
" if any(cells):\n",
" table.append(cells)\n",
" cells = []\n",
" elif all(row):\n",
" # 如果一行全不为空,则本条为新行,上一条结束\n",
" if any(cells):\n",
" table.append(cells)\n",
" cells = []\n",
" table.append(row)\n",
" else:\n",
" if len(cells) == 0:\n",
" cells = row\n",
" else:\n",
" for i in range(len(row)):\n",
" if row[i] is not None and not cells[i]:\n",
" cells[i] = row[i]\n",
" elif row[i] is not None and cells[i]: \n",
" table.append(cells)\n",
" cells = row\n",
" #cells[i] = row[i]\n",
" break\n",
" elif row[i] is None and cells[i]:\n",
" continue\n",
" elif row[i] is None and not cells[i]:\n",
" #cells[i] = row[i]\n",
" continue \n",
" # cells[i] = row[i] if cells[i] is None else cells[i] + row[i]\n",
" \n",
" list2 = [] \n",
" for row in table:\n",
" data =[re.sub('\\s+', '', cell) if cell is not None else None for cell in row]\n",
" print(data)\n",
" \n",
" #print(json.dumps(data_list, indent=2, ensure_ascii=False))\n",
" #with open('Test1.json','a',encoding=\"utf-8\") as file: # json文件的存放位置\n",
" # json.dump(data_list, file, ensure_ascii=False)\n",
" list2.append(data)\n",
" if len(list2) > 0:\n",
" dict1[n] = list2\n",
" #print(dict1)\n",
" #list1.append(list2)\n",
" n +=1\n",
"with open('Test1.json','w',encoding=\"utf-8\") as file:\n",
" json.dump(dict1, file, ensure_ascii=False)\n",
" \n",
"pdf.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b0764366-12cf-4880-8e0f-9ae18b46d392",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pdfplumber\n",
"name = 'file/01730823-邵希强.pdf'\n",
"\n",
"pdf = pdfplumber.open(name)\n",
"tables =pdf.pages[1].extract_tables()\n",
"df1 = tables\n",
"for item in df1:\n",
" print(item)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "04dd6436-8374-4096-a2a3-d79f4d93a360",
"metadata": {},
"outputs": [],
"source": [
"import pdfplumber\n",
"name = 'file/03499391-宋文路.pdf'\n",
"pdf = pdfplumber.open(name)\n",
"text = pdf.pages[1].extract_text()#######页码从0开始计数\n",
"#print(text)\n",
"list1 = text.split('\\n')\n",
"print(list1)\n",
"for item in list1:\n",
" if '测试标准 国民体质测定标准' in item:\n",
" list_min = list1.index(item)\n",
" if '请注意:以上测试项目' in item:\n",
" list_max = list1.index(item)\n",
"print(list_min,list_max)\n",
"for i in range(list_min+1,list_max):\n",
" print(list1[i].split(' ')[0])\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ac034bf9-cba9-4a72-87ba-d77721cc7097",
"metadata": {},
"outputs": [],
"source": [
"import pdfplumber\n",
"name = 'file/03499391-宋文路.pdf'\n",
"pdf = pdfplumber.open(name)\n",
"text = pdf.pages[1].extract_text()#######页码从0开始计数\n",
"#print(text)\n",
"list1 = text.split('\\n')\n",
"print(list1)\n",
"for item in list1:\n",
" if '感谢您完成测试' in item:\n",
" i = list1.index(item)\n",
" ss = ''.join(list1[i:])\n",
" print(ss)"
]
},
{
"cell_type": "markdown",
"id": "26623725-92da-48d9-a476-78a1befd06ef",
"metadata": {},
"source": [
"## 合并PDF文件"
]
},
{
"cell_type": "code",
"execution_count": 12,
"id": "9b37b17e-66d3-4acc-8edb-760ea17123b4",
"metadata": {
"execution": {
"iopub.execute_input": "2026-01-17T11:13:57.369456Z",
"iopub.status.busy": "2026-01-17T11:13:57.368880Z",
"iopub.status.idle": "2026-01-17T11:13:59.005588Z",
"shell.execute_reply": "2026-01-17T11:13:59.004975Z",
"shell.execute_reply.started": "2026-01-17T11:13:57.369401Z"
}
},
"outputs": [],
"source": [
"from pypdf import PdfWriter\n",
"import glob\n",
"\n",
"fi_path = 'file/2025/'\n",
"fls = glob.glob(f'{fi_path}*.pdf')\n",
"fls.sort()\n",
"merger = PdfWriter()\n",
"for pdf in fls:\n",
" merger.append(pdf)\n",
"merger.write(\"file/2025年明细账.pdf\")\n",
"merger.close()\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6d5c9dd3-bd09-49b3-8a22-07bb4440aace",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.12.3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}