1095 lines
32 KiB
Plaintext
Executable File
1095 lines
32 KiB
Plaintext
Executable File
{
|
||
"cells": [
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "333a5431-a697-4330-a587-455d6d18ebb0",
|
||
"metadata": {},
|
||
"source": [
|
||
"# 文件管理"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "bcd48bb6-4756-4748-ac8b-854eeee92a3b",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 数字文件名转换为文本文件名"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "54b320e8-c3c1-4214-9a9e-db13c2558604",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os,sys,shutil\n",
|
||
"import openpyxl\n",
|
||
"import math\n",
|
||
"\n",
|
||
"fi_xls = os.getcwd() + '/data/高新区.xlsx'\n",
|
||
"fi_path = os.getcwd() + '/file/220720'\n",
|
||
"old = []\n",
|
||
"dict1 = {}\n",
|
||
"\n",
|
||
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||
"sheet = wb.active\n",
|
||
"depart = []\n",
|
||
"for n in range(2,sheet.max_row+1): \n",
|
||
" m_name = str(sheet.cell(n,1).value)\n",
|
||
" m_depart = sheet.cell(n,5).value\n",
|
||
" if m_depart not in depart:\n",
|
||
" depart.append(sheet.cell(n,5).value)\n",
|
||
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
|
||
" \n",
|
||
"# 创建部门办公室 \n",
|
||
"m_path = os.getcwd()+'/file/220427/new'\n",
|
||
"for pn in depart:\n",
|
||
" if not os.path.exists(m_path + '/' + pn):\n",
|
||
" os.mkdir(m_path + '/' + pn)\n",
|
||
"fl=os.listdir(fi_path)\n",
|
||
"for fn in fl:\n",
|
||
" if os.path.isfile(fi_path + '/' + fn):\n",
|
||
" ofn = int(fn.split('.')[0])\n",
|
||
" old.append(ofn) \n",
|
||
"old.sort()\n",
|
||
"for n in old: \n",
|
||
" o_name = f'{fi_path}/{n}.pdf'\n",
|
||
" n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n",
|
||
" if not os.path.exists(n_name):\n",
|
||
" shutil.copyfile(o_name,n_name)\n",
|
||
" print(n_name)\n",
|
||
"#print(dict1)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "4a43668e-47ba-491f-a29c-e289885badb0",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os,sys,shutil\n",
|
||
"import openpyxl\n",
|
||
"import math\n",
|
||
"\n",
|
||
"fi_xls = os.getcwd() + '/data/高新区.xlsx'\n",
|
||
"fi_path = os.getcwd() + '/file/220720'\n",
|
||
"old = []\n",
|
||
"dict1 = {}\n",
|
||
"\n",
|
||
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||
"sheet = wb.active\n",
|
||
"depart = []\n",
|
||
"for n in range(2,sheet.max_row+1): \n",
|
||
" m_name = str(sheet.cell(n,1).value)\n",
|
||
" m_depart = sheet.cell(n,5).value\n",
|
||
" if m_depart not in depart:\n",
|
||
" depart.append(sheet.cell(n,5).value)\n",
|
||
" dict1[sheet.cell(n,4).value] = [sheet.cell(n,6).value,sheet.cell(n,3).value]\n",
|
||
"#print(dict1)\n",
|
||
"fl=os.listdir(fi_path)\n",
|
||
"for fn in fl:\n",
|
||
" if os.path.isfile(fi_path + '/' + fn):\n",
|
||
" ofn = int(fn.split('.')[0])\n",
|
||
" old.append(ofn) \n",
|
||
"old.sort()\n",
|
||
"\n",
|
||
"\n",
|
||
"wb = openpyxl.Workbook()\n",
|
||
"sheet = wb.active\n",
|
||
"i = 0\n",
|
||
" \n",
|
||
"for item in old:\n",
|
||
" m_bh = item\n",
|
||
" m_name = dict1[item][0]\n",
|
||
" m_dep = dict1[item][1]\n",
|
||
" i += 1\n",
|
||
" sheet[f'A{i}'] = m_bh\n",
|
||
" sheet[f'B{i}'] = m_name\n",
|
||
" sheet[f'C{i}'] = m_dep\n",
|
||
" \n",
|
||
" \n",
|
||
"wb.save('data/高新区报告.xlsx') "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "72866db8-89be-4498-b854-a3aefe4c897a",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 低碳院人员信息转换"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "433c3f52-f4ae-4544-8d06-3636feae71ba",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os,sys,shutil\n",
|
||
"import openpyxl\n",
|
||
"import math\n",
|
||
"\n",
|
||
"fi_xls = os.getcwd() + '/data/北京低碳监测.xlsx'\n",
|
||
"fi_path = os.getcwd() + '/file/220724'\n",
|
||
"old = []\n",
|
||
"dict1 = {}\n",
|
||
"\n",
|
||
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||
"sheet = wb.active\n",
|
||
"depart = []\n",
|
||
"for n in range(2,sheet.max_row): \n",
|
||
" m_name = sheet.cell(n,6).value\n",
|
||
" m_depart = sheet.cell(n,3).value\n",
|
||
" if sheet.cell(n,4).value is not None:\n",
|
||
" dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n",
|
||
"\n",
|
||
"m_path = os.getcwd()+'/file/220724/new'\n",
|
||
"\n",
|
||
"fl=os.listdir(fi_path)\n",
|
||
"for fn in fl:\n",
|
||
" if os.path.isfile(fi_path + '/' + fn):\n",
|
||
" ofn = int(fn.split('.')[0])\n",
|
||
" old.append(ofn) \n",
|
||
"old.sort()\n",
|
||
"for n in old: \n",
|
||
" o_name = f'{fi_path}/{n}.pdf'\n",
|
||
" n_name = f'{fi_path}/new/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n",
|
||
" if not os.path.exists(n_name):\n",
|
||
" shutil.copyfile(o_name,n_name)\n",
|
||
" print(n_name)\n",
|
||
"#print(dict1)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "cda8fbed-c0f2-4bfc-a23f-6b412bbcd0ba",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 低碳院人员信息核验"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "a34c9f6b-4ef3-4678-89bb-8b692e183968",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os,sys,shutil\n",
|
||
"import openpyxl\n",
|
||
"import math\n",
|
||
"import json\n",
|
||
"\n",
|
||
"fi_xls = os.getcwd() + '/data/北京低碳监测.xlsx'\n",
|
||
"fi_path = os.getcwd() + '/file/220724'\n",
|
||
"old = []\n",
|
||
"dict1 = {}\n",
|
||
"dict2 = {}\n",
|
||
"dict3 = {}\n",
|
||
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||
"sheet = wb.active\n",
|
||
"depart = []\n",
|
||
"for n in range(2,sheet.max_row+1): \n",
|
||
" m_name = sheet.cell(n,6).value\n",
|
||
" m_depart = sheet.cell(n,3).value\n",
|
||
" if sheet.cell(n,4).value is not None:\n",
|
||
" dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n",
|
||
"\n",
|
||
"fi_xls = os.getcwd() + '/data/1658883945637.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(2,sheet.max_row+1): \n",
|
||
" m_dm = sheet.cell(n,3).value\n",
|
||
" m_depart = sheet.cell(n,2).value\n",
|
||
" m_name = sheet.cell(n,4).value\n",
|
||
" if sheet.cell(n,3).value is not None:\n",
|
||
" dict2[sheet.cell(n,3).value] = [m_name,m_depart,m_dm]\n",
|
||
"#print(dict2)\n",
|
||
"fl=os.listdir(fi_path)\n",
|
||
"for fn in fl:\n",
|
||
" if os.path.isfile(fi_path + '/' + fn):\n",
|
||
" ofn = int(fn.split('.')[0])\n",
|
||
" old.append(ofn) \n",
|
||
"old.sort()\n",
|
||
"for n in old:\n",
|
||
" m_name = dict1[n][0]\n",
|
||
" m_depart = dict1[n][1]\n",
|
||
" for item in dict2.values():\n",
|
||
" if m_name in item and m_depart in item:\n",
|
||
" dict3[n] = [m_name,m_depart,item[2]]\n",
|
||
" break\n",
|
||
"for k in old:\n",
|
||
" if k not in dict3.keys():\n",
|
||
" print(k)\n",
|
||
"with open('data/低碳院报告.json','w') as fl2:\n",
|
||
" json.dump(dict3,fl2,ensure_ascii=False) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "3244663e-890c-48de-ae6a-1f666cb1bb76",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 低碳院人员附件核验"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "8e20ead8-754a-4535-95ea-b103264e3783",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import os\n",
|
||
"\n",
|
||
"fl_name = 'data/低碳院报告.json'\n",
|
||
"\n",
|
||
"with open(fl_name,'r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k, v in m_xx.items():\n",
|
||
" m_bh = str(k).rjust(5,\"0\")\n",
|
||
" fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n",
|
||
" if os.path.exists(fl_name):\n",
|
||
" print(fl_name)\n",
|
||
" else:\n",
|
||
" print(f'{v[0]}不存在')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "a7976e2e-23fa-4464-89f5-0e96f721bc95",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 低碳院人员导出"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "d77c3ff5-91fb-4449-80fb-57bc467c27bd",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
" \n",
|
||
"\n",
|
||
"fl_name = 'data/低碳院报告.json'\n",
|
||
"\n",
|
||
"with open(fl_name,'r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
" \n",
|
||
"wb = openpyxl.Workbook()\n",
|
||
"sheet = wb.active\n",
|
||
"i = 0\n",
|
||
" \n",
|
||
"for k, v in m_xx.items():\n",
|
||
" m_bh = str(k).rjust(5,\"0\")\n",
|
||
" m_name = v[0]\n",
|
||
" fl = f'{m_bh}-{v[0]}.pdf'\n",
|
||
" m_rec = v[2]+'@ceic.com'\n",
|
||
" i += 1\n",
|
||
" sheet[f'A{i}'] = i\n",
|
||
" sheet[f'B{i}'] = m_bh\n",
|
||
" sheet[f'C{i}'] = m_name\n",
|
||
" sheet[f'D{i}'] = m_rec \n",
|
||
" sheet[f'E{i}'] = fl\n",
|
||
" \n",
|
||
"wb.save('data/低碳院邮件.xlsx') \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "bbf0b425-c0ce-4608-ae52-85a78ddcd6c0",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 批量更换扩展名"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "2c2bd1d1-4d84-4639-ba20-c97c218ae275",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import glob\n",
|
||
"import os,shutil\n",
|
||
"\n",
|
||
"filepath = 'file/yimo/'\n",
|
||
"old = 'SGF'\n",
|
||
"new = 'sgf'\n",
|
||
"files = glob.glob(f'{filepath}*.{old}')\n",
|
||
"for f in files:\n",
|
||
" new_file =os.path.splitext(f)[0]+'.'+new\n",
|
||
" os.rename(f,new_file)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "69712868-6786-430b-8591-f6828d781588",
|
||
"metadata": {},
|
||
"source": [
|
||
"## PDF文件压缩"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "8492dc14-739e-4c78-a302-f4b57ae571dd",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import fitz\n",
|
||
"import os\n",
|
||
"\n",
|
||
"\n",
|
||
"def covert2pic(zoom):\n",
|
||
" if os.path.exists('.pdf'): # 临时文件,需为空\n",
|
||
" os.removedirs('.pdf')\n",
|
||
" os.mkdir('.pdf')\n",
|
||
" for pg in range(totaling):\n",
|
||
" page = doc[pg]\n",
|
||
" zoom = int(zoom) #值越大,分辨率越高,文件越清晰\n",
|
||
" rotate = int(0)\n",
|
||
" print(page)\n",
|
||
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0).preRotate(rotate)\n",
|
||
" pm = page.getPixmap(matrix=trans, alpha=False)\n",
|
||
" \n",
|
||
" lurl='.pdf/%s.jpg' % str(pg+1)\n",
|
||
" pm.writePNG(lurl)\n",
|
||
" doc.close()\n",
|
||
"\n",
|
||
"def pic2pdf(obj):\n",
|
||
" doc = fitz.open()\n",
|
||
" for pg in range(totaling):\n",
|
||
" img = '.pdf/%s.jpg' % str(pg+1)\n",
|
||
" imgdoc = fitz.open(img) # 打开图片\n",
|
||
" pdfbytes = imgdoc.convertToPDF() # 使用图片创建单页的 PDF\n",
|
||
" os.remove(img) \n",
|
||
" imgpdf = fitz.open(\"pdf\", pdfbytes)\n",
|
||
" doc.insertPDF(imgpdf) # 将当前页插入文档\n",
|
||
" if os.path.exists(obj): # 若文件存在先删除\n",
|
||
" os.remove(obj)\n",
|
||
" doc.save(obj) # 保存pdf文件\n",
|
||
" doc.close()\n",
|
||
"\n",
|
||
"\n",
|
||
"def pdfz(sor, obj, zoom): \n",
|
||
" covert2pic(zoom)\n",
|
||
" pic2pdf(obj)\n",
|
||
" \n",
|
||
"\n",
|
||
"\n",
|
||
"sor = \"5.pdf\" # 需要压缩的PDF文件\n",
|
||
"obj = \"new-\" + sor\n",
|
||
"doc = fitz.open(sor) \n",
|
||
"totaling = doc.pageCount\n",
|
||
"\n",
|
||
"zoom = 150 # 清晰度调节,缩放比率\n",
|
||
"pdfz(sor, obj, zoom)\n",
|
||
"os.removedirs('.pdf')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "88d84849-5b98-48cb-86b1-e96757a8615f",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"from pdf2image import convert_from_path, convert_from_bytes\n",
|
||
"import os,sys\n",
|
||
"import tempfile\n",
|
||
"from pdf2image.exceptions import (\n",
|
||
" PDFInfoNotInstalledError,\n",
|
||
" PDFPageCountError,\n",
|
||
" PDFSyntaxError\n",
|
||
")\n",
|
||
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
|
||
"with tempfile.TemporaryDirectory() as path:\n",
|
||
" images_from_path = convert_from_path('5.pdf', dpi=100,fmt='jpg', output_folder='./pic')\n",
|
||
"print(path)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "058a40ba-4948-4675-8400-e4c762e836ee",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import img2pdf \n",
|
||
"import os,sys\n",
|
||
"import glob\n",
|
||
"\n",
|
||
"fl=glob.glob('./pic/*.jpg')\n",
|
||
"fl.sort()\n",
|
||
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
|
||
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
|
||
"with open(\"name.pdf\",\"wb\") as f:\n",
|
||
" f.write(img2pdf.convert(fl,layout_fun=layout_fun))"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "e997ad75-b2d9-43bf-a1de-8833cfe0cf6a",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import glob\n",
|
||
"import fitz # 导入本模块需安装pymupdf库\n",
|
||
"import os,sys\n",
|
||
"\n",
|
||
"def pic2pdf_1(img_path, pdf_path, pdf_name):\n",
|
||
" doc = fitz.open()\n",
|
||
" fl=os.listdir(img_path)\n",
|
||
" fl.sort()\n",
|
||
" width, height = fitz.PaperSize(\"a4\")\n",
|
||
" for img in fl:\n",
|
||
" fn = img_path+'/'+img\n",
|
||
" if os.path.isfile(fn):\n",
|
||
" imgdoc = fitz.open(img_path+'/'+img)\n",
|
||
" pdfbytes = imgdoc.convertToPDF()\n",
|
||
" imgpdf = fitz.open(\"pdf\", pdfbytes,width = width, height = height)\n",
|
||
" doc.insertPDF(imgpdf)\n",
|
||
" doc.save(pdf_path +'/'+ pdf_name)\n",
|
||
" doc.close()\n",
|
||
"\n",
|
||
"img_path = os.getcwd() +'/pic'\n",
|
||
"pdf_path = os.getcwd()\n",
|
||
"pic2pdf_1(img_path=img_path, pdf_path=pdf_path, pdf_name='1.pdf')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "1f38e4b6-ef20-4fef-801c-ff47951d2ad6",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 图片文件扫描识别"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "8c990b6a-7f8e-4527-a150-65f60bd131ac",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"#将指定目录下图片文件进行文字识别\n",
|
||
"import os,sys\n",
|
||
"from aip import AipOcr\n",
|
||
"import glob\n",
|
||
"\n",
|
||
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||
"APP_ID = '17553946'\n",
|
||
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||
"\n",
|
||
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||
"def get_file_content(filePath):\n",
|
||
" with open(filePath, 'rb') as fp:\n",
|
||
" return fp.read()\n",
|
||
"\n",
|
||
"options = {}\n",
|
||
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||
"options[\"detect_direction\"] = \"true\"\n",
|
||
"options[\"detect_language\"] = \"true\"\n",
|
||
"options[\"probability\"] = \"true\"\n",
|
||
"\n",
|
||
"fi_path = os.getcwd()+'/pic'\n",
|
||
"fl = glob.glob(f'{fi_path}/*.jpg')\n",
|
||
"fl.sort()\n",
|
||
"for fl1 in fl:\n",
|
||
" file_name = fl1\n",
|
||
" image = get_file_content(file_name)\n",
|
||
" result= client.basicGeneral(image, options)\n",
|
||
" if 'words_result' in result:\n",
|
||
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
|
||
" print('\\n')\n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "d936ca99-269b-46f8-8582-fb02ed2cd3cc",
|
||
"metadata": {},
|
||
"source": [
|
||
"## word文档读取"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "d8043425-be9e-4e9e-bbfc-3a2d8885a860",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import docx\n",
|
||
"import json\n",
|
||
"Doc = docx.Document(r\"症状对症处方选穴.docx\")\n",
|
||
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
|
||
"testList = []\n",
|
||
"text1 = []\n",
|
||
"dict1 = {}\n",
|
||
"for text in Doc.paragraphs:\n",
|
||
" testList.append(text)\n",
|
||
"i = 0\n",
|
||
"for pg in testList:\n",
|
||
" \n",
|
||
" if len(pg.text) >0:\n",
|
||
" s = ''.join(pg.text.split()) \n",
|
||
" if s[0:1] == '第':\n",
|
||
" i += 1\n",
|
||
" n = 0\n",
|
||
" \n",
|
||
" dict1.setdefault(i,{})\n",
|
||
" dict1[i]['title'] = pg.text.split()[1]\n",
|
||
" dict1[i]['sub'] = {}\n",
|
||
" \n",
|
||
" #dict1[i]['title'].setdefault(i,{})\n",
|
||
" elif s[0:1] == '(':\n",
|
||
" sub = s.split(')')[1]\n",
|
||
" dict2 = {}\n",
|
||
" dict1[i]['sub'].setdefault(sub,[])\n",
|
||
" else:\n",
|
||
" dict1[i]['sub'][sub].append(s)\n",
|
||
"filename = '症状对症处方选穴.json'\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(dict1, fl,ensure_ascii=False) \n",
|
||
"print('ok') "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "c440b599-9c80-4072-8543-d9c381214f1d",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import docx\n",
|
||
"import json\n",
|
||
"Doc = docx.Document(r\"常见疾病辨病处方选穴.docx\")\n",
|
||
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
|
||
"testList = []\n",
|
||
"text1 = []\n",
|
||
"dict1 = {}\n",
|
||
"for text in Doc.paragraphs:\n",
|
||
" testList.append(text)\n",
|
||
"i = 0\n",
|
||
"for pg in testList:\n",
|
||
" \n",
|
||
" if len(pg.text) >0:\n",
|
||
" s = ''.join(pg.text.split()) \n",
|
||
" if s[0:1] == '第':\n",
|
||
" i += 1\n",
|
||
" n = 0\n",
|
||
" \n",
|
||
" dict1.setdefault(i,{})\n",
|
||
" dict1[i]['title'] = pg.text.split('节')[1]\n",
|
||
" dict1[i]['sub'] = {}\n",
|
||
" \n",
|
||
" #dict1[i]['title'].setdefault(i,{})\n",
|
||
" elif s[0:1] == '(':\n",
|
||
" sub = s.split(')')[1]\n",
|
||
" dict2 = {}\n",
|
||
" dict1[i]['sub'].setdefault(sub,[])\n",
|
||
" else:\n",
|
||
" dict1[i]['sub'][sub].append(s)\n",
|
||
"filename = '常见疾病辨病处方选穴.json'\n",
|
||
"with open(filename,'w') as fl:\n",
|
||
" json.dump(dict1, fl,ensure_ascii=False)\n",
|
||
"print('ok') "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "b5e500b5-14d4-4417-8b5e-46df71a6b6b8",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import re\n",
|
||
"filename = 'file/ticemuban.json'\n",
|
||
"\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl) \n",
|
||
"print(dict1)\n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "24d3fdb2-65ac-4a2b-b65a-f64a01a3fdf5",
|
||
"metadata": {},
|
||
"source": [
|
||
"## word使用模板生成文档"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 33,
|
||
"id": "9957a90c-7735-4224-b5b5-4a4b1b7688c9",
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2022-12-07T06:24:28.191719Z",
|
||
"iopub.status.busy": "2022-12-07T06:24:28.191155Z",
|
||
"iopub.status.idle": "2022-12-07T06:24:28.291897Z",
|
||
"shell.execute_reply": "2022-12-07T06:24:28.290876Z",
|
||
"shell.execute_reply.started": "2022-12-07T06:24:28.191671Z"
|
||
},
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import openpyxl\n",
|
||
"from docxtpl import DocxTemplate,InlineImage\n",
|
||
"\n",
|
||
"list1 = []\n",
|
||
"filename = 'file/ticemuban.json'\n",
|
||
"tpl = DocxTemplate(\"file/体测报告模板.docx\")\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" dict1 = json.load(fl) \n",
|
||
"dict1['img_radar'] = InlineImage(tpl, image_descriptor=dict1['radar'])\n",
|
||
"list1.append(dict1)\n",
|
||
"\n",
|
||
"context['info'] = list1\n",
|
||
"\n",
|
||
"tpl.render(dict1)\n",
|
||
"tpl.save('file/test1.docx')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 36,
|
||
"id": "42122edc-eb77-4cfc-bcd8-3f844ffcca58",
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2022-12-07T07:30:29.738226Z",
|
||
"iopub.status.busy": "2022-12-07T07:30:29.737684Z",
|
||
"iopub.status.idle": "2022-12-07T07:30:30.656573Z",
|
||
"shell.execute_reply": "2022-12-07T07:30:30.655231Z",
|
||
"shell.execute_reply.started": "2022-12-07T07:30:29.738176Z"
|
||
},
|
||
"tags": []
|
||
},
|
||
"outputs": [
|
||
{
|
||
"name": "stdout",
|
||
"output_type": "stream",
|
||
"text": [
|
||
"转换完成\n"
|
||
]
|
||
}
|
||
],
|
||
"source": [
|
||
"from subprocess import Popen\n",
|
||
"\n",
|
||
"file = 'file/test1.docx'\n",
|
||
"Popen(['abiword', '-t', 'pdf', file]).communicate()\n",
|
||
"print('转换完成')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "8a555d4f-8b2c-4f2c-af7e-fe8dfbd33f5f",
|
||
"metadata": {},
|
||
"source": [
|
||
"## PDF文件读取"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "308511e7-0f2d-40bc-a151-68f26ee635a2",
|
||
"metadata": {},
|
||
"source": [
|
||
"### pdfminer读取PDF文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "0b0d621c-f2be-4dcb-a1f0-4b16196d25ee",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"from newpdfminer.pdfparser import PDFParser, PDFDocument\n",
|
||
"from newpdfminer.pdfinterp import PDFResourceManager, PDFPageInterpreter\n",
|
||
"from newpdfminer.converter import PDFPageAggregator\n",
|
||
"from newpdfminer.layout import LAParams, LTTextBox\n",
|
||
"from newpdfminer.pdfinterp import PDFTextExtractionNotAllowed\n",
|
||
"\n",
|
||
"path = \"file/211027/114.pdf\"\n",
|
||
"\n",
|
||
"# 用文件对象来创建一个pdf文档分析器\n",
|
||
"praser = PDFParser(open(path, 'rb'))\n",
|
||
"# 创建一个PDF文档\n",
|
||
"doc = PDFDocument()\n",
|
||
"# 连接分析器 与文档对象\n",
|
||
"praser.set_document(doc)\n",
|
||
"doc.set_parser(praser)\n",
|
||
"\n",
|
||
"# 提供初始化密码\n",
|
||
"# 如果没有密码 就创建一个空的字符串\n",
|
||
"doc.initialize()\n",
|
||
"\n",
|
||
"# 检测文档是否提供txt转换,不提供就忽略\n",
|
||
"if not doc.is_extractable:\n",
|
||
" raise PDFTextExtractionNotAllowed\n",
|
||
"else:\n",
|
||
" # 创建PDf 资源管理器 来管理共享资源\n",
|
||
" rsrcmgr = PDFResourceManager()\n",
|
||
" # 创建一个PDF设备对象\n",
|
||
" laparams = LAParams()\n",
|
||
" device = PDFPageAggregator(rsrcmgr, laparams=laparams)\n",
|
||
" # 创建一个PDF解释器对象\n",
|
||
" interpreter = PDFPageInterpreter(rsrcmgr, device)\n",
|
||
"\n",
|
||
" # 循环遍历列表,每次处理一个page的内容\n",
|
||
" for page in doc.get_pages():\n",
|
||
" interpreter.process_page(page) \n",
|
||
" # 接受该页面的LTPage对象\n",
|
||
" layout = device.get_result()\n",
|
||
" # 这里layout是一个LTPage对象,里面存放着这个 page 解析出的各种对象\n",
|
||
" # 包括 LTTextBox, LTFigure, LTImage, LTTextBoxHorizontal 等 \n",
|
||
" for x in layout:\n",
|
||
" if isinstance(x, LTTextBox):\n",
|
||
" print(x.get_text().strip())"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "8fff7f1e-b52c-4666-8a2c-e9cde85a8e01",
|
||
"metadata": {},
|
||
"source": [
|
||
"### pdfplumber读取PDF文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "c8685b70-55b3-439e-852d-c4fbb67b6326",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pdfplumber\n",
|
||
"from PIL import Image\n",
|
||
"\n",
|
||
"# 读取 PDF 文档\n",
|
||
"pdf = pdfplumber.open(\"5.pdf\")\n",
|
||
"\n",
|
||
"# 获取页数\n",
|
||
"print(\"总页数:\",len(pdf.pages))\n",
|
||
"print(\"-----------------------------------------\")\n",
|
||
"\n",
|
||
"# 读取第 4 页;索引从 1 开始\n",
|
||
"page = pdf.pages[0] \n",
|
||
"print(\"本页:\",page.page_number + 1)\n",
|
||
"print(\"-----------------------------------------\")\n",
|
||
"# 单页文件存储为图片\n",
|
||
"#im = page.to_image()\n",
|
||
"#im.save('./5_1.png', format='PNG')\n",
|
||
"# 单页文件中原始图片读取,不能读取图表文件\n",
|
||
"#imgs = page.images\n",
|
||
"#for i, img in enumerate(imgs):\n",
|
||
"# size = img['width'], img['height']\n",
|
||
"# data = img['stream'].get_data()\n",
|
||
"# out_path = f'003pdf_images_{i}.png'\n",
|
||
"# with open(out_path, 'wb') as fimg_out:\n",
|
||
"# fimg_out.write(data)\n",
|
||
"# 导出第 4 页文本\n",
|
||
"text = page.extract_text()\n",
|
||
"print(text)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "59e8775b-e2d6-4963-9376-4a944108ec58",
|
||
"metadata": {},
|
||
"source": [
|
||
"### fitz1.8读取PDF文件中的图片"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "5ff3d5dd-b8dd-4707-8e29-26ba6cf97070",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import fitz\n",
|
||
"import re\n",
|
||
"import os\n",
|
||
"\n",
|
||
"file_path = '5.pdf' # PDF 文件路径\n",
|
||
"dir_path = './data' # 存放图片的文件夹\n",
|
||
"\n",
|
||
"def pdf2image1(path, pic_path):\n",
|
||
" checkIM = r\"/Subtype(?= */Image)\"\n",
|
||
" pdf = fitz.open(path)\n",
|
||
" lenXREF = pdf.xref_length()\n",
|
||
" count = 1\n",
|
||
" for i in range(1, lenXREF):\n",
|
||
" text = pdf.xref_object(i)\n",
|
||
" isImage = re.search(checkIM, text)\n",
|
||
" if not isImage:\n",
|
||
" continue\n",
|
||
" pix = fitz.Pixmap(pdf, i)\n",
|
||
" new_name = f\"img_{count}.png\"\n",
|
||
" pix.writePNG(os.path.join(pic_path, new_name))\n",
|
||
" count += 1\n",
|
||
" pix = None\n",
|
||
"\n",
|
||
"pdf2image1(file_path, dir_path)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "7e86692e-43fe-4aaa-a72f-2fe25fc3597e",
|
||
"metadata": {},
|
||
"source": [
|
||
"### Pdf文件拆分"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "c4931fc0-7f0b-4f78-8c8f-151dce930a6b",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os, sys\n",
|
||
"from PyPDF2 import PdfFileWriter, PdfFileReader, PdfFileMerger\n",
|
||
"\n",
|
||
"file = \"file/211027/114.pdf\"\n",
|
||
"pdf_reader = PdfFileReader(file)\n",
|
||
"n = pdf_reader.numPages\n",
|
||
"print(n)\n",
|
||
"pdfWriter = PdfFileWriter()\n",
|
||
"pageObj = pdf_reader.getPage(n-2)\n",
|
||
"pageObj.scaleBy(1)\n",
|
||
"pdfWriter.addPage(pageObj)\n",
|
||
"with open('split_114.pdf', 'wb') as pdfOutputFile:\n",
|
||
" pdfWriter.write(pdfOutputFile)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "2382429a-7b73-42bc-bef6-67a0b69a449d",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import fitz\n",
|
||
"import os\n",
|
||
"\n",
|
||
"file = \"file/211027/114.pdf\"\n",
|
||
"doc = fitz.open(file)\n",
|
||
"n = doc.page_count\n",
|
||
"print(n)\n",
|
||
"doc2 = fitz.open()\n",
|
||
"page = doc[n - 2]\n",
|
||
"#doc2.insert_pdf(doc, to_page = n-2) # first 10 pages\n",
|
||
"doc2.insert_pdf(doc, from_page = n-2,to_page = n-2)\n",
|
||
"#print(type(page))\n",
|
||
"#doc2.insert_pdf(page) # last 10 pages\n",
|
||
"doc2.save(\"first-and-last-10.pdf\")\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"id": "917dfa96-aaea-4c40-8b89-0fe093c65c33",
|
||
"metadata": {},
|
||
"source": [
|
||
"## PDF文件读取表格"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "2f9f474d-265c-4beb-bd0e-53372ae9e57e",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pdfplumber\n",
|
||
"import json\n",
|
||
"import re\n",
|
||
"\n",
|
||
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
|
||
"pdf = pdfplumber.open(file)\n",
|
||
"list1 = []\n",
|
||
"dict1 = {}\n",
|
||
"n = 1\n",
|
||
"for page in pdf.pages:\n",
|
||
" # print(page.extract_text())\n",
|
||
" \n",
|
||
" \n",
|
||
" for pdf_table in page.extract_tables():\n",
|
||
" list2 = []\n",
|
||
" table = []\n",
|
||
" cells = []\n",
|
||
" for row in pdf_table:\n",
|
||
" if not any(row):\n",
|
||
" # 如果一行全为空,则视为一条记录结束\n",
|
||
" if any(cells):\n",
|
||
" table.append(cells)\n",
|
||
" cells = []\n",
|
||
" elif all(row):\n",
|
||
" # 如果一行全不为空,则本条为新行,上一条结束\n",
|
||
" if any(cells):\n",
|
||
" table.append(cells)\n",
|
||
" cells = []\n",
|
||
" table.append(row)\n",
|
||
" else:\n",
|
||
" if len(cells) == 0:\n",
|
||
" cells = row\n",
|
||
" else:\n",
|
||
" for i in range(len(row)):\n",
|
||
" if row[i] is not None and not cells[i]:\n",
|
||
" cells[i] = row[i]\n",
|
||
" elif row[i] is not None and cells[i]: \n",
|
||
" table.append(cells)\n",
|
||
" cells = row\n",
|
||
" #cells[i] = row[i]\n",
|
||
" break\n",
|
||
" elif row[i] is None and cells[i]:\n",
|
||
" continue\n",
|
||
" elif row[i] is None and not cells[i]:\n",
|
||
" #cells[i] = row[i]\n",
|
||
" continue \n",
|
||
" # cells[i] = row[i] if cells[i] is None else cells[i] + row[i]\n",
|
||
" \n",
|
||
" list2 = [] \n",
|
||
" for row in table:\n",
|
||
" data =[re.sub('\\s+', '', cell) if cell is not None else None for cell in row]\n",
|
||
" print(data)\n",
|
||
" \n",
|
||
" #print(json.dumps(data_list, indent=2, ensure_ascii=False))\n",
|
||
" #with open('Test1.json','a',encoding=\"utf-8\") as file: # json文件的存放位置\n",
|
||
" # json.dump(data_list, file, ensure_ascii=False)\n",
|
||
" list2.append(data)\n",
|
||
" if len(list2) > 0:\n",
|
||
" dict1[n] = list2\n",
|
||
" #print(dict1)\n",
|
||
" #list1.append(list2)\n",
|
||
" n +=1\n",
|
||
"with open('Test1.json','w',encoding=\"utf-8\") as file:\n",
|
||
" json.dump(dict1, file, ensure_ascii=False)\n",
|
||
" \n",
|
||
"pdf.close()"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "71e1da77-6744-4d86-a4bc-88feeb3023e8",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pdfplumber\n",
|
||
"import json\n",
|
||
"import re\n",
|
||
"\n",
|
||
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
|
||
"pdf = pdfplumber.open(file)\n",
|
||
"list1 = []\n",
|
||
"dict1 = {}\n",
|
||
"n = 1\n",
|
||
"for page in pdf.pages:\n",
|
||
" for pdf_table in page.extract_tables():\n",
|
||
" list2 = []\n",
|
||
" table = []\n",
|
||
" cells = []\n",
|
||
" for row in pdf_table:\n",
|
||
" print(row)\n",
|
||
" print('******')\n",
|
||
" print('--------')\n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "b0764366-12cf-4880-8e0f-9ae18b46d392",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import camelot\n",
|
||
"import json\n",
|
||
"import re\n",
|
||
"\n",
|
||
"file = \"file/1_中石化北海炼化2020年团体体检报告.pdf\"\n",
|
||
"tables = camelot.read_pdf(file, pages='3',flavor='stream')\n",
|
||
"# 2.导出pdf所有的表格为csv文件\n",
|
||
"tables.export('foo.json', f='json')\n",
|
||
"print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"id": "04dd6436-8374-4096-a2a3-d79f4d93a360",
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
}
|
||
],
|
||
"metadata": {
|
||
"kernelspec": {
|
||
"display_name": "Python 3",
|
||
"language": "python",
|
||
"name": "python3"
|
||
},
|
||
"language_info": {
|
||
"codemirror_mode": {
|
||
"name": "ipython",
|
||
"version": 3
|
||
},
|
||
"file_extension": ".py",
|
||
"mimetype": "text/x-python",
|
||
"name": "python",
|
||
"nbconvert_exporter": "python",
|
||
"pygments_lexer": "ipython3",
|
||
"version": "3.8.10"
|
||
}
|
||
},
|
||
"nbformat": 4,
|
||
"nbformat_minor": 5
|
||
}
|