This commit is contained in:
512song@sina.com committed 2022-11-01 11:27:06 +08:00
1 parent e8dc9dce9d
commit 8542f74bd0
1 file changed
+130 -10
+130 -10
View File
@@ -169,6 +169,14 @@
"## 按照日期进行报告分类" "## 按照日期进行报告分类"
] ]
}, },
{
"cell_type": "markdown",
"id": "3d57ff66-69c3-4c13-811b-f1e13b404077",
"metadata": {},
"source": [
"### 按照体测明细分类"
]
},
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": null,
@@ -218,6 +226,14 @@
"print('ok!')" "print('ok!')"
] ]
}, },
{
"cell_type": "markdown",
"id": "ae2b17bd-38b9-4809-affb-01441a5581eb",
"metadata": {},
"source": [
"### 按照报告生成日期分类"
]
},
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": null,
@@ -229,24 +245,128 @@
"source": [ "source": [
"import json\n", "import json\n",
"import time\n", "import time\n",
"import csv\n",
"import os,sys,shutil\n", "import os,sys,shutil\n",
"import glob\n", "import glob\n",
"\n", "\n",
"fi_list = []\n", "dict1 = {}\n",
"m_time = '20221020'\n", "list1 = []\n",
"m_path = 'file/134/20221020/'\n", "\n",
"fl=glob.glob(f'{m_path}*_????????.pdf')\n", "m_path = 'file/134'\n",
"for fn in fl:\n", "mrq = '20221028'\n",
" f_date = os.path.basename(fn).split('_')[1][:8]\n", "if not os.path.exists(m_path + '/' + mrq):\n",
" if m_time <= f_date:\n", " os.mkdir(m_path + '/' + mrq)\n",
" fi_list.append(fn)\n", "fls = glob.glob(f'file/new/*.pdf')\n",
"fi_list" "for fn in fls:\n",
" #old = os.path.basename(fn).split('.')[0].rjust(8,'0') \n",
" old = os.path.basename(fn).split('.')[0] \n",
" n_name = f'{m_path}/{mrq}/{str(old).rjust(8,\"0\")}_{mrq}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(fn,n_name)\n",
"print('ok!')"
]
},
{
"cell_type": "markdown",
"id": "838ec7d3-137c-4ccd-9f16-7b6312d1ffa6",
"metadata": {},
"source": [
"### PDF文件压缩"
] ]
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": null,
"id": "c758420a-e2a4-4260-a571-ff06ed030076", "id": "a6ab0943-b018-4dd6-9176-85a762f9f60c",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import fitz\n",
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"import img2pdf \n",
"import glob\n",
"import shutil\n",
"\n",
"def covert2pic(old_fn):\n",
" if os.path.exists('.pdf'): # 临时文件,需为空\n",
" shutil.rmtree('.pdf')\n",
" os.mkdir('.pdf')\n",
" with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path(old_fn, dpi=100,fmt='jpg', output_folder='.pdf')\n",
"\n",
"def pic2pdf(new_fn):\n",
" fl1=glob.glob('.pdf/*.jpg')\n",
" fl1.sort()\n",
" a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
" layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
" with open(new_fn,\"wb\") as f:\n",
" f.write(img2pdf.convert(fl1,layout_fun=layout_fun))\n",
" print(f'{new_fn}转换成功!')\n",
" \n",
"\n",
"\n",
"def pdfz(sor, obj, zoom): \n",
" covert2pic(zoom)\n",
" pic2pdf(obj)\n",
" \n",
"fi_path = 'file/134/20221028/'\n",
"fl = glob.glob(f'{fi_path}*.pdf')\n",
"\n",
"for fn in fl:\n",
" new_fn = fi_path+'new/'+os.path.basename(fn)\n",
" covert2pic(fn)\n",
" pic2pdf(new_fn)\n",
" shutil.rmtree('.pdf')\n",
"\n",
"\n"
]
},
{
"cell_type": "markdown",
"id": "72c693e1-dd7c-40fa-8948-4fd6fbea8ddc",
"metadata": {},
"source": [
"### 区分新文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "9d06d4f0-364b-438c-844b-7d5badaf3496",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import glob\n",
"import time\n",
"\n",
"fi_path = 'file/'\n",
"fls = glob.glob(f'{fi_path}*.pdf')\n",
"m_date = time.strptime('2022-10-28','%Y-%m-%d')\n",
"for fn in fls:\n",
" c_time = time.gmtime(os.path.getctime(fn))\n",
" if c_time > m_date:\n",
" n_name = f'{fi_path}new/{os.path.basename(fn)}'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(fn,n_name)\n",
" print(n_name)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "32b61559-4495-4fbe-8eb6-2614e6225b5c",
"metadata": {}, "metadata": {},
"outputs": [], "outputs": [],
"source": [] "source": []