diff --git a/体测单位/天津石化.ipynb b/体测单位/天津石化.ipynb index c69e8a3..895be63 100644 --- a/体测单位/天津石化.ipynb +++ b/体测单位/天津石化.ipynb @@ -169,6 +169,14 @@ "## 按照日期进行报告分类" ] }, + { + "cell_type": "markdown", + "id": "3d57ff66-69c3-4c13-811b-f1e13b404077", + "metadata": {}, + "source": [ + "### 按照体测明细分类" + ] + }, { "cell_type": "code", "execution_count": null, @@ -218,6 +226,14 @@ "print('ok!')" ] }, + { + "cell_type": "markdown", + "id": "ae2b17bd-38b9-4809-affb-01441a5581eb", + "metadata": {}, + "source": [ + "### 按照报告生成日期分类" + ] + }, { "cell_type": "code", "execution_count": null, @@ -229,24 +245,128 @@ "source": [ "import json\n", "import time\n", + "import csv\n", "import os,sys,shutil\n", "import glob\n", "\n", - "fi_list = []\n", - "m_time = '20221020'\n", - "m_path = 'file/134/20221020/'\n", - "fl=glob.glob(f'{m_path}*_????????.pdf')\n", - "for fn in fl:\n", - " f_date = os.path.basename(fn).split('_')[1][:8]\n", - " if m_time <= f_date:\n", - " fi_list.append(fn)\n", - "fi_list" + "dict1 = {}\n", + "list1 = []\n", + "\n", + "m_path = 'file/134'\n", + "mrq = '20221028'\n", + "if not os.path.exists(m_path + '/' + mrq):\n", + " os.mkdir(m_path + '/' + mrq)\n", + "fls = glob.glob(f'file/new/*.pdf')\n", + "for fn in fls:\n", + " #old = os.path.basename(fn).split('.')[0].rjust(8,'0') \n", + " old = os.path.basename(fn).split('.')[0] \n", + " n_name = f'{m_path}/{mrq}/{str(old).rjust(8,\"0\")}_{mrq}.pdf'\n", + " if not os.path.exists(n_name):\n", + " shutil.copyfile(fn,n_name)\n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "id": "838ec7d3-137c-4ccd-9f16-7b6312d1ffa6", + "metadata": {}, + "source": [ + "### PDF文件压缩" ] }, { "cell_type": "code", "execution_count": null, - "id": "c758420a-e2a4-4260-a571-ff06ed030076", + "id": "a6ab0943-b018-4dd6-9176-85a762f9f60c", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import fitz\n", + "from pdf2image import convert_from_path, convert_from_bytes\n", + "import os,sys\n", + "import tempfile\n", + "from pdf2image.exceptions import (\n", + " PDFInfoNotInstalledError,\n", + " PDFPageCountError,\n", + " PDFSyntaxError\n", + ")\n", + "import img2pdf \n", + "import glob\n", + "import shutil\n", + "\n", + "def covert2pic(old_fn):\n", + " if os.path.exists('.pdf'): # 临时文件,需为空\n", + " shutil.rmtree('.pdf')\n", + " os.mkdir('.pdf')\n", + " with tempfile.TemporaryDirectory() as path:\n", + " images_from_path = convert_from_path(old_fn, dpi=100,fmt='jpg', output_folder='.pdf')\n", + "\n", + "def pic2pdf(new_fn):\n", + " fl1=glob.glob('.pdf/*.jpg')\n", + " fl1.sort()\n", + " a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n", + " layout_fun = img2pdf.get_layout_fun(a4inpt)\n", + " with open(new_fn,\"wb\") as f:\n", + " f.write(img2pdf.convert(fl1,layout_fun=layout_fun))\n", + " print(f'{new_fn}转换成功!')\n", + " \n", + "\n", + "\n", + "def pdfz(sor, obj, zoom): \n", + " covert2pic(zoom)\n", + " pic2pdf(obj)\n", + " \n", + "fi_path = 'file/134/20221028/'\n", + "fl = glob.glob(f'{fi_path}*.pdf')\n", + "\n", + "for fn in fl:\n", + " new_fn = fi_path+'new/'+os.path.basename(fn)\n", + " covert2pic(fn)\n", + " pic2pdf(new_fn)\n", + " shutil.rmtree('.pdf')\n", + "\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "id": "72c693e1-dd7c-40fa-8948-4fd6fbea8ddc", + "metadata": {}, + "source": [ + "### 区分新文件" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "9d06d4f0-364b-438c-844b-7d5badaf3496", + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import os,sys,shutil\n", + "import glob\n", + "import time\n", + "\n", + "fi_path = 'file/'\n", + "fls = glob.glob(f'{fi_path}*.pdf')\n", + "m_date = time.strptime('2022-10-28','%Y-%m-%d')\n", + "for fn in fls:\n", + " c_time = time.gmtime(os.path.getctime(fn))\n", + " if c_time > m_date:\n", + " n_name = f'{fi_path}new/{os.path.basename(fn)}'\n", + " if not os.path.exists(n_name):\n", + " shutil.copyfile(fn,n_name)\n", + " print(n_name)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "32b61559-4495-4fbe-8eb6-2614e6225b5c", "metadata": {}, "outputs": [], "source": []