diff --git a/文件管理1.ipynb b/文件管理1.ipynb index 650c382..70d5c41 100755 --- a/文件管理1.ipynb +++ b/文件管理1.ipynb @@ -411,9 +411,60 @@ { "cell_type": "code", "execution_count": null, - "id": "88d84849-5b98-48cb-86b1-e96757a8615f", - "metadata": {}, + "id": "99fff4b6-73ea-4d53-96bc-41b6b664e00a", + "metadata": { + "tags": [] + }, "outputs": [], + "source": [ + "import fitz\n", + "import os\n", + "\n", + "sor = \"中华药膳全书学做药膳不生病.pdf\" # 需要压缩的PDF文件\n", + "\n", + "doc = fitz.open(sor) \n", + "totaling = doc.pageCount\n", + "\n", + "zoom = 300 # 清晰度调节,缩放比率\n", + "if os.path.exists('pdf_cp'): # 临时文件,需为空\n", + " os.removedirs('pdf_cp')\n", + "os.mkdir('pdf_cp')\n", + "for pg in range(6,totaling):\n", + " page = doc[pg]\n", + " zoom = int(zoom) #值越大,分辨率越高,文件越清晰\n", + " rotate = int(0)\n", + " print(page)\n", + " trans = fitz.Matrix(zoom / 100.0, zoom / 100.0).preRotate(rotate)\n", + " pm = page.getPixmap(matrix=trans, alpha=False)\n", + "\n", + " lurl='pdf_cp/%s.jpg' % str(pg+1).rjust(3,\"0\")\n", + " pm.writePNG(lurl)\n", + "doc.close()\n" + ] + }, + { + "cell_type": "code", + "execution_count": 42, + "id": "88d84849-5b98-48cb-86b1-e96757a8615f", + "metadata": { + "execution": { + "iopub.execute_input": "2023-01-05T05:09:41.478369Z", + "iopub.status.busy": "2023-01-05T05:09:41.477814Z", + "iopub.status.idle": "2023-01-05T05:11:07.156560Z", + "shell.execute_reply": "2023-01-05T05:11:07.155155Z", + "shell.execute_reply.started": "2023-01-05T05:09:41.478316Z" + }, + "tags": [] + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "/tmp/tmp0tguht7x\n" + ] + } + ], "source": [ "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", @@ -425,7 +476,7 @@ ")\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "with tempfile.TemporaryDirectory() as path:\n", - " images_from_path = convert_from_path('5.pdf', dpi=100,fmt='jpg', output_folder='./pic')\n", + " images_from_path = convert_from_path('中华药膳全书学做药膳不生病.pdf', dpi=300,fmt='jpg', output_folder='./pic')\n", "print(path)" ] }, @@ -650,16 +701,9 @@ }, { "cell_type": "code", - "execution_count": 38, + "execution_count": null, "id": "9957a90c-7735-4224-b5b5-4a4b1b7688c9", "metadata": { - "execution": { - "iopub.execute_input": "2022-12-12T14:12:26.428992Z", - "iopub.status.busy": "2022-12-12T14:12:26.428458Z", - "iopub.status.idle": "2022-12-12T14:12:26.536571Z", - "shell.execute_reply": "2022-12-12T14:12:26.535612Z", - "shell.execute_reply.started": "2022-12-12T14:12:26.428944Z" - }, "tags": [] }, "outputs": [], @@ -683,27 +727,12 @@ }, { "cell_type": "code", - "execution_count": 36, + "execution_count": null, "id": "42122edc-eb77-4cfc-bcd8-3f844ffcca58", "metadata": { - "execution": { - "iopub.execute_input": "2022-12-07T07:30:29.738226Z", - "iopub.status.busy": "2022-12-07T07:30:29.737684Z", - "iopub.status.idle": "2022-12-07T07:30:30.656573Z", - "shell.execute_reply": "2022-12-07T07:30:30.655231Z", - "shell.execute_reply.started": "2022-12-07T07:30:29.738176Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "转换完成\n" - ] - } - ], + "outputs": [], "source": [ "from subprocess import Popen\n", "\n", diff --git a/百度OCR.ipynb b/百度OCR.ipynb index 759336e..1e724db 100644 --- a/百度OCR.ipynb +++ b/百度OCR.ipynb @@ -49,7 +49,6 @@ "metadata": {}, "outputs": [], "source": [ - "\n", "import os,sys\n", "from aip import AipOcr\n", "\n", @@ -92,6 +91,72 @@ " print('\\n')" ] }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 将指定目录下图片文件进行文字识别保存为json文件" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import os,sys\n", + "from aip import AipOcr\n", + "import json\n", + "\n", + "def get_access_token():\n", + " client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", + " client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n", + " # client_id 为官网获取的AK, client_secret 为官网获取的SK\n", + " host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n", + " client_id, client_secret)\n", + " response = requests.get(host).text\n", + " data = json.loads(response)\n", + " access_token = data['access_token']\n", + " return access_token\n", + "\n", + "\"\"\" 你的 APPID AK SK \"\"\"\n", + "APP_ID = '17553946'\n", + "API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", + "SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", + "\n", + "client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n", + "def get_file_content(filePath):\n", + " with open(filePath, 'rb') as fp:\n", + " return fp.read()\n", + "\n", + "options = {}\n", + "options[\"language_type\"] = \"CHN_ENG\"\n", + "options[\"detect_direction\"] = \"true\"\n", + "options[\"detect_language\"] = \"true\"\n", + "options[\"probability\"] = \"true\"\n", + "dict1 = {}\n", + "fi_path = os.getcwd()+'/pic'\n", + "fl = os.listdir(fi_path)\n", + "fl.sort()\n", + "i = 1\n", + "for fl1 in fl:\n", + " file_name = fi_path+'/' + fl1\n", + " image = get_file_content(file_name)\n", + " result= client.basicGeneral(image, options)\n", + " if 'words_result' in result:\n", + " text = ''.join([w['words'] for w in result['words_result']])\n", + " \n", + " dict1[i] = text\n", + " i+=1\n", + " print(i,fl1)\n", + " #print('\\n'.join([w['words'] for w in result['words_result']]))\n", + " #print('\\n')\n", + "with open('中华药膳全书学做药膳不生病.json','w') as fl2:\n", + " json.dump(dict1,fl2,ensure_ascii=False) " + ] + }, { "cell_type": "markdown", "metadata": {}, @@ -101,26 +166,9 @@ }, { "cell_type": "code", - "execution_count": 2, - "metadata": { - "execution": { - "iopub.execute_input": "2020-11-26T08:59:12.967127Z", - "iopub.status.busy": "2020-11-26T08:59:12.966172Z", - "iopub.status.idle": "2020-11-26T08:59:28.010488Z", - "shell.execute_reply": "2020-11-26T08:59:28.006947Z", - "shell.execute_reply.started": "2020-11-26T08:59:12.967019Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "22917135_2274033\n", - "已完成\n" - ] - } - ], + "execution_count": null, + "metadata": {}, + "outputs": [], "source": [ "import requests\n", "import json\n", @@ -175,26 +223,11 @@ }, { "cell_type": "code", - "execution_count": 2, + "execution_count": null, "metadata": { - "execution": { - "iopub.execute_input": "2023-01-03T00:39:37.591809Z", - "iopub.status.busy": "2023-01-03T00:39:37.591257Z", - "iopub.status.idle": "2023-01-03T00:39:37.763387Z", - "shell.execute_reply": "2023-01-03T00:39:37.761195Z", - "shell.execute_reply.started": "2023-01-03T00:39:37.591761Z" - }, "tags": [] }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "{'refresh_token': '25.e1167a22bcc63f2e736768f11eb893f9.315360000.1988066377.282335-17553946', 'expires_in': 2592000, 'session_key': '9mzdWr+jeUXdQXD/259SQnM94l/zWgp0GyY6hh17Kt+kjRLgEQG9OYATKN1tjVBrQIdg8DnKb0OBJpbiVsQ/7ixzXqxduw==', 'access_token': '24.acdee0098fe862a7c598f62121a2baac.2592000.1675298377.282335-17553946', 'scope': 'brain_ocr_meter brain_doc_analysis brain_ocr_webimage_loc vis-ocr_机动车购车发票识别 brain_ocr_vehicle_invoice brain_formula vis-ocr_行程单识别 brain_ocr_air_ticket public vis-ocr_ocr brain_ocr_scope brain_ocr_general brain_ocr_general_basic vis-ocr_business_license brain_ocr_webimage brain_all_scope brain_ocr_idcard brain_ocr_driving_license brain_ocr_vehicle_license vis-ocr_plate_number brain_solution brain_ocr_plate_number brain_ocr_accurate brain_ocr_accurate_basic brain_ocr_receipt brain_ocr_business_license brain_solution_iocr brain_qrcode brain_ocr_handwriting brain_ocr_passport brain_ocr_vat_invoice brain_numbers brain_ocr_business_card brain_ocr_train_ticket brain_ocr_taxi_receipt vis-ocr_household_register vis-ocr_vis-classify_birth_certificate vis-ocr_台湾通行证 vis-ocr_港澳通行证 vis-ocr_机动车检验合格证识别 vis-ocr_车辆vin码识别 vis-ocr_定额发票识别 vis-ocr_保单识别 brain_ocr_vin brain_ocr_quota_invoice brain_ocr_birth_certificate brain_ocr_household_register brain_ocr_HK_Macau_pass brain_ocr_taiwan_pass brain_ocr_vehicle_certificate brain_ocr_insurance_doc wise_adapt lebo_resource_base lightservice_public hetu_basic lightcms_map_poi kaidian_kaidian ApsMisTest_Test权限 vis-classify_flower lpq_开放 cop_helloScope ApsMis_fangdi_permission smartapp_snsapi_base smartapp_mapp_dev_manage iop_autocar oauth_tp_app smartapp_smart_game_openapi oauth_sessionkey smartapp_swanid_verify smartapp_opensource_openapi smartapp_opensource_recapi fake_face_detect_开放Scope vis-ocr_虚拟人物助理 idl-video_虚拟人物助理 smartapp_component smartapp_search_plugin avatar_video_test b2b_tp_openapi b2b_tp_openapi_online smartapp_gov_aladin_to_xcx', 'session_secret': '20a3564eae255945e2a4a45ecc57c608'}\n" - ] - } - ], + "outputs": [], "source": [ "import requests \n", "\n", @@ -207,25 +240,9 @@ }, { "cell_type": "code", - "execution_count": 14, - "metadata": { - "execution": { - "iopub.execute_input": "2020-11-03T03:24:08.412670Z", - "iopub.status.busy": "2020-11-03T03:24:08.411768Z", - "iopub.status.idle": "2020-11-03T03:24:08.813580Z", - "shell.execute_reply": "2020-11-03T03:24:08.810985Z", - "shell.execute_reply.started": "2020-11-03T03:24:08.412567Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "已完成\n" - ] - } - ], + "execution_count": null, + "metadata": {}, + "outputs": [], "source": [ "import requests\n", "import json\n", @@ -251,28 +268,9 @@ }, { "cell_type": "code", - "execution_count": 80, - "metadata": { - "execution": { - "iopub.execute_input": "2020-11-03T08:43:37.180693Z", - "iopub.status.busy": "2020-11-03T08:43:37.179800Z", - "iopub.status.idle": "2020-11-03T08:43:37.656055Z", - "shell.execute_reply": "2020-11-03T08:43:37.653668Z", - "shell.execute_reply.started": "2020-11-03T08:43:37.180589Z" - } - }, - "outputs": [ - { - "data": { - "text/plain": [ - "list" - ] - }, - "execution_count": 80, - "metadata": {}, - "output_type": "execute_result" - } - ], + "execution_count": null, + "metadata": {}, + "outputs": [], "source": [ "import requests\n", "import json\n", @@ -313,25 +311,9 @@ }, { "cell_type": "code", - "execution_count": 81, - "metadata": { - "execution": { - "iopub.execute_input": "2020-11-03T08:48:26.082167Z", - "iopub.status.busy": "2020-11-03T08:48:26.081266Z", - "iopub.status.idle": "2020-11-03T08:48:26.430039Z", - "shell.execute_reply": "2020-11-03T08:48:26.428236Z", - "shell.execute_reply.started": "2020-11-03T08:48:26.082060Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "已完成\n" - ] - } - ], + "execution_count": null, + "metadata": {}, + "outputs": [], "source": [ "import requests\n", "import json\n",