{ "cells": [ { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from aip import AipOcr\n", "\n", "\"\"\" 你的 APPID AK SK \"\"\"\n", "APP_ID = '17553946'\n", "API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", "SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", "\n", "client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n", "def get_file_content(filePath):\n", " with open(filePath, 'rb') as fp:\n", " return fp.read()\n", "\n", "image = get_file_content('1.jpg')\n", "\n", "\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n", "#client.basicGeneral(image);\n", "\n", "\"\"\" 如果有可选参数 \"\"\"\n", "options = {}\n", "options[\"language_type\"] = \"CHN_ENG\"\n", "options[\"detect_direction\"] = \"true\"\n", "options[\"detect_language\"] = \"true\"\n", "options[\"probability\"] = \"true\"\n", "\n", "\"\"\" 带参数调用通用文字识别, 图片参数为本地图片 \"\"\"\n", "result= client.basicGeneral(image, options)\n", "if 'words_result' in result:\n", " print('\\n'.join([w['words'] for w in result['words_result']]))" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 将指定目录下图片文件进行文字识别" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import os,sys\n", "from aip import AipOcr\n", "\n", "def get_access_token():\n", " client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", " client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n", " # client_id 为官网获取的AK, client_secret 为官网获取的SK\n", " host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n", " client_id, client_secret)\n", " response = requests.get(host).text\n", " data = json.loads(response)\n", " access_token = data['access_token']\n", " return access_token\n", "\n", "\"\"\" 你的 APPID AK SK \"\"\"\n", "APP_ID = '17553946'\n", "API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", "SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", "\n", "client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n", "def get_file_content(filePath):\n", " with open(filePath, 'rb') as fp:\n", " return fp.read()\n", "\n", "options = {}\n", "options[\"language_type\"] = \"CHN_ENG\"\n", "options[\"detect_direction\"] = \"true\"\n", "options[\"detect_language\"] = \"true\"\n", "options[\"probability\"] = \"true\"\n", "\n", "fi_path = os.getcwd()+'/data'\n", "fl = os.listdir(fi_path)\n", "fl.sort()\n", "for fl1 in fl:\n", " file_name = fi_path+'/' + fl1\n", " image = get_file_content(file_name)\n", " result= client.basicGeneral(image, options)\n", " if 'words_result' in result:\n", " print('\\n'.join([w['words'] for w in result['words_result']]))\n", " print('\\n')" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 将指定目录下图片文件进行文字识别保存为json文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import os,sys\n", "from aip import AipOcr\n", "import json\n", "\n", "def get_access_token():\n", " client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", " client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n", " # client_id 为官网获取的AK, client_secret 为官网获取的SK\n", " host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n", " client_id, client_secret)\n", " response = requests.get(host).text\n", " data = json.loads(response)\n", " access_token = data['access_token']\n", " return access_token\n", "\n", "\"\"\" 你的 APPID AK SK \"\"\"\n", "APP_ID = '17553946'\n", "API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", "SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", "\n", "client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n", "def get_file_content(filePath):\n", " with open(filePath, 'rb') as fp:\n", " return fp.read()\n", "\n", "options = {}\n", "options[\"language_type\"] = \"CHN_ENG\"\n", "options[\"detect_direction\"] = \"true\"\n", "options[\"detect_language\"] = \"true\"\n", "options[\"probability\"] = \"true\"\n", "dict1 = {}\n", "fi_path = os.getcwd()+'/pic'\n", "fl = os.listdir(fi_path)\n", "fl.sort()\n", "i = 1\n", "for fl1 in fl:\n", " file_name = fi_path+'/' + fl1\n", " image = get_file_content(file_name)\n", " result= client.basicGeneral(image, options)\n", " if 'words_result' in result:\n", " text = ''.join([w['words'] for w in result['words_result']])\n", " \n", " dict1[i] = text\n", " i+=1\n", " print(i,fl1)\n", " #print('\\n'.join([w['words'] for w in result['words_result']]))\n", " #print('\\n')\n", "with open('中华药膳全书学做药膳不生病.json','w') as fl2:\n", " json.dump(dict1,fl2,ensure_ascii=False) " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 识别图片中表格" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import json\n", "import base64\n", "import time\n", "\n", "def get_access_token():\n", " client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'\n", " client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq' \n", " # client_id 为官网获取的AK, client_secret 为官网获取的SK\n", " host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n", " client_id, client_secret)\n", " response = requests.get(host).text\n", " data = json.loads(response)\n", " access_token = data['access_token']\n", " return access_token\n", "\n", "def get_excel(requests_id, access_token):\n", " headers = {'content-type': 'application/x-www-form-urlencoded'}\n", " pargams = {\n", " 'request_id': requests_id,\n", " 'result_type': 'excel'\n", " }\n", " url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", " url_all = url + \"?access_token=\" + access_token\n", " res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", " info_1 = res.json()['result']['ret_msg']\n", " excel_url=res.json()['result']['result_data']\n", " excel_1=requests.get(excel_url).content\n", " with open('识别结果11.xls','wb+') as f:\n", " f.write(excel_1)\n", " print(info_1)\n", "\n", "\n", "request_url = \"https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request\"\n", "# 二进制方式打开图片文件\n", "f = open('山东大学强基计划(2020).jpg', 'rb')\n", "img = base64.b64encode(f.read())\n", "\n", "params = {\"image\":img}\n", "access_token = get_access_token()\n", "request_url = request_url + \"?access_token=\" + access_token\n", "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", "response = requests.post(request_url, data=params, headers=headers)\n", "if response:\n", " m_xx = response.json()\n", "requests_id = m_xx['result'][0]['request_id'] \n", "print(requests_id)\n", "time.sleep(10)\n", "get_excel(requests_id, access_token)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import requests \n", "\n", "# client_id 为官网获取的AK, client_secret 为官网获取的SK\n", "host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=IMC1ss3Tyo3vEAdVH6jgcdv2&client_secret=Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", "response = requests.get(host)\n", "if response:\n", " print(response.json())" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import json\n", "import base64\n", "access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n", "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", "pargams = {\n", " 'request_id': '22917135_2227436',\n", " 'result_type': 'excel'\n", "}\n", "\n", "\n", "url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", "url_all = url + \"?access_token=\" + access_token\n", "res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", "info_1 = res.json()['result']['ret_msg']\n", "excel_url=res.json()['result']['result_data']\n", "excel_1=requests.get(excel_url).content\n", "with open('识别结果12.xls','wb+') as f:\n", " f.write(excel_1)\n", "print(info_1)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import json\n", "import base64\n", "import demjson\n", "access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n", "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", "pargams = {\n", " 'request_id': '22917135_2227436',\n", " 'result_type': 'json'\n", "}\n", "\n", "\n", "url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", "url_all = url + \"?access_token=\" + access_token\n", "res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", "#info_1 = res.json()['result']['ret_msg']\n", "excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n", "type(excel_1)\n", "#excel_new = demjson.decode(excel_1)\n", "#for m_col in excel_new['forms'][0]['body']:\n", "# print(m_col)\n", "m_xx =json.loads(excel_1)\n", "\n", "#with open('识别结果12.json','w') as fl:\n", "# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n", "#print(info_1)\n", "#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))\n", "print(m_xx['forms'][0]['body'])" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 识别保存为json文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import json\n", "import base64\n", "import demjson\n", "access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n", "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", "pargams = {\n", " 'request_id': '22917135_2227436',\n", " 'result_type': 'json'\n", "}\n", "\n", "\n", "url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", "url_all = url + \"?access_token=\" + access_token\n", "res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", "#info_1 = res.json()['result']['ret_msg']\n", "excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n", "type(excel_1)\n", "#excel_new = demjson.decode(excel_1)\n", "#for m_col in excel_new['forms'][0]['body']:\n", "# print(m_col)\n", "m_xx =json.loads(excel_1)\n", "\n", "with open('识别结果12.json','w') as fl:\n", " json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n", "print(info_1)\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.8.10" }, "toc-autonumbering": true, "toc-showmarkdowntxt": false, "toc-showtags": false }, "nbformat": 4, "nbformat_minor": 4 }