Files
jupyter/百度OCR.ipynb
2023-02-09 07:38:10 +08:00

400 lines
13 KiB
Plaintext
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from aip import AipOcr\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"image = get_file_content('1.jpg')\n",
"\n",
"\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
"#client.basicGeneral(image);\n",
"\n",
"\"\"\" 如果有可选参数 \"\"\"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"\n",
"\"\"\" 带参数调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
"result= client.basicGeneral(image, options)\n",
"if 'words_result' in result:\n",
" print('\\n'.join([w['words'] for w in result['words_result']]))"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 将指定目录下图片文件进行文字识别"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os,sys\n",
"from aip import AipOcr\n",
"\n",
"def get_access_token():\n",
" client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
" client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n",
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
" client_id, client_secret)\n",
" response = requests.get(host).text\n",
" data = json.loads(response)\n",
" access_token = data['access_token']\n",
" return access_token\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"\n",
"fi_path = os.getcwd()+'/data'\n",
"fl = os.listdir(fi_path)\n",
"fl.sort()\n",
"for fl1 in fl:\n",
" file_name = fi_path+'/' + fl1\n",
" image = get_file_content(file_name)\n",
" result= client.basicGeneral(image, options)\n",
" if 'words_result' in result:\n",
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
" print('\\n')"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 将指定目录下图片文件进行文字识别保存为json文件"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {
"execution": {
"iopub.execute_input": "2023-01-31T07:25:58.845092Z",
"iopub.status.busy": "2023-01-31T07:25:58.844564Z",
"iopub.status.idle": "2023-01-31T07:26:05.074689Z",
"shell.execute_reply": "2023-01-31T07:26:05.072976Z",
"shell.execute_reply.started": "2023-01-31T07:25:58.845044Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"2 扫描全能王 2023-01-31 15.17_1.jpg\n",
"3 扫描全能王 2023-01-31 15.17_2.jpg\n",
"4 扫描全能王 2023-01-31 15.17_3.jpg\n",
"5 扫描全能王 2023-01-31 15.17_4.jpg\n",
"6 扫描全能王 2023-01-31 15.17_5.jpg\n",
"7 扫描全能王 2023-01-31 15.17_6.jpg\n",
"8 扫描全能王 2023-01-31 15.17_7.jpg\n"
]
}
],
"source": [
"import os,sys\n",
"from aip import AipOcr\n",
"import json\n",
"\n",
"def get_access_token():\n",
" client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
" client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n",
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
" client_id, client_secret)\n",
" response = requests.get(host).text\n",
" data = json.loads(response)\n",
" access_token = data['access_token']\n",
" return access_token\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"dict1 = {}\n",
"fi_path = os.getcwd()+'/data/pic'\n",
"fl = os.listdir(fi_path)\n",
"fl.sort()\n",
"i = 1\n",
"for fl1 in fl:\n",
" file_name = fi_path+'/' + fl1\n",
" image = get_file_content(file_name)\n",
" result= client.basicGeneral(image, options)\n",
" if 'words_result' in result:\n",
" text = ''.join([w['words'] for w in result['words_result']])\n",
" \n",
" dict1[i] = text\n",
" i+=1\n",
" print(i,fl1)\n",
" #print('\\n'.join([w['words'] for w in result['words_result']]))\n",
" #print('\\n')\n",
"with open('合同.json','w') as fl2:\n",
" json.dump(dict1,fl2,ensure_ascii=False) "
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 识别图片中表格"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"import time\n",
"\n",
"def get_access_token():\n",
" client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'\n",
" client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq' \n",
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
" client_id, client_secret)\n",
" response = requests.get(host).text\n",
" data = json.loads(response)\n",
" access_token = data['access_token']\n",
" return access_token\n",
"\n",
"def get_excel(requests_id, access_token):\n",
" headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
" pargams = {\n",
" 'request_id': requests_id,\n",
" 'result_type': 'excel'\n",
" }\n",
" url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
" url_all = url + \"?access_token=\" + access_token\n",
" res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
" info_1 = res.json()['result']['ret_msg']\n",
" excel_url=res.json()['result']['result_data']\n",
" excel_1=requests.get(excel_url).content\n",
" with open('识别结果11.xls','wb+') as f:\n",
" f.write(excel_1)\n",
" print(info_1)\n",
"\n",
"\n",
"request_url = \"https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request\"\n",
"# 二进制方式打开图片文件\n",
"f = open('山东大学强基计划(2020).jpg', 'rb')\n",
"img = base64.b64encode(f.read())\n",
"\n",
"params = {\"image\":img}\n",
"access_token = get_access_token()\n",
"request_url = request_url + \"?access_token=\" + access_token\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"response = requests.post(request_url, data=params, headers=headers)\n",
"if response:\n",
" m_xx = response.json()\n",
"requests_id = m_xx['result'][0]['request_id'] \n",
"print(requests_id)\n",
"time.sleep(10)\n",
"get_excel(requests_id, access_token)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import requests \n",
"\n",
"# client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
"host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=IMC1ss3Tyo3vEAdVH6jgcdv2&client_secret=Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"response = requests.get(host)\n",
"if response:\n",
" print(response.json())"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"pargams = {\n",
" 'request_id': '22917135_2227436',\n",
" 'result_type': 'excel'\n",
"}\n",
"\n",
"\n",
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
"url_all = url + \"?access_token=\" + access_token\n",
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
"info_1 = res.json()['result']['ret_msg']\n",
"excel_url=res.json()['result']['result_data']\n",
"excel_1=requests.get(excel_url).content\n",
"with open('识别结果12.xls','wb+') as f:\n",
" f.write(excel_1)\n",
"print(info_1)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"import demjson\n",
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"pargams = {\n",
" 'request_id': '22917135_2227436',\n",
" 'result_type': 'json'\n",
"}\n",
"\n",
"\n",
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
"url_all = url + \"?access_token=\" + access_token\n",
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
"#info_1 = res.json()['result']['ret_msg']\n",
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
"type(excel_1)\n",
"#excel_new = demjson.decode(excel_1)\n",
"#for m_col in excel_new['forms'][0]['body']:\n",
"# print(m_col)\n",
"m_xx =json.loads(excel_1)\n",
"\n",
"#with open('识别结果12.json','w') as fl:\n",
"# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
"#print(info_1)\n",
"#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))\n",
"print(m_xx['forms'][0]['body'])"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 识别保存为json文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"import demjson\n",
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"pargams = {\n",
" 'request_id': '22917135_2227436',\n",
" 'result_type': 'json'\n",
"}\n",
"\n",
"\n",
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
"url_all = url + \"?access_token=\" + access_token\n",
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
"#info_1 = res.json()['result']['ret_msg']\n",
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
"type(excel_1)\n",
"#excel_new = demjson.decode(excel_1)\n",
"#for m_col in excel_new['forms'][0]['body']:\n",
"# print(m_col)\n",
"m_xx =json.loads(excel_1)\n",
"\n",
"with open('识别结果12.json','w') as fl:\n",
" json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
"print(info_1)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
},
"toc-autonumbering": true,
"toc-showmarkdowntxt": false,
"toc-showtags": false
},
"nbformat": 4,
"nbformat_minor": 4
}