379 lines
13 KiB
Plaintext
379 lines
13 KiB
Plaintext
{
|
||
"cells": [
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"from aip import AipOcr\n",
|
||
"\n",
|
||
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||
"APP_ID = '17553946'\n",
|
||
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||
"\n",
|
||
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||
"def get_file_content(filePath):\n",
|
||
" with open(filePath, 'rb') as fp:\n",
|
||
" return fp.read()\n",
|
||
"\n",
|
||
"image = get_file_content('1.jpg')\n",
|
||
"\n",
|
||
"\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
|
||
"#client.basicGeneral(image);\n",
|
||
"\n",
|
||
"\"\"\" 如果有可选参数 \"\"\"\n",
|
||
"options = {}\n",
|
||
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||
"options[\"detect_direction\"] = \"true\"\n",
|
||
"options[\"detect_language\"] = \"true\"\n",
|
||
"options[\"probability\"] = \"true\"\n",
|
||
"\n",
|
||
"\"\"\" 带参数调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
|
||
"result= client.basicGeneral(image, options)\n",
|
||
"if 'words_result' in result:\n",
|
||
" print('\\n'.join([w['words'] for w in result['words_result']]))"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 将指定目录下图片文件进行文字识别"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os,sys\n",
|
||
"from aip import AipOcr\n",
|
||
"\n",
|
||
"def get_access_token():\n",
|
||
" client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||
" client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n",
|
||
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
|
||
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
|
||
" client_id, client_secret)\n",
|
||
" response = requests.get(host).text\n",
|
||
" data = json.loads(response)\n",
|
||
" access_token = data['access_token']\n",
|
||
" return access_token\n",
|
||
"\n",
|
||
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||
"APP_ID = '17553946'\n",
|
||
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||
"\n",
|
||
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||
"def get_file_content(filePath):\n",
|
||
" with open(filePath, 'rb') as fp:\n",
|
||
" return fp.read()\n",
|
||
"\n",
|
||
"options = {}\n",
|
||
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||
"options[\"detect_direction\"] = \"true\"\n",
|
||
"options[\"detect_language\"] = \"true\"\n",
|
||
"options[\"probability\"] = \"true\"\n",
|
||
"\n",
|
||
"fi_path = os.getcwd()+'/data'\n",
|
||
"fl = os.listdir(fi_path)\n",
|
||
"fl.sort()\n",
|
||
"for fl1 in fl:\n",
|
||
" file_name = fi_path+'/' + fl1\n",
|
||
" image = get_file_content(file_name)\n",
|
||
" result= client.basicGeneral(image, options)\n",
|
||
" if 'words_result' in result:\n",
|
||
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
|
||
" print('\\n')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 将指定目录下图片文件进行文字识别保存为json文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import os,sys\n",
|
||
"from aip import AipOcr\n",
|
||
"import json\n",
|
||
"\n",
|
||
"def get_access_token():\n",
|
||
" client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||
" client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ' \n",
|
||
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
|
||
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
|
||
" client_id, client_secret)\n",
|
||
" response = requests.get(host).text\n",
|
||
" data = json.loads(response)\n",
|
||
" access_token = data['access_token']\n",
|
||
" return access_token\n",
|
||
"\n",
|
||
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||
"APP_ID = '17553946'\n",
|
||
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||
"\n",
|
||
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||
"def get_file_content(filePath):\n",
|
||
" with open(filePath, 'rb') as fp:\n",
|
||
" return fp.read()\n",
|
||
"\n",
|
||
"options = {}\n",
|
||
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||
"options[\"detect_direction\"] = \"true\"\n",
|
||
"options[\"detect_language\"] = \"true\"\n",
|
||
"options[\"probability\"] = \"true\"\n",
|
||
"dict1 = {}\n",
|
||
"fi_path = os.getcwd()+'/pic'\n",
|
||
"fl = os.listdir(fi_path)\n",
|
||
"fl.sort()\n",
|
||
"i = 1\n",
|
||
"for fl1 in fl:\n",
|
||
" file_name = fi_path+'/' + fl1\n",
|
||
" image = get_file_content(file_name)\n",
|
||
" result= client.basicGeneral(image, options)\n",
|
||
" if 'words_result' in result:\n",
|
||
" text = ''.join([w['words'] for w in result['words_result']])\n",
|
||
" \n",
|
||
" dict1[i] = text\n",
|
||
" i+=1\n",
|
||
" print(i,fl1)\n",
|
||
" #print('\\n'.join([w['words'] for w in result['words_result']]))\n",
|
||
" #print('\\n')\n",
|
||
"with open('中华药膳全书学做药膳不生病.json','w') as fl2:\n",
|
||
" json.dump(dict1,fl2,ensure_ascii=False) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 识别图片中表格"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import json\n",
|
||
"import base64\n",
|
||
"import time\n",
|
||
"\n",
|
||
"def get_access_token():\n",
|
||
" client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'\n",
|
||
" client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq' \n",
|
||
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
|
||
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
|
||
" client_id, client_secret)\n",
|
||
" response = requests.get(host).text\n",
|
||
" data = json.loads(response)\n",
|
||
" access_token = data['access_token']\n",
|
||
" return access_token\n",
|
||
"\n",
|
||
"def get_excel(requests_id, access_token):\n",
|
||
" headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||
" pargams = {\n",
|
||
" 'request_id': requests_id,\n",
|
||
" 'result_type': 'excel'\n",
|
||
" }\n",
|
||
" url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||
" url_all = url + \"?access_token=\" + access_token\n",
|
||
" res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||
" info_1 = res.json()['result']['ret_msg']\n",
|
||
" excel_url=res.json()['result']['result_data']\n",
|
||
" excel_1=requests.get(excel_url).content\n",
|
||
" with open('识别结果11.xls','wb+') as f:\n",
|
||
" f.write(excel_1)\n",
|
||
" print(info_1)\n",
|
||
"\n",
|
||
"\n",
|
||
"request_url = \"https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request\"\n",
|
||
"# 二进制方式打开图片文件\n",
|
||
"f = open('山东大学强基计划(2020).jpg', 'rb')\n",
|
||
"img = base64.b64encode(f.read())\n",
|
||
"\n",
|
||
"params = {\"image\":img}\n",
|
||
"access_token = get_access_token()\n",
|
||
"request_url = request_url + \"?access_token=\" + access_token\n",
|
||
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||
"response = requests.post(request_url, data=params, headers=headers)\n",
|
||
"if response:\n",
|
||
" m_xx = response.json()\n",
|
||
"requests_id = m_xx['result'][0]['request_id'] \n",
|
||
"print(requests_id)\n",
|
||
"time.sleep(10)\n",
|
||
"get_excel(requests_id, access_token)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests \n",
|
||
"\n",
|
||
"# client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
|
||
"host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=IMC1ss3Tyo3vEAdVH6jgcdv2&client_secret=Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||
"response = requests.get(host)\n",
|
||
"if response:\n",
|
||
" print(response.json())"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import json\n",
|
||
"import base64\n",
|
||
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
|
||
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||
"pargams = {\n",
|
||
" 'request_id': '22917135_2227436',\n",
|
||
" 'result_type': 'excel'\n",
|
||
"}\n",
|
||
"\n",
|
||
"\n",
|
||
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||
"url_all = url + \"?access_token=\" + access_token\n",
|
||
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||
"info_1 = res.json()['result']['ret_msg']\n",
|
||
"excel_url=res.json()['result']['result_data']\n",
|
||
"excel_1=requests.get(excel_url).content\n",
|
||
"with open('识别结果12.xls','wb+') as f:\n",
|
||
" f.write(excel_1)\n",
|
||
"print(info_1)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import json\n",
|
||
"import base64\n",
|
||
"import demjson\n",
|
||
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
|
||
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||
"pargams = {\n",
|
||
" 'request_id': '22917135_2227436',\n",
|
||
" 'result_type': 'json'\n",
|
||
"}\n",
|
||
"\n",
|
||
"\n",
|
||
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||
"url_all = url + \"?access_token=\" + access_token\n",
|
||
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||
"#info_1 = res.json()['result']['ret_msg']\n",
|
||
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
|
||
"type(excel_1)\n",
|
||
"#excel_new = demjson.decode(excel_1)\n",
|
||
"#for m_col in excel_new['forms'][0]['body']:\n",
|
||
"# print(m_col)\n",
|
||
"m_xx =json.loads(excel_1)\n",
|
||
"\n",
|
||
"#with open('识别结果12.json','w') as fl:\n",
|
||
"# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
|
||
"#print(info_1)\n",
|
||
"#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))\n",
|
||
"print(m_xx['forms'][0]['body'])"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 识别保存为json文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import json\n",
|
||
"import base64\n",
|
||
"import demjson\n",
|
||
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
|
||
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||
"pargams = {\n",
|
||
" 'request_id': '22917135_2227436',\n",
|
||
" 'result_type': 'json'\n",
|
||
"}\n",
|
||
"\n",
|
||
"\n",
|
||
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||
"url_all = url + \"?access_token=\" + access_token\n",
|
||
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||
"#info_1 = res.json()['result']['ret_msg']\n",
|
||
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
|
||
"type(excel_1)\n",
|
||
"#excel_new = demjson.decode(excel_1)\n",
|
||
"#for m_col in excel_new['forms'][0]['body']:\n",
|
||
"# print(m_col)\n",
|
||
"m_xx =json.loads(excel_1)\n",
|
||
"\n",
|
||
"with open('识别结果12.json','w') as fl:\n",
|
||
" json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
|
||
"print(info_1)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
}
|
||
],
|
||
"metadata": {
|
||
"kernelspec": {
|
||
"display_name": "Python 3",
|
||
"language": "python",
|
||
"name": "python3"
|
||
},
|
||
"language_info": {
|
||
"codemirror_mode": {
|
||
"name": "ipython",
|
||
"version": 3
|
||
},
|
||
"file_extension": ".py",
|
||
"mimetype": "text/x-python",
|
||
"name": "python",
|
||
"nbconvert_exporter": "python",
|
||
"pygments_lexer": "ipython3",
|
||
"version": "3.8.10"
|
||
},
|
||
"toc-autonumbering": true,
|
||
"toc-showmarkdowntxt": false,
|
||
"toc-showtags": false
|
||
},
|
||
"nbformat": 4,
|
||
"nbformat_minor": 4
|
||
}
|