13 KiB
13 KiB
In [ ]:
from aip import AipOcr
""" 你的 APPID AK SK """
APP_ID = '17553946'
API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
client = AipOcr(APP_ID, API_KEY, SECRET_KEY)
def get_file_content(filePath):
with open(filePath, 'rb') as fp:
return fp.read()
image = get_file_content('1.jpg')
""" 调用通用文字识别, 图片参数为本地图片 """
#client.basicGeneral(image);
""" 如果有可选参数 """
options = {}
options["language_type"] = "CHN_ENG"
options["detect_direction"] = "true"
options["detect_language"] = "true"
options["probability"] = "true"
""" 带参数调用通用文字识别, 图片参数为本地图片 """
result= client.basicGeneral(image, options)
if 'words_result' in result:
print('\n'.join([w['words'] for w in result['words_result']]))In [ ]:
import os,sys
from aip import AipOcr
def get_access_token():
client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
# client_id 为官网获取的AK, client_secret 为官网获取的SK
host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(
client_id, client_secret)
response = requests.get(host).text
data = json.loads(response)
access_token = data['access_token']
return access_token
""" 你的 APPID AK SK """
APP_ID = '17553946'
API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
client = AipOcr(APP_ID, API_KEY, SECRET_KEY)
def get_file_content(filePath):
with open(filePath, 'rb') as fp:
return fp.read()
options = {}
options["language_type"] = "CHN_ENG"
options["detect_direction"] = "true"
options["detect_language"] = "true"
options["probability"] = "true"
fi_path = os.getcwd()+'/data'
fl = os.listdir(fi_path)
fl.sort()
for fl1 in fl:
file_name = fi_path+'/' + fl1
image = get_file_content(file_name)
result= client.basicGeneral(image, options)
if 'words_result' in result:
print('\n'.join([w['words'] for w in result['words_result']]))
print('\n')In [ ]:
import os,sys
from aip import AipOcr
import json
def get_access_token():
client_id = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
client_secret = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
# client_id 为官网获取的AK, client_secret 为官网获取的SK
host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(
client_id, client_secret)
response = requests.get(host).text
data = json.loads(response)
access_token = data['access_token']
return access_token
""" 你的 APPID AK SK """
APP_ID = '17553946'
API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
client = AipOcr(APP_ID, API_KEY, SECRET_KEY)
def get_file_content(filePath):
with open(filePath, 'rb') as fp:
return fp.read()
options = {}
options["language_type"] = "CHN_ENG"
options["detect_direction"] = "true"
options["detect_language"] = "true"
options["probability"] = "true"
dict1 = {}
fi_path = os.getcwd()+'/pic'
fl = os.listdir(fi_path)
fl.sort()
i = 1
for fl1 in fl:
file_name = fi_path+'/' + fl1
image = get_file_content(file_name)
result= client.basicGeneral(image, options)
if 'words_result' in result:
text = ''.join([w['words'] for w in result['words_result']])
dict1[i] = text
i+=1
print(i,fl1)
#print('\n'.join([w['words'] for w in result['words_result']]))
#print('\n')
with open('中华药膳全书学做药膳不生病.json','w') as fl2:
json.dump(dict1,fl2,ensure_ascii=False) In [ ]:
import requests
import json
import base64
import time
def get_access_token():
client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'
client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq'
# client_id 为官网获取的AK, client_secret 为官网获取的SK
host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(
client_id, client_secret)
response = requests.get(host).text
data = json.loads(response)
access_token = data['access_token']
return access_token
def get_excel(requests_id, access_token):
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': requests_id,
'result_type': 'excel'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
info_1 = res.json()['result']['ret_msg']
excel_url=res.json()['result']['result_data']
excel_1=requests.get(excel_url).content
with open('识别结果11.xls','wb+') as f:
f.write(excel_1)
print(info_1)
request_url = "https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request"
# 二进制方式打开图片文件
f = open('山东大学强基计划(2020).jpg', 'rb')
img = base64.b64encode(f.read())
params = {"image":img}
access_token = get_access_token()
request_url = request_url + "?access_token=" + access_token
headers = {'content-type': 'application/x-www-form-urlencoded'}
response = requests.post(request_url, data=params, headers=headers)
if response:
m_xx = response.json()
requests_id = m_xx['result'][0]['request_id']
print(requests_id)
time.sleep(10)
get_excel(requests_id, access_token)In [ ]:
import requests
# client_id 为官网获取的AK, client_secret 为官网获取的SK
host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=IMC1ss3Tyo3vEAdVH6jgcdv2&client_secret=Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
response = requests.get(host)
if response:
print(response.json())In [ ]:
import requests
import json
import base64
access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': '22917135_2227436',
'result_type': 'excel'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
info_1 = res.json()['result']['ret_msg']
excel_url=res.json()['result']['result_data']
excel_1=requests.get(excel_url).content
with open('识别结果12.xls','wb+') as f:
f.write(excel_1)
print(info_1)In [ ]:
import requests
import json
import base64
import demjson
access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': '22917135_2227436',
'result_type': 'json'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
#info_1 = res.json()['result']['ret_msg']
excel_1=res.json()['result']['result_data']#['forms'][0]['body']
type(excel_1)
#excel_new = demjson.decode(excel_1)
#for m_col in excel_new['forms'][0]['body']:
# print(m_col)
m_xx =json.loads(excel_1)
#with open('识别结果12.json','w') as fl:
# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)
#print(info_1)
#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))
print(m_xx['forms'][0]['body'])In [ ]:
import requests
import json
import base64
import demjson
access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': '22917135_2227436',
'result_type': 'json'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
#info_1 = res.json()['result']['ret_msg']
excel_1=res.json()['result']['result_data']#['forms'][0]['body']
type(excel_1)
#excel_new = demjson.decode(excel_1)
#for m_col in excel_new['forms'][0]['body']:
# print(m_col)
m_xx =json.loads(excel_1)
with open('识别结果12.json','w') as fl:
json.dump(m_xx['forms'][0],fl,ensure_ascii=False)
print(info_1)
In [ ]: