14 KiB
14 KiB
In [ ]:
from aip import AipOcr
""" 你的 APPID AK SK """
APP_ID = '17553946'
API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
client = AipOcr(APP_ID, API_KEY, SECRET_KEY)
def get_file_content(filePath):
with open(filePath, 'rb') as fp:
return fp.read()
image = get_file_content('1.jpg')
""" 调用通用文字识别, 图片参数为本地图片 """
#client.basicGeneral(image);
""" 如果有可选参数 """
options = {}
options["language_type"] = "CHN_ENG"
options["detect_direction"] = "true"
options["detect_language"] = "true"
options["probability"] = "true"
""" 带参数调用通用文字识别, 图片参数为本地图片 """
result= client.basicGeneral(image, options)
if 'words_result' in result:
print('\n'.join([w['words'] for w in result['words_result']]))In [ ]:
#将指定目录下图片文件进行文字识别
import os,sys
from aip import AipOcr
""" 你的 APPID AK SK """
APP_ID = '17553946'
API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'
SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'
client = AipOcr(APP_ID, API_KEY, SECRET_KEY)
def get_file_content(filePath):
with open(filePath, 'rb') as fp:
return fp.read()
options = {}
options["language_type"] = "CHN_ENG"
options["detect_direction"] = "true"
options["detect_language"] = "true"
options["probability"] = "true"
fi_path = os.getcwd()+'/data'
fl = os.listdir(fi_path)
fl.sort()
for fl1 in fl:
file_name = fi_path+'/' + fl1
image = get_file_content(file_name)
result= client.basicGeneral(image, options)
if 'words_result' in result:
print('\n'.join([w['words'] for w in result['words_result']]))
print('\n')In [2]:
import requests
import json
import base64
import time
def get_access_token():
client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'
client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq'
# client_id 为官网获取的AK, client_secret 为官网获取的SK
host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(
client_id, client_secret)
response = requests.get(host).text
data = json.loads(response)
access_token = data['access_token']
return access_token
def get_excel(requests_id, access_token):
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': requests_id,
'result_type': 'excel'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
info_1 = res.json()['result']['ret_msg']
excel_url=res.json()['result']['result_data']
excel_1=requests.get(excel_url).content
with open('识别结果11.xls','wb+') as f:
f.write(excel_1)
print(info_1)
request_url = "https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request"
# 二进制方式打开图片文件
f = open('山东大学强基计划(2020).jpg', 'rb')
img = base64.b64encode(f.read())
params = {"image":img}
access_token = get_access_token()
request_url = request_url + "?access_token=" + access_token
headers = {'content-type': 'application/x-www-form-urlencoded'}
response = requests.post(request_url, data=params, headers=headers)
if response:
m_xx = response.json()
requests_id = m_xx['result'][0]['request_id']
print(requests_id)
time.sleep(10)
get_excel(requests_id, access_token)22917135_2274033 已完成
In [1]:
import requests
# client_id 为官网获取的AK, client_secret 为官网获取的SK
host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=KwXkGawxh0sjOQdF9Ae9LeLb&client_secret=siprEKMp5UcRTOAngEfIOOe9x6xkqGXq'
response = requests.get(host)
if response:
print(response.json()){'refresh_token': '25.1dfb7a14cd15d03051976c8db246fa92.315360000.1919729184.282335-22917135', 'expires_in': 2592000, 'session_key': '9mzdCSFczT8Mv7ZN07SjsPo0dZr0AvxAeDt6Kjn1Z9jiiotA6kW2TZYzslnuYOFd1ZCx75mmzW0TRF+nxYAHzPOb5fmjlw==', 'access_token': '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135', 'scope': 'public vis-ocr_ocr brain_ocr_scope brain_ocr_general brain_ocr_general_basic vis-ocr_business_license brain_ocr_webimage brain_all_scope brain_ocr_idcard brain_ocr_driving_license brain_ocr_vehicle_license vis-ocr_plate_number brain_solution brain_ocr_plate_number brain_ocr_accurate brain_ocr_accurate_basic brain_ocr_receipt brain_ocr_business_license brain_solution_iocr brain_qrcode brain_ocr_handwriting brain_ocr_passport brain_ocr_vat_invoice brain_numbers brain_ocr_business_card brain_ocr_train_ticket brain_ocr_taxi_receipt vis-ocr_household_register vis-ocr_vis-classify_birth_certificate vis-ocr_台湾通行证 vis-ocr_港澳通行证 vis-ocr_机动车购车发票识别 vis-ocr_机动车检验合格证识别 vis-ocr_车辆vin码识别 vis-ocr_定额发票识别 vis-ocr_保单识别 vis-ocr_机打发票识别 vis-ocr_行程单识别 brain_ocr_vin brain_ocr_quota_invoice brain_ocr_birth_certificate brain_ocr_household_register brain_ocr_HK_Macau_pass brain_ocr_taiwan_pass brain_ocr_vehicle_invoice brain_ocr_vehicle_certificate brain_ocr_air_ticket brain_ocr_invoice brain_ocr_insurance_doc brain_formula brain_ocr_meter brain_doc_analysis brain_ocr_webimage_loc wise_adapt lebo_resource_base lightservice_public hetu_basic lightcms_map_poi kaidian_kaidian ApsMisTest_Test权限 vis-classify_flower lpq_开放 cop_helloScope ApsMis_fangdi_permission smartapp_snsapi_base smartapp_mapp_dev_manage iop_autocar oauth_tp_app smartapp_smart_game_openapi oauth_sessionkey smartapp_swanid_verify smartapp_opensource_openapi smartapp_opensource_recapi fake_face_detect_开放Scope vis-ocr_虚拟人物助理 idl-video_虚拟人物助理 smartapp_component', 'session_secret': '6ac230d6705697808e228241c519f303'}
In [14]:
import requests
import json
import base64
access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': '22917135_2227436',
'result_type': 'excel'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
info_1 = res.json()['result']['ret_msg']
excel_url=res.json()['result']['result_data']
excel_1=requests.get(excel_url).content
with open('识别结果12.xls','wb+') as f:
f.write(excel_1)
print(info_1)已完成
In [80]:
import requests
import json
import base64
import demjson
access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': '22917135_2227436',
'result_type': 'json'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
#info_1 = res.json()['result']['ret_msg']
excel_1=res.json()['result']['result_data']#['forms'][0]['body']
type(excel_1)
#excel_new = demjson.decode(excel_1)
#for m_col in excel_new['forms'][0]['body']:
# print(m_col)
m_xx =json.loads(excel_1)
#with open('识别结果12.json','w') as fl:
# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)
#print(info_1)
#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))
print(m_xx['forms'][0]['body'])Out [80]:
list
In [81]:
import requests
import json
import base64
import demjson
access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'
headers = {'content-type': 'application/x-www-form-urlencoded'}
pargams = {
'request_id': '22917135_2227436',
'result_type': 'json'
}
url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'
url_all = url + "?access_token=" + access_token
res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页
#info_1 = res.json()['result']['ret_msg']
excel_1=res.json()['result']['result_data']#['forms'][0]['body']
type(excel_1)
#excel_new = demjson.decode(excel_1)
#for m_col in excel_new['forms'][0]['body']:
# print(m_col)
m_xx =json.loads(excel_1)
with open('识别结果12.json','w') as fl:
json.dump(m_xx['forms'][0],fl,ensure_ascii=False)
print(info_1)
已完成
In [ ]: