Files
2024-07-15 23:14:31 +08:00

153 lines
4.4 KiB
Plaintext

{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"id": "f02a8ab8-2e27-493a-91d9-e0dffae99a62",
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "code",
"execution_count": null,
"id": "0d209bb5-6977-4c7c-b6d9-414e5ce7e9f1",
"metadata": {},
"outputs": [],
"source": [
"from aip import AipOcr\n",
"import glob\n",
"from opencc import OpenCC\n",
"import time\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"\n",
"fi_path = './pic'\n",
"fls = glob.glob(f'{fi_path}/*.jpg')\n",
"fls.sort()\n",
"\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
"#client.basicGeneral(image);\n",
"\n",
"\"\"\" 如果有可选参数 \"\"\"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"#options[\"detect_direction\"] = \"true\"\n",
"cc = OpenCC('t2s')\n",
"for fi in fls:\n",
" image = get_file_content(fi)\n",
" result= client.basicAccurate(image, options)\n",
" print(fi)\n",
" list1 = []\n",
" if 'words_result' in result:\n",
" for w in result['words_result']:\n",
" list1.append(cc.convert(w['words']))\n",
" print('\\n'.join(list1))\n",
" time.sleep(5)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "75e8d016-deba-4177-9036-01bfd77002a9",
"metadata": {},
"outputs": [],
"source": [
"s = '宋文帝使人齎藥賜王景文死'\n",
"print(cc.convert(s))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d4df9553-0ded-4c40-bf89-38ee90feecbd",
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import types\n",
"from tencentcloud.common import credential\n",
"from tencentcloud.common.profile.client_profile import ClientProfile\n",
"from tencentcloud.common.profile.http_profile import HttpProfile\n",
"from tencentcloud.common.exception.tencent_cloud_sdk_exception import TencentCloudSDKException\n",
"from tencentcloud.ocr.v20181119 import ocr_client, models\n",
"import base64\n",
" \n",
"cred = credential.Credential(\"AKIDoZHHEB2lHbluEZvN8uYHwcdvacqfmDeJ\",\"fEkr2VucwEQXXVN8jCd0CFBxdAe6Xny9\")\n",
"httpProfile = HttpProfile()\n",
"httpProfile.endpoint = \"ocr.tencentcloudapi.com\" \n",
"clientProfile = ClientProfile()\n",
"clientProfile.httpProfile = httpProfile \n",
"client = ocr_client.OcrClient(cred, \"ap-beijing\", clientProfile) \n",
"req = models.GeneralAccurateOCRRequest()\n",
"with open(\"pic/5002.jpg\",\"rb\") as f:\n",
" img_data = f.read()\n",
"img_base64 = base64.b64encode(img_data)\n",
"params = {\n",
" \"ImageBase64\": img_base64.decode('utf-8')\n",
"}\n",
"req.from_json_string(json.dumps(params))\n",
"\n",
"# 返回的resp是一个GeneralAccurateOCRResponse的实例,与请求对象对应\n",
"resp = client.GeneralAccurateOCR(req)\n",
"txt = resp.to_json_string()\n",
"print(type(txt))\n",
"\n",
"#except TencentCloudSDKException as err:\n",
"# print(err)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ed33ec27-e4b3-4966-a64a-2d5079d57cf3",
"metadata": {},
"outputs": [],
"source": [
"dict1 = json.loads(txt)\n",
"for item in dict1['TextDetections']:\n",
" print(item['DetectedText'])"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "fd1235fb-213f-443c-8b24-66d245d354c2",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.12.3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}