Files
jupyter/文件操作.ipynb
T
2024-09-18 10:19:45 +08:00

1724 lines
49 KiB
Plaintext

{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 数字文件名转换为文本文件名"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 数字文件名转换为文本文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd()+'/file/北海石化监测花名册2022 .xlsx'\n",
"fi_name = {}\n",
"fi_path = os.getcwd()+'/file/221103'\n",
"old = []\n",
"new = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1):\n",
" if sheet.cell(n,3).value is not None: \n",
" m_name = sheet.cell(n,6).value.strip()\n",
" m_depart = sheet.cell(n,3).value.strip() \n",
" depart.append(m_depart) \n",
" dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n",
" #print()\n",
"# 创建部门办公室 \n",
"m_path = os.getcwd()+'/file/221103/new'\n",
"for pn in depart:\n",
" if not os.path.exists(m_path + '/' + pn):\n",
" os.mkdir(m_path + '/' + pn)\n",
"#print(dict1)\n",
"\n",
"\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn)\n",
" #print(fn)\n",
"old.sort()\n",
" #print(str(nfn)+'.pdf')\n",
"\n",
"for n in old:\n",
" \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(old)\n",
"\n",
"#print(dict1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 目录文件按照文件名排序"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"\n",
"import os,sys\n",
"\n",
"\n",
"fi_xls = 'test1.xlsx'\n",
"fi_name = {}\n",
"#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n",
"fi_path = os.getcwd()+'/data'\n",
"old = []\n",
"new = []\n",
"\n",
"fl=os.listdir(fi_path)\n",
"fl.sort()\n",
"n = 0\n",
"for i in fl:\n",
" oldname=fl[n]\n",
" name, suffix = os.path.splitext(oldname)\n",
" if name in old:\n",
" new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" old_name = fi_path+ os.sep + fl[n]\n",
" os.rename(old_name,new_name)\n",
" n+= 1\n",
"fl"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 将pdf文件转为图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n",
"print(path)\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 新建二维码图片pdf文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from fpdf import FPDF\n",
"import warnings\n",
"\n",
"\n",
"warnings.filterwarnings(\"ignore\")\n",
"pdf = FPDF()\n",
"# compression is not yet supported in py3k version\n",
"pdf.compress = False\n",
"pdf.add_page()\n",
"# Unicode is not yet supported in the py3k version; use windows-1252 standard font\n",
"pdf.add_font('ali-regular', '', r\"fonts/AlibabaPuHuiTi-2-55-Regular.ttf\", uni=True)\n",
"pdf.set_font('ali-regular', '', 14) \n",
"#pdf.ln(10)\n",
"#pdf.write(5, '欢迎')\n",
"pdf.text(20,280,'欢迎来到中国,中国欢迎您!')\n",
"pdf.image(\"data/water/平和体质.png\", 50, 250,20)\n",
"\n",
"pdf.output('py3k.pdf', 'F')"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 图像文件夹打包"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import zipfile\n",
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"import shutil\n",
"import time\n",
"\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"def addfile(zipfilename, dirname):\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"def make_path(p):\n",
" if os.path.exists(p): # 判断文件夹是否存在\n",
" shutil.rmtree(p) # 删除文件夹\n",
" os.mkdir(p) \n",
"pdf_file = '2.pdf'\n",
"output_folder='./pic1'\n",
"zip_file = 'ribenweiqishihua.zip'\n",
"make_path(output_folder)\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n",
"compress_file(zip_file, output_folder) # 执行函数\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from opencc import OpenCC\n",
"\n",
"cc = OpenCC('t2s')\n",
"cc.convert(\"此亦寶力菱成上所不可缺者兹舉其基本變化十\")\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 文本文件操作"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 基本读取"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import re\n",
"file_name = 'data/2012.txt'\n",
"with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n",
" for l in fl:\n",
" l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n",
" if l.split():\n",
" print(l)\n",
" fl1.write(l)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 读取csv文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import csv\n",
"import re\n",
"\n",
"filename = 'data/130.csv'\n",
"with open(filename,'r',newline='') as csv_file:\n",
" fl = csv.reader(csv_file,delimiter=',')\n",
" header = next(fl) \n",
" for line in fl:\n",
" #line = re.sub('[\\r\\n\\f ]{1,}', '', line)\n",
" print(line)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 读取分隔符分割文件,导入MongoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"import re\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"city\"]\n",
"m_mongo = {}\n",
"m_xx = []\n",
"fl_name = 'china-city-list.txt'\n",
"n = 0\n",
"with open(fl_name,'r') as fl:\n",
" for l in fl:\n",
" n += 1\n",
" if n >6:\n",
" m_mongo = {}\n",
" m_xx = re.sub('[ ]{1,}', '', l).split('|')\n",
" #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n",
" m_mongo['name'] = m_xx[3]\n",
" m_mongo['code'] = m_xx[1]\n",
" m_mongo['sheng'] = m_xx[8]\n",
" m_mongo['shi'] = m_xx[10]\n",
" m_mongo['jing'] = m_xx[11]\n",
" m_mongo['wei'] = m_xx[12]\n",
" mycol.insert_one(m_mongo) \n",
" #print(m_mongo)\n",
"print('ok!')\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 整理文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import re\n",
"file_name = 'file/三国演义.txt'\n",
"list1 = []\n",
"with open(file_name,'r') as fl:\n",
" for l in fl:\n",
" if len(l.strip())>0:\n",
" list1.append(l.strip()+'\\n')\n",
"with open('file/三国演义_new.txt','w') as fl1:\n",
" fl1.writelines(list1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Excel文件修改"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import openpyxl\n",
"import json\n",
"\n",
"filename = 'data/122.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"dict2 = {}\n",
"for k, v in dict1.items():\n",
" for item in v:\n",
" bh = item['avatar_id']\n",
" dict2[bh] = item\n",
"print(dict2) \n",
" \n",
"wb = openpyxl.load_workbook('data/济南炼化职工分组表.xlsx')\n",
"sheet = wb.active\n",
"person = {}\n",
"for n in range(2, sheet.max_row+1):\n",
" if sheet.cell(n, 14).value is not None and sheet.cell(n, 14).value in dict2.keys():\n",
" code = sheet.cell(n, 14).value\n",
" sheet[f'F{n}']=dict2[code]['birth']\n",
"wb.save('data/济南炼化职工分组表1.xlsx') \n",
"print('ok!') \n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Excel文件内容生成markdown格式文本"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"\n",
"fi_xls = 'data/对局目录.xlsx'\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb['卷18'] \n",
"\n",
"\n",
"for n in range(sheet.max_row - 1,sheet.max_row+1): \n",
" print('## 第'+sheet.cell(n,1).value+'局 ')\n",
" print(sheet.cell(n,2).value+' ')\n",
" print(sheet.cell(n,3).value+' ')\n",
" print('共'+sheet.cell(n,4).value+'着 ')\n",
" if sheet.cell(n,5).value is not None:\n",
" print(sheet.cell(n,5).value+' ')\n",
" print(sheet.cell(n,6).value+' ')\n",
" print('\\n')"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"\n",
"fi_xls = 'data/对局目录.xlsx'\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb['卷24'] \n",
"\n",
"\n",
"for n in range(50,562): \n",
" print(sheet.cell(n,1).value+' ')\n",
" if sheet.cell(n,2).value is not None:\n",
" print(sheet.cell(n,2).value+' ')\n",
" if sheet.cell(n,3).value is not None:\n",
" print(sheet.cell(n,3).value+' ')\n",
" print('\\n')"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"\n",
"fi_xls = 'data/对局目录.xlsx'\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb['卷22'] \n",
"\n",
"\n",
"for n in range(2,sheet.max_row+1): \n",
" print('## 第'+sheet.cell(n,1).value+'局 ')\n",
" print('第'+sheet.cell(n,2).value+'局 ')\n",
" print('共'+sheet.cell(n,3).value+'着 ')\n",
" print(sheet.cell(n,4).value+' ')\n",
" print(sheet.cell(n,5).value+' ')\n",
" print('\\n')"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"\n",
"fi_xls = 'data/对局目录.xlsx'\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb['卷24'] \n",
"\n",
"\n",
"for n in range(50,562): \n",
" #print(sheet.cell(n,1).value+' ')\n",
" if sheet.cell(n,2).value is not None:\n",
" print(sheet.cell(n,1).value+' '+sheet.cell(n,2).value+' '+sheet.cell(n,3).value+'局')\n",
" else:\n",
" print(sheet.cell(n,1).value+' ')\n",
" \n",
" "
]
},
{
"cell_type": "code",
"execution_count": 14,
"metadata": {
"execution": {
"iopub.execute_input": "2024-09-01T03:46:45.192219Z",
"iopub.status.busy": "2024-09-01T03:46:45.191408Z",
"iopub.status.idle": "2024-09-01T03:46:45.305652Z",
"shell.execute_reply": "2024-09-01T03:46:45.305039Z",
"shell.execute_reply.started": "2024-09-01T03:46:45.192134Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"苏揆之 \n",
"邵文远 \n",
"朱缵公 \n",
"范胜林 \n",
"邱商贤 \n",
"胡肇麟 盐商,性酷嗜弈,有胡铁头之称。\n",
"僧贯如 著有《贯如弈谱》\n",
"金仲柳 \n",
"姜杰士 \n",
"金廷三 \n",
"金玠文 \n",
"赖乔山 \n",
"沈衡之 \n",
"李锦文 \n",
"臧念宣 著有《弈理析疑》行世。\n",
"陈苑游 \n",
"吴凤来 \n",
"韩学元 \n",
"黄及侣 \n",
"俞永嘉 字长候,湘人。弈品第三,为范西屏之师,施襄夏幼时亦当师之。\n",
"童和衷 \n",
"张丹九 著有《三张弈谱》。\n",
"郭璜友 \n",
"陈九如 \n",
"洪羽翔 \n",
"吴莼圃 \n",
"罗允升 \n",
"卞子兰 \n",
"江星若 \n",
"朱天植 \n",
"郑涟漪 \n",
"李步青 吴修圃云:忆三十年前,步青遇西屏于金陵,受二子共六局,胜负参半。越二年,步青复于吴越对弈四局,受先亦互有胜负,近年所诣益,未知视西屏奚若矣。\n",
"周春来 \n",
"何右林 名耕书,为范西屏高足弟子。\n",
"吴冠三 \n",
"金在田 \n",
"谭揆士 \n",
"钮亮周 \n",
"郭溶川 \n",
"洪羽翔 (注:重复)\n",
"范紫纶 \n",
"卜沧如 \n",
"姚聘三 \n",
"顾审音 \n",
"黄友功 \n",
"张廷彦 \n",
"胡敬孚 疑即胡肇麟\n",
"黄掌纶 \n",
"陈㻙琪 \n",
"沈重伦 \n",
"祥成 \n",
"格臈林阿 \n",
"方俊臣 \n",
"王小兰 \n",
"陈德崇 \n",
"僧慢得 辑有《空中楼阁弈谱》。\n",
"袁岭崇 \n",
"陈复初 \n",
"永治庵 \n",
"德乐庵 \n",
"于清和 \n",
"王公锡 \n",
"吴是中 \n",
"郑液池 \n",
"徐履中 \n",
"何朂庵 \n",
"陈炳麟 \n",
"张景福 \n",
"纪懋斋 \n",
"章芝楣 与楚桐隐合评《潘景斋弈谱约选》。\n",
"关介田 \n",
"严德音 \n",
"朱锦川 \n",
"陈天汉 \n",
"高九锡 \n",
"温佩良 粤东人,辑有《弈彙》上下册。\n",
"徐艺斋 \n",
"刘勤垣 \n",
"丁礼民 \n",
"史莲叔 \n",
"何见周 \n",
"徐鹤年 \n",
"孙春沂 \n",
"方荫臣 \n",
"戴星门 \n",
"郭云海 \n",
"丁剑侯 \n",
"方秋客 \n",
"刘春圃 \n",
"何不药 \n",
"曹丽塘 \n",
"潘秀甫 \n",
"钱芳圃 著有《钱芳圃弈谱》。\n",
"刘叔伦 名绪,江西南丰人,咸丰庚申进士,刑部郎中,官至大理寺少卿,寿八十有奇,著有《如松堂弈谱》一卷。\n",
"松茂亭 名桂,满洲人。\n",
"刘云峰 直隶人,居京师。\n",
"汪叙诗 安徽歙县人。著有《汪叙诗弈谱》。\n",
"诸弈者姓名 同前,以时代为次\n",
"王性有 \n",
"嵇修五 \n",
"叶岐山 \n",
"江溯岷 \n",
"孙茂臣 \n",
"黄隐求 \n",
"虞象明 \n",
"林葛陂 \n",
"凌葛坡 疑即林葛陂之误。\n",
"尤子成 \n",
"曹禹玉 \n",
"高景于 \n",
"程九仪 \n",
"马又白 \n",
"陆韶九 \n",
"沈子尚 \n",
"高峻宇 疑即高景于之误。\n",
"王六吉 \n",
"徐景周 \n",
"李梅崖 \n",
"王季修 \n",
"张履贞 \n",
"张兼善 \n",
"吴佩六 \n",
"程鸿儒 \n",
"杨景先 \n",
"周芳政 \n",
"赵端木 \n",
"邵次珩 \n",
"顾廷光 \n",
"周东艺 \n",
"吴虞臣 \n",
"邵含章 \n",
"程乐田 \n",
"李景文 \n",
"刘继武 \n",
"张公书 \n",
"钱东汇 著有《残局类选》。\n",
"陈耆年 \n",
"邵翼圣 \n",
"倪克让 \n",
"段朗卿 \n",
"李义宾 \n",
"黄友南 \n",
"郭虹如 \n",
"韩柱臣 \n",
"汪掌纶 \n",
"卞立言 著有《弈萃官子》二卷。\n",
"黄经文 \n",
"萧洛理 \n",
"胡位三 \n",
"张振西 \n",
"陈洪 \n",
"石某 \n",
"周果亭 \n",
"王月山 \n",
"宋鸣歧 \n",
"陈廷桂 \n",
"马立亭 \n",
"王莘臣 \n",
"云脉 \n",
"陆士经 \n",
"刘宽夫 \n",
"达顺成 \n",
"陈冶夫 \n",
"王晋风 \n",
"萧菊如 \n",
"许蔬庵 \n",
"巴德符 \n",
"唐蟾之 一作蝉枝。\n",
"党苍云 \n",
"王廷璧 \n",
"方作周 \n",
"程序东 \n",
"刘馨斋 \n",
"刘克有 \n",
"僧广智 \n",
"严弈山 \n",
"杨介亭 \n",
"宋致堂 \n",
"善庆斋 \n",
"傅崧泉 \n",
"格 疑即格臈林阿\n",
"汪季庵 \n",
"宋粹生 \n",
"邵莘甫 \n",
"赵晋卿 \n",
"谭和伯 \n",
"日本弈者姓字 \n",
"林门入 \n",
"道硕 \n",
"林朴入 \n",
"算哲 \n",
"算智 \n",
"因策 \n",
"哲斋 \n",
"道悦 \n",
"板垣 \n",
"山家 \n",
"愚硕 \n",
"知哲 \n",
"道策 \n",
"南里 \n",
"朝寻 \n",
"俊知 \n",
"道砂 \n",
"道的 \n",
"道节 \n",
"八硕 \n",
"道智 \n",
"可儿 \n",
"因入 \n",
"井上安节 \n",
"丈和 \n",
"立彻 \n",
"奥贯智策 \n",
"松原南冈 \n",
"舟桥元美 \n",
"水谷琢顺 \n",
"琉球弈者姓字 \n",
"津波 \n",
"元才 \n",
"宫城 \n",
"祝岭 \n",
"孙思忠 \n",
"向盛保 \n",
"杨光裕 \n",
"武邦瑞 \n"
]
}
],
"source": [
"import openpyxl\n",
"\n",
"fi_xls = 'data/对局目录.xlsx'\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb['Sheet2'] \n",
"\n",
"\n",
"for n in range(2,sheet.max_row+1): \n",
" #print(sheet.cell(n,1).value+' ')\n",
" if sheet.cell(n,2).value is not None:\n",
" print(sheet.cell(n,1).value+' '+sheet.cell(n,2).value)\n",
" else:\n",
" print(sheet.cell(n,1).value+' ')"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## YALM文件操作"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### YALM文件读取转存JSON文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import yaml\n",
"import json\n",
"\n",
"filename = 'data/全维健康中医体质检测问卷.yaml'\n",
"with open(filename, 'r', encoding='utf-8') as f:\n",
" result = yaml.load(f.read(), Loader=yaml.FullLoader)\n",
"i = 1\n",
"#print(result)\n",
"filename = 'data/全维健康中医体质检测问卷.json'\n",
"with open(filename, 'w') as fl:\n",
" json.dump(result, fl, ensure_ascii=False)\n",
"print('ok')"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 按照年龄分组生成心理问卷"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import os,sys,shutil\n",
"import openpyxl\n",
"\n",
"list1 = [\n",
" [1,'完全不正确','有点正确','多数正确','完全正确'],\n",
" [2,'从不','极少','偶尔','经常','频繁','非常频繁','每天'],\n",
" [3,'不符合','有时符合','常常符合','总是符合'],\n",
" [4,'从不','很少','有时','经常','总是'],\n",
" [5,'非常不满意','不满意','一般','满意','非常满意'],\n",
" [6,'没有或很少时间(少于1天)','一部分时间(1-2天)','相当多时间(3-4天)','绝大部分时间(5-7天)'],\n",
" [7,'从来没有这种感觉','很少有这种感觉','有时有这种感觉','经常有这种感觉'],\n",
" [8,'完全不同意','不同意','同意','完全同意'], \n",
" [10,'晚多了','晚一点','差不多','早一点','早多了']\n",
"]\n",
"dict1 = {}\n",
"for item in list1:\n",
" #dict1.setdefault(item[0],[])\n",
" list2 =[]\n",
" for i in range(1,len(item)):\n",
" list2.append(item[i])\n",
" dict1[item[0]] = list2\n",
"dict_choiced = {}\n",
"print(dict1)\n",
"wj = {}\n",
"wj[\"login\"] = \"phone\"\n",
"wj[\"locale\"] = \"zh-cn\"\n",
"wj[\"title\"] = '社区心理测评问卷'\n",
"list2 =[]\n",
"dict2 = {\n",
" \"type\": \"radiogroup\",\n",
" \"name\": \"Gender\",\n",
" \"title\": {\n",
" \"zh-cn\": \"您的性别\"\n",
" },\n",
" \"isRequired\": True,\n",
" \"choices\": [\n",
" {\n",
" \"value\": \"m\",\n",
" \"text\": \"男性\"\n",
" },\n",
" {\n",
" \"value\": \"f\",\n",
" \"text\": \"女性\"\n",
" }\n",
" ] \n",
"}\n",
"list2.append(dict2)\n",
"dict2 = {}\n",
"dict3 = {}\n",
"dict2 = {\n",
" \"type\": \"radiogroup\",\n",
" \"name\": \"Age\",\n",
" \"title\": {\n",
" \"zh-cn\": \"您的年龄\"\n",
" },\n",
" \"isRequired\": True,\n",
" \"choices\": [\n",
" {\n",
" \"value\": \"Y\",\n",
" \"text\": \"8-17岁\"\n",
" },\n",
" {\n",
" \"value\": \"M\",\n",
" \"text\": \"15-59岁\"\n",
" },\n",
" {\n",
" \"value\": \"O\",\n",
" \"text\": \"60岁及以上\"\n",
" }\n",
" ] \n",
"}\n",
"list2.append(dict2)\n",
"\n",
"wb = openpyxl.load_workbook('data/心理问卷.xlsx')\n",
"sheet = wb.active\n",
"data1 =list(sheet.values)\n",
"del data1[0]\n",
"for item in data1:\n",
" \n",
" dict2 = {} \n",
" dict2['type'] = item[4] \n",
" dict2['name'] = 'q'+str(item[0])+item[3]\n",
" dict2['visibleIf'] = '{Age}=\"'+item[3]+'\"'\n",
" dict3 = {}\n",
" dict3[\"zh-cn\"] = item[1]\n",
" dict2['title'] =dict3 \n",
" dict2['isRequired'] = True\n",
" if item[4] == 'rating':\n",
" dict2['rateMin'] = 0\n",
" dict2['rateMax'] = 7\n",
" dict2['minRateDescription'] = { 'zh-cn':'没有'}\n",
" dict2['maxRateDescription'] = { 'zh-cn':'非常同意'}\n",
" else:\n",
" if int(item[2]) in dict_choiced.keys():\n",
" dict2['choicesFromQuestion'] = dict_choiced[int(item[2])]\n",
" else:\n",
" list3 = []\n",
" i = 1\n",
" for xm in dict1[int(item[2])]:\n",
" dict3 = {}\n",
" dict3 = {'value': i,'text' : xm}\n",
" list3.append(dict3)\n",
" i+=1\n",
" dict2['choices'] = list3\n",
" dict_choiced[int(item[2])] = dict2['name']\n",
" list2.append(dict2)\n",
"dict3 = {}\n",
"dict3[\"elements\"] = list2\n",
"wj[\"pages\"] =[]\n",
"\n",
"wj[\"pages\"].append(dict3)\n",
"#filename = 'data/心理问卷1.json'\n",
"#with open(filename, 'w') as fl:\n",
"# json.dump(wj, fl, ensure_ascii=False)\n",
"#print('ok')\n",
"with open('心理问卷.yml', 'w', encoding='utf-8') as f:\n",
" yaml.dump(data=wj, stream=f, allow_unicode=True)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import yaml\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 邮件管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 使用网易邮箱群发邮件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.mime.multipart import MIMEMultipart\n",
"from email.mime.application import MIMEApplication\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"import json\n",
"\n",
"\n",
" \n",
"\n",
"fl_name = 'data/低碳院报告.json'\n",
"\n",
"with open(fl_name,'r') as fl:\n",
" m_xx = json.load(fl)\n",
"for k, v in m_xx.items():\n",
" m_bh = str(k).rjust(5,\"0\")\n",
" fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n",
" fl = f'{m_bh}-{v[0]}.pdf'\n",
" m_rec = v[2]+'@ceic.com'\n",
" \n",
" from_addr = 'kmingedu@163.com' #发件邮箱\n",
" password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n",
" smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" msg = MIMEMultipart()\n",
" msg['Subject'] = Header(\"低碳清洁能源研究院体质监测报告\",'utf-8')\n",
" msg['From'] = Header('北京坤铭体质监测评估中心')\n",
" msg['To'] = Header(m_rec)\n",
"\n",
" from_addr = 'kmingedu@163.com' #发件邮箱\n",
" password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n",
" to_addr = m_rec #收件邮箱\n",
" att1 =MIMEApplication(open(fl_name, 'rb').read())\n",
" #att1[\"Content-Type\"] = 'application/octet-stream'\n",
" # 这里的filename可以任意写,写什么名字,邮件中显示什么名字\n",
" att1.add_header('Content-Disposition','attachment',filename=fl)\n",
" msg.attach(att1)\n",
" try:\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit\n",
" time.sleep(5)\n",
" print(f'{fl}邮件发送成功!')\n",
" except smtplib.SMTPException:\n",
" print (\"Error: 无法发送邮件\")\n",
" \n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 使用网易邮箱群发测试链接"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.mime.multipart import MIMEMultipart\n",
"from email.mime.application import MIMEApplication\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"import json\n",
"import jwt\n",
"\n",
"key = \"Xixi1234\"\n",
"encoded = jwt.encode({\"email\": \"eygsoft@gmail.com\"}, key, algorithm='HS256')\n",
"#print(encoded)\n",
"\n",
"\n",
" \n",
"list1 = [\n",
" {'name':'lillianwong','E_mail':'lillianwong73@gmail.com'},\n",
" {'name':'杨玉冰','E_mail':'lkyyb@126.com'}]\n",
"\n",
"\n",
"for item in list1:\n",
" m_mail = item['E_mail']\n",
" m_name = item['name']\n",
" encoded = jwt.encode({\"email\": m_mail}, key, algorithm='HS256')\n",
" lianjie = f'https://www.forraid.com/survey/token/{encoded}/tcm0'\n",
" mail_msg = f\"\"\"\n",
"<p>欢迎参加北京坤铭体质监测评估中心中医体质检测</p>\n",
"<p><a href=\"{lianjie}\">点击参加测试</a></p>\n",
"\"\"\"\n",
" from_addr = 'kmingedu@163.com' #发件邮箱\n",
" password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n",
" smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" msg = MIMEText(mail_msg, 'html', 'utf-8')\n",
" msg['Subject'] = Header(\"中医体质检测测试链接\",'utf-8')\n",
" msg['From'] = Header('北京坤铭体质监测评估中心')\n",
" msg['To'] = Header(m_mail)\n",
"\n",
" \n",
" to_addr = m_mail #收件邮箱\n",
" try:\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit\n",
" time.sleep(5)\n",
" print('邮件发送成功!')\n",
" except smtplib.SMTPException:\n",
" print (\"Error: 无法发送邮件\")\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"try:\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit\n",
" time.sleep(5)\n",
" print(f'{fl}邮件发送成功!')\n",
" except smtplib.SMTPException:\n",
" print (\"Error: 无法发送邮件\")\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# Twilio使用"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"from twilio.rest import Client\n",
"\n",
"\n",
"# Your Account Sid and Auth Token from twilio.com/console\n",
"# and set the environment variables. See http://twil.io/secure\n",
"account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n",
"auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n",
"client = Client(account_sid, auth_token)\n",
"\n",
"message = client.messages \\\n",
" .create(\n",
" body=\"I'm back.\",\n",
" from_='+12056066931',\n",
" to='+8613793180751'\n",
" )\n",
"\n",
"print(message.sid)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import time\n",
"\n",
"localtime = time.localtime(time.time())\n",
"#type(localtime)\n",
"print (\"本地时间为 :\", localtime)\n",
"jyr = '12345'\n",
"if time.strftime(\"%w\", time.localtime()) in jyr:\n",
" print('ok')\n",
"else:\n",
" print('今日不是交易日!')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"\n",
"def sendmail(message):\n",
" msg = MIMEText(message,'plain','utf-8')\n",
" msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
" msg['From'] = Header('512song@sina.com')\n",
" msg['To'] = Header('songyi@yeah.net','utf-8')\n",
"\n",
" from_addr = '512song@sina.com' #发件邮箱\n",
" password = '409fe5d8471da663' #邮箱密码\n",
" to_addr = 'songyi@yeah.net' #收件邮箱\n",
" smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit() \n",
" \n",
" \n",
"\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"data = strhtml.text\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
"#print(data2)\n",
"with open('price.txt','r') as fl:\n",
" for line in fl:\n",
" p_high = line.split(',')[0]\n",
" p_low = line.split(',')[1]\n",
"m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
"while time.strftime(\"%w\", time.localtime()) in '12345':\n",
" \n",
" print(p_high,p_low)\n",
" time.sleep(10)\n",
" strhtml = requests.get(url)\n",
" data = strhtml.text\n",
" if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
" sendmail(m_message)\n",
" p_high = str(float(p_high) + 0.04) \n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元卖出价:{}'.format(data2[2])\n",
" p_low = str(float(p_low) - 0.04)\n",
" sendmail(m_message)\n",
" time.sleep(900)\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# AWS应用"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS获取sns信息"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Create an SNS client\n",
"sns = boto3.client('sns')\n",
"\n",
"# Call SNS to list topics\n",
"response = sns.list_topics()\n",
"\n",
"# Get a list of all topic ARNs from the response\n",
"topics = [topic['TopicArn'] for topic in response['Topics']]\n",
"\n",
"# Print out the topic list\n",
"print(\"Topic List: %s\" % topics)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS操作DynamoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"# Create the DynamoDB table.\n",
"table = dynamodb.create_table(\n",
" TableName='waihui',\n",
" \n",
" AttributeDefinitions=[ \n",
" {\n",
" 'AttributeName': 'code',\n",
" 'AttributeType': 'S'\n",
" }\n",
" \n",
" \n",
" ],\n",
" KeySchema=[\n",
" {\n",
" 'AttributeName': 'code',\n",
" 'KeyType': 'HASH'\n",
" }\n",
" \n",
" ],\n",
" ProvisionedThroughput={\n",
" 'ReadCapacityUnits': 5,\n",
" 'WriteCapacityUnits': 5\n",
" }\n",
" \n",
")\n",
"\n",
"# Wait until the table exists.\n",
"table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n",
"\n",
"# Print out some data about the table.\n",
"print(table.item_count)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.put_item(\n",
" Item={\n",
" 'code': 'USDCAD',\n",
" 'high': Decimal('1.3200'),\n",
" 'low': Decimal('1.3000'),\n",
" }\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"response = table.get_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n",
"item = response['Item']\n",
"print(item)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.delete_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"table.update_item(\n",
" Key={\n",
" 'code': 'USDCAD'\n",
" },\n",
" UpdateExpression='SET low = :val1',\n",
" ExpressionAttributeValues={\n",
" ':val1': decimal.Decimal('1.2900')\n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Create SQS client\n",
"sqs = boto3.client('sqs')\n",
"\n",
"queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n",
"\n",
"# Receive message from SQS queue\n",
"response = sqs.receive_message(\n",
" QueueUrl=queue_url,\n",
" AttributeNames=[\n",
" 'SentTimestamp'\n",
" ],\n",
" MaxNumberOfMessages=1,\n",
" MessageAttributeNames=[\n",
" 'All'\n",
" ],\n",
" VisibilityTimeout=0,\n",
" WaitTimeSeconds=0\n",
")\n",
"\n",
"message = response['Messages'][0]\n",
"receipt_handle = message['ReceiptHandle']\n",
"\n",
"# Delete received message from queue\n",
"sqs.delete_message(\n",
" QueueUrl=queue_url,\n",
" ReceiptHandle=receipt_handle\n",
")\n",
"print('Received and deleted message: %s' % message)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# MongoDB系统GridFS文件管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 文件上传"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(self,file_coll,_id,out_name):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"m_dir = './data/tmp'\n",
"fls=os.listdir(m_dir)\n",
"n = 0\n",
"for fl in fls:\n",
" #oldname=fl[n]\n",
" name, suffix = os.path.splitext(fl)\n",
" #if name in old:\n",
" # new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" # old_name = fi_path+ os.sep + fl[n]\n",
" # os.rename(old_name,new_name)\n",
" #print(os.path.basename(fl))\n",
" #print(fl,suffix[1:])\n",
" full_path = m_dir+ '/' + fl\n",
" upLoadFile(\"document\",full_path,\"\")\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": file_name, \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(file_coll,_id,out_name):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n",
"print (ll)\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n",
"#print (ll)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 腾讯文字识别"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"import types\n",
"from tencentcloud.common import credential\n",
"from tencentcloud.common.profile.client_profile import ClientProfile\n",
"from tencentcloud.common.profile.http_profile import HttpProfile\n",
"from tencentcloud.common.exception.tencent_cloud_sdk_exception import TencentCloudSDKException\n",
"from tencentcloud.ocr.v20181119 import ocr_client, models\n",
"import base64\n",
" \n",
"cred = credential.Credential(\"AKIDoZHHEB2lHbluEZvN8uYHwcdvacqfmDeJ\",\"fEkr2VucwEQXXVN8jCd0CFBxdAe6Xny9\")\n",
"httpProfile = HttpProfile()\n",
"httpProfile.endpoint = \"ocr.tencentcloudapi.com\" \n",
"clientProfile = ClientProfile()\n",
"clientProfile.httpProfile = httpProfile \n",
"client = ocr_client.OcrClient(cred, \"ap-beijing\", clientProfile) \n",
"req = models.GeneralAccurateOCRRequest()\n",
"with open(\"001.jpg\",\"rb\") as f:\n",
" img_data = f.read()\n",
"img_base64 = base64.b64encode(img_data)\n",
"params = {\n",
" \"ImageBase64\": img_base64.decode('utf-8')\n",
"}\n",
"req.from_json_string(json.dumps(params))\n",
"\n",
"# 返回的resp是一个GeneralAccurateOCRResponse的实例,与请求对象对应\n",
"resp = client.GeneralAccurateOCR(req)\n",
"print(resp.to_json_string())\n",
"\n",
"#except TencentCloudSDKException as err:\n",
"# print(err)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.12.3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}