Jupyterlab
Jupyterlab
This commit is contained in:
commit
4e8c1b3254
12 files changed
+6769
No files matched your search
@@ -0,0 +1,86 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "4925cb3a-bea3-4a4d-9590-6b91b62ca5b4",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 体质检测管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "2a85feec-695b-4ea8-8093-91a59ed34c7f",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 体测单位信息导入"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 5,
|
||||||
|
"id": "c5a0568c-3833-4335-a4b9-600310e1a541",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"['综合办公室(党委办公室)', '销运管理部', '财务资产部', '财务管理部', '企业管理部(法律事务部)', '发展计划部', '生产技术部', '设备工程部', '安全环保部', '党委组织部(资源管理部)', '审计管理部', '党群文化部(党委宣传部、工会、团委)', '纪委(监督部)', '检验计量中心', '物资采购中心', '信息中心', '行政事务中心', '消防救援支队', '电气仪表中心', '炼油运行一部', '炼油运行二部', '炼油运行三部', '炼油运行四部', '炼油运行五部', '化工运行部', '热电运行部', '水务运行部', '储运运行部', '编组站']\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import openpyxl\n",
|
||||||
|
"import json\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"tice\"]\n",
|
||||||
|
"mycol = mydb[\"unit\"]\n",
|
||||||
|
"wb = openpyxl.load_workbook('./data/shijiazhuang.xlsx')\n",
|
||||||
|
"sheet = wb.active\n",
|
||||||
|
"#sheets = wb.sheetnames\n",
|
||||||
|
"depart = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"new_code = []\n",
|
||||||
|
"for n in range(2,sheet.max_row+1):\n",
|
||||||
|
" m_depart = sheet.cell(n,3).value\n",
|
||||||
|
" if m_depart not in depart:\n",
|
||||||
|
" depart.append(sheet.cell(n,3).value)\n",
|
||||||
|
" #print(sheet.cell(n,5).value)\n",
|
||||||
|
"print(depart)\n",
|
||||||
|
" "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "aab6612f-4aad-46f0-89cd-0384ae7ac0c9",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 5
|
||||||
|
}
|
||||||
+1354
File diff suppressed because it is too large.
Load diff
+130
@@ -0,0 +1,130 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"cursor.execute(\"SELECT VERSION()\")\n",
|
||||||
|
"data = cursor.fetchone()\n",
|
||||||
|
"print (\"Database version : %s \" % data)\n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"sql = \"select id,name from sports_person where name= %s\"\n",
|
||||||
|
"cursor.execute(sql, ('张联红',))\n",
|
||||||
|
"result = cursor.fetchone()\n",
|
||||||
|
"print(result)\n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"sql = \"select id,name from sports_person\"\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"result = cursor.fetchone()\n",
|
||||||
|
"print(\"人员信息:\")\n",
|
||||||
|
"print(result)\n",
|
||||||
|
"sql = \"select id,name from sports_item\"\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"result = cursor.fetchone()\n",
|
||||||
|
"print(\"\\n运动项目信息:\")\n",
|
||||||
|
"print(result)\n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"sql = \"select a.target_id,b.name from sports_item_target as a,sports_target as b where a.target_id=b.id\"\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"re_ta = {}\n",
|
||||||
|
"print(\"\\n运动项目信息:\")\n",
|
||||||
|
"results = cursor.fetchall()\n",
|
||||||
|
"for result in results:\n",
|
||||||
|
"# print (\"%d--%s\" %(result[0],result[1]))\n",
|
||||||
|
" re_ta[result[0]] = result[1]\n",
|
||||||
|
" print(\"re_ta[%d]= #%s\" %(result[0],result[1]))\n",
|
||||||
|
"print(re_ta)\n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"person_id = \n",
|
||||||
|
"record_date = ''\n",
|
||||||
|
"item_id =\n",
|
||||||
|
"re_ta = {}\n",
|
||||||
|
"re_ta[1]= #运动距离\n",
|
||||||
|
"re_ta[2]= #运动时间\n",
|
||||||
|
"re_ta[3]= #平均配速\n",
|
||||||
|
"re_ta[4]= #平均速度\n",
|
||||||
|
"re_ta[5]= #平均步频\n",
|
||||||
|
"re_ta[6]= #平均步幅\n",
|
||||||
|
"re_ta[7]= #步数\n",
|
||||||
|
"re_ta[8]= #平均心率\n",
|
||||||
|
"re_ta[9]= #最大心率\n",
|
||||||
|
"re_ta[10]= #无氧耐力\n",
|
||||||
|
"re_ta[11]= #有氧耐力\n",
|
||||||
|
"re_ta[12]= #最大步频\n",
|
||||||
|
"re_ta[13]= #最快配速\n",
|
||||||
|
"sql = \"select * from sports_person\"\n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.5"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+880
@@ -0,0 +1,880 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 数字文件名转换为文本文件名"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 数字文件名转换为文本文件名"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import os,sys,shutil\n",
|
||||||
|
"import openpyxl\n",
|
||||||
|
"import math\n",
|
||||||
|
"\n",
|
||||||
|
"fi_xls = os.getcwd()+'/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n",
|
||||||
|
"fi_name = {}\n",
|
||||||
|
"fi_path = os.getcwd()+'/file/210924'\n",
|
||||||
|
"old = []\n",
|
||||||
|
"new = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"\n",
|
||||||
|
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||||||
|
"sheet = wb.active\n",
|
||||||
|
"depart = []\n",
|
||||||
|
"for n in range(2,sheet.max_row+1):\n",
|
||||||
|
" \n",
|
||||||
|
" m_name = sheet.cell(n,1).value.strip()\n",
|
||||||
|
" m_depart = sheet.cell(n,5).value\n",
|
||||||
|
" if m_depart not in depart:\n",
|
||||||
|
" depart.append(sheet.cell(n,5).value)\n",
|
||||||
|
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
|
||||||
|
" #print()\n",
|
||||||
|
"# 创建部门办公室 \n",
|
||||||
|
"#m_path = fi_path = os.getcwd()+'/file/210924/new'\n",
|
||||||
|
"#for pn in depart:\n",
|
||||||
|
"# if not os.path.exists(m_path + '/' + pn):\n",
|
||||||
|
"# os.mkdir(m_path + '/' + pn)\n",
|
||||||
|
"#print(dict1)\n",
|
||||||
|
"fl=os.listdir(fi_path)\n",
|
||||||
|
"for fn in fl:\n",
|
||||||
|
" if os.path.isfile(fi_path + '/' + fn):\n",
|
||||||
|
" ofn = int(fn.split('.')[0])\n",
|
||||||
|
" old.append(ofn)\n",
|
||||||
|
" #print(fn)\n",
|
||||||
|
"old.sort()\n",
|
||||||
|
" #print(str(nfn)+'.pdf')\n",
|
||||||
|
"'''\n",
|
||||||
|
"for n in old:\n",
|
||||||
|
" \n",
|
||||||
|
" o_name = f'{fi_path}/{n}.pdf'\n",
|
||||||
|
" n_name = f'{fi_path}/new/{dict1[n][1]}/{dict1[n][0]}.pdf'\n",
|
||||||
|
" if not os.path.exists(n_name):\n",
|
||||||
|
" shutil.copyfile(o_name,n_name)\n",
|
||||||
|
" print(n_name)\n",
|
||||||
|
"#print(old)\n",
|
||||||
|
"'''\n",
|
||||||
|
"print(dict1)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 目录文件按照文件名排序"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"fi_xls = 'test1.xlsx'\n",
|
||||||
|
"fi_name = {}\n",
|
||||||
|
"#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n",
|
||||||
|
"fi_path = os.getcwd()+'/data'\n",
|
||||||
|
"old = []\n",
|
||||||
|
"new = []\n",
|
||||||
|
"\n",
|
||||||
|
"fl=os.listdir(fi_path)\n",
|
||||||
|
"fl.sort()\n",
|
||||||
|
"n = 0\n",
|
||||||
|
"for i in fl:\n",
|
||||||
|
" oldname=fl[n]\n",
|
||||||
|
" name, suffix = os.path.splitext(oldname)\n",
|
||||||
|
" if name in old:\n",
|
||||||
|
" new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
|
||||||
|
" old_name = fi_path+ os.sep + fl[n]\n",
|
||||||
|
" os.rename(old_name,new_name)\n",
|
||||||
|
" n+= 1\n",
|
||||||
|
"fl"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## 将pdf文件转为图片"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from pdf2image import convert_from_path, convert_from_bytes\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"import tempfile\n",
|
||||||
|
"from pdf2image.exceptions import (\n",
|
||||||
|
" PDFInfoNotInstalledError,\n",
|
||||||
|
" PDFPageCountError,\n",
|
||||||
|
" PDFSyntaxError\n",
|
||||||
|
")\n",
|
||||||
|
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
|
||||||
|
"with tempfile.TemporaryDirectory() as path:\n",
|
||||||
|
" images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n",
|
||||||
|
"print(path)\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## 图像文件夹打包"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import zipfile\n",
|
||||||
|
"from pdf2image import convert_from_path, convert_from_bytes\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"import tempfile\n",
|
||||||
|
"import shutil\n",
|
||||||
|
"import time\n",
|
||||||
|
"\n",
|
||||||
|
"from pdf2image.exceptions import (\n",
|
||||||
|
" PDFInfoNotInstalledError,\n",
|
||||||
|
" PDFPageCountError,\n",
|
||||||
|
" PDFSyntaxError\n",
|
||||||
|
")\n",
|
||||||
|
"def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n",
|
||||||
|
" if os.path.isfile(dirname):\n",
|
||||||
|
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
|
||||||
|
" z.write(dirname)\n",
|
||||||
|
" else:\n",
|
||||||
|
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
|
||||||
|
" for root, dirs, files in os.walk(dirname):\n",
|
||||||
|
" for single_file in files:\n",
|
||||||
|
" if single_file != zipfilename:\n",
|
||||||
|
" filepath = os.path.join(root, single_file)\n",
|
||||||
|
" z.write(filepath)\n",
|
||||||
|
"\n",
|
||||||
|
"def addfile(zipfilename, dirname):\n",
|
||||||
|
" if os.path.isfile(dirname):\n",
|
||||||
|
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
|
||||||
|
" z.write(dirname)\n",
|
||||||
|
" else:\n",
|
||||||
|
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
|
||||||
|
" for root, dirs, files in os.walk(dirname):\n",
|
||||||
|
" for single_file in files:\n",
|
||||||
|
" if single_file != zipfilename:\n",
|
||||||
|
" filepath = os.path.join(root, single_file)\n",
|
||||||
|
" z.write(filepath)\n",
|
||||||
|
"\n",
|
||||||
|
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
|
||||||
|
"def make_path(p):\n",
|
||||||
|
" if os.path.exists(p): # 判断文件夹是否存在\n",
|
||||||
|
" shutil.rmtree(p) # 删除文件夹\n",
|
||||||
|
" os.mkdir(p) \n",
|
||||||
|
"pdf_file = '2.pdf'\n",
|
||||||
|
"output_folder='./pic1'\n",
|
||||||
|
"zip_file = 'ribenweiqishihua.zip'\n",
|
||||||
|
"make_path(output_folder)\n",
|
||||||
|
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n",
|
||||||
|
"with tempfile.TemporaryDirectory() as path:\n",
|
||||||
|
" images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n",
|
||||||
|
"compress_file(zip_file, output_folder) # 执行函数\n",
|
||||||
|
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## 文本文件操作"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"### 基本读取"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import re\n",
|
||||||
|
"file_name = 'data/2012.txt'\n",
|
||||||
|
"with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n",
|
||||||
|
" for l in fl:\n",
|
||||||
|
" l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n",
|
||||||
|
" if l.split():\n",
|
||||||
|
" print(l)\n",
|
||||||
|
" fl1.write(l)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"### 读取分隔符分割文件,导入MongoDB"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"import re\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"city\"]\n",
|
||||||
|
"m_mongo = {}\n",
|
||||||
|
"m_xx = []\n",
|
||||||
|
"fl_name = 'china-city-list.txt'\n",
|
||||||
|
"n = 0\n",
|
||||||
|
"with open(fl_name,'r') as fl:\n",
|
||||||
|
" for l in fl:\n",
|
||||||
|
" n += 1\n",
|
||||||
|
" if n >6:\n",
|
||||||
|
" m_mongo = {}\n",
|
||||||
|
" m_xx = re.sub('[ ]{1,}', '', l).split('|')\n",
|
||||||
|
" #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n",
|
||||||
|
" m_mongo['name'] = m_xx[3]\n",
|
||||||
|
" m_mongo['code'] = m_xx[1]\n",
|
||||||
|
" m_mongo['sheng'] = m_xx[8]\n",
|
||||||
|
" m_mongo['shi'] = m_xx[10]\n",
|
||||||
|
" m_mongo['jing'] = m_xx[11]\n",
|
||||||
|
" m_mongo['wei'] = m_xx[12]\n",
|
||||||
|
" mycol.insert_one(m_mongo) \n",
|
||||||
|
" #print(m_mongo)\n",
|
||||||
|
"print('ok!')\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"# Twilio使用"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import os\n",
|
||||||
|
"from twilio.rest import Client\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"# Your Account Sid and Auth Token from twilio.com/console\n",
|
||||||
|
"# and set the environment variables. See http://twil.io/secure\n",
|
||||||
|
"account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n",
|
||||||
|
"auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n",
|
||||||
|
"client = Client(account_sid, auth_token)\n",
|
||||||
|
"\n",
|
||||||
|
"message = client.messages \\\n",
|
||||||
|
" .create(\n",
|
||||||
|
" body=\"I'm back.\",\n",
|
||||||
|
" from_='+12056066931',\n",
|
||||||
|
" to='+8613793180751'\n",
|
||||||
|
" )\n",
|
||||||
|
"\n",
|
||||||
|
"print(message.sid)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import time\n",
|
||||||
|
"\n",
|
||||||
|
"localtime = time.localtime(time.time())\n",
|
||||||
|
"#type(localtime)\n",
|
||||||
|
"print (\"本地时间为 :\", localtime)\n",
|
||||||
|
"jyr = '12345'\n",
|
||||||
|
"if time.strftime(\"%w\", time.localtime()) in jyr:\n",
|
||||||
|
" print('ok')\n",
|
||||||
|
"else:\n",
|
||||||
|
" print('今日不是交易日!')\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from email.mime.text import MIMEText\n",
|
||||||
|
"from email.header import Header\n",
|
||||||
|
"import smtplib\n",
|
||||||
|
"import requests\n",
|
||||||
|
"import time\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"def sendmail(message):\n",
|
||||||
|
" msg = MIMEText(message,'plain','utf-8')\n",
|
||||||
|
" msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
|
||||||
|
" msg['From'] = Header('512song@sina.com')\n",
|
||||||
|
" msg['To'] = Header('songyi@yeah.net','utf-8')\n",
|
||||||
|
"\n",
|
||||||
|
" from_addr = '512song@sina.com' #发件邮箱\n",
|
||||||
|
" password = '409fe5d8471da663' #邮箱密码\n",
|
||||||
|
" to_addr = 'songyi@yeah.net' #收件邮箱\n",
|
||||||
|
" smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
|
||||||
|
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
|
||||||
|
" server.login(from_addr,password) #登录邮箱\n",
|
||||||
|
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
|
||||||
|
" server.quit() \n",
|
||||||
|
" \n",
|
||||||
|
" \n",
|
||||||
|
"\n",
|
||||||
|
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
|
||||||
|
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
|
||||||
|
"strhtml = requests.get(url)\n",
|
||||||
|
"data = strhtml.text\n",
|
||||||
|
"if pattern.findall(data):\n",
|
||||||
|
" for data1 in pattern.findall(data):\n",
|
||||||
|
" data2 = data1.split(',')\n",
|
||||||
|
"#print(data2)\n",
|
||||||
|
"with open('price.txt','r') as fl:\n",
|
||||||
|
" for line in fl:\n",
|
||||||
|
" p_high = line.split(',')[0]\n",
|
||||||
|
" p_low = line.split(',')[1]\n",
|
||||||
|
"m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
|
||||||
|
"while time.strftime(\"%w\", time.localtime()) in '12345':\n",
|
||||||
|
" \n",
|
||||||
|
" print(p_high,p_low)\n",
|
||||||
|
" time.sleep(10)\n",
|
||||||
|
" strhtml = requests.get(url)\n",
|
||||||
|
" data = strhtml.text\n",
|
||||||
|
" if pattern.findall(data):\n",
|
||||||
|
" for data1 in pattern.findall(data):\n",
|
||||||
|
" data2 = data1.split(',')\n",
|
||||||
|
" if float(data2[1]) > float(p_high):\n",
|
||||||
|
" m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
|
||||||
|
" sendmail(m_message)\n",
|
||||||
|
" p_high = str(float(p_high) + 0.04) \n",
|
||||||
|
" if float(data2[1]) > float(p_high):\n",
|
||||||
|
" m_message = '当前美元加元卖出价:{}'.format(data2[2])\n",
|
||||||
|
" p_low = str(float(p_low) - 0.04)\n",
|
||||||
|
" sendmail(m_message)\n",
|
||||||
|
" time.sleep(900)\n",
|
||||||
|
" "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"# AWS应用"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## AWS获取sns信息"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 8,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"Topic List: ['arn:aws:sns:us-east-1:915521803346:MyTopic', 'arn:aws:sns:us-east-1:915521803346:dynamodb']\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"\n",
|
||||||
|
"# Create an SNS client\n",
|
||||||
|
"sns = boto3.client('sns')\n",
|
||||||
|
"\n",
|
||||||
|
"# Call SNS to list topics\n",
|
||||||
|
"response = sns.list_topics()\n",
|
||||||
|
"\n",
|
||||||
|
"# Get a list of all topic ARNs from the response\n",
|
||||||
|
"topics = [topic['TopicArn'] for topic in response['Topics']]\n",
|
||||||
|
"\n",
|
||||||
|
"# Print out the topic list\n",
|
||||||
|
"print(\"Topic List: %s\" % topics)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## AWS操作DynamoDB"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"\n",
|
||||||
|
"# Get the service resource.\n",
|
||||||
|
"dynamodb = boto3.resource('dynamodb')\n",
|
||||||
|
"\n",
|
||||||
|
"# Create the DynamoDB table.\n",
|
||||||
|
"table = dynamodb.create_table(\n",
|
||||||
|
" TableName='waihui',\n",
|
||||||
|
" \n",
|
||||||
|
" AttributeDefinitions=[ \n",
|
||||||
|
" {\n",
|
||||||
|
" 'AttributeName': 'code',\n",
|
||||||
|
" 'AttributeType': 'S'\n",
|
||||||
|
" }\n",
|
||||||
|
" \n",
|
||||||
|
" \n",
|
||||||
|
" ],\n",
|
||||||
|
" KeySchema=[\n",
|
||||||
|
" {\n",
|
||||||
|
" 'AttributeName': 'code',\n",
|
||||||
|
" 'KeyType': 'HASH'\n",
|
||||||
|
" }\n",
|
||||||
|
" \n",
|
||||||
|
" ],\n",
|
||||||
|
" ProvisionedThroughput={\n",
|
||||||
|
" 'ReadCapacityUnits': 5,\n",
|
||||||
|
" 'WriteCapacityUnits': 5\n",
|
||||||
|
" }\n",
|
||||||
|
" \n",
|
||||||
|
")\n",
|
||||||
|
"\n",
|
||||||
|
"# Wait until the table exists.\n",
|
||||||
|
"table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n",
|
||||||
|
"\n",
|
||||||
|
"# Print out some data about the table.\n",
|
||||||
|
"print(table.item_count)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"import decimal\n",
|
||||||
|
"# Get the service resource.\n",
|
||||||
|
"dynamodb = boto3.resource('dynamodb')\n",
|
||||||
|
"\n",
|
||||||
|
"table = dynamodb.Table('waihui')\n",
|
||||||
|
"\n",
|
||||||
|
"table.put_item(\n",
|
||||||
|
" Item={\n",
|
||||||
|
" 'code': 'USDCAD',\n",
|
||||||
|
" 'high': Decimal('1.3200'),\n",
|
||||||
|
" 'low': Decimal('1.3000'),\n",
|
||||||
|
" }\n",
|
||||||
|
")"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 7,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"{'low': Decimal('1.29'), 'code': 'USDCAD', 'high': Decimal('1.32')}\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"# Get the service resource.\n",
|
||||||
|
"dynamodb = boto3.resource('dynamodb')\n",
|
||||||
|
"\n",
|
||||||
|
"table = dynamodb.Table('waihui')\n",
|
||||||
|
"\n",
|
||||||
|
"response = table.get_item(\n",
|
||||||
|
" Key={\n",
|
||||||
|
" 'code': 'USDCAD' \n",
|
||||||
|
" }\n",
|
||||||
|
")\n",
|
||||||
|
"item = response['Item']\n",
|
||||||
|
"print(item)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"# Get the service resource.\n",
|
||||||
|
"dynamodb = boto3.resource('dynamodb')\n",
|
||||||
|
"\n",
|
||||||
|
"table = dynamodb.Table('waihui')\n",
|
||||||
|
"\n",
|
||||||
|
"table.delete_item(\n",
|
||||||
|
" Key={\n",
|
||||||
|
" 'code': 'USDCAD' \n",
|
||||||
|
" }\n",
|
||||||
|
")\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 6,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"data": {
|
||||||
|
"text/plain": [
|
||||||
|
"{'ResponseMetadata': {'RequestId': 'LCHMG55BG6A89FF2KBSDM7JRHNVV4KQNSO5AEMVJF66Q9ASUAAJG',\n",
|
||||||
|
" 'HTTPStatusCode': 200,\n",
|
||||||
|
" 'HTTPHeaders': {'server': 'Server',\n",
|
||||||
|
" 'date': 'Thu, 30 Sep 2021 05:04:36 GMT',\n",
|
||||||
|
" 'content-type': 'application/x-amz-json-1.0',\n",
|
||||||
|
" 'content-length': '2',\n",
|
||||||
|
" 'connection': 'keep-alive',\n",
|
||||||
|
" 'x-amzn-requestid': 'LCHMG55BG6A89FF2KBSDM7JRHNVV4KQNSO5AEMVJF66Q9ASUAAJG',\n",
|
||||||
|
" 'x-amz-crc32': '2745614147'},\n",
|
||||||
|
" 'RetryAttempts': 0}}"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"execution_count": 6,
|
||||||
|
"metadata": {},
|
||||||
|
"output_type": "execute_result"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"import decimal\n",
|
||||||
|
"# Get the service resource.\n",
|
||||||
|
"dynamodb = boto3.resource('dynamodb')\n",
|
||||||
|
"\n",
|
||||||
|
"table = dynamodb.Table('waihui')\n",
|
||||||
|
"table.update_item(\n",
|
||||||
|
" Key={\n",
|
||||||
|
" 'code': 'USDCAD'\n",
|
||||||
|
" },\n",
|
||||||
|
" UpdateExpression='SET low = :val1',\n",
|
||||||
|
" ExpressionAttributeValues={\n",
|
||||||
|
" ':val1': decimal.Decimal('1.2900')\n",
|
||||||
|
" }\n",
|
||||||
|
")\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 15,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"Received and deleted message: {'MessageId': 'ecd656af-72e2-4bfc-9deb-3043dbe1bb80', 'ReceiptHandle': 'AQEBTx6u3PRfCgUy0SfLyOZcsD4fa5ge7AZvDTeQ/P4sOslg+sDMRYDZelJQOvxDK2XsLvy24DvD/n6GxBa2RwN55b/bnN7PxeYvv42d8GgkIy1dOPEBTytVKNIf+Z7t9+jMZMT2kq2ypXSa7jLFPFN00ArUkd0GI7nZ5SIS133s8exP46gRUGjgaS04CzAwPocTpWNt9GZpA114bvlicx2gNMYTIBsF65K2MzcyBoNVjLzmSJ1Pi0OCcQrAYaHvwzfnkriOBJLUX8BU4ksjWGJg0m8wh8hlPVMTLX+qxMneLzLawbf3cM/ricQR1XtooXrhsAuwAXA+VXaq5TzlpFJZaDZu5BvebF0Yj3rst7j54L/O3DwY0WfVmIrgv2yjOqL7', 'MD5OfBody': '9794e060a0eae4992dbf298549c42dc5', 'Body': 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1', 'Attributes': {'SentTimestamp': '1632980620260'}}\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import boto3\n",
|
||||||
|
"\n",
|
||||||
|
"# Create SQS client\n",
|
||||||
|
"sqs = boto3.client('sqs')\n",
|
||||||
|
"\n",
|
||||||
|
"queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n",
|
||||||
|
"\n",
|
||||||
|
"# Receive message from SQS queue\n",
|
||||||
|
"response = sqs.receive_message(\n",
|
||||||
|
" QueueUrl=queue_url,\n",
|
||||||
|
" AttributeNames=[\n",
|
||||||
|
" 'SentTimestamp'\n",
|
||||||
|
" ],\n",
|
||||||
|
" MaxNumberOfMessages=1,\n",
|
||||||
|
" MessageAttributeNames=[\n",
|
||||||
|
" 'All'\n",
|
||||||
|
" ],\n",
|
||||||
|
" VisibilityTimeout=0,\n",
|
||||||
|
" WaitTimeSeconds=0\n",
|
||||||
|
")\n",
|
||||||
|
"\n",
|
||||||
|
"message = response['Messages'][0]\n",
|
||||||
|
"receipt_handle = message['ReceiptHandle']\n",
|
||||||
|
"\n",
|
||||||
|
"# Delete received message from queue\n",
|
||||||
|
"sqs.delete_message(\n",
|
||||||
|
" QueueUrl=queue_url,\n",
|
||||||
|
" ReceiptHandle=receipt_handle\n",
|
||||||
|
")\n",
|
||||||
|
"print('Received and deleted message: %s' % message)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# MongoDB系统GridFS文件管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 文件上传"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"tags": []
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"from gridfs import GridFS\n",
|
||||||
|
"from bson.objectid import ObjectId\n",
|
||||||
|
"import os\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"college\"]\n",
|
||||||
|
"\n",
|
||||||
|
"UploadCache = \"uploadcache\"\n",
|
||||||
|
"dbURL = \"mongodb://localhost:27017\"\n",
|
||||||
|
"\n",
|
||||||
|
"#上传文件\n",
|
||||||
|
"def upLoadFile(file_coll,file_name,data_link):\n",
|
||||||
|
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"\n",
|
||||||
|
" db = client[\"gaokao\"]\n",
|
||||||
|
"\n",
|
||||||
|
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
|
||||||
|
" gridfs_col = GridFS(db, collection=file_coll)\n",
|
||||||
|
" file_ = \"0\"\n",
|
||||||
|
" query = {\"filename\":\"\"}\n",
|
||||||
|
" query[\"filename\"] = file_name\n",
|
||||||
|
"\n",
|
||||||
|
" if gridfs_col.exists(query):\n",
|
||||||
|
" print('已经存在该文件')\n",
|
||||||
|
" else:\n",
|
||||||
|
"\n",
|
||||||
|
" with open(file_name, 'rb') as file_r:\n",
|
||||||
|
" file_data = file_r.read()\n",
|
||||||
|
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
|
||||||
|
"\n",
|
||||||
|
" print(file_)\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
" return file_ \n",
|
||||||
|
"# 按文件名获取文档\n",
|
||||||
|
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
|
||||||
|
" client = pymongo.MongoClient(self.dbURL)\n",
|
||||||
|
"\n",
|
||||||
|
" db = client[\"store\"]\n",
|
||||||
|
"\n",
|
||||||
|
" gridfs_col = GridFS(db, collection=file_coll)\n",
|
||||||
|
"\n",
|
||||||
|
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
|
||||||
|
"\n",
|
||||||
|
" with open(out_name, 'wb') as file_w:\n",
|
||||||
|
" file_w.write(file_data)\n",
|
||||||
|
"\n",
|
||||||
|
"# 按文件_Id获取文档 \n",
|
||||||
|
"def downLoadFilebyID(self,file_coll,_id,out_name):\n",
|
||||||
|
" client = pymongo.MongoClient(self.dbURL)\n",
|
||||||
|
"\n",
|
||||||
|
" db = client[\"store\"]\n",
|
||||||
|
"\n",
|
||||||
|
" gridfs_col = GridFS(db, collection=file_coll)\n",
|
||||||
|
"\n",
|
||||||
|
" O_Id = ObjectId(_id)\n",
|
||||||
|
"\n",
|
||||||
|
" gf = gridfs_col.get(file_id=O_Id)\n",
|
||||||
|
" file_data = gf.read()\n",
|
||||||
|
" with open(out_name, 'wb') as file_w:\n",
|
||||||
|
"\n",
|
||||||
|
" file_w.write(file_data) \n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
" return gf.filename \n",
|
||||||
|
"m_dir = './data/tmp'\n",
|
||||||
|
"fls=os.listdir(m_dir)\n",
|
||||||
|
"n = 0\n",
|
||||||
|
"for fl in fls:\n",
|
||||||
|
" #oldname=fl[n]\n",
|
||||||
|
" name, suffix = os.path.splitext(fl)\n",
|
||||||
|
" #if name in old:\n",
|
||||||
|
" # new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
|
||||||
|
" # old_name = fi_path+ os.sep + fl[n]\n",
|
||||||
|
" # os.rename(old_name,new_name)\n",
|
||||||
|
" #print(os.path.basename(fl))\n",
|
||||||
|
" #print(fl,suffix[1:])\n",
|
||||||
|
" full_path = m_dir+ '/' + fl\n",
|
||||||
|
" upLoadFile(\"document\",full_path,\"\")\n",
|
||||||
|
"#a = MongoGridFS(\"\")\n",
|
||||||
|
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
|
||||||
|
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
|
||||||
|
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n",
|
||||||
|
"#print (ll)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"from gridfs import GridFS\n",
|
||||||
|
"from bson.objectid import ObjectId\n",
|
||||||
|
"import os\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"college\"]\n",
|
||||||
|
"\n",
|
||||||
|
"UploadCache = \"uploadcache\"\n",
|
||||||
|
"dbURL = \"mongodb://localhost:27017\"\n",
|
||||||
|
"\n",
|
||||||
|
"#上传文件\n",
|
||||||
|
"def upLoadFile(file_coll,file_name,data_link):\n",
|
||||||
|
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"\n",
|
||||||
|
" db = client[\"gaokao\"]\n",
|
||||||
|
"\n",
|
||||||
|
" filter_condition = {\"filename\": file_name, \"url\": data_link}\n",
|
||||||
|
" gridfs_col = GridFS(db, collection=file_coll)\n",
|
||||||
|
" file_ = \"0\"\n",
|
||||||
|
" query = {\"filename\":\"\"}\n",
|
||||||
|
" query[\"filename\"] = file_name\n",
|
||||||
|
"\n",
|
||||||
|
" if gridfs_col.exists(query):\n",
|
||||||
|
" print('已经存在该文件')\n",
|
||||||
|
" else:\n",
|
||||||
|
"\n",
|
||||||
|
" with open(file_name, 'rb') as file_r:\n",
|
||||||
|
" file_data = file_r.read()\n",
|
||||||
|
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
|
||||||
|
"\n",
|
||||||
|
" print(file_)\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
" return file_ \n",
|
||||||
|
"# 按文件名获取文档\n",
|
||||||
|
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
|
||||||
|
" client = pymongo.MongoClient(self.dbURL)\n",
|
||||||
|
"\n",
|
||||||
|
" db = client[\"store\"]\n",
|
||||||
|
"\n",
|
||||||
|
" gridfs_col = GridFS(db, collection=file_coll)\n",
|
||||||
|
"\n",
|
||||||
|
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
|
||||||
|
"\n",
|
||||||
|
" with open(out_name, 'wb') as file_w:\n",
|
||||||
|
" file_w.write(file_data)\n",
|
||||||
|
"\n",
|
||||||
|
"# 按文件_Id获取文档 \n",
|
||||||
|
"def downLoadFilebyID(file_coll,_id,out_name):\n",
|
||||||
|
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"\n",
|
||||||
|
" db = client[\"gaokao\"]\n",
|
||||||
|
"\n",
|
||||||
|
" gridfs_col = GridFS(db, collection=file_coll)\n",
|
||||||
|
"\n",
|
||||||
|
" O_Id = ObjectId(_id)\n",
|
||||||
|
"\n",
|
||||||
|
" gf = gridfs_col.get(file_id=O_Id)\n",
|
||||||
|
" file_data = gf.read()\n",
|
||||||
|
" with open(out_name, 'wb') as file_w:\n",
|
||||||
|
"\n",
|
||||||
|
" file_w.write(file_data) \n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
" return gf.filename \n",
|
||||||
|
"ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n",
|
||||||
|
"print (ll)\n",
|
||||||
|
"#a = MongoGridFS(\"\")\n",
|
||||||
|
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
|
||||||
|
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
|
||||||
|
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n",
|
||||||
|
"#print (ll)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+439
@@ -0,0 +1,439 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "333a5431-a697-4330-a587-455d6d18ebb0",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 文件管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "bcd48bb6-4756-4748-ac8b-854eeee92a3b",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 数字文件名转换为文本文件名"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "54b320e8-c3c1-4214-9a9e-db13c2558604",
|
||||||
|
"metadata": {
|
||||||
|
"tags": []
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import os,sys,shutil\n",
|
||||||
|
"import openpyxl\n",
|
||||||
|
"import math\n",
|
||||||
|
"\n",
|
||||||
|
"fi_xls = os.getcwd() + '/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n",
|
||||||
|
"fi_path = os.getcwd() + '/file/210924'\n",
|
||||||
|
"old = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"\n",
|
||||||
|
"wb = openpyxl.load_workbook(fi_xls)\n",
|
||||||
|
"sheet = wb.active\n",
|
||||||
|
"depart = []\n",
|
||||||
|
"for n in range(2,sheet.max_row+1): \n",
|
||||||
|
" m_name = sheet.cell(n,1).value.strip()\n",
|
||||||
|
" m_depart = sheet.cell(n,5).value\n",
|
||||||
|
" if m_depart not in depart:\n",
|
||||||
|
" depart.append(sheet.cell(n,5).value)\n",
|
||||||
|
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
|
||||||
|
" \n",
|
||||||
|
"# 创建部门办公室 \n",
|
||||||
|
"m_path = fi_path = os.getcwd()+'/file/210924/new'\n",
|
||||||
|
"for pn in depart:\n",
|
||||||
|
" if not os.path.exists(m_path + '/' + pn):\n",
|
||||||
|
" os.mkdir(m_path + '/' + pn)\n",
|
||||||
|
"fl=os.listdir(fi_path)\n",
|
||||||
|
"for fn in fl:\n",
|
||||||
|
" if os.path.isfile(fi_path + '/' + fn):\n",
|
||||||
|
" ofn = int(fn.split('.')[0])\n",
|
||||||
|
" old.append(ofn) \n",
|
||||||
|
"old.sort()\n",
|
||||||
|
"for n in old: \n",
|
||||||
|
" o_name = f'{fi_path}/{n}.pdf'\n",
|
||||||
|
" n_name = f'{fi_path}/new/{dict1[n][1]}/{dict1[n][0]}.pdf'\n",
|
||||||
|
" if not os.path.exists(n_name):\n",
|
||||||
|
" shutil.copyfile(o_name,n_name)\n",
|
||||||
|
" print(n_name)\n",
|
||||||
|
"#print(dict1)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "69712868-6786-430b-8591-f6828d781588",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## PDF文件压缩"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "8492dc14-739e-4c78-a302-f4b57ae571dd",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import fitz\n",
|
||||||
|
"import os\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"def covert2pic(zoom):\n",
|
||||||
|
" if os.path.exists('.pdf'): # 临时文件,需为空\n",
|
||||||
|
" os.removedirs('.pdf')\n",
|
||||||
|
" os.mkdir('.pdf')\n",
|
||||||
|
" for pg in range(totaling):\n",
|
||||||
|
" page = doc[pg]\n",
|
||||||
|
" zoom = int(zoom) #值越大,分辨率越高,文件越清晰\n",
|
||||||
|
" rotate = int(0)\n",
|
||||||
|
" print(page)\n",
|
||||||
|
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0).preRotate(rotate)\n",
|
||||||
|
" pm = page.getPixmap(matrix=trans, alpha=False)\n",
|
||||||
|
" \n",
|
||||||
|
" lurl='.pdf/%s.jpg' % str(pg+1)\n",
|
||||||
|
" pm.writePNG(lurl)\n",
|
||||||
|
" doc.close()\n",
|
||||||
|
"\n",
|
||||||
|
"def pic2pdf(obj):\n",
|
||||||
|
" doc = fitz.open()\n",
|
||||||
|
" for pg in range(totaling):\n",
|
||||||
|
" img = '.pdf/%s.jpg' % str(pg+1)\n",
|
||||||
|
" imgdoc = fitz.open(img) # 打开图片\n",
|
||||||
|
" pdfbytes = imgdoc.convertToPDF() # 使用图片创建单页的 PDF\n",
|
||||||
|
" os.remove(img) \n",
|
||||||
|
" imgpdf = fitz.open(\"pdf\", pdfbytes)\n",
|
||||||
|
" doc.insertPDF(imgpdf) # 将当前页插入文档\n",
|
||||||
|
" if os.path.exists(obj): # 若文件存在先删除\n",
|
||||||
|
" os.remove(obj)\n",
|
||||||
|
" doc.save(obj) # 保存pdf文件\n",
|
||||||
|
" doc.close()\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"def pdfz(sor, obj, zoom): \n",
|
||||||
|
" covert2pic(zoom)\n",
|
||||||
|
" pic2pdf(obj)\n",
|
||||||
|
" \n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"sor = \"5.pdf\" # 需要压缩的PDF文件\n",
|
||||||
|
"obj = \"new-\" + sor\n",
|
||||||
|
"doc = fitz.open(sor) \n",
|
||||||
|
"totaling = doc.pageCount\n",
|
||||||
|
"\n",
|
||||||
|
"zoom = 150 # 清晰度调节,缩放比率\n",
|
||||||
|
"pdfz(sor, obj, zoom)\n",
|
||||||
|
"os.removedirs('.pdf')\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "88d84849-5b98-48cb-86b1-e96757a8615f",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from pdf2image import convert_from_path, convert_from_bytes\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"import tempfile\n",
|
||||||
|
"from pdf2image.exceptions import (\n",
|
||||||
|
" PDFInfoNotInstalledError,\n",
|
||||||
|
" PDFPageCountError,\n",
|
||||||
|
" PDFSyntaxError\n",
|
||||||
|
")\n",
|
||||||
|
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
|
||||||
|
"with tempfile.TemporaryDirectory() as path:\n",
|
||||||
|
" images_from_path = convert_from_path('5.pdf', dpi=100,fmt='jpg', output_folder='./pic')\n",
|
||||||
|
"print(path)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "058a40ba-4948-4675-8400-e4c762e836ee",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import img2pdf \n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"import glob\n",
|
||||||
|
"\n",
|
||||||
|
"fl=glob.glob('./pic/*.jpg')\n",
|
||||||
|
"fl.sort()\n",
|
||||||
|
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
|
||||||
|
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
|
||||||
|
"with open(\"name.pdf\",\"wb\") as f:\n",
|
||||||
|
" f.write(img2pdf.convert(fl,layout_fun=layout_fun))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "e997ad75-b2d9-43bf-a1de-8833cfe0cf6a",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import glob\n",
|
||||||
|
"import fitz # 导入本模块需安装pymupdf库\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"\n",
|
||||||
|
"def pic2pdf_1(img_path, pdf_path, pdf_name):\n",
|
||||||
|
" doc = fitz.open()\n",
|
||||||
|
" fl=os.listdir(img_path)\n",
|
||||||
|
" fl.sort()\n",
|
||||||
|
" width, height = fitz.PaperSize(\"a4\")\n",
|
||||||
|
" for img in fl:\n",
|
||||||
|
" fn = img_path+'/'+img\n",
|
||||||
|
" if os.path.isfile(fn):\n",
|
||||||
|
" imgdoc = fitz.open(img_path+'/'+img)\n",
|
||||||
|
" pdfbytes = imgdoc.convertToPDF()\n",
|
||||||
|
" imgpdf = fitz.open(\"pdf\", pdfbytes,width = width, height = height)\n",
|
||||||
|
" doc.insertPDF(imgpdf)\n",
|
||||||
|
" doc.save(pdf_path +'/'+ pdf_name)\n",
|
||||||
|
" doc.close()\n",
|
||||||
|
"\n",
|
||||||
|
"img_path = os.getcwd() +'/pic'\n",
|
||||||
|
"pdf_path = os.getcwd()\n",
|
||||||
|
"pic2pdf_1(img_path=img_path, pdf_path=pdf_path, pdf_name='1.pdf')"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "1f38e4b6-ef20-4fef-801c-ff47951d2ad6",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 图片文件扫描识别"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "8c990b6a-7f8e-4527-a150-65f60bd131ac",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"#将指定目录下图片文件进行文字识别\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"from aip import AipOcr\n",
|
||||||
|
"import glob\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||||||
|
"APP_ID = '17553946'\n",
|
||||||
|
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||||||
|
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||||||
|
"\n",
|
||||||
|
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||||||
|
"def get_file_content(filePath):\n",
|
||||||
|
" with open(filePath, 'rb') as fp:\n",
|
||||||
|
" return fp.read()\n",
|
||||||
|
"\n",
|
||||||
|
"options = {}\n",
|
||||||
|
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||||||
|
"options[\"detect_direction\"] = \"true\"\n",
|
||||||
|
"options[\"detect_language\"] = \"true\"\n",
|
||||||
|
"options[\"probability\"] = \"true\"\n",
|
||||||
|
"\n",
|
||||||
|
"fi_path = os.getcwd()+'/pic'\n",
|
||||||
|
"fl = glob.glob(f'{fi_path}/*.jpg')\n",
|
||||||
|
"fl.sort()\n",
|
||||||
|
"for fl1 in fl:\n",
|
||||||
|
" file_name = fl1\n",
|
||||||
|
" image = get_file_content(file_name)\n",
|
||||||
|
" result= client.basicGeneral(image, options)\n",
|
||||||
|
" if 'words_result' in result:\n",
|
||||||
|
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
|
||||||
|
" print('\\n')\n",
|
||||||
|
" \n",
|
||||||
|
" "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "d936ca99-269b-46f8-8582-fb02ed2cd3cc",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## word文档读取"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 47,
|
||||||
|
"id": "d8043425-be9e-4e9e-bbfc-3a2d8885a860",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"ok\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import docx\n",
|
||||||
|
"import json\n",
|
||||||
|
"Doc = docx.Document(r\"症状对症处方选穴.docx\")\n",
|
||||||
|
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
|
||||||
|
"testList = []\n",
|
||||||
|
"text1 = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"for text in Doc.paragraphs:\n",
|
||||||
|
" testList.append(text)\n",
|
||||||
|
"i = 0\n",
|
||||||
|
"for pg in testList:\n",
|
||||||
|
" \n",
|
||||||
|
" if len(pg.text) >0:\n",
|
||||||
|
" s = ''.join(pg.text.split()) \n",
|
||||||
|
" if s[0:1] == '第':\n",
|
||||||
|
" i += 1\n",
|
||||||
|
" n = 0\n",
|
||||||
|
" \n",
|
||||||
|
" dict1.setdefault(i,{})\n",
|
||||||
|
" dict1[i]['title'] = pg.text.split()[1]\n",
|
||||||
|
" dict1[i]['sub'] = {}\n",
|
||||||
|
" \n",
|
||||||
|
" #dict1[i]['title'].setdefault(i,{})\n",
|
||||||
|
" elif s[0:1] == '(':\n",
|
||||||
|
" sub = s.split(')')[1]\n",
|
||||||
|
" dict2 = {}\n",
|
||||||
|
" dict1[i]['sub'].setdefault(sub,[])\n",
|
||||||
|
" else:\n",
|
||||||
|
" dict1[i]['sub'][sub].append(s)\n",
|
||||||
|
"filename = '症状对症处方选穴.json'\n",
|
||||||
|
"with open(filename,'w') as fl:\n",
|
||||||
|
" json.dump(dict1, fl,ensure_ascii=False) \n",
|
||||||
|
"print('ok') "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 2,
|
||||||
|
"id": "c440b599-9c80-4072-8543-d9c381214f1d",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"ok\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import docx\n",
|
||||||
|
"import json\n",
|
||||||
|
"Doc = docx.Document(r\"常见疾病辨病处方选穴.docx\")\n",
|
||||||
|
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
|
||||||
|
"testList = []\n",
|
||||||
|
"text1 = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"for text in Doc.paragraphs:\n",
|
||||||
|
" testList.append(text)\n",
|
||||||
|
"i = 0\n",
|
||||||
|
"for pg in testList:\n",
|
||||||
|
" \n",
|
||||||
|
" if len(pg.text) >0:\n",
|
||||||
|
" s = ''.join(pg.text.split()) \n",
|
||||||
|
" if s[0:1] == '第':\n",
|
||||||
|
" i += 1\n",
|
||||||
|
" n = 0\n",
|
||||||
|
" \n",
|
||||||
|
" dict1.setdefault(i,{})\n",
|
||||||
|
" dict1[i]['title'] = pg.text.split('节')[1]\n",
|
||||||
|
" dict1[i]['sub'] = {}\n",
|
||||||
|
" \n",
|
||||||
|
" #dict1[i]['title'].setdefault(i,{})\n",
|
||||||
|
" elif s[0:1] == '(':\n",
|
||||||
|
" sub = s.split(')')[1]\n",
|
||||||
|
" dict2 = {}\n",
|
||||||
|
" dict1[i]['sub'].setdefault(sub,[])\n",
|
||||||
|
" else:\n",
|
||||||
|
" dict1[i]['sub'][sub].append(s)\n",
|
||||||
|
"filename = '常见疾病辨病处方选穴.json'\n",
|
||||||
|
"with open(filename,'w') as fl:\n",
|
||||||
|
" json.dump(dict1, fl,ensure_ascii=False)\n",
|
||||||
|
"print('ok') "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 69,
|
||||||
|
"id": "b5e500b5-14d4-4417-8b5e-46df71a6b6b8",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"ok\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import json\n",
|
||||||
|
"import re\n",
|
||||||
|
"filename = '症状对症处方选穴.json'\n",
|
||||||
|
"pattern = r'[\\d\\.]'\n",
|
||||||
|
"mo =r'[\\u4e00-\\u9fa5]+'\n",
|
||||||
|
"dict2 = {}\n",
|
||||||
|
"with open(filename,'r') as fl:\n",
|
||||||
|
" dict1 = json.load(fl) \n",
|
||||||
|
"for k,v in dict1.items():\n",
|
||||||
|
" m_title = v['title']\n",
|
||||||
|
" for k1,v1 in v['sub'].items():\n",
|
||||||
|
" sub_title = k1 \n",
|
||||||
|
" s = ''\n",
|
||||||
|
" for mx in v1: \n",
|
||||||
|
" s = s + mx\n",
|
||||||
|
" list1 = re.split(pattern, s)\n",
|
||||||
|
" list2 = []\n",
|
||||||
|
" for ss in list1:\n",
|
||||||
|
" if ss != '':\n",
|
||||||
|
" list3 = re.findall(mo,ss)\n",
|
||||||
|
" list2.append(list3)\n",
|
||||||
|
" \n",
|
||||||
|
" #print(m_title,sub_title,list2)\n",
|
||||||
|
" dict2.setdefault(m_title,{})\n",
|
||||||
|
" dict2[m_title][sub_title] = list2\n",
|
||||||
|
"filename = 'new_症状对症处方选穴.json'\n",
|
||||||
|
"with open(filename,'w') as fl:\n",
|
||||||
|
" json.dump(dict2, fl,ensure_ascii=False) \n",
|
||||||
|
"print('ok') \n",
|
||||||
|
" "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "eeea1429-e574-4d3b-88fd-0ddb3c090947",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 5
|
||||||
|
}
|
||||||
+377
@@ -0,0 +1,377 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from aip import AipOcr\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||||||
|
"APP_ID = '17553946'\n",
|
||||||
|
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||||||
|
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||||||
|
"\n",
|
||||||
|
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||||||
|
"def get_file_content(filePath):\n",
|
||||||
|
" with open(filePath, 'rb') as fp:\n",
|
||||||
|
" return fp.read()\n",
|
||||||
|
"\n",
|
||||||
|
"image = get_file_content('1.jpg')\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
|
||||||
|
"#client.basicGeneral(image);\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 如果有可选参数 \"\"\"\n",
|
||||||
|
"options = {}\n",
|
||||||
|
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||||||
|
"options[\"detect_direction\"] = \"true\"\n",
|
||||||
|
"options[\"detect_language\"] = \"true\"\n",
|
||||||
|
"options[\"probability\"] = \"true\"\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 带参数调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
|
||||||
|
"result= client.basicGeneral(image, options)\n",
|
||||||
|
"if 'words_result' in result:\n",
|
||||||
|
" print('\\n'.join([w['words'] for w in result['words_result']]))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"#将指定目录下图片文件进行文字识别\n",
|
||||||
|
"import os,sys\n",
|
||||||
|
"from aip import AipOcr\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||||||
|
"APP_ID = '17553946'\n",
|
||||||
|
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
|
||||||
|
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
|
||||||
|
"\n",
|
||||||
|
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
|
||||||
|
"def get_file_content(filePath):\n",
|
||||||
|
" with open(filePath, 'rb') as fp:\n",
|
||||||
|
" return fp.read()\n",
|
||||||
|
"\n",
|
||||||
|
"options = {}\n",
|
||||||
|
"options[\"language_type\"] = \"CHN_ENG\"\n",
|
||||||
|
"options[\"detect_direction\"] = \"true\"\n",
|
||||||
|
"options[\"detect_language\"] = \"true\"\n",
|
||||||
|
"options[\"probability\"] = \"true\"\n",
|
||||||
|
"\n",
|
||||||
|
"fi_path = os.getcwd()+'/data'\n",
|
||||||
|
"fl = os.listdir(fi_path)\n",
|
||||||
|
"fl.sort()\n",
|
||||||
|
"for fl1 in fl:\n",
|
||||||
|
" file_name = fi_path+'/' + fl1\n",
|
||||||
|
" image = get_file_content(file_name)\n",
|
||||||
|
" result= client.basicGeneral(image, options)\n",
|
||||||
|
" if 'words_result' in result:\n",
|
||||||
|
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
|
||||||
|
" print('\\n')"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 识别图片中表格"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 2,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-26T08:59:12.967127Z",
|
||||||
|
"iopub.status.busy": "2020-11-26T08:59:12.966172Z",
|
||||||
|
"iopub.status.idle": "2020-11-26T08:59:28.010488Z",
|
||||||
|
"shell.execute_reply": "2020-11-26T08:59:28.006947Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-26T08:59:12.967019Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"22917135_2274033\n",
|
||||||
|
"已完成\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import json\n",
|
||||||
|
"import base64\n",
|
||||||
|
"import time\n",
|
||||||
|
"\n",
|
||||||
|
"def get_access_token():\n",
|
||||||
|
" client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'\n",
|
||||||
|
" client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq' \n",
|
||||||
|
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
|
||||||
|
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
|
||||||
|
" client_id, client_secret)\n",
|
||||||
|
" response = requests.get(host).text\n",
|
||||||
|
" data = json.loads(response)\n",
|
||||||
|
" access_token = data['access_token']\n",
|
||||||
|
" return access_token\n",
|
||||||
|
"\n",
|
||||||
|
"def get_excel(requests_id, access_token):\n",
|
||||||
|
" headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||||||
|
" pargams = {\n",
|
||||||
|
" 'request_id': requests_id,\n",
|
||||||
|
" 'result_type': 'excel'\n",
|
||||||
|
" }\n",
|
||||||
|
" url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||||||
|
" url_all = url + \"?access_token=\" + access_token\n",
|
||||||
|
" res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||||||
|
" info_1 = res.json()['result']['ret_msg']\n",
|
||||||
|
" excel_url=res.json()['result']['result_data']\n",
|
||||||
|
" excel_1=requests.get(excel_url).content\n",
|
||||||
|
" with open('识别结果11.xls','wb+') as f:\n",
|
||||||
|
" f.write(excel_1)\n",
|
||||||
|
" print(info_1)\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"request_url = \"https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request\"\n",
|
||||||
|
"# 二进制方式打开图片文件\n",
|
||||||
|
"f = open('山东大学强基计划(2020).jpg', 'rb')\n",
|
||||||
|
"img = base64.b64encode(f.read())\n",
|
||||||
|
"\n",
|
||||||
|
"params = {\"image\":img}\n",
|
||||||
|
"access_token = get_access_token()\n",
|
||||||
|
"request_url = request_url + \"?access_token=\" + access_token\n",
|
||||||
|
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||||||
|
"response = requests.post(request_url, data=params, headers=headers)\n",
|
||||||
|
"if response:\n",
|
||||||
|
" m_xx = response.json()\n",
|
||||||
|
"requests_id = m_xx['result'][0]['request_id'] \n",
|
||||||
|
"print(requests_id)\n",
|
||||||
|
"time.sleep(10)\n",
|
||||||
|
"get_excel(requests_id, access_token)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 1,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-03T02:06:23.901979Z",
|
||||||
|
"iopub.status.busy": "2020-11-03T02:06:23.900925Z",
|
||||||
|
"iopub.status.idle": "2020-11-03T02:06:24.312339Z",
|
||||||
|
"shell.execute_reply": "2020-11-03T02:06:24.309116Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-03T02:06:23.901714Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"{'refresh_token': '25.1dfb7a14cd15d03051976c8db246fa92.315360000.1919729184.282335-22917135', 'expires_in': 2592000, 'session_key': '9mzdCSFczT8Mv7ZN07SjsPo0dZr0AvxAeDt6Kjn1Z9jiiotA6kW2TZYzslnuYOFd1ZCx75mmzW0TRF+nxYAHzPOb5fmjlw==', 'access_token': '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135', 'scope': 'public vis-ocr_ocr brain_ocr_scope brain_ocr_general brain_ocr_general_basic vis-ocr_business_license brain_ocr_webimage brain_all_scope brain_ocr_idcard brain_ocr_driving_license brain_ocr_vehicle_license vis-ocr_plate_number brain_solution brain_ocr_plate_number brain_ocr_accurate brain_ocr_accurate_basic brain_ocr_receipt brain_ocr_business_license brain_solution_iocr brain_qrcode brain_ocr_handwriting brain_ocr_passport brain_ocr_vat_invoice brain_numbers brain_ocr_business_card brain_ocr_train_ticket brain_ocr_taxi_receipt vis-ocr_household_register vis-ocr_vis-classify_birth_certificate vis-ocr_台湾通行证 vis-ocr_港澳通行证 vis-ocr_机动车购车发票识别 vis-ocr_机动车检验合格证识别 vis-ocr_车辆vin码识别 vis-ocr_定额发票识别 vis-ocr_保单识别 vis-ocr_机打发票识别 vis-ocr_行程单识别 brain_ocr_vin brain_ocr_quota_invoice brain_ocr_birth_certificate brain_ocr_household_register brain_ocr_HK_Macau_pass brain_ocr_taiwan_pass brain_ocr_vehicle_invoice brain_ocr_vehicle_certificate brain_ocr_air_ticket brain_ocr_invoice brain_ocr_insurance_doc brain_formula brain_ocr_meter brain_doc_analysis brain_ocr_webimage_loc wise_adapt lebo_resource_base lightservice_public hetu_basic lightcms_map_poi kaidian_kaidian ApsMisTest_Test权限 vis-classify_flower lpq_开放 cop_helloScope ApsMis_fangdi_permission smartapp_snsapi_base smartapp_mapp_dev_manage iop_autocar oauth_tp_app smartapp_smart_game_openapi oauth_sessionkey smartapp_swanid_verify smartapp_opensource_openapi smartapp_opensource_recapi fake_face_detect_开放Scope vis-ocr_虚拟人物助理 idl-video_虚拟人物助理 smartapp_component', 'session_secret': '6ac230d6705697808e228241c519f303'}\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import requests \n",
|
||||||
|
"\n",
|
||||||
|
"# client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
|
||||||
|
"host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=KwXkGawxh0sjOQdF9Ae9LeLb&client_secret=siprEKMp5UcRTOAngEfIOOe9x6xkqGXq'\n",
|
||||||
|
"response = requests.get(host)\n",
|
||||||
|
"if response:\n",
|
||||||
|
" print(response.json())"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 14,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-03T03:24:08.412670Z",
|
||||||
|
"iopub.status.busy": "2020-11-03T03:24:08.411768Z",
|
||||||
|
"iopub.status.idle": "2020-11-03T03:24:08.813580Z",
|
||||||
|
"shell.execute_reply": "2020-11-03T03:24:08.810985Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-03T03:24:08.412567Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"已完成\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import json\n",
|
||||||
|
"import base64\n",
|
||||||
|
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
|
||||||
|
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||||||
|
"pargams = {\n",
|
||||||
|
" 'request_id': '22917135_2227436',\n",
|
||||||
|
" 'result_type': 'excel'\n",
|
||||||
|
"}\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||||||
|
"url_all = url + \"?access_token=\" + access_token\n",
|
||||||
|
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||||||
|
"info_1 = res.json()['result']['ret_msg']\n",
|
||||||
|
"excel_url=res.json()['result']['result_data']\n",
|
||||||
|
"excel_1=requests.get(excel_url).content\n",
|
||||||
|
"with open('识别结果12.xls','wb+') as f:\n",
|
||||||
|
" f.write(excel_1)\n",
|
||||||
|
"print(info_1)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 80,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-03T08:43:37.180693Z",
|
||||||
|
"iopub.status.busy": "2020-11-03T08:43:37.179800Z",
|
||||||
|
"iopub.status.idle": "2020-11-03T08:43:37.656055Z",
|
||||||
|
"shell.execute_reply": "2020-11-03T08:43:37.653668Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-03T08:43:37.180589Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"data": {
|
||||||
|
"text/plain": [
|
||||||
|
"list"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"execution_count": 80,
|
||||||
|
"metadata": {},
|
||||||
|
"output_type": "execute_result"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import json\n",
|
||||||
|
"import base64\n",
|
||||||
|
"import demjson\n",
|
||||||
|
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
|
||||||
|
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||||||
|
"pargams = {\n",
|
||||||
|
" 'request_id': '22917135_2227436',\n",
|
||||||
|
" 'result_type': 'json'\n",
|
||||||
|
"}\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||||||
|
"url_all = url + \"?access_token=\" + access_token\n",
|
||||||
|
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||||||
|
"#info_1 = res.json()['result']['ret_msg']\n",
|
||||||
|
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
|
||||||
|
"type(excel_1)\n",
|
||||||
|
"#excel_new = demjson.decode(excel_1)\n",
|
||||||
|
"#for m_col in excel_new['forms'][0]['body']:\n",
|
||||||
|
"# print(m_col)\n",
|
||||||
|
"m_xx =json.loads(excel_1)\n",
|
||||||
|
"\n",
|
||||||
|
"#with open('识别结果12.json','w') as fl:\n",
|
||||||
|
"# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
|
||||||
|
"#print(info_1)\n",
|
||||||
|
"#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))\n",
|
||||||
|
"print(m_xx['forms'][0]['body'])"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 识别保存为json文件"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 81,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-03T08:48:26.082167Z",
|
||||||
|
"iopub.status.busy": "2020-11-03T08:48:26.081266Z",
|
||||||
|
"iopub.status.idle": "2020-11-03T08:48:26.430039Z",
|
||||||
|
"shell.execute_reply": "2020-11-03T08:48:26.428236Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-03T08:48:26.082060Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"已完成\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import json\n",
|
||||||
|
"import base64\n",
|
||||||
|
"import demjson\n",
|
||||||
|
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
|
||||||
|
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
|
||||||
|
"pargams = {\n",
|
||||||
|
" 'request_id': '22917135_2227436',\n",
|
||||||
|
" 'result_type': 'json'\n",
|
||||||
|
"}\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
|
||||||
|
"url_all = url + \"?access_token=\" + access_token\n",
|
||||||
|
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
|
||||||
|
"#info_1 = res.json()['result']['ret_msg']\n",
|
||||||
|
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
|
||||||
|
"type(excel_1)\n",
|
||||||
|
"#excel_new = demjson.decode(excel_1)\n",
|
||||||
|
"#for m_col in excel_new['forms'][0]['body']:\n",
|
||||||
|
"# print(m_col)\n",
|
||||||
|
"m_xx =json.loads(excel_1)\n",
|
||||||
|
"\n",
|
||||||
|
"with open('识别结果12.json','w') as fl:\n",
|
||||||
|
" json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
|
||||||
|
"print(info_1)\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
},
|
||||||
|
"toc-autonumbering": true,
|
||||||
|
"toc-showmarkdowntxt": false,
|
||||||
|
"toc-showtags": false
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+499
@@ -0,0 +1,499 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 股票管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## 股票信息导入"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import baostock as bs\n",
|
||||||
|
"import pandas as pd\n",
|
||||||
|
"\n",
|
||||||
|
"#### 登陆系统 ####\n",
|
||||||
|
"lg = bs.login()\n",
|
||||||
|
"# 显示登陆返回信息\n",
|
||||||
|
"print('login respond error_code:'+lg.error_code)\n",
|
||||||
|
"print('login respond error_msg:'+lg.error_msg)\n",
|
||||||
|
"\n",
|
||||||
|
"#### 获取证券信息 ####\n",
|
||||||
|
"rs = bs.query_all_stock(day=\"2020-10-20\")\n",
|
||||||
|
"print('query_all_stock respond error_code:'+rs.error_code)\n",
|
||||||
|
"print('query_all_stock respond error_msg:'+rs.error_msg)\n",
|
||||||
|
"\n",
|
||||||
|
"#### 打印结果集 ####\n",
|
||||||
|
"data_list = []\n",
|
||||||
|
"while (rs.error_code == '0') & rs.next():\n",
|
||||||
|
" # 获取一条记录,将记录合并在一起\n",
|
||||||
|
" data_list.append(rs.get_row_data())\n",
|
||||||
|
"#results = pd.DataFrame(data_list, columns=rs.fields)\n",
|
||||||
|
"\n",
|
||||||
|
"#### 结果集输出到csv文件 #### \n",
|
||||||
|
"#result.to_csv(\"all_stock.csv\", encoding=\"utf-8\", index=False)\n",
|
||||||
|
"for result in data_list:\n",
|
||||||
|
" print(result)\n",
|
||||||
|
"\n",
|
||||||
|
"#### 登出系统 ####\n",
|
||||||
|
"bs.logout()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import baostock as bs\n",
|
||||||
|
"import pandas as pd\n",
|
||||||
|
"import pymysql\n",
|
||||||
|
"\n",
|
||||||
|
"# 登陆系统\n",
|
||||||
|
"lg = bs.login()\n",
|
||||||
|
"# 显示登陆返回信息\n",
|
||||||
|
"print('login respond error_code:'+lg.error_code)\n",
|
||||||
|
"print('login respond error_msg:'+lg.error_msg)\n",
|
||||||
|
"\n",
|
||||||
|
"# 获取证券基本资料\n",
|
||||||
|
"rs = bs.query_stock_basic(code=\"\")\n",
|
||||||
|
"# rs = bs.query_stock_basic(code_name=\"浦发银行\") # 支持模糊查询\n",
|
||||||
|
"print('query_stock_basic respond error_code:'+rs.error_code)\n",
|
||||||
|
"print('query_stock_basic respond error_msg:'+rs.error_msg)\n",
|
||||||
|
"\n",
|
||||||
|
"# 打印结果集\n",
|
||||||
|
"data_list = []\n",
|
||||||
|
"while (rs.error_code == '0') & rs.next():\n",
|
||||||
|
" # 获取一条记录,将记录合并在一起\n",
|
||||||
|
" data_list.append(rs.get_row_data())\n",
|
||||||
|
"#result = pd.DataFrame(data_list, columns=rs.fields)\n",
|
||||||
|
"# 结果集输出到csv文件\n",
|
||||||
|
"#result.to_csv(\"D:/stock_basic.csv\", encoding=\"gbk\", index=False)\n",
|
||||||
|
"\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"\n",
|
||||||
|
"sql = \"insert into stock_info (code,name,ipoDate,outDate,type,status) values(%s,%s,%s,%s,%s,%s)\"\n",
|
||||||
|
"try:\n",
|
||||||
|
" cursor.executemany(sql,tuple(data_list))\n",
|
||||||
|
" db.commit()\n",
|
||||||
|
" print(\"ok!\")\n",
|
||||||
|
"except:\n",
|
||||||
|
" # 如果发生错误则回滚\n",
|
||||||
|
" print(\"error!\")\n",
|
||||||
|
" db.rollback() \n",
|
||||||
|
"db.close()\n",
|
||||||
|
"# 登出系统\n",
|
||||||
|
"bs.logout()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 每日持有标的信息导入"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"tags": []
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import csv\n",
|
||||||
|
"import pymysql\n",
|
||||||
|
"\n",
|
||||||
|
"def read_data(filename):\n",
|
||||||
|
" detail = {}\n",
|
||||||
|
" with open(filename) as f:\n",
|
||||||
|
" reader = csv.reader(f)\n",
|
||||||
|
" #header_row =next(reader)\n",
|
||||||
|
" for row in reader:\n",
|
||||||
|
" detail.setdefault(row[0],[])\n",
|
||||||
|
" detail[row[0]].append(row[1])\n",
|
||||||
|
" return detail\n",
|
||||||
|
"m_xx = []\n",
|
||||||
|
"m_stock = {}\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"sql = 'select code,name from stock_info where type=\"1\"'\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"results = cursor.fetchall()\n",
|
||||||
|
"for result in results:\n",
|
||||||
|
" m_stock[result[0]] = result[1]\n",
|
||||||
|
"#print(m_stock)\n",
|
||||||
|
"filename = '每日标的信息.csv'\n",
|
||||||
|
"detail = read_data(filename)\n",
|
||||||
|
"for k,v in detail.items():\n",
|
||||||
|
" m_rq = '2020-' + k[0:2] + '-' + k[2:]\n",
|
||||||
|
" i = 1\n",
|
||||||
|
" for m_dm in v:\n",
|
||||||
|
" if m_dm[0:1] == '6':\n",
|
||||||
|
" m_dm = 'sh.' + m_dm\n",
|
||||||
|
" else:\n",
|
||||||
|
" m_dm = 'sz.' + m_dm\n",
|
||||||
|
" print('\\t'+m_dm + '\\t'+m_stock[m_dm])\n",
|
||||||
|
" m_xx.append((m_rq,m_dm,i))\n",
|
||||||
|
" i += 1\n",
|
||||||
|
"choice = input('以上为本日数据,是否导入?(y/n)')\n",
|
||||||
|
"if choice.upper() == \"Y\":\n",
|
||||||
|
" sql = \"insert into daily_item (rq,code,ord) values(%s,%s,%s)\"\n",
|
||||||
|
" try:\n",
|
||||||
|
" cursor.executemany(sql,m_xx)\n",
|
||||||
|
" db.commit()\n",
|
||||||
|
" print(\"ok!\")\n",
|
||||||
|
" except:\n",
|
||||||
|
" # 如果发生错误则回滚\n",
|
||||||
|
" print(\"error!\")\n",
|
||||||
|
" db.rollback() \n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 每日调仓信息导入"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import csv\n",
|
||||||
|
"import pymysql\n",
|
||||||
|
"\n",
|
||||||
|
"def read_data(filename):\n",
|
||||||
|
" detail = {}\n",
|
||||||
|
" \n",
|
||||||
|
" with open(filename) as f:\n",
|
||||||
|
" reader = csv.reader(f)\n",
|
||||||
|
"# header_row =next(reader)\n",
|
||||||
|
" for row in reader:\n",
|
||||||
|
" detail1 = {}\n",
|
||||||
|
" detail.setdefault(row[3],[])\n",
|
||||||
|
" detail1 = {'dm':row[0],'zj':row[1],'ly':row[2],'cb':row[4]}\n",
|
||||||
|
" detail[row[3]].append(detail1)\n",
|
||||||
|
" return detail\n",
|
||||||
|
"m_xx = []\n",
|
||||||
|
"m_add = []\n",
|
||||||
|
"m_sub = []\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"filename = '调仓明细.csv'\n",
|
||||||
|
"detail = read_data(filename)\n",
|
||||||
|
"#print(detail)\n",
|
||||||
|
"for rq in sorted(detail.keys()):\n",
|
||||||
|
" m_rq = '2020-' + rq[0:2] + '-' + rq[2:]\n",
|
||||||
|
" for xx_move in detail[rq]:\n",
|
||||||
|
" m_dm = xx_move['dm']\n",
|
||||||
|
" if m_dm[0:1] == '6':\n",
|
||||||
|
" m_dm = 'sh.' + m_dm\n",
|
||||||
|
" else:\n",
|
||||||
|
" m_dm = 'sz.' + m_dm\n",
|
||||||
|
" m_xx.append((m_rq,m_dm,int(xx_move['zj']),xx_move['ly']))\n",
|
||||||
|
" if xx_move['zj'] == '1':\n",
|
||||||
|
" m_add.append((m_dm,xx_move['ly'],float(xx_move['cb'])))\n",
|
||||||
|
" else:\n",
|
||||||
|
" m_sub.append((m_dm))\n",
|
||||||
|
"#print(m_xx)\n",
|
||||||
|
"sql = \"insert into change_item (rq,code,pos,reason) values(%s,%s,%s,%s)\"\n",
|
||||||
|
"try:\n",
|
||||||
|
" cursor.executemany(sql,m_xx)\n",
|
||||||
|
" if len(m_add) > 0:\n",
|
||||||
|
" sql_add = \"insert into stock_item (code,reason,cost) values(%s,%s,%s)\"\n",
|
||||||
|
" cursor.executemany(sql_add,m_add)\n",
|
||||||
|
" if len(m_sub) > 0:\n",
|
||||||
|
" sql_sub = \"update stock_item set status=0 where code=%s\"\n",
|
||||||
|
" cursor.executemany(sql_sub,m_sub) \n",
|
||||||
|
" db.commit()\n",
|
||||||
|
" print(\"导入成功!\")\n",
|
||||||
|
"except:\n",
|
||||||
|
" # 如果发生错误则回滚\n",
|
||||||
|
" print(\"error!\")\n",
|
||||||
|
" db.rollback() \n",
|
||||||
|
"\n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## 持仓股票信息导入"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"sql = 'SELECT a.unit,a.CODE,b.cost FROM daily_item AS a,daily_cost as b WHERE a.rq=(select max(rq) FROM daily_item) AND a.code=b.code'\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"results = cursor.fetchall()\n",
|
||||||
|
"sql = \"insert into stock_item(item,code,cost) values(%s,%s,%s)\"\n",
|
||||||
|
"try:\n",
|
||||||
|
" cursor.executemany(sql,list(results))\n",
|
||||||
|
" #db.commit()\n",
|
||||||
|
" print(\"ok!\")\n",
|
||||||
|
"except:\n",
|
||||||
|
" # 如果发生错误则回滚\n",
|
||||||
|
" print(\"error!\")\n",
|
||||||
|
" db.rollback() \n",
|
||||||
|
"sql = 'SELECT a.reason,a.code from change_item AS a WHERE a.pos=1'\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"results = cursor.fetchall()\n",
|
||||||
|
"sql = \"update stock_item set reason=%s where code=%s \"\n",
|
||||||
|
"try:\n",
|
||||||
|
" cursor.executemany(sql,list(results))\n",
|
||||||
|
" #db.commit()\n",
|
||||||
|
" print(\"ok!\")\n",
|
||||||
|
"except:\n",
|
||||||
|
" # 如果发生错误则回滚\n",
|
||||||
|
" print(\"error!\")\n",
|
||||||
|
" db.rollback() \n",
|
||||||
|
"db.close()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 外汇实时数据采集"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import time\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"#pattern = re.compile('(?<=\\\").*(?=\\\")')\n",
|
||||||
|
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
|
||||||
|
"#myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"#mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"#mycol = mydb[\"news\"]\n",
|
||||||
|
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
|
||||||
|
"strhtml = requests.get(url)\n",
|
||||||
|
"#strhtml.encoding = 'utf8'\n",
|
||||||
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
"data = strhtml.text\n",
|
||||||
|
"#data1 = data.split(\"\\r\")\n",
|
||||||
|
"#data = soup.select('schoolList')\n",
|
||||||
|
"#for item1 in data:\n",
|
||||||
|
"# print(item1.get_text())\n",
|
||||||
|
"if pattern.findall(data):\n",
|
||||||
|
" for data1 in pattern.findall(data):\n",
|
||||||
|
" data2 = data1.split(',')\n",
|
||||||
|
" print(data2)\n",
|
||||||
|
" print('当前买入价:',data2[1])\n",
|
||||||
|
" print('当前卖出价:',data2[2])\n",
|
||||||
|
" print('昨收价:',data2[3])\n",
|
||||||
|
" print('今开价:',data2[5])\n",
|
||||||
|
" print('最高价:',data2[6])\n",
|
||||||
|
" print('最低价:',data2[7])\n",
|
||||||
|
" print(float(data2[7]))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from email.mime.text import MIMEText\n",
|
||||||
|
"from email.header import Header\n",
|
||||||
|
"import smtplib\n",
|
||||||
|
"import requests\n",
|
||||||
|
"import time\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
|
||||||
|
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
|
||||||
|
"strhtml = requests.get(url)\n",
|
||||||
|
"data = strhtml.text\n",
|
||||||
|
"if pattern.findall(data):\n",
|
||||||
|
" for data1 in pattern.findall(data):\n",
|
||||||
|
" data2 = data1.split(',')\n",
|
||||||
|
"#print(data2)\n",
|
||||||
|
"with open('price.txt','r') as fl:\n",
|
||||||
|
" for line in fi:\n",
|
||||||
|
" p_high = line.split(',')[0]\n",
|
||||||
|
" p_low = line.split(',')[1]\n",
|
||||||
|
"\n",
|
||||||
|
"message ='当前美元加元买入价:{},卖出价:{}'.format(data2[1],data2[2])\n",
|
||||||
|
"msg = MIMEText(message,'plain','utf-8')\n",
|
||||||
|
"msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
|
||||||
|
"msg['From'] = Header('512song@sina.com')\n",
|
||||||
|
"msg['To'] = Header('491525765@qq.com','utf-8')\n",
|
||||||
|
"\n",
|
||||||
|
"from_addr = '512song@sina.com' #发件邮箱\n",
|
||||||
|
"password = '409fe5d8471da663' #邮箱密码\n",
|
||||||
|
"to_addr = 'songyi@yeah.net' #收件邮箱\n",
|
||||||
|
"smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
|
||||||
|
"try:\n",
|
||||||
|
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
|
||||||
|
" print('开始登录')\n",
|
||||||
|
" server.set_debuglevel(1) \n",
|
||||||
|
" server.login(from_addr,password) #登录邮箱\n",
|
||||||
|
" print('登录成功')\n",
|
||||||
|
" print(\"邮件开始发送\")\n",
|
||||||
|
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
|
||||||
|
" server.quit()\n",
|
||||||
|
" print(\"邮件发送成功\")\n",
|
||||||
|
"except smtplib.SMTPException as e:\n",
|
||||||
|
" print(\"邮件发送失败\",e)\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from email.mime.text import MIMEText\n",
|
||||||
|
"from email.header import Header\n",
|
||||||
|
"import smtplib\n",
|
||||||
|
"import requests\n",
|
||||||
|
"import time\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
|
||||||
|
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
|
||||||
|
"strhtml = requests.get(url)\n",
|
||||||
|
"data = strhtml.text\n",
|
||||||
|
"if pattern.findall(data):\n",
|
||||||
|
" for data1 in pattern.findall(data):\n",
|
||||||
|
" data2 = data1.split(',')\n",
|
||||||
|
"print(data2)\n",
|
||||||
|
"\n",
|
||||||
|
"message ='当前美元加元最高价:{},最低价:{}'.formtat()\n",
|
||||||
|
"msg = MIMEText(message,'plain','utf-8')\n",
|
||||||
|
"\n",
|
||||||
|
"msg['Subject'] = Header(\"测试smtp邮件\",'utf-8')\n",
|
||||||
|
"msg['From'] = Header('512song@sina.com')\n",
|
||||||
|
"msg['To'] = Header('491525765@qq.com','utf-8')\n",
|
||||||
|
"\n",
|
||||||
|
"from_addr = '512song@sina.com' #发件邮箱\n",
|
||||||
|
"password = '409fe5d8471da663' #邮箱密码\n",
|
||||||
|
"to_addr = '491525765@qq.com' #收件邮箱\n",
|
||||||
|
"\n",
|
||||||
|
"smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
|
||||||
|
"try:\n",
|
||||||
|
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
|
||||||
|
" print('开始登录')\n",
|
||||||
|
" server.set_debuglevel(1) \n",
|
||||||
|
" server.login(from_addr,password) #登录邮箱\n",
|
||||||
|
" print('登录成功')\n",
|
||||||
|
" print(\"邮件开始发送\")\n",
|
||||||
|
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
|
||||||
|
" server.quit()\n",
|
||||||
|
" print(\"邮件发送成功\")\n",
|
||||||
|
"except smtplib.SMTPException as e:\n",
|
||||||
|
" print(\"邮件发送失败\",e)\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"toc-hr-collapsed": true,
|
||||||
|
"toc-nb-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"# pygal图表"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pygal\n",
|
||||||
|
"bar_chart = pygal.Bar(height=300)\n",
|
||||||
|
"bar_chart.add('Fibonacci', [0, 1, 1, 2, 3, 5, 8, 13, 21, 34, 55])\n",
|
||||||
|
"bar_chart.add('Padovan', [1, 1, 1, 2, 2, 3, 4, 5, 7, 9, 12])\n",
|
||||||
|
"svg = bar_chart.render()"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from IPython.display import SVG\n",
|
||||||
|
"SVG(svg)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.5"
|
||||||
|
},
|
||||||
|
"toc-autonumbering": true,
|
||||||
|
"toc-showcode": true,
|
||||||
|
"toc-showmarkdowntxt": true
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+74
@@ -0,0 +1,74 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"#gTTS语音\n",
|
||||||
|
"from gtts import gTTS\n",
|
||||||
|
"#engine = pyttsx3.init('espeak')\n",
|
||||||
|
"tts = gTTS(text=\"军士上前,将英玉兰架起,两个抓着脚踝,两个托住肩头,一起用力,英玉兰无奈分开两条大浪腿,露出骚屄,被举下\",lang='zh-cn')\n",
|
||||||
|
"#engine.save_to_file(\"欢迎使用百度语音合成,本次测试为Python接口\",'./test')\n",
|
||||||
|
"tts.save(\"./test.mp3\")\n",
|
||||||
|
"print('ok')"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"#百度语音在线\n",
|
||||||
|
"from aip import AipSpeech\n",
|
||||||
|
"\n",
|
||||||
|
"\"\"\" 你的 APPID AK SK \"\"\"\n",
|
||||||
|
"APP_ID = '17553946'\n",
|
||||||
|
"API_KEY = 'i5LBXalkBn2KHTqdifesA1EB'\n",
|
||||||
|
"SECRET_KEY = '1ktP6qDH7nMFHjotRdKpnI13w41Bvk59'\n",
|
||||||
|
"\n",
|
||||||
|
"client = AipSpeech(APP_ID, API_KEY, SECRET_KEY)\n",
|
||||||
|
"\n",
|
||||||
|
"result = client.synthesis('军士上前,将英玉兰架起,两个抓着脚踝,两个托住肩头,一起用力,英玉兰无奈分开两条大浪腿,露出骚屄,被举下。', 'zh', 5, {\n",
|
||||||
|
" 'vol': 5,'per': 4,\n",
|
||||||
|
"})\n",
|
||||||
|
"\n",
|
||||||
|
"# 识别正确返回语音二进制 错误则返回dict 参照下面错误码\n",
|
||||||
|
"if not isinstance(result, dict):\n",
|
||||||
|
" with open('./test3.mp3', 'wb') as f:\n",
|
||||||
|
" f.write(result)\n",
|
||||||
|
"else:print(result)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.5"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+235
@@ -0,0 +1,235 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 重点高校信息管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 强基计划信息管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"### 录入强基计划信息"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 3,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-27T07:02:30.612429Z",
|
||||||
|
"iopub.status.busy": "2020-11-27T07:02:30.611524Z",
|
||||||
|
"iopub.status.idle": "2020-11-27T07:02:31.128055Z",
|
||||||
|
"shell.execute_reply": "2020-11-27T07:02:31.124461Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-27T07:02:30.612326Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import time\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"qiangji\"]\n",
|
||||||
|
"m_content = ''\n",
|
||||||
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||||||
|
"url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n",
|
||||||
|
"strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
"strhtml.encoding = 'utf8'\n",
|
||||||
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
"#data = strhtml.text\n",
|
||||||
|
"#data1 = data.split(\"\\r\")\n",
|
||||||
|
"data = soup.select('body > div.container.content > div > div > div.col-md-9.col-sm-8 > div')\n",
|
||||||
|
"for item1 in data:\n",
|
||||||
|
" m_content +=item1.get_text()\n",
|
||||||
|
"#print(m_content)\n",
|
||||||
|
"m_title = soup.select('body > div.y_tit_box > div > div > div > h1')\n",
|
||||||
|
"m_name = '中国人民大学'\n",
|
||||||
|
"m_code = 'A002'\n",
|
||||||
|
"m_year = '2020'\n",
|
||||||
|
"myquery = {'code':m_code}\n",
|
||||||
|
"x = mycol.count_documents(myquery)\n",
|
||||||
|
"#print(x)\n",
|
||||||
|
"m_id = time.strftime(\"%Y%m%d%H%M%S\", time.localtime())\n",
|
||||||
|
"if x > 0: \n",
|
||||||
|
" m_mg = {} \n",
|
||||||
|
" m_mg.setdefault(m_id,{})\n",
|
||||||
|
" m_mg[m_id]['title'] = m_title[0].text\n",
|
||||||
|
" m_mg[m_id]['content'] = m_content\n",
|
||||||
|
" m_mg[m_id]['url'] = url\n",
|
||||||
|
" mycol.update_one(myquery,{'$push':{m_year:m_mg}})\n",
|
||||||
|
" print(x,s)\n",
|
||||||
|
"else:\n",
|
||||||
|
" m_mg = {}\n",
|
||||||
|
" m_mg['name'] = m_name\n",
|
||||||
|
" m_mg['code'] = m_code\n",
|
||||||
|
" m_mg.setdefault(m_year,[])\n",
|
||||||
|
" m_mg1 = {}\n",
|
||||||
|
" m_mg1.setdefault(m_id,{})\n",
|
||||||
|
" m_mg1[m_id]['title'] = m_title[0].text\n",
|
||||||
|
" m_mg1[m_id]['content'] = m_content\n",
|
||||||
|
" m_mg1[m_id]['url'] = url\n",
|
||||||
|
" m_mg[m_year].append(m_mg1)\n",
|
||||||
|
" mycol.insert_one(m_mg) \n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"### 录入强基计划附件"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 73,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2020-11-26T10:35:49.756505Z",
|
||||||
|
"iopub.status.busy": "2020-11-26T10:35:49.755575Z",
|
||||||
|
"iopub.status.idle": "2020-11-26T10:35:50.001279Z",
|
||||||
|
"shell.execute_reply": "2020-11-26T10:35:49.998646Z",
|
||||||
|
"shell.execute_reply.started": "2020-11-26T10:35:49.756395Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"1 gaokao.qiangji.2020\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import time\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import os\n",
|
||||||
|
"from gridfs import GridFS\n",
|
||||||
|
"\n",
|
||||||
|
"def upLoadFile(file_coll,file_name,data_link): \n",
|
||||||
|
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
|
||||||
|
" gridfs_col = GridFS(mydb, collection=file_coll)\n",
|
||||||
|
" file_ = \"0\"\n",
|
||||||
|
" query = {\"filename\":\"\"}\n",
|
||||||
|
" query[\"filename\"] = file_name\n",
|
||||||
|
" if gridfs_col.exists(query):\n",
|
||||||
|
" print('已经存在该文件')\n",
|
||||||
|
" else:\n",
|
||||||
|
" with open(file_name, 'rb') as file_r:\n",
|
||||||
|
" file_data = file_r.read()\n",
|
||||||
|
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
|
||||||
|
"\n",
|
||||||
|
" #print(file_)\n",
|
||||||
|
" return file_ \n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"qiangji\"]\n",
|
||||||
|
"m_content = ''\n",
|
||||||
|
"url = 'https://www.bkzs.sdu.edu.cn/info/1036/1635.htm'\n",
|
||||||
|
"m_title = ''\n",
|
||||||
|
"m_name = '山东大学'\n",
|
||||||
|
"m_code = 'A422'\n",
|
||||||
|
"m_year = '2020'\n",
|
||||||
|
"myquery = {'code':m_code}\n",
|
||||||
|
"x = mycol.count_documents(myquery)\n",
|
||||||
|
"m_id = time.strftime(\"%Y%m%d%H%M%S\", time.localtime())\n",
|
||||||
|
"m_dir = './data/tmp'\n",
|
||||||
|
"fls=os.listdir(m_dir)\n",
|
||||||
|
"l_fls =[]\n",
|
||||||
|
"for fl in fls: \n",
|
||||||
|
" full_path = m_dir+ '/' + fl\n",
|
||||||
|
" l_fls.append(upLoadFile(\"document\",full_path,\"\"))\n",
|
||||||
|
"m_mg = {}\n",
|
||||||
|
"if x > 0: \n",
|
||||||
|
" m_mg.setdefault(m_id,{})\n",
|
||||||
|
" m_mg[m_id]['title'] = m_title\n",
|
||||||
|
" m_mg[m_id]['files'] = l_fls\n",
|
||||||
|
" m_mg[m_id]['url'] = url\n",
|
||||||
|
" mycol.update_one(myquery,{'$push':{m_year:m_mg}})\n",
|
||||||
|
"else:\n",
|
||||||
|
" m_mg['name'] = m_name\n",
|
||||||
|
" m_mg['code'] = m_code\n",
|
||||||
|
" m_mg.setdefault(m_year,[])\n",
|
||||||
|
" m_mg1 = {}\n",
|
||||||
|
" m_mg1.setdefault(m_id,{})\n",
|
||||||
|
" m_mg1[m_id]['title'] = m_title\n",
|
||||||
|
" m_mg[m_id]['files'] = l_fls\n",
|
||||||
|
" m_mg1[m_id]['url'] = url\n",
|
||||||
|
" m_mg[m_year].append(m_mg1)\n",
|
||||||
|
" mycol.insert_one(m_mg) \n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 1,
|
||||||
|
"metadata": {
|
||||||
|
"execution": {
|
||||||
|
"iopub.execute_input": "2021-02-24T08:13:56.382398Z",
|
||||||
|
"iopub.status.busy": "2021-02-24T08:13:56.381289Z",
|
||||||
|
"iopub.status.idle": "2021-02-24T08:13:56.397601Z",
|
||||||
|
"shell.execute_reply": "2021-02-24T08:13:56.395607Z",
|
||||||
|
"shell.execute_reply.started": "2021-02-24T08:13:56.382135Z"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"20210224161356\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import time\n",
|
||||||
|
"\n",
|
||||||
|
"print(time.strftime(\"%Y%m%d%H%M%S\", time.localtime()))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
},
|
||||||
|
"toc-autonumbering": false
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+150
@@ -0,0 +1,150 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "488c74e2-3b75-4274-8d9a-4503f367b5ca",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 高考志愿查询"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "8913f77b-bbd5-4959-b5a4-3214732f8480",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 按位次模糊查询专业"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 80,
|
||||||
|
"id": "be2371be-89f5-4879-82a9-4a1d9e6376d3",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"1 南京中医药大学 中医学(本硕连读5+3一体化) 11019\n",
|
||||||
|
"2 天津医科大学 预防医学 11108\n",
|
||||||
|
"3 哈尔滨医科大学 临床医学 11227\n",
|
||||||
|
"4 空军军医大学 基础医学 11232\n",
|
||||||
|
"5 南京医科大学 预防医学 11275\n",
|
||||||
|
"6 温州医科大学 眼视光医学(5+3一体化) 11564\n",
|
||||||
|
"7 上海中医药大学 中医学(5+3一体化针灸推拿英语方向) 11620\n",
|
||||||
|
"8 兰州大学 临床医学类 11662\n",
|
||||||
|
"9 天津中医药大学 中医学(5+3一体化) 11825\n",
|
||||||
|
"10 哈尔滨医科大学 临床医学(5+3一体化,儿科学硕士) 11837\n",
|
||||||
|
"11 温州医科大学 临床医学(5+3一体化) 11869\n",
|
||||||
|
"12 东北大学 智能医学工程 11916\n",
|
||||||
|
"13 海军军医大学 中医学(中医临床医师) 11955\n",
|
||||||
|
"14 苏州大学 预防医学 11991\n",
|
||||||
|
"15 暨南大学 临床医学 12045\n",
|
||||||
|
"16 中国医科大学 医学影像学 12059\n",
|
||||||
|
"17 南昌大学 临床医学 12252\n",
|
||||||
|
"18 天津医科大学 医学影像技术 12272\n",
|
||||||
|
"19 吉林大学 预防医学 12320\n",
|
||||||
|
"20 南京航空航天大学 生物医学工程 12504\n",
|
||||||
|
"21 江南大学 临床医学 12526\n",
|
||||||
|
"22 广州中医药大学 中医学(5+3一体化) 12715\n",
|
||||||
|
"23 中国医科大学 基础医学 12755\n",
|
||||||
|
"24 重庆医科大学 医学影像学 12784\n",
|
||||||
|
"25 大连医科大学 临床医学(5+3一体化) 13012\n",
|
||||||
|
"26 天津医科大学 智能医学工程 13103\n",
|
||||||
|
"27 南京中医药大学 中医学 13301\n",
|
||||||
|
"28 温州医科大学 眼视光医学 13370\n",
|
||||||
|
"29 天津医科大学 医学检验技术 13371\n",
|
||||||
|
"30 天津医科大学 生物医学工程 13396\n",
|
||||||
|
"31 温州医科大学 临床医学 13421\n",
|
||||||
|
"32 中国医科大学 预防医学 13769\n",
|
||||||
|
"33 汕头大学 临床医学(5+3一体化) 13808\n",
|
||||||
|
"34 郑州大学 临床医学类 13981\n",
|
||||||
|
"35 郑州大学 口腔医学 14147\n",
|
||||||
|
"36 南方医科大学 基础医学 14160\n",
|
||||||
|
"37 哈尔滨医科大学 基础医学 14331\n",
|
||||||
|
"38 南方医科大学 预防医学 14366\n",
|
||||||
|
"39 大连医科大学 临床医学 14423\n",
|
||||||
|
"40 西北大学 临床医学 14492\n",
|
||||||
|
"41 天津中医药大学 中医学(5+3一体化中医儿科学) 14708\n",
|
||||||
|
"42 南京医科大学 智能医学工程 14721\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"admission_2020\"]\n",
|
||||||
|
"\n",
|
||||||
|
"m_zhuanye = '医学'\n",
|
||||||
|
"m_rank1 = 11000\n",
|
||||||
|
"m_rank2 = 15000\n",
|
||||||
|
"\n",
|
||||||
|
"myquery = {'spe_name':re.compile(m_zhuanye),'rank_min':{\"$gte\": m_rank1,\"$lte\": m_rank2}}\n",
|
||||||
|
"i = 1\n",
|
||||||
|
"for x in mycol.find(myquery,{\"_id\": 0, }):\n",
|
||||||
|
" print(i,x['col_name'],x['spe_name'],x['rank_min'])\n",
|
||||||
|
" i += 1\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"id": "2731820f-b81e-4d7e-8f53-a2501d9c943f",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 按分数模糊查询专业"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"id": "07dbb641-d594-4c7c-92da-045a5370e6e9",
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"admission_2020\"]\n",
|
||||||
|
"\n",
|
||||||
|
"m_zhuanye = '医学'\n",
|
||||||
|
"m_fenshu = 600\n",
|
||||||
|
"\n",
|
||||||
|
"myquery = {'spe_name':re.compile(m_zhuanye),'num_min':{\"$gte\": m_fenshu}}\n",
|
||||||
|
"i = 1\n",
|
||||||
|
"for x in mycol.find(myquery,{\"_id\": 0, }):\n",
|
||||||
|
" print(i,x['col_name'],x['spe_name'],x['num_min'])\n",
|
||||||
|
" i += 1\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
},
|
||||||
|
"toc-autonumbering": true,
|
||||||
|
"toc-showmarkdowntxt": true,
|
||||||
|
"toc-showtags": false
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 5
|
||||||
|
}
|
||||||
+690
@@ -0,0 +1,690 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"tags": [],
|
||||||
|
"toc-hr-collapsed": true
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"# 高考志愿管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 2020年高考录取信息导入MongoDB"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"tags": []
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymysql\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import decimal\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"college\"]\n",
|
||||||
|
"mycol1 = mydb[\"admission_2020\"]\n",
|
||||||
|
"\n",
|
||||||
|
"m_col = {}\n",
|
||||||
|
"m_spe = {}\n",
|
||||||
|
"m_xx = {}\n",
|
||||||
|
"for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n",
|
||||||
|
" m_col[x['code']] = x['name']\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n",
|
||||||
|
"cursor = db.cursor()\n",
|
||||||
|
"sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2020\"'\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"results = cursor.fetchall()\n",
|
||||||
|
"for result in results:\n",
|
||||||
|
" m_spe.setdefault(result[1],{}) \n",
|
||||||
|
" m_spe[result[1]][result[0]] = result[2]\n",
|
||||||
|
"sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'\n",
|
||||||
|
"cursor.execute(sql)\n",
|
||||||
|
"results = cursor.fetchall()\n",
|
||||||
|
"i = 0\n",
|
||||||
|
"m_min = 0\n",
|
||||||
|
"ii = 0\n",
|
||||||
|
"for result in results:\n",
|
||||||
|
" m_xx.clear()\n",
|
||||||
|
" if result[6] == m_min:\n",
|
||||||
|
" ii = ii\n",
|
||||||
|
" i = i+1\n",
|
||||||
|
" else:\n",
|
||||||
|
" i = i+1\n",
|
||||||
|
" ii = i\n",
|
||||||
|
" m_min = result[6]\n",
|
||||||
|
" m_xx['pos'] = ii\n",
|
||||||
|
" m_xx['col_code'] = result[1]\n",
|
||||||
|
" m_xx['col_name'] = m_col[result[1]]\n",
|
||||||
|
" m_xx['spe_code'] = result[2]\n",
|
||||||
|
" m_xx['spe_name'] = m_spe[result[1]][result[2]]\n",
|
||||||
|
" m_xx['plan'] = result[3]\n",
|
||||||
|
" m_xx['dispense'] = result[5]\n",
|
||||||
|
" m_xx['num_min'] = result[6]\n",
|
||||||
|
" m_xx['num_avg'] = int(result[7])\n",
|
||||||
|
" m_xx['rank_min'] = result[8]\n",
|
||||||
|
" m_xx['nian'] = '2020' \n",
|
||||||
|
" mycol1.insert_one(m_xx)\n",
|
||||||
|
"#print(m_spe)\n",
|
||||||
|
"\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 计算志愿分数概率"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import random\n",
|
||||||
|
"\n",
|
||||||
|
"array2 = []\n",
|
||||||
|
"for i in range(300000):\n",
|
||||||
|
" array1 = []\n",
|
||||||
|
" s = 0\n",
|
||||||
|
" for ii in range(60):\n",
|
||||||
|
" m1 = random.randint(580,595)\n",
|
||||||
|
" array1.append(m1)\n",
|
||||||
|
" s = s + m1 \n",
|
||||||
|
" m_avg = round(s/60,2)\n",
|
||||||
|
" if m_avg == 582.8 and (580 in array1) and (595 in array1):\n",
|
||||||
|
" #print(array1)\n",
|
||||||
|
" array2.extend(array1)\n",
|
||||||
|
" \n",
|
||||||
|
"#print(array2)\n",
|
||||||
|
"m_set = set(array2)\n",
|
||||||
|
"for m in m_set:\n",
|
||||||
|
" print(m,array2.count(m))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 导入山东大学录取明细"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import openpyxl\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"admission_college\"]\n",
|
||||||
|
"wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')\n",
|
||||||
|
"sheet = wb.active\n",
|
||||||
|
"#sheets = wb.sheetnames\n",
|
||||||
|
"code = 'A422'\n",
|
||||||
|
"name = '山东大学'\n",
|
||||||
|
"new_col = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"\n",
|
||||||
|
"new_code = []\n",
|
||||||
|
"dict1['code'] = code\n",
|
||||||
|
"dict1['name'] = name\n",
|
||||||
|
"for n in range(1,sheet.max_row+1):\n",
|
||||||
|
" nian = str(sheet.cell(n,1).value)\n",
|
||||||
|
" dict2 = {}\n",
|
||||||
|
" \n",
|
||||||
|
" dict1.setdefault(nian,[])\n",
|
||||||
|
" if sheet.cell(n,2).value =='理工':\n",
|
||||||
|
" m_lb = 'l'\n",
|
||||||
|
" elif sheet.cell(n,2).value =='文史':\n",
|
||||||
|
" m_lb = 'w'\n",
|
||||||
|
" else:\n",
|
||||||
|
" m_lb = 'z' \n",
|
||||||
|
" dict2['type'] = m_lb\n",
|
||||||
|
" dict2['spe_name'] = sheet.cell(n,4).value\n",
|
||||||
|
" dict2['max_score'] = sheet.cell(n,5).value\n",
|
||||||
|
" dict2['min_score'] = sheet.cell(n,6).value\n",
|
||||||
|
" dict2['avg_score'] = sheet.cell(n,7).value\n",
|
||||||
|
" dict2['dispense'] = sheet.cell(n,8).value\n",
|
||||||
|
" dict1[nian].append(dict2)\n",
|
||||||
|
"mycol.insert_one(dict1)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 导入山东师范大学录取明细"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import openpyxl\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"admission_college\"]\n",
|
||||||
|
"wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')\n",
|
||||||
|
"sheet = wb.active\n",
|
||||||
|
"#sheets = wb.sheetnames\n",
|
||||||
|
"code = 'A423'\n",
|
||||||
|
"name = '中国海洋大学'\n",
|
||||||
|
"new_col = []\n",
|
||||||
|
"dict1 = {}\n",
|
||||||
|
"\n",
|
||||||
|
"new_code = []\n",
|
||||||
|
"dict1['code'] = code\n",
|
||||||
|
"dict1['name'] = name\n",
|
||||||
|
"for n in range(1,sheet.max_row+1):\n",
|
||||||
|
" nian = str(sheet.cell(n,1).value)\n",
|
||||||
|
" dict2 = {}\n",
|
||||||
|
" \n",
|
||||||
|
" dict1.setdefault(nian,[])\n",
|
||||||
|
" if sheet.cell(n,2).value =='理工':\n",
|
||||||
|
" m_lb = 'l'\n",
|
||||||
|
" elif sheet.cell(n,2).value =='文史':\n",
|
||||||
|
" m_lb = 'w'\n",
|
||||||
|
" else:\n",
|
||||||
|
" m_lb = 'z'\n",
|
||||||
|
" dict2['type'] = m_lb\n",
|
||||||
|
" dict2['spe_name'] = sheet.cell(n,4).value\n",
|
||||||
|
" dict2['max_score'] = sheet.cell(n,7).value\n",
|
||||||
|
" dict2['min_score'] = sheet.cell(n,5).value\n",
|
||||||
|
" dict2['avg_score'] = sheet.cell(n,6).value\n",
|
||||||
|
" if sheet.cell(n,8).value:\n",
|
||||||
|
" dict2['dispense'] = sheet.cell(n,8).value\n",
|
||||||
|
" dict1[nian].append(dict2)\n",
|
||||||
|
"mycol.insert_one(dict1)\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 按地区列示高校"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"college\"]\n",
|
||||||
|
"m_city = []\n",
|
||||||
|
"m_xx = {}\n",
|
||||||
|
"myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}\n",
|
||||||
|
"colleges = mycol.find(myquery,{ \"_id\": 0, \"name\": 1, \"code\": 1,'city':1 })\n",
|
||||||
|
"for x in colleges:\n",
|
||||||
|
" m_xx1 = []\n",
|
||||||
|
" c = x['city']\n",
|
||||||
|
" m_xx.setdefault(c,[])\n",
|
||||||
|
" m_xx1.append(x['code'])\n",
|
||||||
|
" m_xx1.append(x['name'])\n",
|
||||||
|
" m_xx[c].append(m_xx1)\n",
|
||||||
|
"#print(m_city)\n",
|
||||||
|
"#按照原顺序对高校所在城市排序\n",
|
||||||
|
"'''\n",
|
||||||
|
"city = list(set(m_city))\n",
|
||||||
|
"city.sort(key=m_city.index)\n",
|
||||||
|
"for c in city:\n",
|
||||||
|
"# m_xx['city'] = c\n",
|
||||||
|
" m_xx.setdefault(c,[])\n",
|
||||||
|
" for y in colleges:\n",
|
||||||
|
" print(y['code'],y['name'])\n",
|
||||||
|
" \n",
|
||||||
|
"#m_xx \n",
|
||||||
|
"''' \n",
|
||||||
|
"for k,v in m_xx.items():\n",
|
||||||
|
" print(k)\n",
|
||||||
|
" for mm in v:\n",
|
||||||
|
" print(mm[0],mm[1])"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 显示学校2020年招生信息"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"admission_2020\"]\n",
|
||||||
|
"code = 'A422'\n",
|
||||||
|
"m_city = []\n",
|
||||||
|
"m_xx = {}\n",
|
||||||
|
"myquery = {'col_code':code}\n",
|
||||||
|
"colleges = mycol.find(myquery,{ \"_id\": 0 , \"col_code\":0,\"col_name\":0,'plan':0,'nian':0}).sort('pos')\n",
|
||||||
|
"for x in colleges:\n",
|
||||||
|
" print(x)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 2021年拟在山东招生普通高校专业(类)选考科目要求"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from os import mkdir\n",
|
||||||
|
"from time import sleep\n",
|
||||||
|
"from re import findall,sub,S\n",
|
||||||
|
"from os.path import isdir,isfile\n",
|
||||||
|
"from urllib.request import urlopen\n",
|
||||||
|
"from urllib.parse import urlencode,quote\n",
|
||||||
|
"from openpyxl import Workbook\n",
|
||||||
|
"import ssl\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"ssl._create_default_https_context = ssl._create_unverified_context\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"xuankaokemu\"]\n",
|
||||||
|
"list2 = []\n",
|
||||||
|
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
|
||||||
|
" list2.append(x['code'])\n",
|
||||||
|
"#m_xx = dict()\n",
|
||||||
|
"start_url = 'https://xkkm.sdzk.cn/web/xx.html'\n",
|
||||||
|
"with urlopen(start_url) as fp:\n",
|
||||||
|
" content = fp.read().decode('utf8')\n",
|
||||||
|
" \n",
|
||||||
|
"pattern = (r'<tr>.*?<td.+?</td>.*?<td.+?>(.+?)</td>'\n",
|
||||||
|
" '.*?<td.+?>(.+?)</td>.*?<td.+?>(.+?)</td>')\n",
|
||||||
|
"\n",
|
||||||
|
"for item in findall(pattern,content,S):\n",
|
||||||
|
" if len(item[0]) > 5:\n",
|
||||||
|
" continue\n",
|
||||||
|
" \n",
|
||||||
|
" shengfen,dm,mc = item\n",
|
||||||
|
" print('学校代码:', dm)\n",
|
||||||
|
" print('学校名称:', mc)\n",
|
||||||
|
" m_xx = {}\n",
|
||||||
|
" m_xx['code'] = dm\n",
|
||||||
|
" m_xx['name'] = mc\n",
|
||||||
|
" m_xx.setdefault('zhuanye',{})\n",
|
||||||
|
" if dm in list2:\n",
|
||||||
|
" continue\n",
|
||||||
|
" url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'\n",
|
||||||
|
" data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')\n",
|
||||||
|
" with urlopen(url,data) as fp:\n",
|
||||||
|
" xuexiao_content = fp.read().decode()\n",
|
||||||
|
" soup = BeautifulSoup(xuexiao_content,'lxml')\n",
|
||||||
|
" #data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')\n",
|
||||||
|
" data = soup.select('#ccc > div > table > tbody > tr ')\n",
|
||||||
|
" for data1 in data:\n",
|
||||||
|
" list1 =[]\n",
|
||||||
|
" m_xx1 = {}\n",
|
||||||
|
" for item in data1.stripped_strings: \n",
|
||||||
|
" list1.append(item)\n",
|
||||||
|
" #del list1[0]\n",
|
||||||
|
" #print('层次:',list1[1])\n",
|
||||||
|
" #print('专业(类)名称:',list1[2])\n",
|
||||||
|
" #print('选考科目范围:',list1[3])\n",
|
||||||
|
" #print('类中所含专业:',list1[4:])\n",
|
||||||
|
" code = list1[0]\n",
|
||||||
|
" m_xx['zhuanye'].setdefault(code,{})\n",
|
||||||
|
" \n",
|
||||||
|
" m_xx['zhuanye'][code]['name'] = list1[2]\n",
|
||||||
|
" m_xx['zhuanye'][code]['level'] = list1[1]\n",
|
||||||
|
" m_xx['zhuanye'][code]['fanwei'] = list1[3]\n",
|
||||||
|
" m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]\n",
|
||||||
|
" mycol.insert_one(m_xx) \n",
|
||||||
|
" \n",
|
||||||
|
" print('ok')\n",
|
||||||
|
" \n",
|
||||||
|
" # list1.clear\n",
|
||||||
|
" \n",
|
||||||
|
"\n",
|
||||||
|
" sleep(5)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import pymongo\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"xuankaokemu\"]\n",
|
||||||
|
"list2 = []\n",
|
||||||
|
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
|
||||||
|
" list2.append(x['code'])\n",
|
||||||
|
"print(list2)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# 招生简章管理"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 各大学招生简章采集"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from time import sleep\n",
|
||||||
|
"import ssl\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import requests\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
||||||
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||||||
|
"#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'\n",
|
||||||
|
"\n",
|
||||||
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
||||||
|
"for i in sf:\n",
|
||||||
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
|
||||||
|
" strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
||||||
|
" \n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" dict1 = {}\n",
|
||||||
|
" dict1['name'] = item.text.strip()\n",
|
||||||
|
" dict1['url'] = item.get('href')\n",
|
||||||
|
" if item.get('style') =='color:gray':\n",
|
||||||
|
" dict1['bz'] = 0\n",
|
||||||
|
" else:\n",
|
||||||
|
" dict1['bz'] = 1\n",
|
||||||
|
" mycol.insert_one(dict1) \n",
|
||||||
|
"\n",
|
||||||
|
"\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 采集单个学校招生简章"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from time import sleep\n",
|
||||||
|
"from re import findall,sub,S\n",
|
||||||
|
"import ssl\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import requests\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
||||||
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||||||
|
"url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'\n",
|
||||||
|
"strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
"strhtml.encoding = 'utf8'\n",
|
||||||
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
"#data = strhtml.text\n",
|
||||||
|
"#data1 = data.split(\"\\r\")\n",
|
||||||
|
"#print(soup)\n",
|
||||||
|
"data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
||||||
|
"for item in data:\n",
|
||||||
|
" url1 = item.get('href')\n",
|
||||||
|
"url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
|
||||||
|
"strhtml = requests.get(url1,headers = headers)\n",
|
||||||
|
"strhtml.encoding = 'utf8'\n",
|
||||||
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
"data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
||||||
|
"nr = ''\n",
|
||||||
|
"for item in data:\n",
|
||||||
|
" nr += item.text+'\\n'\n",
|
||||||
|
"print(nr)\n",
|
||||||
|
" \n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 第一次采集招生简章"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from time import sleep\n",
|
||||||
|
"from re import findall,sub,S\n",
|
||||||
|
"import ssl\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import requests\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
||||||
|
"for x in mycol.find({\"bz\": 1 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
|
||||||
|
" url = 'https://gaokao.chsi.com.cn'+x['url']\n",
|
||||||
|
" col_name = x['name']\n",
|
||||||
|
" strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" #data = strhtml.text\n",
|
||||||
|
" #data1 = data.split(\"\\r\")\n",
|
||||||
|
" #print(soup)\n",
|
||||||
|
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" url1 = item.get('href')\n",
|
||||||
|
" url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
|
||||||
|
" strhtml = requests.get(url1,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
||||||
|
" nr = ''\n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" nr += item.text+'\\n' \n",
|
||||||
|
" myquery = {'name':col_name}\n",
|
||||||
|
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
|
||||||
|
" sleep(5)\n",
|
||||||
|
" "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 检查新增的学校及招生简章"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from time import sleep\n",
|
||||||
|
"from re import findall,sub,S\n",
|
||||||
|
"import ssl\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import requests\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
||||||
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||||||
|
"l_name = []\n",
|
||||||
|
"for x in mycol.find({},{ \"_id\": 0, \"name\": 1}):\n",
|
||||||
|
" l_name.append(x['name'])\n",
|
||||||
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
||||||
|
"n_url = []\n",
|
||||||
|
"for i in sf:\n",
|
||||||
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
|
||||||
|
" strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
||||||
|
" \n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" m_url = item.get('href')\n",
|
||||||
|
" m_name = item.text.strip()\n",
|
||||||
|
" if m_name not in l_name:\n",
|
||||||
|
" url = 'https://gaokao.chsi.com.cn'+m_url\n",
|
||||||
|
" col_name = m_name \n",
|
||||||
|
" dict1 = {}\n",
|
||||||
|
" dict1['name'] = m_name\n",
|
||||||
|
" dict1['url'] = m_url\n",
|
||||||
|
" dict1['bz'] = 0 \n",
|
||||||
|
" mycol.insert_one(dict1) \n",
|
||||||
|
" print(dict1)\n",
|
||||||
|
" sleep(3)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 检查、新增招生简章"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 9,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from time import sleep\n",
|
||||||
|
"from re import findall,sub,S\n",
|
||||||
|
"import ssl\n",
|
||||||
|
"from bs4 import BeautifulSoup\n",
|
||||||
|
"import pymongo\n",
|
||||||
|
"import requests\n",
|
||||||
|
"\n",
|
||||||
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||||||
|
"mydb = myclient[\"gaokao\"]\n",
|
||||||
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
||||||
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||||||
|
"l_name = []\n",
|
||||||
|
"for x in mycol.find({\"bz\": 0 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
|
||||||
|
" l_name.append(x['name'])\n",
|
||||||
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
||||||
|
"n_name = []\n",
|
||||||
|
"for i in sf:\n",
|
||||||
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk' \n",
|
||||||
|
" strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
||||||
|
" \n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" m_url = item.get('href')\n",
|
||||||
|
" col_name = item.text.strip()\n",
|
||||||
|
" #print(m_url)\n",
|
||||||
|
" if (item.get('style') !='color:gray') and (col_name in l_name):\n",
|
||||||
|
" url = 'https://gaokao.chsi.com.cn' + m_url\n",
|
||||||
|
" \n",
|
||||||
|
" strhtml = requests.get(url,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" url1 = item.get('href')\n",
|
||||||
|
" print(url1)\n",
|
||||||
|
" url1 = 'https://gaokao.chsi.com.cn' + url1\n",
|
||||||
|
" strhtml = requests.get(url1,headers = headers)\n",
|
||||||
|
" strhtml.encoding = 'utf8'\n",
|
||||||
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||||||
|
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
||||||
|
" nr = ''\n",
|
||||||
|
" for item in data:\n",
|
||||||
|
" nr += item.text+'\\n' \n",
|
||||||
|
" myquery = {'name':col_name}\n",
|
||||||
|
" mycol.update_one(myquery,{'$set':{'bz':1}})\n",
|
||||||
|
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
|
||||||
|
" print(col_name)\n",
|
||||||
|
" sleep(3)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.8.10"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 4
|
||||||
|
}
|
||||||
+1855
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user