Jupyterlab

Jupyterlab
This commit is contained in:
512song@sina.com committed 2021-10-09 15:12:24 +08:00
commit 4e8c1b3254
12 files changed
+6769

No files matched your search

+86
View File
@@ -0,0 +1,86 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "4925cb3a-bea3-4a4d-9590-6b91b62ca5b4",
"metadata": {},
"source": [
"# 体质检测管理"
]
},
{
"cell_type": "markdown",
"id": "2a85feec-695b-4ea8-8093-91a59ed34c7f",
"metadata": {},
"source": [
"## 体测单位信息导入"
]
},
{
"cell_type": "code",
"execution_count": 5,
"id": "c5a0568c-3833-4335-a4b9-600310e1a541",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"['综合办公室(党委办公室)', '销运管理部', '财务资产部', '财务管理部', '企业管理部(法律事务部)', '发展计划部', '生产技术部', '设备工程部', '安全环保部', '党委组织部(资源管理部)', '审计管理部', '党群文化部(党委宣传部、工会、团委)', '纪委(监督部)', '检验计量中心', '物资采购中心', '信息中心', '行政事务中心', '消防救援支队', '电气仪表中心', '炼油运行一部', '炼油运行二部', '炼油运行三部', '炼油运行四部', '炼油运行五部', '化工运行部', '热电运行部', '水务运行部', '储运运行部', '编组站']\n"
]
}
],
"source": [
"import openpyxl\n",
"import json\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"tice\"]\n",
"mycol = mydb[\"unit\"]\n",
"wb = openpyxl.load_workbook('./data/shijiazhuang.xlsx')\n",
"sheet = wb.active\n",
"#sheets = wb.sheetnames\n",
"depart = []\n",
"dict1 = {}\n",
"new_code = []\n",
"for n in range(2,sheet.max_row+1):\n",
" m_depart = sheet.cell(n,3).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,3).value)\n",
" #print(sheet.cell(n,5).value)\n",
"print(depart)\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "aab6612f-4aad-46f0-89cd-0384ae7ac0c9",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
+1354
View File
File diff suppressed because it is too large. Load diff
+130
View File
@@ -0,0 +1,130 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymysql\n",
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
"cursor = db.cursor()\n",
"cursor.execute(\"SELECT VERSION()\")\n",
"data = cursor.fetchone()\n",
"print (\"Database version : %s \" % data)\n",
"db.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymysql\n",
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
"cursor = db.cursor()\n",
"sql = \"select id,name from sports_person where name= %s\"\n",
"cursor.execute(sql, ('张联红',))\n",
"result = cursor.fetchone()\n",
"print(result)\n",
"db.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymysql\n",
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
"cursor = db.cursor()\n",
"sql = \"select id,name from sports_person\"\n",
"cursor.execute(sql)\n",
"result = cursor.fetchone()\n",
"print(\"人员信息:\")\n",
"print(result)\n",
"sql = \"select id,name from sports_item\"\n",
"cursor.execute(sql)\n",
"result = cursor.fetchone()\n",
"print(\"\\n运动项目信息:\")\n",
"print(result)\n",
"db.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymysql\n",
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
"cursor = db.cursor()\n",
"sql = \"select a.target_id,b.name from sports_item_target as a,sports_target as b where a.target_id=b.id\"\n",
"cursor.execute(sql)\n",
"re_ta = {}\n",
"print(\"\\n运动项目信息:\")\n",
"results = cursor.fetchall()\n",
"for result in results:\n",
"# print (\"%d--%s\" %(result[0],result[1]))\n",
" re_ta[result[0]] = result[1]\n",
" print(\"re_ta[%d]= #%s\" %(result[0],result[1]))\n",
"print(re_ta)\n",
"db.close()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymysql\n",
"db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n",
"cursor = db.cursor()\n",
"person_id = \n",
"record_date = ''\n",
"item_id =\n",
"re_ta = {}\n",
"re_ta[1]= #运动距离\n",
"re_ta[2]= #运动时间\n",
"re_ta[3]= #平均配速\n",
"re_ta[4]= #平均速度\n",
"re_ta[5]= #平均步频\n",
"re_ta[6]= #平均步幅\n",
"re_ta[7]= #步数\n",
"re_ta[8]= #平均心率\n",
"re_ta[9]= #最大心率\n",
"re_ta[10]= #无氧耐力\n",
"re_ta[11]= #有氧耐力\n",
"re_ta[12]= #最大步频\n",
"re_ta[13]= #最快配速\n",
"sql = \"select * from sports_person\"\n",
"db.close()"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.5"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
+880
View File
@@ -0,0 +1,880 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 数字文件名转换为文本文件名"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 数字文件名转换为文本文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd()+'/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n",
"fi_name = {}\n",
"fi_path = os.getcwd()+'/file/210924'\n",
"old = []\n",
"new = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1):\n",
" \n",
" m_name = sheet.cell(n,1).value.strip()\n",
" m_depart = sheet.cell(n,5).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,5).value)\n",
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
" #print()\n",
"# 创建部门办公室 \n",
"#m_path = fi_path = os.getcwd()+'/file/210924/new'\n",
"#for pn in depart:\n",
"# if not os.path.exists(m_path + '/' + pn):\n",
"# os.mkdir(m_path + '/' + pn)\n",
"#print(dict1)\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn)\n",
" #print(fn)\n",
"old.sort()\n",
" #print(str(nfn)+'.pdf')\n",
"'''\n",
"for n in old:\n",
" \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{dict1[n][1]}/{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(old)\n",
"'''\n",
"print(dict1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 目录文件按照文件名排序"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"\n",
"import os,sys\n",
"\n",
"\n",
"fi_xls = 'test1.xlsx'\n",
"fi_name = {}\n",
"#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n",
"fi_path = os.getcwd()+'/data'\n",
"old = []\n",
"new = []\n",
"\n",
"fl=os.listdir(fi_path)\n",
"fl.sort()\n",
"n = 0\n",
"for i in fl:\n",
" oldname=fl[n]\n",
" name, suffix = os.path.splitext(oldname)\n",
" if name in old:\n",
" new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" old_name = fi_path+ os.sep + fl[n]\n",
" os.rename(old_name,new_name)\n",
" n+= 1\n",
"fl"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 将pdf文件转为图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n",
"print(path)\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 图像文件夹打包"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import zipfile\n",
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"import shutil\n",
"import time\n",
"\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"def addfile(zipfilename, dirname):\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"def make_path(p):\n",
" if os.path.exists(p): # 判断文件夹是否存在\n",
" shutil.rmtree(p) # 删除文件夹\n",
" os.mkdir(p) \n",
"pdf_file = '2.pdf'\n",
"output_folder='./pic1'\n",
"zip_file = 'ribenweiqishihua.zip'\n",
"make_path(output_folder)\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n",
"compress_file(zip_file, output_folder) # 执行函数\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 文本文件操作"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 基本读取"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import re\n",
"file_name = 'data/2012.txt'\n",
"with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n",
" for l in fl:\n",
" l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n",
" if l.split():\n",
" print(l)\n",
" fl1.write(l)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 读取分隔符分割文件,导入MongoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"import re\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"city\"]\n",
"m_mongo = {}\n",
"m_xx = []\n",
"fl_name = 'china-city-list.txt'\n",
"n = 0\n",
"with open(fl_name,'r') as fl:\n",
" for l in fl:\n",
" n += 1\n",
" if n >6:\n",
" m_mongo = {}\n",
" m_xx = re.sub('[ ]{1,}', '', l).split('|')\n",
" #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n",
" m_mongo['name'] = m_xx[3]\n",
" m_mongo['code'] = m_xx[1]\n",
" m_mongo['sheng'] = m_xx[8]\n",
" m_mongo['shi'] = m_xx[10]\n",
" m_mongo['jing'] = m_xx[11]\n",
" m_mongo['wei'] = m_xx[12]\n",
" mycol.insert_one(m_mongo) \n",
" #print(m_mongo)\n",
"print('ok!')\n",
"\n",
"\n",
"\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# Twilio使用"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"from twilio.rest import Client\n",
"\n",
"\n",
"# Your Account Sid and Auth Token from twilio.com/console\n",
"# and set the environment variables. See http://twil.io/secure\n",
"account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n",
"auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n",
"client = Client(account_sid, auth_token)\n",
"\n",
"message = client.messages \\\n",
" .create(\n",
" body=\"I'm back.\",\n",
" from_='+12056066931',\n",
" to='+8613793180751'\n",
" )\n",
"\n",
"print(message.sid)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import time\n",
"\n",
"localtime = time.localtime(time.time())\n",
"#type(localtime)\n",
"print (\"本地时间为 :\", localtime)\n",
"jyr = '12345'\n",
"if time.strftime(\"%w\", time.localtime()) in jyr:\n",
" print('ok')\n",
"else:\n",
" print('今日不是交易日!')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"\n",
"def sendmail(message):\n",
" msg = MIMEText(message,'plain','utf-8')\n",
" msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
" msg['From'] = Header('512song@sina.com')\n",
" msg['To'] = Header('songyi@yeah.net','utf-8')\n",
"\n",
" from_addr = '512song@sina.com' #发件邮箱\n",
" password = '409fe5d8471da663' #邮箱密码\n",
" to_addr = 'songyi@yeah.net' #收件邮箱\n",
" smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit() \n",
" \n",
" \n",
"\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"data = strhtml.text\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
"#print(data2)\n",
"with open('price.txt','r') as fl:\n",
" for line in fl:\n",
" p_high = line.split(',')[0]\n",
" p_low = line.split(',')[1]\n",
"m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
"while time.strftime(\"%w\", time.localtime()) in '12345':\n",
" \n",
" print(p_high,p_low)\n",
" time.sleep(10)\n",
" strhtml = requests.get(url)\n",
" data = strhtml.text\n",
" if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
" sendmail(m_message)\n",
" p_high = str(float(p_high) + 0.04) \n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元卖出价:{}'.format(data2[2])\n",
" p_low = str(float(p_low) - 0.04)\n",
" sendmail(m_message)\n",
" time.sleep(900)\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# AWS应用"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS获取sns信息"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Topic List: ['arn:aws:sns:us-east-1:915521803346:MyTopic', 'arn:aws:sns:us-east-1:915521803346:dynamodb']\n"
]
}
],
"source": [
"import boto3\n",
"\n",
"# Create an SNS client\n",
"sns = boto3.client('sns')\n",
"\n",
"# Call SNS to list topics\n",
"response = sns.list_topics()\n",
"\n",
"# Get a list of all topic ARNs from the response\n",
"topics = [topic['TopicArn'] for topic in response['Topics']]\n",
"\n",
"# Print out the topic list\n",
"print(\"Topic List: %s\" % topics)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS操作DynamoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"# Create the DynamoDB table.\n",
"table = dynamodb.create_table(\n",
" TableName='waihui',\n",
" \n",
" AttributeDefinitions=[ \n",
" {\n",
" 'AttributeName': 'code',\n",
" 'AttributeType': 'S'\n",
" }\n",
" \n",
" \n",
" ],\n",
" KeySchema=[\n",
" {\n",
" 'AttributeName': 'code',\n",
" 'KeyType': 'HASH'\n",
" }\n",
" \n",
" ],\n",
" ProvisionedThroughput={\n",
" 'ReadCapacityUnits': 5,\n",
" 'WriteCapacityUnits': 5\n",
" }\n",
" \n",
")\n",
"\n",
"# Wait until the table exists.\n",
"table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n",
"\n",
"# Print out some data about the table.\n",
"print(table.item_count)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.put_item(\n",
" Item={\n",
" 'code': 'USDCAD',\n",
" 'high': Decimal('1.3200'),\n",
" 'low': Decimal('1.3000'),\n",
" }\n",
")"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"{'low': Decimal('1.29'), 'code': 'USDCAD', 'high': Decimal('1.32')}\n"
]
}
],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"response = table.get_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n",
"item = response['Item']\n",
"print(item)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.delete_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'ResponseMetadata': {'RequestId': 'LCHMG55BG6A89FF2KBSDM7JRHNVV4KQNSO5AEMVJF66Q9ASUAAJG',\n",
" 'HTTPStatusCode': 200,\n",
" 'HTTPHeaders': {'server': 'Server',\n",
" 'date': 'Thu, 30 Sep 2021 05:04:36 GMT',\n",
" 'content-type': 'application/x-amz-json-1.0',\n",
" 'content-length': '2',\n",
" 'connection': 'keep-alive',\n",
" 'x-amzn-requestid': 'LCHMG55BG6A89FF2KBSDM7JRHNVV4KQNSO5AEMVJF66Q9ASUAAJG',\n",
" 'x-amz-crc32': '2745614147'},\n",
" 'RetryAttempts': 0}}"
]
},
"execution_count": 6,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"table.update_item(\n",
" Key={\n",
" 'code': 'USDCAD'\n",
" },\n",
" UpdateExpression='SET low = :val1',\n",
" ExpressionAttributeValues={\n",
" ':val1': decimal.Decimal('1.2900')\n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": 15,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Received and deleted message: {'MessageId': 'ecd656af-72e2-4bfc-9deb-3043dbe1bb80', 'ReceiptHandle': 'AQEBTx6u3PRfCgUy0SfLyOZcsD4fa5ge7AZvDTeQ/P4sOslg+sDMRYDZelJQOvxDK2XsLvy24DvD/n6GxBa2RwN55b/bnN7PxeYvv42d8GgkIy1dOPEBTytVKNIf+Z7t9+jMZMT2kq2ypXSa7jLFPFN00ArUkd0GI7nZ5SIS133s8exP46gRUGjgaS04CzAwPocTpWNt9GZpA114bvlicx2gNMYTIBsF65K2MzcyBoNVjLzmSJ1Pi0OCcQrAYaHvwzfnkriOBJLUX8BU4ksjWGJg0m8wh8hlPVMTLX+qxMneLzLawbf3cM/ricQR1XtooXrhsAuwAXA+VXaq5TzlpFJZaDZu5BvebF0Yj3rst7j54L/O3DwY0WfVmIrgv2yjOqL7', 'MD5OfBody': '9794e060a0eae4992dbf298549c42dc5', 'Body': 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1', 'Attributes': {'SentTimestamp': '1632980620260'}}\n"
]
}
],
"source": [
"import boto3\n",
"\n",
"# Create SQS client\n",
"sqs = boto3.client('sqs')\n",
"\n",
"queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n",
"\n",
"# Receive message from SQS queue\n",
"response = sqs.receive_message(\n",
" QueueUrl=queue_url,\n",
" AttributeNames=[\n",
" 'SentTimestamp'\n",
" ],\n",
" MaxNumberOfMessages=1,\n",
" MessageAttributeNames=[\n",
" 'All'\n",
" ],\n",
" VisibilityTimeout=0,\n",
" WaitTimeSeconds=0\n",
")\n",
"\n",
"message = response['Messages'][0]\n",
"receipt_handle = message['ReceiptHandle']\n",
"\n",
"# Delete received message from queue\n",
"sqs.delete_message(\n",
" QueueUrl=queue_url,\n",
" ReceiptHandle=receipt_handle\n",
")\n",
"print('Received and deleted message: %s' % message)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# MongoDB系统GridFS文件管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 文件上传"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(self,file_coll,_id,out_name):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"m_dir = './data/tmp'\n",
"fls=os.listdir(m_dir)\n",
"n = 0\n",
"for fl in fls:\n",
" #oldname=fl[n]\n",
" name, suffix = os.path.splitext(fl)\n",
" #if name in old:\n",
" # new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" # old_name = fi_path+ os.sep + fl[n]\n",
" # os.rename(old_name,new_name)\n",
" #print(os.path.basename(fl))\n",
" #print(fl,suffix[1:])\n",
" full_path = m_dir+ '/' + fl\n",
" upLoadFile(\"document\",full_path,\"\")\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": file_name, \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(file_coll,_id,out_name):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n",
"print (ll)\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
+439
View File
@@ -0,0 +1,439 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "333a5431-a697-4330-a587-455d6d18ebb0",
"metadata": {},
"source": [
"# 文件管理"
]
},
{
"cell_type": "markdown",
"id": "bcd48bb6-4756-4748-ac8b-854eeee92a3b",
"metadata": {},
"source": [
"## 数字文件名转换为文本文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "54b320e8-c3c1-4214-9a9e-db13c2558604",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd() + '/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n",
"fi_path = os.getcwd() + '/file/210924'\n",
"old = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1): \n",
" m_name = sheet.cell(n,1).value.strip()\n",
" m_depart = sheet.cell(n,5).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,5).value)\n",
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
" \n",
"# 创建部门办公室 \n",
"m_path = fi_path = os.getcwd()+'/file/210924/new'\n",
"for pn in depart:\n",
" if not os.path.exists(m_path + '/' + pn):\n",
" os.mkdir(m_path + '/' + pn)\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn) \n",
"old.sort()\n",
"for n in old: \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{dict1[n][1]}/{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(dict1)"
]
},
{
"cell_type": "markdown",
"id": "69712868-6786-430b-8591-f6828d781588",
"metadata": {},
"source": [
"## PDF文件压缩"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8492dc14-739e-4c78-a302-f4b57ae571dd",
"metadata": {},
"outputs": [],
"source": [
"import fitz\n",
"import os\n",
"\n",
"\n",
"def covert2pic(zoom):\n",
" if os.path.exists('.pdf'): # 临时文件,需为空\n",
" os.removedirs('.pdf')\n",
" os.mkdir('.pdf')\n",
" for pg in range(totaling):\n",
" page = doc[pg]\n",
" zoom = int(zoom) #值越大,分辨率越高,文件越清晰\n",
" rotate = int(0)\n",
" print(page)\n",
" trans = fitz.Matrix(zoom / 100.0, zoom / 100.0).preRotate(rotate)\n",
" pm = page.getPixmap(matrix=trans, alpha=False)\n",
" \n",
" lurl='.pdf/%s.jpg' % str(pg+1)\n",
" pm.writePNG(lurl)\n",
" doc.close()\n",
"\n",
"def pic2pdf(obj):\n",
" doc = fitz.open()\n",
" for pg in range(totaling):\n",
" img = '.pdf/%s.jpg' % str(pg+1)\n",
" imgdoc = fitz.open(img) # 打开图片\n",
" pdfbytes = imgdoc.convertToPDF() # 使用图片创建单页的 PDF\n",
" os.remove(img) \n",
" imgpdf = fitz.open(\"pdf\", pdfbytes)\n",
" doc.insertPDF(imgpdf) # 将当前页插入文档\n",
" if os.path.exists(obj): # 若文件存在先删除\n",
" os.remove(obj)\n",
" doc.save(obj) # 保存pdf文件\n",
" doc.close()\n",
"\n",
"\n",
"def pdfz(sor, obj, zoom): \n",
" covert2pic(zoom)\n",
" pic2pdf(obj)\n",
" \n",
"\n",
"\n",
"sor = \"5.pdf\" # 需要压缩的PDF文件\n",
"obj = \"new-\" + sor\n",
"doc = fitz.open(sor) \n",
"totaling = doc.pageCount\n",
"\n",
"zoom = 150 # 清晰度调节,缩放比率\n",
"pdfz(sor, obj, zoom)\n",
"os.removedirs('.pdf')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "88d84849-5b98-48cb-86b1-e96757a8615f",
"metadata": {},
"outputs": [],
"source": [
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path('5.pdf', dpi=100,fmt='jpg', output_folder='./pic')\n",
"print(path)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "058a40ba-4948-4675-8400-e4c762e836ee",
"metadata": {},
"outputs": [],
"source": [
"import img2pdf \n",
"import os,sys\n",
"import glob\n",
"\n",
"fl=glob.glob('./pic/*.jpg')\n",
"fl.sort()\n",
"a4inpt = (img2pdf.mm_to_pt(210),img2pdf.mm_to_pt(297))\n",
"layout_fun = img2pdf.get_layout_fun(a4inpt)\n",
"with open(\"name.pdf\",\"wb\") as f:\n",
" f.write(img2pdf.convert(fl,layout_fun=layout_fun))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e997ad75-b2d9-43bf-a1de-8833cfe0cf6a",
"metadata": {},
"outputs": [],
"source": [
"import glob\n",
"import fitz # 导入本模块需安装pymupdf库\n",
"import os,sys\n",
"\n",
"def pic2pdf_1(img_path, pdf_path, pdf_name):\n",
" doc = fitz.open()\n",
" fl=os.listdir(img_path)\n",
" fl.sort()\n",
" width, height = fitz.PaperSize(\"a4\")\n",
" for img in fl:\n",
" fn = img_path+'/'+img\n",
" if os.path.isfile(fn):\n",
" imgdoc = fitz.open(img_path+'/'+img)\n",
" pdfbytes = imgdoc.convertToPDF()\n",
" imgpdf = fitz.open(\"pdf\", pdfbytes,width = width, height = height)\n",
" doc.insertPDF(imgpdf)\n",
" doc.save(pdf_path +'/'+ pdf_name)\n",
" doc.close()\n",
"\n",
"img_path = os.getcwd() +'/pic'\n",
"pdf_path = os.getcwd()\n",
"pic2pdf_1(img_path=img_path, pdf_path=pdf_path, pdf_name='1.pdf')"
]
},
{
"cell_type": "markdown",
"id": "1f38e4b6-ef20-4fef-801c-ff47951d2ad6",
"metadata": {},
"source": [
"## 图片文件扫描识别"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8c990b6a-7f8e-4527-a150-65f60bd131ac",
"metadata": {},
"outputs": [],
"source": [
"#将指定目录下图片文件进行文字识别\n",
"import os,sys\n",
"from aip import AipOcr\n",
"import glob\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"\n",
"fi_path = os.getcwd()+'/pic'\n",
"fl = glob.glob(f'{fi_path}/*.jpg')\n",
"fl.sort()\n",
"for fl1 in fl:\n",
" file_name = fl1\n",
" image = get_file_content(file_name)\n",
" result= client.basicGeneral(image, options)\n",
" if 'words_result' in result:\n",
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
" print('\\n')\n",
" \n",
" "
]
},
{
"cell_type": "markdown",
"id": "d936ca99-269b-46f8-8582-fb02ed2cd3cc",
"metadata": {},
"source": [
"## word文档读取"
]
},
{
"cell_type": "code",
"execution_count": 47,
"id": "d8043425-be9e-4e9e-bbfc-3a2d8885a860",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [
"import docx\n",
"import json\n",
"Doc = docx.Document(r\"症状对症处方选穴.docx\")\n",
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
"testList = []\n",
"text1 = []\n",
"dict1 = {}\n",
"for text in Doc.paragraphs:\n",
" testList.append(text)\n",
"i = 0\n",
"for pg in testList:\n",
" \n",
" if len(pg.text) >0:\n",
" s = ''.join(pg.text.split()) \n",
" if s[0:1] == '第':\n",
" i += 1\n",
" n = 0\n",
" \n",
" dict1.setdefault(i,{})\n",
" dict1[i]['title'] = pg.text.split()[1]\n",
" dict1[i]['sub'] = {}\n",
" \n",
" #dict1[i]['title'].setdefault(i,{})\n",
" elif s[0:1] == '(':\n",
" sub = s.split(')')[1]\n",
" dict2 = {}\n",
" dict1[i]['sub'].setdefault(sub,[])\n",
" else:\n",
" dict1[i]['sub'][sub].append(s)\n",
"filename = '症状对症处方选穴.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl,ensure_ascii=False) \n",
"print('ok') "
]
},
{
"cell_type": "code",
"execution_count": 2,
"id": "c440b599-9c80-4072-8543-d9c381214f1d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [
"import docx\n",
"import json\n",
"Doc = docx.Document(r\"常见疾病辨病处方选穴.docx\")\n",
"#print(\"檔案內含段落數:\",len(Doc.paragraphs),\"\\n\")\n",
"testList = []\n",
"text1 = []\n",
"dict1 = {}\n",
"for text in Doc.paragraphs:\n",
" testList.append(text)\n",
"i = 0\n",
"for pg in testList:\n",
" \n",
" if len(pg.text) >0:\n",
" s = ''.join(pg.text.split()) \n",
" if s[0:1] == '第':\n",
" i += 1\n",
" n = 0\n",
" \n",
" dict1.setdefault(i,{})\n",
" dict1[i]['title'] = pg.text.split('节')[1]\n",
" dict1[i]['sub'] = {}\n",
" \n",
" #dict1[i]['title'].setdefault(i,{})\n",
" elif s[0:1] == '(':\n",
" sub = s.split(')')[1]\n",
" dict2 = {}\n",
" dict1[i]['sub'].setdefault(sub,[])\n",
" else:\n",
" dict1[i]['sub'][sub].append(s)\n",
"filename = '常见疾病辨病处方选穴.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl,ensure_ascii=False)\n",
"print('ok') "
]
},
{
"cell_type": "code",
"execution_count": 69,
"id": "b5e500b5-14d4-4417-8b5e-46df71a6b6b8",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [
"import json\n",
"import re\n",
"filename = '症状对症处方选穴.json'\n",
"pattern = r'[\\d\\.]'\n",
"mo =r'[\\u4e00-\\u9fa5]+'\n",
"dict2 = {}\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl) \n",
"for k,v in dict1.items():\n",
" m_title = v['title']\n",
" for k1,v1 in v['sub'].items():\n",
" sub_title = k1 \n",
" s = ''\n",
" for mx in v1: \n",
" s = s + mx\n",
" list1 = re.split(pattern, s)\n",
" list2 = []\n",
" for ss in list1:\n",
" if ss != '':\n",
" list3 = re.findall(mo,ss)\n",
" list2.append(list3)\n",
" \n",
" #print(m_title,sub_title,list2)\n",
" dict2.setdefault(m_title,{})\n",
" dict2[m_title][sub_title] = list2\n",
"filename = 'new_症状对症处方选穴.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict2, fl,ensure_ascii=False) \n",
"print('ok') \n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "eeea1429-e574-4d3b-88fd-0ddb3c090947",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
+377
View File
@@ -0,0 +1,377 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from aip import AipOcr\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"image = get_file_content('1.jpg')\n",
"\n",
"\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
"#client.basicGeneral(image);\n",
"\n",
"\"\"\" 如果有可选参数 \"\"\"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"\n",
"\"\"\" 带参数调用通用文字识别, 图片参数为本地图片 \"\"\"\n",
"result= client.basicGeneral(image, options)\n",
"if 'words_result' in result:\n",
" print('\\n'.join([w['words'] for w in result['words_result']]))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"#将指定目录下图片文件进行文字识别\n",
"import os,sys\n",
"from aip import AipOcr\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n",
"SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n",
"\n",
"client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n",
"def get_file_content(filePath):\n",
" with open(filePath, 'rb') as fp:\n",
" return fp.read()\n",
"\n",
"options = {}\n",
"options[\"language_type\"] = \"CHN_ENG\"\n",
"options[\"detect_direction\"] = \"true\"\n",
"options[\"detect_language\"] = \"true\"\n",
"options[\"probability\"] = \"true\"\n",
"\n",
"fi_path = os.getcwd()+'/data'\n",
"fl = os.listdir(fi_path)\n",
"fl.sort()\n",
"for fl1 in fl:\n",
" file_name = fi_path+'/' + fl1\n",
" image = get_file_content(file_name)\n",
" result= client.basicGeneral(image, options)\n",
" if 'words_result' in result:\n",
" print('\\n'.join([w['words'] for w in result['words_result']]))\n",
" print('\\n')"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 识别图片中表格"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-26T08:59:12.967127Z",
"iopub.status.busy": "2020-11-26T08:59:12.966172Z",
"iopub.status.idle": "2020-11-26T08:59:28.010488Z",
"shell.execute_reply": "2020-11-26T08:59:28.006947Z",
"shell.execute_reply.started": "2020-11-26T08:59:12.967019Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"22917135_2274033\n",
"已完成\n"
]
}
],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"import time\n",
"\n",
"def get_access_token():\n",
" client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'\n",
" client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq' \n",
" # client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
" host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n",
" client_id, client_secret)\n",
" response = requests.get(host).text\n",
" data = json.loads(response)\n",
" access_token = data['access_token']\n",
" return access_token\n",
"\n",
"def get_excel(requests_id, access_token):\n",
" headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
" pargams = {\n",
" 'request_id': requests_id,\n",
" 'result_type': 'excel'\n",
" }\n",
" url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
" url_all = url + \"?access_token=\" + access_token\n",
" res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
" info_1 = res.json()['result']['ret_msg']\n",
" excel_url=res.json()['result']['result_data']\n",
" excel_1=requests.get(excel_url).content\n",
" with open('识别结果11.xls','wb+') as f:\n",
" f.write(excel_1)\n",
" print(info_1)\n",
"\n",
"\n",
"request_url = \"https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request\"\n",
"# 二进制方式打开图片文件\n",
"f = open('山东大学强基计划(2020).jpg', 'rb')\n",
"img = base64.b64encode(f.read())\n",
"\n",
"params = {\"image\":img}\n",
"access_token = get_access_token()\n",
"request_url = request_url + \"?access_token=\" + access_token\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"response = requests.post(request_url, data=params, headers=headers)\n",
"if response:\n",
" m_xx = response.json()\n",
"requests_id = m_xx['result'][0]['request_id'] \n",
"print(requests_id)\n",
"time.sleep(10)\n",
"get_excel(requests_id, access_token)"
]
},
{
"cell_type": "code",
"execution_count": 1,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-03T02:06:23.901979Z",
"iopub.status.busy": "2020-11-03T02:06:23.900925Z",
"iopub.status.idle": "2020-11-03T02:06:24.312339Z",
"shell.execute_reply": "2020-11-03T02:06:24.309116Z",
"shell.execute_reply.started": "2020-11-03T02:06:23.901714Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"{'refresh_token': '25.1dfb7a14cd15d03051976c8db246fa92.315360000.1919729184.282335-22917135', 'expires_in': 2592000, 'session_key': '9mzdCSFczT8Mv7ZN07SjsPo0dZr0AvxAeDt6Kjn1Z9jiiotA6kW2TZYzslnuYOFd1ZCx75mmzW0TRF+nxYAHzPOb5fmjlw==', 'access_token': '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135', 'scope': 'public vis-ocr_ocr brain_ocr_scope brain_ocr_general brain_ocr_general_basic vis-ocr_business_license brain_ocr_webimage brain_all_scope brain_ocr_idcard brain_ocr_driving_license brain_ocr_vehicle_license vis-ocr_plate_number brain_solution brain_ocr_plate_number brain_ocr_accurate brain_ocr_accurate_basic brain_ocr_receipt brain_ocr_business_license brain_solution_iocr brain_qrcode brain_ocr_handwriting brain_ocr_passport brain_ocr_vat_invoice brain_numbers brain_ocr_business_card brain_ocr_train_ticket brain_ocr_taxi_receipt vis-ocr_household_register vis-ocr_vis-classify_birth_certificate vis-ocr_台湾通行证 vis-ocr_港澳通行证 vis-ocr_机动车购车发票识别 vis-ocr_机动车检验合格证识别 vis-ocr_车辆vin码识别 vis-ocr_定额发票识别 vis-ocr_保单识别 vis-ocr_机打发票识别 vis-ocr_行程单识别 brain_ocr_vin brain_ocr_quota_invoice brain_ocr_birth_certificate brain_ocr_household_register brain_ocr_HK_Macau_pass brain_ocr_taiwan_pass brain_ocr_vehicle_invoice brain_ocr_vehicle_certificate brain_ocr_air_ticket brain_ocr_invoice brain_ocr_insurance_doc brain_formula brain_ocr_meter brain_doc_analysis brain_ocr_webimage_loc wise_adapt lebo_resource_base lightservice_public hetu_basic lightcms_map_poi kaidian_kaidian ApsMisTest_Test权限 vis-classify_flower lpq_开放 cop_helloScope ApsMis_fangdi_permission smartapp_snsapi_base smartapp_mapp_dev_manage iop_autocar oauth_tp_app smartapp_smart_game_openapi oauth_sessionkey smartapp_swanid_verify smartapp_opensource_openapi smartapp_opensource_recapi fake_face_detect_开放Scope vis-ocr_虚拟人物助理 idl-video_虚拟人物助理 smartapp_component', 'session_secret': '6ac230d6705697808e228241c519f303'}\n"
]
}
],
"source": [
"import requests \n",
"\n",
"# client_id 为官网获取的AK, client_secret 为官网获取的SK\n",
"host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=KwXkGawxh0sjOQdF9Ae9LeLb&client_secret=siprEKMp5UcRTOAngEfIOOe9x6xkqGXq'\n",
"response = requests.get(host)\n",
"if response:\n",
" print(response.json())"
]
},
{
"cell_type": "code",
"execution_count": 14,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-03T03:24:08.412670Z",
"iopub.status.busy": "2020-11-03T03:24:08.411768Z",
"iopub.status.idle": "2020-11-03T03:24:08.813580Z",
"shell.execute_reply": "2020-11-03T03:24:08.810985Z",
"shell.execute_reply.started": "2020-11-03T03:24:08.412567Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"已完成\n"
]
}
],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"pargams = {\n",
" 'request_id': '22917135_2227436',\n",
" 'result_type': 'excel'\n",
"}\n",
"\n",
"\n",
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
"url_all = url + \"?access_token=\" + access_token\n",
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
"info_1 = res.json()['result']['ret_msg']\n",
"excel_url=res.json()['result']['result_data']\n",
"excel_1=requests.get(excel_url).content\n",
"with open('识别结果12.xls','wb+') as f:\n",
" f.write(excel_1)\n",
"print(info_1)"
]
},
{
"cell_type": "code",
"execution_count": 80,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-03T08:43:37.180693Z",
"iopub.status.busy": "2020-11-03T08:43:37.179800Z",
"iopub.status.idle": "2020-11-03T08:43:37.656055Z",
"shell.execute_reply": "2020-11-03T08:43:37.653668Z",
"shell.execute_reply.started": "2020-11-03T08:43:37.180589Z"
}
},
"outputs": [
{
"data": {
"text/plain": [
"list"
]
},
"execution_count": 80,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"import demjson\n",
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"pargams = {\n",
" 'request_id': '22917135_2227436',\n",
" 'result_type': 'json'\n",
"}\n",
"\n",
"\n",
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
"url_all = url + \"?access_token=\" + access_token\n",
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
"#info_1 = res.json()['result']['ret_msg']\n",
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
"type(excel_1)\n",
"#excel_new = demjson.decode(excel_1)\n",
"#for m_col in excel_new['forms'][0]['body']:\n",
"# print(m_col)\n",
"m_xx =json.loads(excel_1)\n",
"\n",
"#with open('识别结果12.json','w') as fl:\n",
"# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
"#print(info_1)\n",
"#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))\n",
"print(m_xx['forms'][0]['body'])"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 识别保存为json文件"
]
},
{
"cell_type": "code",
"execution_count": 81,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-03T08:48:26.082167Z",
"iopub.status.busy": "2020-11-03T08:48:26.081266Z",
"iopub.status.idle": "2020-11-03T08:48:26.430039Z",
"shell.execute_reply": "2020-11-03T08:48:26.428236Z",
"shell.execute_reply.started": "2020-11-03T08:48:26.082060Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"已完成\n"
]
}
],
"source": [
"import requests\n",
"import json\n",
"import base64\n",
"import demjson\n",
"access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n",
"headers = {'content-type': 'application/x-www-form-urlencoded'}\n",
"pargams = {\n",
" 'request_id': '22917135_2227436',\n",
" 'result_type': 'json'\n",
"}\n",
"\n",
"\n",
"url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n",
"url_all = url + \"?access_token=\" + access_token\n",
"res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n",
"#info_1 = res.json()['result']['ret_msg']\n",
"excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n",
"type(excel_1)\n",
"#excel_new = demjson.decode(excel_1)\n",
"#for m_col in excel_new['forms'][0]['body']:\n",
"# print(m_col)\n",
"m_xx =json.loads(excel_1)\n",
"\n",
"with open('识别结果12.json','w') as fl:\n",
" json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n",
"print(info_1)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
},
"toc-autonumbering": true,
"toc-showmarkdowntxt": false,
"toc-showtags": false
},
"nbformat": 4,
"nbformat_minor": 4
}
+499
View File
@@ -0,0 +1,499 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 股票管理"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 股票信息导入"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import baostock as bs\n",
"import pandas as pd\n",
"\n",
"#### 登陆系统 ####\n",
"lg = bs.login()\n",
"# 显示登陆返回信息\n",
"print('login respond error_code:'+lg.error_code)\n",
"print('login respond error_msg:'+lg.error_msg)\n",
"\n",
"#### 获取证券信息 ####\n",
"rs = bs.query_all_stock(day=\"2020-10-20\")\n",
"print('query_all_stock respond error_code:'+rs.error_code)\n",
"print('query_all_stock respond error_msg:'+rs.error_msg)\n",
"\n",
"#### 打印结果集 ####\n",
"data_list = []\n",
"while (rs.error_code == '0') & rs.next():\n",
" # 获取一条记录,将记录合并在一起\n",
" data_list.append(rs.get_row_data())\n",
"#results = pd.DataFrame(data_list, columns=rs.fields)\n",
"\n",
"#### 结果集输出到csv文件 #### \n",
"#result.to_csv(\"all_stock.csv\", encoding=\"utf-8\", index=False)\n",
"for result in data_list:\n",
" print(result)\n",
"\n",
"#### 登出系统 ####\n",
"bs.logout()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import baostock as bs\n",
"import pandas as pd\n",
"import pymysql\n",
"\n",
"# 登陆系统\n",
"lg = bs.login()\n",
"# 显示登陆返回信息\n",
"print('login respond error_code:'+lg.error_code)\n",
"print('login respond error_msg:'+lg.error_msg)\n",
"\n",
"# 获取证券基本资料\n",
"rs = bs.query_stock_basic(code=\"\")\n",
"# rs = bs.query_stock_basic(code_name=\"浦发银行\") # 支持模糊查询\n",
"print('query_stock_basic respond error_code:'+rs.error_code)\n",
"print('query_stock_basic respond error_msg:'+rs.error_msg)\n",
"\n",
"# 打印结果集\n",
"data_list = []\n",
"while (rs.error_code == '0') & rs.next():\n",
" # 获取一条记录,将记录合并在一起\n",
" data_list.append(rs.get_row_data())\n",
"#result = pd.DataFrame(data_list, columns=rs.fields)\n",
"# 结果集输出到csv文件\n",
"#result.to_csv(\"D:/stock_basic.csv\", encoding=\"gbk\", index=False)\n",
"\n",
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
"cursor = db.cursor()\n",
"\n",
"sql = \"insert into stock_info (code,name,ipoDate,outDate,type,status) values(%s,%s,%s,%s,%s,%s)\"\n",
"try:\n",
" cursor.executemany(sql,tuple(data_list))\n",
" db.commit()\n",
" print(\"ok!\")\n",
"except:\n",
" # 如果发生错误则回滚\n",
" print(\"error!\")\n",
" db.rollback() \n",
"db.close()\n",
"# 登出系统\n",
"bs.logout()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 每日持有标的信息导入"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import csv\n",
"import pymysql\n",
"\n",
"def read_data(filename):\n",
" detail = {}\n",
" with open(filename) as f:\n",
" reader = csv.reader(f)\n",
" #header_row =next(reader)\n",
" for row in reader:\n",
" detail.setdefault(row[0],[])\n",
" detail[row[0]].append(row[1])\n",
" return detail\n",
"m_xx = []\n",
"m_stock = {}\n",
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
"cursor = db.cursor()\n",
"sql = 'select code,name from stock_info where type=\"1\"'\n",
"cursor.execute(sql)\n",
"results = cursor.fetchall()\n",
"for result in results:\n",
" m_stock[result[0]] = result[1]\n",
"#print(m_stock)\n",
"filename = '每日标的信息.csv'\n",
"detail = read_data(filename)\n",
"for k,v in detail.items():\n",
" m_rq = '2020-' + k[0:2] + '-' + k[2:]\n",
" i = 1\n",
" for m_dm in v:\n",
" if m_dm[0:1] == '6':\n",
" m_dm = 'sh.' + m_dm\n",
" else:\n",
" m_dm = 'sz.' + m_dm\n",
" print('\\t'+m_dm + '\\t'+m_stock[m_dm])\n",
" m_xx.append((m_rq,m_dm,i))\n",
" i += 1\n",
"choice = input('以上为本日数据,是否导入?(y/n)')\n",
"if choice.upper() == \"Y\":\n",
" sql = \"insert into daily_item (rq,code,ord) values(%s,%s,%s)\"\n",
" try:\n",
" cursor.executemany(sql,m_xx)\n",
" db.commit()\n",
" print(\"ok!\")\n",
" except:\n",
" # 如果发生错误则回滚\n",
" print(\"error!\")\n",
" db.rollback() \n",
"db.close()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 每日调仓信息导入"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import csv\n",
"import pymysql\n",
"\n",
"def read_data(filename):\n",
" detail = {}\n",
" \n",
" with open(filename) as f:\n",
" reader = csv.reader(f)\n",
"# header_row =next(reader)\n",
" for row in reader:\n",
" detail1 = {}\n",
" detail.setdefault(row[3],[])\n",
" detail1 = {'dm':row[0],'zj':row[1],'ly':row[2],'cb':row[4]}\n",
" detail[row[3]].append(detail1)\n",
" return detail\n",
"m_xx = []\n",
"m_add = []\n",
"m_sub = []\n",
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
"cursor = db.cursor()\n",
"filename = '调仓明细.csv'\n",
"detail = read_data(filename)\n",
"#print(detail)\n",
"for rq in sorted(detail.keys()):\n",
" m_rq = '2020-' + rq[0:2] + '-' + rq[2:]\n",
" for xx_move in detail[rq]:\n",
" m_dm = xx_move['dm']\n",
" if m_dm[0:1] == '6':\n",
" m_dm = 'sh.' + m_dm\n",
" else:\n",
" m_dm = 'sz.' + m_dm\n",
" m_xx.append((m_rq,m_dm,int(xx_move['zj']),xx_move['ly']))\n",
" if xx_move['zj'] == '1':\n",
" m_add.append((m_dm,xx_move['ly'],float(xx_move['cb'])))\n",
" else:\n",
" m_sub.append((m_dm))\n",
"#print(m_xx)\n",
"sql = \"insert into change_item (rq,code,pos,reason) values(%s,%s,%s,%s)\"\n",
"try:\n",
" cursor.executemany(sql,m_xx)\n",
" if len(m_add) > 0:\n",
" sql_add = \"insert into stock_item (code,reason,cost) values(%s,%s,%s)\"\n",
" cursor.executemany(sql_add,m_add)\n",
" if len(m_sub) > 0:\n",
" sql_sub = \"update stock_item set status=0 where code=%s\"\n",
" cursor.executemany(sql_sub,m_sub) \n",
" db.commit()\n",
" print(\"导入成功!\")\n",
"except:\n",
" # 如果发生错误则回滚\n",
" print(\"error!\")\n",
" db.rollback() \n",
"\n",
"db.close()"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 持仓股票信息导入"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymysql\n",
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n",
"cursor = db.cursor()\n",
"sql = 'SELECT a.unit,a.CODE,b.cost FROM daily_item AS a,daily_cost as b WHERE a.rq=(select max(rq) FROM daily_item) AND a.code=b.code'\n",
"cursor.execute(sql)\n",
"results = cursor.fetchall()\n",
"sql = \"insert into stock_item(item,code,cost) values(%s,%s,%s)\"\n",
"try:\n",
" cursor.executemany(sql,list(results))\n",
" #db.commit()\n",
" print(\"ok!\")\n",
"except:\n",
" # 如果发生错误则回滚\n",
" print(\"error!\")\n",
" db.rollback() \n",
"sql = 'SELECT a.reason,a.code from change_item AS a WHERE a.pos=1'\n",
"cursor.execute(sql)\n",
"results = cursor.fetchall()\n",
"sql = \"update stock_item set reason=%s where code=%s \"\n",
"try:\n",
" cursor.executemany(sql,list(results))\n",
" #db.commit()\n",
" print(\"ok!\")\n",
"except:\n",
" # 如果发生错误则回滚\n",
" print(\"error!\")\n",
" db.rollback() \n",
"db.close()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 外汇实时数据采集"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import requests\n",
"import time\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import re\n",
"\n",
"#pattern = re.compile('(?<=\\\").*(?=\\\")')\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"#myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"#mydb = myclient[\"gaokao\"]\n",
"#mycol = mydb[\"news\"]\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"#strhtml.encoding = 'utf8'\n",
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
"data = strhtml.text\n",
"#data1 = data.split(\"\\r\")\n",
"#data = soup.select('schoolList')\n",
"#for item1 in data:\n",
"# print(item1.get_text())\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
" print(data2)\n",
" print('当前买入价:',data2[1])\n",
" print('当前卖出价:',data2[2])\n",
" print('昨收价:',data2[3])\n",
" print('今开价:',data2[5])\n",
" print('最高价:',data2[6])\n",
" print('最低价:',data2[7])\n",
" print(float(data2[7]))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"data = strhtml.text\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
"#print(data2)\n",
"with open('price.txt','r') as fl:\n",
" for line in fi:\n",
" p_high = line.split(',')[0]\n",
" p_low = line.split(',')[1]\n",
"\n",
"message ='当前美元加元买入价:{},卖出价:{}'.format(data2[1],data2[2])\n",
"msg = MIMEText(message,'plain','utf-8')\n",
"msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
"msg['From'] = Header('512song@sina.com')\n",
"msg['To'] = Header('491525765@qq.com','utf-8')\n",
"\n",
"from_addr = '512song@sina.com' #发件邮箱\n",
"password = '409fe5d8471da663' #邮箱密码\n",
"to_addr = 'songyi@yeah.net' #收件邮箱\n",
"smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
"try:\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" print('开始登录')\n",
" server.set_debuglevel(1) \n",
" server.login(from_addr,password) #登录邮箱\n",
" print('登录成功')\n",
" print(\"邮件开始发送\")\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit()\n",
" print(\"邮件发送成功\")\n",
"except smtplib.SMTPException as e:\n",
" print(\"邮件发送失败\",e)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"data = strhtml.text\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
"print(data2)\n",
"\n",
"message ='当前美元加元最高价:{},最低价:{}'.formtat()\n",
"msg = MIMEText(message,'plain','utf-8')\n",
"\n",
"msg['Subject'] = Header(\"测试smtp邮件\",'utf-8')\n",
"msg['From'] = Header('512song@sina.com')\n",
"msg['To'] = Header('491525765@qq.com','utf-8')\n",
"\n",
"from_addr = '512song@sina.com' #发件邮箱\n",
"password = '409fe5d8471da663' #邮箱密码\n",
"to_addr = '491525765@qq.com' #收件邮箱\n",
"\n",
"smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
"try:\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" print('开始登录')\n",
" server.set_debuglevel(1) \n",
" server.login(from_addr,password) #登录邮箱\n",
" print('登录成功')\n",
" print(\"邮件开始发送\")\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit()\n",
" print(\"邮件发送成功\")\n",
"except smtplib.SMTPException as e:\n",
" print(\"邮件发送失败\",e)\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# pygal图表"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pygal\n",
"bar_chart = pygal.Bar(height=300)\n",
"bar_chart.add('Fibonacci', [0, 1, 1, 2, 3, 5, 8, 13, 21, 34, 55])\n",
"bar_chart.add('Padovan', [1, 1, 1, 2, 2, 3, 4, 5, 7, 9, 12])\n",
"svg = bar_chart.render()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from IPython.display import SVG\n",
"SVG(svg)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.5"
},
"toc-autonumbering": true,
"toc-showcode": true,
"toc-showmarkdowntxt": true
},
"nbformat": 4,
"nbformat_minor": 4
}
+74
View File
@@ -0,0 +1,74 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"#gTTS语音\n",
"from gtts import gTTS\n",
"#engine = pyttsx3.init('espeak')\n",
"tts = gTTS(text=\"军士上前,将英玉兰架起,两个抓着脚踝,两个托住肩头,一起用力,英玉兰无奈分开两条大浪腿,露出骚屄,被举下\",lang='zh-cn')\n",
"#engine.save_to_file(\"欢迎使用百度语音合成,本次测试为Python接口\",'./test')\n",
"tts.save(\"./test.mp3\")\n",
"print('ok')"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"#百度语音在线\n",
"from aip import AipSpeech\n",
"\n",
"\"\"\" 你的 APPID AK SK \"\"\"\n",
"APP_ID = '17553946'\n",
"API_KEY = 'i5LBXalkBn2KHTqdifesA1EB'\n",
"SECRET_KEY = '1ktP6qDH7nMFHjotRdKpnI13w41Bvk59'\n",
"\n",
"client = AipSpeech(APP_ID, API_KEY, SECRET_KEY)\n",
"\n",
"result = client.synthesis('军士上前,将英玉兰架起,两个抓着脚踝,两个托住肩头,一起用力,英玉兰无奈分开两条大浪腿,露出骚屄,被举下。', 'zh', 5, {\n",
" 'vol': 5,'per': 4,\n",
"})\n",
"\n",
"# 识别正确返回语音二进制 错误则返回dict 参照下面错误码\n",
"if not isinstance(result, dict):\n",
" with open('./test3.mp3', 'wb') as f:\n",
" f.write(result)\n",
"else:print(result)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.5"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
+235
View File
@@ -0,0 +1,235 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 重点高校信息管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 强基计划信息管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 录入强基计划信息"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-27T07:02:30.612429Z",
"iopub.status.busy": "2020-11-27T07:02:30.611524Z",
"iopub.status.idle": "2020-11-27T07:02:31.128055Z",
"shell.execute_reply": "2020-11-27T07:02:31.124461Z",
"shell.execute_reply.started": "2020-11-27T07:02:30.612326Z"
}
},
"outputs": [],
"source": [
"import requests\n",
"import time\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"qiangji\"]\n",
"m_content = ''\n",
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
"url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n",
"strhtml = requests.get(url,headers = headers)\n",
"strhtml.encoding = 'utf8'\n",
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
"#data = strhtml.text\n",
"#data1 = data.split(\"\\r\")\n",
"data = soup.select('body > div.container.content > div > div > div.col-md-9.col-sm-8 > div')\n",
"for item1 in data:\n",
" m_content +=item1.get_text()\n",
"#print(m_content)\n",
"m_title = soup.select('body > div.y_tit_box > div > div > div > h1')\n",
"m_name = '中国人民大学'\n",
"m_code = 'A002'\n",
"m_year = '2020'\n",
"myquery = {'code':m_code}\n",
"x = mycol.count_documents(myquery)\n",
"#print(x)\n",
"m_id = time.strftime(\"%Y%m%d%H%M%S\", time.localtime())\n",
"if x > 0: \n",
" m_mg = {} \n",
" m_mg.setdefault(m_id,{})\n",
" m_mg[m_id]['title'] = m_title[0].text\n",
" m_mg[m_id]['content'] = m_content\n",
" m_mg[m_id]['url'] = url\n",
" mycol.update_one(myquery,{'$push':{m_year:m_mg}})\n",
" print(x,s)\n",
"else:\n",
" m_mg = {}\n",
" m_mg['name'] = m_name\n",
" m_mg['code'] = m_code\n",
" m_mg.setdefault(m_year,[])\n",
" m_mg1 = {}\n",
" m_mg1.setdefault(m_id,{})\n",
" m_mg1[m_id]['title'] = m_title[0].text\n",
" m_mg1[m_id]['content'] = m_content\n",
" m_mg1[m_id]['url'] = url\n",
" m_mg[m_year].append(m_mg1)\n",
" mycol.insert_one(m_mg) \n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 录入强基计划附件"
]
},
{
"cell_type": "code",
"execution_count": 73,
"metadata": {
"execution": {
"iopub.execute_input": "2020-11-26T10:35:49.756505Z",
"iopub.status.busy": "2020-11-26T10:35:49.755575Z",
"iopub.status.idle": "2020-11-26T10:35:50.001279Z",
"shell.execute_reply": "2020-11-26T10:35:49.998646Z",
"shell.execute_reply.started": "2020-11-26T10:35:49.756395Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"1 gaokao.qiangji.2020\n"
]
}
],
"source": [
"import requests\n",
"import time\n",
"import pymongo\n",
"import os\n",
"from gridfs import GridFS\n",
"\n",
"def upLoadFile(file_coll,file_name,data_link): \n",
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
" gridfs_col = GridFS(mydb, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" #print(file_)\n",
" return file_ \n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"qiangji\"]\n",
"m_content = ''\n",
"url = 'https://www.bkzs.sdu.edu.cn/info/1036/1635.htm'\n",
"m_title = ''\n",
"m_name = '山东大学'\n",
"m_code = 'A422'\n",
"m_year = '2020'\n",
"myquery = {'code':m_code}\n",
"x = mycol.count_documents(myquery)\n",
"m_id = time.strftime(\"%Y%m%d%H%M%S\", time.localtime())\n",
"m_dir = './data/tmp'\n",
"fls=os.listdir(m_dir)\n",
"l_fls =[]\n",
"for fl in fls: \n",
" full_path = m_dir+ '/' + fl\n",
" l_fls.append(upLoadFile(\"document\",full_path,\"\"))\n",
"m_mg = {}\n",
"if x > 0: \n",
" m_mg.setdefault(m_id,{})\n",
" m_mg[m_id]['title'] = m_title\n",
" m_mg[m_id]['files'] = l_fls\n",
" m_mg[m_id]['url'] = url\n",
" mycol.update_one(myquery,{'$push':{m_year:m_mg}})\n",
"else:\n",
" m_mg['name'] = m_name\n",
" m_mg['code'] = m_code\n",
" m_mg.setdefault(m_year,[])\n",
" m_mg1 = {}\n",
" m_mg1.setdefault(m_id,{})\n",
" m_mg1[m_id]['title'] = m_title\n",
" m_mg[m_id]['files'] = l_fls\n",
" m_mg1[m_id]['url'] = url\n",
" m_mg[m_year].append(m_mg1)\n",
" mycol.insert_one(m_mg) \n"
]
},
{
"cell_type": "code",
"execution_count": 1,
"metadata": {
"execution": {
"iopub.execute_input": "2021-02-24T08:13:56.382398Z",
"iopub.status.busy": "2021-02-24T08:13:56.381289Z",
"iopub.status.idle": "2021-02-24T08:13:56.397601Z",
"shell.execute_reply": "2021-02-24T08:13:56.395607Z",
"shell.execute_reply.started": "2021-02-24T08:13:56.382135Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"20210224161356\n"
]
}
],
"source": [
"import time\n",
"\n",
"print(time.strftime(\"%Y%m%d%H%M%S\", time.localtime()))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
},
"toc-autonumbering": false
},
"nbformat": 4,
"nbformat_minor": 4
}
+150
View File
@@ -0,0 +1,150 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "488c74e2-3b75-4274-8d9a-4503f367b5ca",
"metadata": {},
"source": [
"# 高考志愿查询"
]
},
{
"cell_type": "markdown",
"id": "8913f77b-bbd5-4959-b5a4-3214732f8480",
"metadata": {},
"source": [
"## 按位次模糊查询专业"
]
},
{
"cell_type": "code",
"execution_count": 80,
"id": "be2371be-89f5-4879-82a9-4a1d9e6376d3",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"1 南京中医药大学 中医学(本硕连读5+3一体化) 11019\n",
"2 天津医科大学 预防医学 11108\n",
"3 哈尔滨医科大学 临床医学 11227\n",
"4 空军军医大学 基础医学 11232\n",
"5 南京医科大学 预防医学 11275\n",
"6 温州医科大学 眼视光医学(5+3一体化) 11564\n",
"7 上海中医药大学 中医学(5+3一体化针灸推拿英语方向) 11620\n",
"8 兰州大学 临床医学类 11662\n",
"9 天津中医药大学 中医学(5+3一体化) 11825\n",
"10 哈尔滨医科大学 临床医学(5+3一体化,儿科学硕士) 11837\n",
"11 温州医科大学 临床医学(5+3一体化) 11869\n",
"12 东北大学 智能医学工程 11916\n",
"13 海军军医大学 中医学(中医临床医师) 11955\n",
"14 苏州大学 预防医学 11991\n",
"15 暨南大学 临床医学 12045\n",
"16 中国医科大学 医学影像学 12059\n",
"17 南昌大学 临床医学 12252\n",
"18 天津医科大学 医学影像技术 12272\n",
"19 吉林大学 预防医学 12320\n",
"20 南京航空航天大学 生物医学工程 12504\n",
"21 江南大学 临床医学 12526\n",
"22 广州中医药大学 中医学(5+3一体化) 12715\n",
"23 中国医科大学 基础医学 12755\n",
"24 重庆医科大学 医学影像学 12784\n",
"25 大连医科大学 临床医学(5+3一体化) 13012\n",
"26 天津医科大学 智能医学工程 13103\n",
"27 南京中医药大学 中医学 13301\n",
"28 温州医科大学 眼视光医学 13370\n",
"29 天津医科大学 医学检验技术 13371\n",
"30 天津医科大学 生物医学工程 13396\n",
"31 温州医科大学 临床医学 13421\n",
"32 中国医科大学 预防医学 13769\n",
"33 汕头大学 临床医学(5+3一体化) 13808\n",
"34 郑州大学 临床医学类 13981\n",
"35 郑州大学 口腔医学 14147\n",
"36 南方医科大学 基础医学 14160\n",
"37 哈尔滨医科大学 基础医学 14331\n",
"38 南方医科大学 预防医学 14366\n",
"39 大连医科大学 临床医学 14423\n",
"40 西北大学 临床医学 14492\n",
"41 天津中医药大学 中医学(5+3一体化中医儿科学) 14708\n",
"42 南京医科大学 智能医学工程 14721\n"
]
}
],
"source": [
"import pymongo\n",
"import re\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"admission_2020\"]\n",
"\n",
"m_zhuanye = '医学'\n",
"m_rank1 = 11000\n",
"m_rank2 = 15000\n",
"\n",
"myquery = {'spe_name':re.compile(m_zhuanye),'rank_min':{\"$gte\": m_rank1,\"$lte\": m_rank2}}\n",
"i = 1\n",
"for x in mycol.find(myquery,{\"_id\": 0, }):\n",
" print(i,x['col_name'],x['spe_name'],x['rank_min'])\n",
" i += 1\n"
]
},
{
"cell_type": "markdown",
"id": "2731820f-b81e-4d7e-8f53-a2501d9c943f",
"metadata": {},
"source": [
"## 按分数模糊查询专业"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "07dbb641-d594-4c7c-92da-045a5370e6e9",
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"import re\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"admission_2020\"]\n",
"\n",
"m_zhuanye = '医学'\n",
"m_fenshu = 600\n",
"\n",
"myquery = {'spe_name':re.compile(m_zhuanye),'num_min':{\"$gte\": m_fenshu}}\n",
"i = 1\n",
"for x in mycol.find(myquery,{\"_id\": 0, }):\n",
" print(i,x['col_name'],x['spe_name'],x['num_min'])\n",
" i += 1\n"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
},
"toc-autonumbering": true,
"toc-showmarkdowntxt": true,
"toc-showtags": false
},
"nbformat": 4,
"nbformat_minor": 5
}
+690
View File
@@ -0,0 +1,690 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {
"tags": [],
"toc-hr-collapsed": true
},
"source": [
"# 高考志愿管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 2020年高考录取信息导入MongoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pymysql\n",
"import pymongo\n",
"import decimal\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"mycol1 = mydb[\"admission_2020\"]\n",
"\n",
"m_col = {}\n",
"m_spe = {}\n",
"m_xx = {}\n",
"for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n",
" m_col[x['code']] = x['name']\n",
"\n",
"\n",
"db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n",
"cursor = db.cursor()\n",
"sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2020\"'\n",
"cursor.execute(sql)\n",
"results = cursor.fetchall()\n",
"for result in results:\n",
" m_spe.setdefault(result[1],{}) \n",
" m_spe[result[1]][result[0]] = result[2]\n",
"sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'\n",
"cursor.execute(sql)\n",
"results = cursor.fetchall()\n",
"i = 0\n",
"m_min = 0\n",
"ii = 0\n",
"for result in results:\n",
" m_xx.clear()\n",
" if result[6] == m_min:\n",
" ii = ii\n",
" i = i+1\n",
" else:\n",
" i = i+1\n",
" ii = i\n",
" m_min = result[6]\n",
" m_xx['pos'] = ii\n",
" m_xx['col_code'] = result[1]\n",
" m_xx['col_name'] = m_col[result[1]]\n",
" m_xx['spe_code'] = result[2]\n",
" m_xx['spe_name'] = m_spe[result[1]][result[2]]\n",
" m_xx['plan'] = result[3]\n",
" m_xx['dispense'] = result[5]\n",
" m_xx['num_min'] = result[6]\n",
" m_xx['num_avg'] = int(result[7])\n",
" m_xx['rank_min'] = result[8]\n",
" m_xx['nian'] = '2020' \n",
" mycol1.insert_one(m_xx)\n",
"#print(m_spe)\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 计算志愿分数概率"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import random\n",
"\n",
"array2 = []\n",
"for i in range(300000):\n",
" array1 = []\n",
" s = 0\n",
" for ii in range(60):\n",
" m1 = random.randint(580,595)\n",
" array1.append(m1)\n",
" s = s + m1 \n",
" m_avg = round(s/60,2)\n",
" if m_avg == 582.8 and (580 in array1) and (595 in array1):\n",
" #print(array1)\n",
" array2.extend(array1)\n",
" \n",
"#print(array2)\n",
"m_set = set(array2)\n",
"for m in m_set:\n",
" print(m,array2.count(m))"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 导入山东大学录取明细"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"admission_college\"]\n",
"wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')\n",
"sheet = wb.active\n",
"#sheets = wb.sheetnames\n",
"code = 'A422'\n",
"name = '山东大学'\n",
"new_col = []\n",
"dict1 = {}\n",
"\n",
"new_code = []\n",
"dict1['code'] = code\n",
"dict1['name'] = name\n",
"for n in range(1,sheet.max_row+1):\n",
" nian = str(sheet.cell(n,1).value)\n",
" dict2 = {}\n",
" \n",
" dict1.setdefault(nian,[])\n",
" if sheet.cell(n,2).value =='理工':\n",
" m_lb = 'l'\n",
" elif sheet.cell(n,2).value =='文史':\n",
" m_lb = 'w'\n",
" else:\n",
" m_lb = 'z' \n",
" dict2['type'] = m_lb\n",
" dict2['spe_name'] = sheet.cell(n,4).value\n",
" dict2['max_score'] = sheet.cell(n,5).value\n",
" dict2['min_score'] = sheet.cell(n,6).value\n",
" dict2['avg_score'] = sheet.cell(n,7).value\n",
" dict2['dispense'] = sheet.cell(n,8).value\n",
" dict1[nian].append(dict2)\n",
"mycol.insert_one(dict1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 导入山东师范大学录取明细"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"admission_college\"]\n",
"wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')\n",
"sheet = wb.active\n",
"#sheets = wb.sheetnames\n",
"code = 'A423'\n",
"name = '中国海洋大学'\n",
"new_col = []\n",
"dict1 = {}\n",
"\n",
"new_code = []\n",
"dict1['code'] = code\n",
"dict1['name'] = name\n",
"for n in range(1,sheet.max_row+1):\n",
" nian = str(sheet.cell(n,1).value)\n",
" dict2 = {}\n",
" \n",
" dict1.setdefault(nian,[])\n",
" if sheet.cell(n,2).value =='理工':\n",
" m_lb = 'l'\n",
" elif sheet.cell(n,2).value =='文史':\n",
" m_lb = 'w'\n",
" else:\n",
" m_lb = 'z'\n",
" dict2['type'] = m_lb\n",
" dict2['spe_name'] = sheet.cell(n,4).value\n",
" dict2['max_score'] = sheet.cell(n,7).value\n",
" dict2['min_score'] = sheet.cell(n,5).value\n",
" dict2['avg_score'] = sheet.cell(n,6).value\n",
" if sheet.cell(n,8).value:\n",
" dict2['dispense'] = sheet.cell(n,8).value\n",
" dict1[nian].append(dict2)\n",
"mycol.insert_one(dict1)\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 按地区列示高校"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"m_city = []\n",
"m_xx = {}\n",
"myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}\n",
"colleges = mycol.find(myquery,{ \"_id\": 0, \"name\": 1, \"code\": 1,'city':1 })\n",
"for x in colleges:\n",
" m_xx1 = []\n",
" c = x['city']\n",
" m_xx.setdefault(c,[])\n",
" m_xx1.append(x['code'])\n",
" m_xx1.append(x['name'])\n",
" m_xx[c].append(m_xx1)\n",
"#print(m_city)\n",
"#按照原顺序对高校所在城市排序\n",
"'''\n",
"city = list(set(m_city))\n",
"city.sort(key=m_city.index)\n",
"for c in city:\n",
"# m_xx['city'] = c\n",
" m_xx.setdefault(c,[])\n",
" for y in colleges:\n",
" print(y['code'],y['name'])\n",
" \n",
"#m_xx \n",
"''' \n",
"for k,v in m_xx.items():\n",
" print(k)\n",
" for mm in v:\n",
" print(mm[0],mm[1])"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 显示学校2020年招生信息"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"admission_2020\"]\n",
"code = 'A422'\n",
"m_city = []\n",
"m_xx = {}\n",
"myquery = {'col_code':code}\n",
"colleges = mycol.find(myquery,{ \"_id\": 0 , \"col_code\":0,\"col_name\":0,'plan':0,'nian':0}).sort('pos')\n",
"for x in colleges:\n",
" print(x)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 2021年拟在山东招生普通高校专业(类)选考科目要求"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from os import mkdir\n",
"from time import sleep\n",
"from re import findall,sub,S\n",
"from os.path import isdir,isfile\n",
"from urllib.request import urlopen\n",
"from urllib.parse import urlencode,quote\n",
"from openpyxl import Workbook\n",
"import ssl\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"ssl._create_default_https_context = ssl._create_unverified_context\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"xuankaokemu\"]\n",
"list2 = []\n",
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
" list2.append(x['code'])\n",
"#m_xx = dict()\n",
"start_url = 'https://xkkm.sdzk.cn/web/xx.html'\n",
"with urlopen(start_url) as fp:\n",
" content = fp.read().decode('utf8')\n",
" \n",
"pattern = (r'<tr>.*?<td.+?</td>.*?<td.+?>(.+?)</td>'\n",
" '.*?<td.+?>(.+?)</td>.*?<td.+?>(.+?)</td>')\n",
"\n",
"for item in findall(pattern,content,S):\n",
" if len(item[0]) > 5:\n",
" continue\n",
" \n",
" shengfen,dm,mc = item\n",
" print('学校代码:', dm)\n",
" print('学校名称:', mc)\n",
" m_xx = {}\n",
" m_xx['code'] = dm\n",
" m_xx['name'] = mc\n",
" m_xx.setdefault('zhuanye',{})\n",
" if dm in list2:\n",
" continue\n",
" url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'\n",
" data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')\n",
" with urlopen(url,data) as fp:\n",
" xuexiao_content = fp.read().decode()\n",
" soup = BeautifulSoup(xuexiao_content,'lxml')\n",
" #data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')\n",
" data = soup.select('#ccc > div > table > tbody > tr ')\n",
" for data1 in data:\n",
" list1 =[]\n",
" m_xx1 = {}\n",
" for item in data1.stripped_strings: \n",
" list1.append(item)\n",
" #del list1[0]\n",
" #print('层次:',list1[1])\n",
" #print('专业(类)名称:',list1[2])\n",
" #print('选考科目范围:',list1[3])\n",
" #print('类中所含专业:',list1[4:])\n",
" code = list1[0]\n",
" m_xx['zhuanye'].setdefault(code,{})\n",
" \n",
" m_xx['zhuanye'][code]['name'] = list1[2]\n",
" m_xx['zhuanye'][code]['level'] = list1[1]\n",
" m_xx['zhuanye'][code]['fanwei'] = list1[3]\n",
" m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]\n",
" mycol.insert_one(m_xx) \n",
" \n",
" print('ok')\n",
" \n",
" # list1.clear\n",
" \n",
"\n",
" sleep(5)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"xuankaokemu\"]\n",
"list2 = []\n",
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
" list2.append(x['code'])\n",
"print(list2)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 招生简章管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 各大学招生简章采集"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from time import sleep\n",
"import ssl\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import requests\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
"#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'\n",
"\n",
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
"for i in sf:\n",
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
" strhtml = requests.get(url,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
" \n",
" for item in data:\n",
" dict1 = {}\n",
" dict1['name'] = item.text.strip()\n",
" dict1['url'] = item.get('href')\n",
" if item.get('style') =='color:gray':\n",
" dict1['bz'] = 0\n",
" else:\n",
" dict1['bz'] = 1\n",
" mycol.insert_one(dict1) \n",
"\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 采集单个学校招生简章"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from time import sleep\n",
"from re import findall,sub,S\n",
"import ssl\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import requests\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
"url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'\n",
"strhtml = requests.get(url,headers = headers)\n",
"strhtml.encoding = 'utf8'\n",
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
"#data = strhtml.text\n",
"#data1 = data.split(\"\\r\")\n",
"#print(soup)\n",
"data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
"for item in data:\n",
" url1 = item.get('href')\n",
"url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
"strhtml = requests.get(url1,headers = headers)\n",
"strhtml.encoding = 'utf8'\n",
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
"data = soup.select('body > div.width1000.border.gery > div>p')\n",
"nr = ''\n",
"for item in data:\n",
" nr += item.text+'\\n'\n",
"print(nr)\n",
" \n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 第一次采集招生简章"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from time import sleep\n",
"from re import findall,sub,S\n",
"import ssl\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import requests\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
"for x in mycol.find({\"bz\": 1 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
" url = 'https://gaokao.chsi.com.cn'+x['url']\n",
" col_name = x['name']\n",
" strhtml = requests.get(url,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" #data = strhtml.text\n",
" #data1 = data.split(\"\\r\")\n",
" #print(soup)\n",
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
" for item in data:\n",
" url1 = item.get('href')\n",
" url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
" strhtml = requests.get(url1,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
" nr = ''\n",
" for item in data:\n",
" nr += item.text+'\\n' \n",
" myquery = {'name':col_name}\n",
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
" sleep(5)\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 检查新增的学校及招生简章"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from time import sleep\n",
"from re import findall,sub,S\n",
"import ssl\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import requests\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
"l_name = []\n",
"for x in mycol.find({},{ \"_id\": 0, \"name\": 1}):\n",
" l_name.append(x['name'])\n",
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
"n_url = []\n",
"for i in sf:\n",
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
" strhtml = requests.get(url,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
" \n",
" for item in data:\n",
" m_url = item.get('href')\n",
" m_name = item.text.strip()\n",
" if m_name not in l_name:\n",
" url = 'https://gaokao.chsi.com.cn'+m_url\n",
" col_name = m_name \n",
" dict1 = {}\n",
" dict1['name'] = m_name\n",
" dict1['url'] = m_url\n",
" dict1['bz'] = 0 \n",
" mycol.insert_one(dict1) \n",
" print(dict1)\n",
" sleep(3)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 检查、新增招生简章"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [],
"source": [
"from time import sleep\n",
"from re import findall,sub,S\n",
"import ssl\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import requests\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
"l_name = []\n",
"for x in mycol.find({\"bz\": 0 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
" l_name.append(x['name'])\n",
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
"n_name = []\n",
"for i in sf:\n",
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk' \n",
" strhtml = requests.get(url,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
" \n",
" for item in data:\n",
" m_url = item.get('href')\n",
" col_name = item.text.strip()\n",
" #print(m_url)\n",
" if (item.get('style') !='color:gray') and (col_name in l_name):\n",
" url = 'https://gaokao.chsi.com.cn' + m_url\n",
" \n",
" strhtml = requests.get(url,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
" for item in data:\n",
" url1 = item.get('href')\n",
" print(url1)\n",
" url1 = 'https://gaokao.chsi.com.cn' + url1\n",
" strhtml = requests.get(url1,headers = headers)\n",
" strhtml.encoding = 'utf8'\n",
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
" nr = ''\n",
" for item in data:\n",
" nr += item.text+'\\n' \n",
" myquery = {'name':col_name}\n",
" mycol.update_one(myquery,{'$set':{'bz':1}})\n",
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
" print(col_name)\n",
" sleep(3)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
File diff suppressed because it is too large. Load diff