Files
jupyter/文件操作.ipynb
T
2023-04-27 19:28:57 +08:00

1256 lines
36 KiB
Plaintext

{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 数字文件名转换为文本文件名"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 数字文件名转换为文本文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd()+'/file/北海石化监测花名册2022 .xlsx'\n",
"fi_name = {}\n",
"fi_path = os.getcwd()+'/file/221103'\n",
"old = []\n",
"new = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1):\n",
" if sheet.cell(n,3).value is not None: \n",
" m_name = sheet.cell(n,6).value.strip()\n",
" m_depart = sheet.cell(n,3).value.strip() \n",
" depart.append(m_depart) \n",
" dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n",
" #print()\n",
"# 创建部门办公室 \n",
"m_path = os.getcwd()+'/file/221103/new'\n",
"for pn in depart:\n",
" if not os.path.exists(m_path + '/' + pn):\n",
" os.mkdir(m_path + '/' + pn)\n",
"#print(dict1)\n",
"\n",
"\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn)\n",
" #print(fn)\n",
"old.sort()\n",
" #print(str(nfn)+'.pdf')\n",
"\n",
"for n in old:\n",
" \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(old)\n",
"\n",
"#print(dict1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 目录文件按照文件名排序"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"\n",
"import os,sys\n",
"\n",
"\n",
"fi_xls = 'test1.xlsx'\n",
"fi_name = {}\n",
"#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n",
"fi_path = os.getcwd()+'/data'\n",
"old = []\n",
"new = []\n",
"\n",
"fl=os.listdir(fi_path)\n",
"fl.sort()\n",
"n = 0\n",
"for i in fl:\n",
" oldname=fl[n]\n",
" name, suffix = os.path.splitext(oldname)\n",
" if name in old:\n",
" new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" old_name = fi_path+ os.sep + fl[n]\n",
" os.rename(old_name,new_name)\n",
" n+= 1\n",
"fl"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 将pdf文件转为图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n",
"print(path)\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 图像文件夹打包"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import zipfile\n",
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"import shutil\n",
"import time\n",
"\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"def addfile(zipfilename, dirname):\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"def make_path(p):\n",
" if os.path.exists(p): # 判断文件夹是否存在\n",
" shutil.rmtree(p) # 删除文件夹\n",
" os.mkdir(p) \n",
"pdf_file = '2.pdf'\n",
"output_folder='./pic1'\n",
"zip_file = 'ribenweiqishihua.zip'\n",
"make_path(output_folder)\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n",
"compress_file(zip_file, output_folder) # 执行函数\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 文本文件操作"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 基本读取"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import re\n",
"file_name = 'data/2012.txt'\n",
"with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n",
" for l in fl:\n",
" l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n",
" if l.split():\n",
" print(l)\n",
" fl1.write(l)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 读取csv文件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import csv\n",
"import re\n",
"\n",
"filename = 'data/130.csv'\n",
"with open(filename,'r',newline='') as csv_file:\n",
" fl = csv.reader(csv_file,delimiter=',')\n",
" header = next(fl) \n",
" for line in fl:\n",
" #line = re.sub('[\\r\\n\\f ]{1,}', '', line)\n",
" print(line)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 读取分隔符分割文件,导入MongoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"import re\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"city\"]\n",
"m_mongo = {}\n",
"m_xx = []\n",
"fl_name = 'china-city-list.txt'\n",
"n = 0\n",
"with open(fl_name,'r') as fl:\n",
" for l in fl:\n",
" n += 1\n",
" if n >6:\n",
" m_mongo = {}\n",
" m_xx = re.sub('[ ]{1,}', '', l).split('|')\n",
" #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n",
" m_mongo['name'] = m_xx[3]\n",
" m_mongo['code'] = m_xx[1]\n",
" m_mongo['sheng'] = m_xx[8]\n",
" m_mongo['shi'] = m_xx[10]\n",
" m_mongo['jing'] = m_xx[11]\n",
" m_mongo['wei'] = m_xx[12]\n",
" mycol.insert_one(m_mongo) \n",
" #print(m_mongo)\n",
"print('ok!')\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Excel文件修改"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import openpyxl\n",
"import json\n",
"\n",
"filename = 'data/122.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"dict2 = {}\n",
"for k, v in dict1.items():\n",
" for item in v:\n",
" bh = item['avatar_id']\n",
" dict2[bh] = item\n",
"print(dict2) \n",
" \n",
"wb = openpyxl.load_workbook('data/济南炼化职工分组表.xlsx')\n",
"sheet = wb.active\n",
"person = {}\n",
"for n in range(2, sheet.max_row+1):\n",
" if sheet.cell(n, 14).value is not None and sheet.cell(n, 14).value in dict2.keys():\n",
" code = sheet.cell(n, 14).value\n",
" sheet[f'F{n}']=dict2[code]['birth']\n",
"wb.save('data/济南炼化职工分组表1.xlsx') \n",
"print('ok!') \n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## YALM文件操作"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### YALM文件读取转存JSON文件"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {
"execution": {
"iopub.execute_input": "2023-04-27T09:20:28.067998Z",
"iopub.status.busy": "2023-04-27T09:20:28.067295Z",
"iopub.status.idle": "2023-04-27T09:20:28.183642Z",
"shell.execute_reply": "2023-04-27T09:20:28.182659Z",
"shell.execute_reply.started": "2023-04-27T09:20:28.067958Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [
"import yaml\n",
"import json\n",
"\n",
"filename = 'data/体质指标问卷.yaml'\n",
"with open(filename, 'r', encoding='utf-8') as f:\n",
" result = yaml.load(f.read(), Loader=yaml.FullLoader)\n",
"i = 1\n",
"#print(result)\n",
"filename = 'data/体质指标问卷.json'\n",
"with open(filename, 'w') as fl:\n",
" json.dump(result, fl, ensure_ascii=False)\n",
"print('ok')"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 按照年龄分组生成心理问卷"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import json\n",
"import os,sys,shutil\n",
"import openpyxl\n",
"\n",
"list1 = [\n",
" [1,'完全不正确','有点正确','多数正确','完全正确'],\n",
" [2,'从不','极少','偶尔','经常','频繁','非常频繁','每天'],\n",
" [3,'不符合','有时符合','常常符合','总是符合'],\n",
" [4,'从不','很少','有时','经常','总是'],\n",
" [5,'非常不满意','不满意','一般','满意','非常满意'],\n",
" [6,'没有或很少时间(少于1天)','一部分时间(1-2天)','相当多时间(3-4天)','绝大部分时间(5-7天)'],\n",
" [7,'从来没有这种感觉','很少有这种感觉','有时有这种感觉','经常有这种感觉'],\n",
" [8,'完全不同意','不同意','同意','完全同意'], \n",
" [10,'晚多了','晚一点','差不多','早一点','早多了']\n",
"]\n",
"dict1 = {}\n",
"for item in list1:\n",
" #dict1.setdefault(item[0],[])\n",
" list2 =[]\n",
" for i in range(1,len(item)):\n",
" list2.append(item[i])\n",
" dict1[item[0]] = list2\n",
"dict_choiced = {}\n",
"print(dict1)\n",
"wj = {}\n",
"wj[\"login\"] = \"phone\"\n",
"wj[\"locale\"] = \"zh-cn\"\n",
"wj[\"title\"] = '社区心理测评问卷'\n",
"list2 =[]\n",
"dict2 = {\n",
" \"type\": \"radiogroup\",\n",
" \"name\": \"Gender\",\n",
" \"title\": {\n",
" \"zh-cn\": \"您的性别\"\n",
" },\n",
" \"isRequired\": True,\n",
" \"choices\": [\n",
" {\n",
" \"value\": \"m\",\n",
" \"text\": \"男性\"\n",
" },\n",
" {\n",
" \"value\": \"f\",\n",
" \"text\": \"女性\"\n",
" }\n",
" ] \n",
"}\n",
"list2.append(dict2)\n",
"dict2 = {}\n",
"dict3 = {}\n",
"dict2 = {\n",
" \"type\": \"radiogroup\",\n",
" \"name\": \"Age\",\n",
" \"title\": {\n",
" \"zh-cn\": \"您的年龄\"\n",
" },\n",
" \"isRequired\": True,\n",
" \"choices\": [\n",
" {\n",
" \"value\": \"Y\",\n",
" \"text\": \"8-17岁\"\n",
" },\n",
" {\n",
" \"value\": \"M\",\n",
" \"text\": \"15-59岁\"\n",
" },\n",
" {\n",
" \"value\": \"O\",\n",
" \"text\": \"60岁及以上\"\n",
" }\n",
" ] \n",
"}\n",
"list2.append(dict2)\n",
"\n",
"wb = openpyxl.load_workbook('data/心理问卷.xlsx')\n",
"sheet = wb.active\n",
"data1 =list(sheet.values)\n",
"del data1[0]\n",
"for item in data1:\n",
" \n",
" dict2 = {} \n",
" dict2['type'] = item[4] \n",
" dict2['name'] = 'q'+str(item[0])+item[3]\n",
" dict2['visibleIf'] = '{Age}=\"'+item[3]+'\"'\n",
" dict3 = {}\n",
" dict3[\"zh-cn\"] = item[1]\n",
" dict2['title'] =dict3 \n",
" dict2['isRequired'] = True\n",
" if item[4] == 'rating':\n",
" dict2['rateMin'] = 0\n",
" dict2['rateMax'] = 7\n",
" dict2['minRateDescription'] = { 'zh-cn':'没有'}\n",
" dict2['maxRateDescription'] = { 'zh-cn':'非常同意'}\n",
" else:\n",
" if int(item[2]) in dict_choiced.keys():\n",
" dict2['choicesFromQuestion'] = dict_choiced[int(item[2])]\n",
" else:\n",
" list3 = []\n",
" i = 1\n",
" for xm in dict1[int(item[2])]:\n",
" dict3 = {}\n",
" dict3 = {'value': i,'text' : xm}\n",
" list3.append(dict3)\n",
" i+=1\n",
" dict2['choices'] = list3\n",
" dict_choiced[int(item[2])] = dict2['name']\n",
" list2.append(dict2)\n",
"dict3 = {}\n",
"dict3[\"elements\"] = list2\n",
"wj[\"pages\"] =[]\n",
"\n",
"wj[\"pages\"].append(dict3)\n",
"#filename = 'data/心理问卷1.json'\n",
"#with open(filename, 'w') as fl:\n",
"# json.dump(wj, fl, ensure_ascii=False)\n",
"#print('ok')\n",
"with open('心理问卷.yml', 'w', encoding='utf-8') as f:\n",
" yaml.dump(data=wj, stream=f, allow_unicode=True)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import yaml\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 邮件管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 使用网易邮箱群发邮件"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.mime.multipart import MIMEMultipart\n",
"from email.mime.application import MIMEApplication\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"import json\n",
"\n",
"\n",
" \n",
"\n",
"fl_name = 'data/低碳院报告.json'\n",
"\n",
"with open(fl_name,'r') as fl:\n",
" m_xx = json.load(fl)\n",
"for k, v in m_xx.items():\n",
" m_bh = str(k).rjust(5,\"0\")\n",
" fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n",
" fl = f'{m_bh}-{v[0]}.pdf'\n",
" m_rec = v[2]+'@ceic.com'\n",
" \n",
" from_addr = 'kmingedu@163.com' #发件邮箱\n",
" password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n",
" smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" msg = MIMEMultipart()\n",
" msg['Subject'] = Header(\"低碳清洁能源研究院体质监测报告\",'utf-8')\n",
" msg['From'] = Header('北京坤铭体质监测评估中心')\n",
" msg['To'] = Header(m_rec)\n",
"\n",
" from_addr = 'kmingedu@163.com' #发件邮箱\n",
" password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n",
" to_addr = m_rec #收件邮箱\n",
" att1 =MIMEApplication(open(fl_name, 'rb').read())\n",
" #att1[\"Content-Type\"] = 'application/octet-stream'\n",
" # 这里的filename可以任意写,写什么名字,邮件中显示什么名字\n",
" att1.add_header('Content-Disposition','attachment',filename=fl)\n",
" msg.attach(att1)\n",
" try:\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit\n",
" time.sleep(5)\n",
" print(f'{fl}邮件发送成功!')\n",
" except smtplib.SMTPException:\n",
" print (\"Error: 无法发送邮件\")\n",
" \n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 使用网易邮箱群发测试链接"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.mime.multipart import MIMEMultipart\n",
"from email.mime.application import MIMEApplication\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"import json\n",
"import jwt\n",
"\n",
"key = \"Xixi1234\"\n",
"encoded = jwt.encode({\"email\": \"eygsoft@gmail.com\"}, key, algorithm='HS256')\n",
"#print(encoded)\n",
"\n",
"\n",
" \n",
"list1 = [\n",
" {'name':'lillianwong','E_mail':'lillianwong73@gmail.com'},\n",
" {'name':'杨玉冰','E_mail':'lkyyb@126.com'}]\n",
"\n",
"\n",
"for item in list1:\n",
" m_mail = item['E_mail']\n",
" m_name = item['name']\n",
" encoded = jwt.encode({\"email\": m_mail}, key, algorithm='HS256')\n",
" lianjie = f'https://www.forraid.com/survey/token/{encoded}/tcm0'\n",
" mail_msg = f\"\"\"\n",
"<p>欢迎参加北京坤铭体质监测评估中心中医体质检测</p>\n",
"<p><a href=\"{lianjie}\">点击参加测试</a></p>\n",
"\"\"\"\n",
" from_addr = 'kmingedu@163.com' #发件邮箱\n",
" password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n",
" smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" msg = MIMEText(mail_msg, 'html', 'utf-8')\n",
" msg['Subject'] = Header(\"中医体质检测测试链接\",'utf-8')\n",
" msg['From'] = Header('北京坤铭体质监测评估中心')\n",
" msg['To'] = Header(m_mail)\n",
"\n",
" \n",
" to_addr = m_mail #收件邮箱\n",
" try:\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit\n",
" time.sleep(5)\n",
" print('邮件发送成功!')\n",
" except smtplib.SMTPException:\n",
" print (\"Error: 无法发送邮件\")\n",
" "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"try:\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit\n",
" time.sleep(5)\n",
" print(f'{fl}邮件发送成功!')\n",
" except smtplib.SMTPException:\n",
" print (\"Error: 无法发送邮件\")\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# Twilio使用"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"from twilio.rest import Client\n",
"\n",
"\n",
"# Your Account Sid and Auth Token from twilio.com/console\n",
"# and set the environment variables. See http://twil.io/secure\n",
"account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n",
"auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n",
"client = Client(account_sid, auth_token)\n",
"\n",
"message = client.messages \\\n",
" .create(\n",
" body=\"I'm back.\",\n",
" from_='+12056066931',\n",
" to='+8613793180751'\n",
" )\n",
"\n",
"print(message.sid)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import time\n",
"\n",
"localtime = time.localtime(time.time())\n",
"#type(localtime)\n",
"print (\"本地时间为 :\", localtime)\n",
"jyr = '12345'\n",
"if time.strftime(\"%w\", time.localtime()) in jyr:\n",
" print('ok')\n",
"else:\n",
" print('今日不是交易日!')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"\n",
"def sendmail(message):\n",
" msg = MIMEText(message,'plain','utf-8')\n",
" msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
" msg['From'] = Header('512song@sina.com')\n",
" msg['To'] = Header('songyi@yeah.net','utf-8')\n",
"\n",
" from_addr = '512song@sina.com' #发件邮箱\n",
" password = '409fe5d8471da663' #邮箱密码\n",
" to_addr = 'songyi@yeah.net' #收件邮箱\n",
" smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit() \n",
" \n",
" \n",
"\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"data = strhtml.text\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
"#print(data2)\n",
"with open('price.txt','r') as fl:\n",
" for line in fl:\n",
" p_high = line.split(',')[0]\n",
" p_low = line.split(',')[1]\n",
"m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
"while time.strftime(\"%w\", time.localtime()) in '12345':\n",
" \n",
" print(p_high,p_low)\n",
" time.sleep(10)\n",
" strhtml = requests.get(url)\n",
" data = strhtml.text\n",
" if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
" sendmail(m_message)\n",
" p_high = str(float(p_high) + 0.04) \n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元卖出价:{}'.format(data2[2])\n",
" p_low = str(float(p_low) - 0.04)\n",
" sendmail(m_message)\n",
" time.sleep(900)\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# AWS应用"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS获取sns信息"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Create an SNS client\n",
"sns = boto3.client('sns')\n",
"\n",
"# Call SNS to list topics\n",
"response = sns.list_topics()\n",
"\n",
"# Get a list of all topic ARNs from the response\n",
"topics = [topic['TopicArn'] for topic in response['Topics']]\n",
"\n",
"# Print out the topic list\n",
"print(\"Topic List: %s\" % topics)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS操作DynamoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"# Create the DynamoDB table.\n",
"table = dynamodb.create_table(\n",
" TableName='waihui',\n",
" \n",
" AttributeDefinitions=[ \n",
" {\n",
" 'AttributeName': 'code',\n",
" 'AttributeType': 'S'\n",
" }\n",
" \n",
" \n",
" ],\n",
" KeySchema=[\n",
" {\n",
" 'AttributeName': 'code',\n",
" 'KeyType': 'HASH'\n",
" }\n",
" \n",
" ],\n",
" ProvisionedThroughput={\n",
" 'ReadCapacityUnits': 5,\n",
" 'WriteCapacityUnits': 5\n",
" }\n",
" \n",
")\n",
"\n",
"# Wait until the table exists.\n",
"table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n",
"\n",
"# Print out some data about the table.\n",
"print(table.item_count)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.put_item(\n",
" Item={\n",
" 'code': 'USDCAD',\n",
" 'high': Decimal('1.3200'),\n",
" 'low': Decimal('1.3000'),\n",
" }\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"response = table.get_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n",
"item = response['Item']\n",
"print(item)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.delete_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"table.update_item(\n",
" Key={\n",
" 'code': 'USDCAD'\n",
" },\n",
" UpdateExpression='SET low = :val1',\n",
" ExpressionAttributeValues={\n",
" ':val1': decimal.Decimal('1.2900')\n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Create SQS client\n",
"sqs = boto3.client('sqs')\n",
"\n",
"queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n",
"\n",
"# Receive message from SQS queue\n",
"response = sqs.receive_message(\n",
" QueueUrl=queue_url,\n",
" AttributeNames=[\n",
" 'SentTimestamp'\n",
" ],\n",
" MaxNumberOfMessages=1,\n",
" MessageAttributeNames=[\n",
" 'All'\n",
" ],\n",
" VisibilityTimeout=0,\n",
" WaitTimeSeconds=0\n",
")\n",
"\n",
"message = response['Messages'][0]\n",
"receipt_handle = message['ReceiptHandle']\n",
"\n",
"# Delete received message from queue\n",
"sqs.delete_message(\n",
" QueueUrl=queue_url,\n",
" ReceiptHandle=receipt_handle\n",
")\n",
"print('Received and deleted message: %s' % message)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# MongoDB系统GridFS文件管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 文件上传"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(self,file_coll,_id,out_name):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"m_dir = './data/tmp'\n",
"fls=os.listdir(m_dir)\n",
"n = 0\n",
"for fl in fls:\n",
" #oldname=fl[n]\n",
" name, suffix = os.path.splitext(fl)\n",
" #if name in old:\n",
" # new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" # old_name = fi_path+ os.sep + fl[n]\n",
" # os.rename(old_name,new_name)\n",
" #print(os.path.basename(fl))\n",
" #print(fl,suffix[1:])\n",
" full_path = m_dir+ '/' + fl\n",
" upLoadFile(\"document\",full_path,\"\")\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": file_name, \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(file_coll,_id,out_name):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n",
"print (ll)\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.10.6"
}
},
"nbformat": 4,
"nbformat_minor": 4
}