{ "cells": [ { "cell_type": "markdown", "metadata": {}, "source": [ "# 数字文件名转换为文本文件名" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 数字文件名转换为文本文件名" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import os,sys,shutil\n", "import openpyxl\n", "import math\n", "\n", "fi_xls = os.getcwd()+'/file/高新区.xlsx'\n", "fi_name = {}\n", "fi_path = os.getcwd()+'/file/220720'\n", "old = []\n", "new = []\n", "dict1 = {}\n", "\n", "wb = openpyxl.load_workbook(fi_xls)\n", "sheet = wb.active\n", "depart = []\n", "for n in range(2,sheet.max_row+1):\n", " if sheet.cell(n,3).value is not None: \n", " m_name = sheet.cell(n,6).value.strip()\n", " m_depart = sheet.cell(n,3).value.strip() \n", " depart.append(m_depart) \n", " dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n", " #print()\n", "# 创建部门办公室 \n", "m_path = os.getcwd()+'/file/220720/new'\n", "for pn in depart:\n", " if not os.path.exists(m_path + '/' + pn):\n", " os.mkdir(m_path + '/' + pn)\n", "#print(dict1)\n", "\n", "\n", "fl=os.listdir(fi_path)\n", "for fn in fl:\n", " if os.path.isfile(fi_path + '/' + fn):\n", " ofn = int(fn.split('.')[0])\n", " old.append(ofn)\n", " #print(fn)\n", "old.sort()\n", " #print(str(nfn)+'.pdf')\n", "\n", "for n in old:\n", " \n", " o_name = f'{fi_path}/{n}.pdf'\n", " n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n", " if not os.path.exists(n_name):\n", " shutil.copyfile(o_name,n_name)\n", " print(n_name)\n", "#print(old)\n", "\n", "#print(dict1)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 目录文件按照文件名排序" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "\n", "import os,sys\n", "\n", "\n", "fi_xls = 'test1.xlsx'\n", "fi_name = {}\n", "#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n", "fi_path = os.getcwd()+'/data'\n", "old = []\n", "new = []\n", "\n", "fl=os.listdir(fi_path)\n", "fl.sort()\n", "n = 0\n", "for i in fl:\n", " oldname=fl[n]\n", " name, suffix = os.path.splitext(oldname)\n", " if name in old:\n", " new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " old_name = fi_path+ os.sep + fl[n]\n", " os.rename(old_name,new_name)\n", " n+= 1\n", "fl" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 将pdf文件转为图片" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", "import tempfile\n", "from pdf2image.exceptions import (\n", " PDFInfoNotInstalledError,\n", " PDFPageCountError,\n", " PDFSyntaxError\n", ")\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "with tempfile.TemporaryDirectory() as path:\n", " images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n", "print(path)\n" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 图像文件夹打包" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import zipfile\n", "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", "import tempfile\n", "import shutil\n", "import time\n", "\n", "from pdf2image.exceptions import (\n", " PDFInfoNotInstalledError,\n", " PDFPageCountError,\n", " PDFSyntaxError\n", ")\n", "def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n", " if os.path.isfile(dirname):\n", " with zipfile.ZipFile(zipfilename, 'w') as z:\n", " z.write(dirname)\n", " else:\n", " with zipfile.ZipFile(zipfilename, 'w') as z:\n", " for root, dirs, files in os.walk(dirname):\n", " for single_file in files:\n", " if single_file != zipfilename:\n", " filepath = os.path.join(root, single_file)\n", " z.write(filepath)\n", "\n", "def addfile(zipfilename, dirname):\n", " if os.path.isfile(dirname):\n", " with zipfile.ZipFile(zipfilename, 'a') as z:\n", " z.write(dirname)\n", " else:\n", " with zipfile.ZipFile(zipfilename, 'a') as z:\n", " for root, dirs, files in os.walk(dirname):\n", " for single_file in files:\n", " if single_file != zipfilename:\n", " filepath = os.path.join(root, single_file)\n", " z.write(filepath)\n", "\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "def make_path(p):\n", " if os.path.exists(p): # 判断文件夹是否存在\n", " shutil.rmtree(p) # 删除文件夹\n", " os.mkdir(p) \n", "pdf_file = '2.pdf'\n", "output_folder='./pic1'\n", "zip_file = 'ribenweiqishihua.zip'\n", "make_path(output_folder)\n", "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n", "with tempfile.TemporaryDirectory() as path:\n", " images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n", "compress_file(zip_file, output_folder) # 执行函数\n", "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 文本文件操作" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 基本读取" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import re\n", "file_name = 'data/2012.txt'\n", "with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n", " for l in fl:\n", " l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n", " if l.split():\n", " print(l)\n", " fl1.write(l)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 读取分隔符分割文件,导入MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"city\"]\n", "m_mongo = {}\n", "m_xx = []\n", "fl_name = 'china-city-list.txt'\n", "n = 0\n", "with open(fl_name,'r') as fl:\n", " for l in fl:\n", " n += 1\n", " if n >6:\n", " m_mongo = {}\n", " m_xx = re.sub('[ ]{1,}', '', l).split('|')\n", " #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n", " m_mongo['name'] = m_xx[3]\n", " m_mongo['code'] = m_xx[1]\n", " m_mongo['sheng'] = m_xx[8]\n", " m_mongo['shi'] = m_xx[10]\n", " m_mongo['jing'] = m_xx[11]\n", " m_mongo['wei'] = m_xx[12]\n", " mycol.insert_one(m_mongo) \n", " #print(m_mongo)\n", "print('ok!')\n", "\n", "\n", "\n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# 邮件管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 使用网易邮箱群发邮件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.mime.multipart import MIMEMultipart\n", "from email.mime.application import MIMEApplication\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "import json\n", "\n", "\n", " \n", "\n", "fl_name = 'data/低碳院报告.json'\n", "\n", "with open(fl_name,'r') as fl:\n", " m_xx = json.load(fl)\n", "for k, v in m_xx.items():\n", " m_bh = str(k).rjust(5,\"0\")\n", " fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n", " fl = f'{m_bh}-{v[0]}.pdf'\n", " m_rec = v[2]+'@ceic.com'\n", " \n", " from_addr = 'kmingedu@163.com' #发件邮箱\n", " password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", " smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n", " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " msg = MIMEMultipart()\n", " msg['Subject'] = Header(\"低碳清洁能源研究院体质监测报告\",'utf-8')\n", " msg['From'] = Header('北京坤铭体质监测评估中心')\n", " msg['To'] = Header(m_rec)\n", "\n", " from_addr = 'kmingedu@163.com' #发件邮箱\n", " password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", " to_addr = m_rec #收件邮箱\n", " att1 =MIMEApplication(open(fl_name, 'rb').read())\n", " #att1[\"Content-Type\"] = 'application/octet-stream'\n", " # 这里的filename可以任意写,写什么名字,邮件中显示什么名字\n", " att1.add_header('Content-Disposition','attachment',filename=fl)\n", " msg.attach(att1)\n", " try:\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit\n", " time.sleep(5)\n", " print(f'{fl}邮件发送成功!')\n", " except smtplib.SMTPException:\n", " print (\"Error: 无法发送邮件\")\n", " \n" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.mime.multipart import MIMEMultipart\n", "from email.mime.application import MIMEApplication\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "import json\n", "\n", "\n", "\n", "from_addr = 'kmingedu@163.com' #发件邮箱\n", "password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", "smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n", "server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " \n", "\n", "fl_name = 'data/低碳院报告.json'\n", "\n", "with open(fl_name,'r') as fl:\n", " m_xx = json.load(fl)\n", "for k, v in m_xx.items():\n", " m_bh = str(k).rjust(5,\"0\")\n", " fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n", " fl = f'{m_bh}-{v[0]}.pdf'\n", " m_rec = v[2]+'@ceic.com'\n", " msg = MIMEMultipart()\n", " msg['Subject'] = Header(\"低碳清洁能源研究院体质监测报告\",'utf-8')\n", " msg['From'] = Header('北京坤铭体质监测评估中心')\n", " msg['To'] = Header(m_rec)\n", "\n", " from_addr = 'kmingedu@163.com' #发件邮箱\n", " password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", " to_addr = v[2] #收件邮箱\n", " att1 =MIMEApplication(open(fl_name, 'rb').read())\n", " #att1[\"Content-Type\"] = 'application/octet-stream'\n", " # 这里的filename可以任意写,写什么名字,邮件中显示什么名字\n", " att1.add_header('Content-Disposition','attachment',filename=fl)\n", " print(m_bh,v[0],m_rec,fl)" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "# Twilio使用" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import os\n", "from twilio.rest import Client\n", "\n", "\n", "# Your Account Sid and Auth Token from twilio.com/console\n", "# and set the environment variables. See http://twil.io/secure\n", "account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n", "auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n", "client = Client(account_sid, auth_token)\n", "\n", "message = client.messages \\\n", " .create(\n", " body=\"I'm back.\",\n", " from_='+12056066931',\n", " to='+8613793180751'\n", " )\n", "\n", "print(message.sid)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import time\n", "\n", "localtime = time.localtime(time.time())\n", "#type(localtime)\n", "print (\"本地时间为 :\", localtime)\n", "jyr = '12345'\n", "if time.strftime(\"%w\", time.localtime()) in jyr:\n", " print('ok')\n", "else:\n", " print('今日不是交易日!')\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "\n", "def sendmail(message):\n", " msg = MIMEText(message,'plain','utf-8')\n", " msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n", " msg['From'] = Header('512song@sina.com')\n", " msg['To'] = Header('songyi@yeah.net','utf-8')\n", "\n", " from_addr = '512song@sina.com' #发件邮箱\n", " password = '409fe5d8471da663' #邮箱密码\n", " to_addr = 'songyi@yeah.net' #收件邮箱\n", " smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n", " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit() \n", " \n", " \n", "\n", "pattern = re.compile(r'\\\"(.*)\\\"')\n", "url = 'http://hq.sinajs.cn/list=USDCAD'\n", "strhtml = requests.get(url)\n", "data = strhtml.text\n", "if pattern.findall(data):\n", " for data1 in pattern.findall(data):\n", " data2 = data1.split(',')\n", "#print(data2)\n", "with open('price.txt','r') as fl:\n", " for line in fl:\n", " p_high = line.split(',')[0]\n", " p_low = line.split(',')[1]\n", "m_message = '当前美元加元买入价:{}'.format(data2[1])\n", "while time.strftime(\"%w\", time.localtime()) in '12345':\n", " \n", " print(p_high,p_low)\n", " time.sleep(10)\n", " strhtml = requests.get(url)\n", " data = strhtml.text\n", " if pattern.findall(data):\n", " for data1 in pattern.findall(data):\n", " data2 = data1.split(',')\n", " if float(data2[1]) > float(p_high):\n", " m_message = '当前美元加元买入价:{}'.format(data2[1])\n", " sendmail(m_message)\n", " p_high = str(float(p_high) + 0.04) \n", " if float(data2[1]) > float(p_high):\n", " m_message = '当前美元加元卖出价:{}'.format(data2[2])\n", " p_low = str(float(p_low) - 0.04)\n", " sendmail(m_message)\n", " time.sleep(900)\n", " " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "# AWS应用" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## AWS获取sns信息" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Create an SNS client\n", "sns = boto3.client('sns')\n", "\n", "# Call SNS to list topics\n", "response = sns.list_topics()\n", "\n", "# Get a list of all topic ARNs from the response\n", "topics = [topic['TopicArn'] for topic in response['Topics']]\n", "\n", "# Print out the topic list\n", "print(\"Topic List: %s\" % topics)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## AWS操作DynamoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "# Create the DynamoDB table.\n", "table = dynamodb.create_table(\n", " TableName='waihui',\n", " \n", " AttributeDefinitions=[ \n", " {\n", " 'AttributeName': 'code',\n", " 'AttributeType': 'S'\n", " }\n", " \n", " \n", " ],\n", " KeySchema=[\n", " {\n", " 'AttributeName': 'code',\n", " 'KeyType': 'HASH'\n", " }\n", " \n", " ],\n", " ProvisionedThroughput={\n", " 'ReadCapacityUnits': 5,\n", " 'WriteCapacityUnits': 5\n", " }\n", " \n", ")\n", "\n", "# Wait until the table exists.\n", "table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n", "\n", "# Print out some data about the table.\n", "print(table.item_count)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "import decimal\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "table.put_item(\n", " Item={\n", " 'code': 'USDCAD',\n", " 'high': Decimal('1.3200'),\n", " 'low': Decimal('1.3000'),\n", " }\n", ")" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "response = table.get_item(\n", " Key={\n", " 'code': 'USDCAD' \n", " }\n", ")\n", "item = response['Item']\n", "print(item)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "table.delete_item(\n", " Key={\n", " 'code': 'USDCAD' \n", " }\n", ")\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "import decimal\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "table.update_item(\n", " Key={\n", " 'code': 'USDCAD'\n", " },\n", " UpdateExpression='SET low = :val1',\n", " ExpressionAttributeValues={\n", " ':val1': decimal.Decimal('1.2900')\n", " }\n", ")\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Create SQS client\n", "sqs = boto3.client('sqs')\n", "\n", "queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n", "\n", "# Receive message from SQS queue\n", "response = sqs.receive_message(\n", " QueueUrl=queue_url,\n", " AttributeNames=[\n", " 'SentTimestamp'\n", " ],\n", " MaxNumberOfMessages=1,\n", " MessageAttributeNames=[\n", " 'All'\n", " ],\n", " VisibilityTimeout=0,\n", " WaitTimeSeconds=0\n", ")\n", "\n", "message = response['Messages'][0]\n", "receipt_handle = message['ReceiptHandle']\n", "\n", "# Delete received message from queue\n", "sqs.delete_message(\n", " QueueUrl=queue_url,\n", " ReceiptHandle=receipt_handle\n", ")\n", "print('Received and deleted message: %s' % message)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# MongoDB系统GridFS文件管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 文件上传" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import pymongo\n", "from gridfs import GridFS\n", "from bson.objectid import ObjectId\n", "import os\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "\n", "UploadCache = \"uploadcache\"\n", "dbURL = \"mongodb://localhost:27017\"\n", "\n", "#上传文件\n", "def upLoadFile(file_coll,file_name,data_link):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n", " gridfs_col = GridFS(db, collection=file_coll)\n", " file_ = \"0\"\n", " query = {\"filename\":\"\"}\n", " query[\"filename\"] = file_name\n", "\n", " if gridfs_col.exists(query):\n", " print('已经存在该文件')\n", " else:\n", "\n", " with open(file_name, 'rb') as file_r:\n", " file_data = file_r.read()\n", " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", "\n", " print(file_)\n", "\n", "\n", " return file_ \n", "# 按文件名获取文档\n", "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", "\n", " with open(out_name, 'wb') as file_w:\n", " file_w.write(file_data)\n", "\n", "# 按文件_Id获取文档 \n", "def downLoadFilebyID(self,file_coll,_id,out_name):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " O_Id = ObjectId(_id)\n", "\n", " gf = gridfs_col.get(file_id=O_Id)\n", " file_data = gf.read()\n", " with open(out_name, 'wb') as file_w:\n", "\n", " file_w.write(file_data) \n", "\n", "\n", " return gf.filename \n", "m_dir = './data/tmp'\n", "fls=os.listdir(m_dir)\n", "n = 0\n", "for fl in fls:\n", " #oldname=fl[n]\n", " name, suffix = os.path.splitext(fl)\n", " #if name in old:\n", " # new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " # old_name = fi_path+ os.sep + fl[n]\n", " # os.rename(old_name,new_name)\n", " #print(os.path.basename(fl))\n", " #print(fl,suffix[1:])\n", " full_path = m_dir+ '/' + fl\n", " upLoadFile(\"document\",full_path,\"\")\n", "#a = MongoGridFS(\"\")\n", "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n", "#print (ll)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "from gridfs import GridFS\n", "from bson.objectid import ObjectId\n", "import os\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "\n", "UploadCache = \"uploadcache\"\n", "dbURL = \"mongodb://localhost:27017\"\n", "\n", "#上传文件\n", "def upLoadFile(file_coll,file_name,data_link):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " filter_condition = {\"filename\": file_name, \"url\": data_link}\n", " gridfs_col = GridFS(db, collection=file_coll)\n", " file_ = \"0\"\n", " query = {\"filename\":\"\"}\n", " query[\"filename\"] = file_name\n", "\n", " if gridfs_col.exists(query):\n", " print('已经存在该文件')\n", " else:\n", "\n", " with open(file_name, 'rb') as file_r:\n", " file_data = file_r.read()\n", " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", "\n", " print(file_)\n", "\n", "\n", " return file_ \n", "# 按文件名获取文档\n", "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", "\n", " with open(out_name, 'wb') as file_w:\n", " file_w.write(file_data)\n", "\n", "# 按文件_Id获取文档 \n", "def downLoadFilebyID(file_coll,_id,out_name):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " O_Id = ObjectId(_id)\n", "\n", " gf = gridfs_col.get(file_id=O_Id)\n", " file_data = gf.read()\n", " with open(out_name, 'wb') as file_w:\n", "\n", " file_w.write(file_data) \n", "\n", "\n", " return gf.filename \n", "ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n", "print (ll)\n", "#a = MongoGridFS(\"\")\n", "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n", "#print (ll)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.8.10" } }, "nbformat": 4, "nbformat_minor": 4 }