{ "cells": [ { "cell_type": "markdown", "metadata": {}, "source": [ "# 数字文件名转换为文本文件名" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 数字文件名转换为文本文件名" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import os,sys\n", "import xlrd\n", "import math\n", "\n", "fi_xls = 'test1.xlsx'\n", "fi_name = {}\n", "fi_path = os.getcwd()+'/pdf'\n", "old = []\n", "new = []\n", "wb = xlrd.open_workbook(fi_xls)\n", "sheet1 = wb.sheet_by_index(0)\n", "for r in range(sheet1.nrows):\n", " col = []\n", " m1 = str(sheet1.cell(r,0).value)\n", " m1 = str(math.floor(eval(m1)))\n", " m2 = str(sheet1.cell(r,1).value)\n", " fi_name[m1] = m2\n", "old = fi_name.keys()\n", "new = fi_name.values()\n", "fl=os.listdir(fi_path)\n", "n = 0\n", "for i in fl:\n", " oldname=fl[n]\n", " name, suffix = os.path.splitext(oldname)\n", " if name in old:\n", " new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " old_name = fi_path+ os.sep + fl[n]\n", " os.rename(old_name,new_name)\n", " n+= 1\n", "print(n)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 目录文件按照文件名排序" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "\n", "import os,sys\n", "\n", "\n", "fi_xls = 'test1.xlsx'\n", "fi_name = {}\n", "#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n", "fi_path = os.getcwd()+'/data'\n", "old = []\n", "new = []\n", "\n", "fl=os.listdir(fi_path)\n", "fl.sort()\n", "n = 0\n", "for i in fl:\n", " oldname=fl[n]\n", " name, suffix = os.path.splitext(oldname)\n", " if name in old:\n", " new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " old_name = fi_path+ os.sep + fl[n]\n", " os.rename(old_name,new_name)\n", " n+= 1\n", "fl" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 将pdf文件转为图片" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", "import tempfile\n", "from pdf2image.exceptions import (\n", " PDFInfoNotInstalledError,\n", " PDFPageCountError,\n", " PDFSyntaxError\n", ")\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "with tempfile.TemporaryDirectory() as path:\n", " images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n", "print(path)\n" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 图像文件夹打包" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import zipfile\n", "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", "import tempfile\n", "import shutil\n", "import time\n", "\n", "from pdf2image.exceptions import (\n", " PDFInfoNotInstalledError,\n", " PDFPageCountError,\n", " PDFSyntaxError\n", ")\n", "def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n", " if os.path.isfile(dirname):\n", " with zipfile.ZipFile(zipfilename, 'w') as z:\n", " z.write(dirname)\n", " else:\n", " with zipfile.ZipFile(zipfilename, 'w') as z:\n", " for root, dirs, files in os.walk(dirname):\n", " for single_file in files:\n", " if single_file != zipfilename:\n", " filepath = os.path.join(root, single_file)\n", " z.write(filepath)\n", "\n", "def addfile(zipfilename, dirname):\n", " if os.path.isfile(dirname):\n", " with zipfile.ZipFile(zipfilename, 'a') as z:\n", " z.write(dirname)\n", " else:\n", " with zipfile.ZipFile(zipfilename, 'a') as z:\n", " for root, dirs, files in os.walk(dirname):\n", " for single_file in files:\n", " if single_file != zipfilename:\n", " filepath = os.path.join(root, single_file)\n", " z.write(filepath)\n", "\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "def make_path(p):\n", " if os.path.exists(p): # 判断文件夹是否存在\n", " shutil.rmtree(p) # 删除文件夹\n", " os.mkdir(p) \n", "pdf_file = '2.pdf'\n", "output_folder='./pic1'\n", "zip_file = 'ribenweiqishihua.zip'\n", "make_path(output_folder)\n", "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n", "with tempfile.TemporaryDirectory() as path:\n", " images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n", "compress_file(zip_file, output_folder) # 执行函数\n", "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 文本文件操作" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 基本读取" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import re\n", "file_name = 'data/2012.txt'\n", "with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n", " for l in fl:\n", " l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n", " if l.split():\n", " print(l)\n", " fl1.write(l)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 读取分隔符分割文件,导入MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"city\"]\n", "m_mongo = {}\n", "m_xx = []\n", "fl_name = 'china-city-list.txt'\n", "n = 0\n", "with open(fl_name,'r') as fl:\n", " for l in fl:\n", " n += 1\n", " if n >6:\n", " m_mongo = {}\n", " m_xx = re.sub('[ ]{1,}', '', l).split('|')\n", " #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n", " m_mongo['name'] = m_xx[3]\n", " m_mongo['code'] = m_xx[1]\n", " m_mongo['sheng'] = m_xx[8]\n", " m_mongo['shi'] = m_xx[10]\n", " m_mongo['jing'] = m_xx[11]\n", " m_mongo['wei'] = m_xx[12]\n", " mycol.insert_one(m_mongo) \n", " #print(m_mongo)\n", "print('ok!')\n", "\n", "\n", "\n", "\n" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "# Twilio使用" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import os\n", "from twilio.rest import Client\n", "\n", "\n", "# Your Account Sid and Auth Token from twilio.com/console\n", "# and set the environment variables. See http://twil.io/secure\n", "account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n", "auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n", "client = Client(account_sid, auth_token)\n", "\n", "message = client.messages \\\n", " .create(\n", " body=\"I'm back.\",\n", " from_='+12056066931',\n", " to='+8613793180751'\n", " )\n", "\n", "print(message.sid)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import time\n", "\n", "localtime = time.localtime(time.time())\n", "#type(localtime)\n", "print (\"本地时间为 :\", localtime)\n", "jyr = '12345'\n", "if time.strftime(\"%w\", time.localtime()) in jyr:\n", " print('ok')\n", "else:\n", " print('今日不是交易日!')\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "\n", "def sendmail(message):\n", " msg = MIMEText(message,'plain','utf-8')\n", " msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n", " msg['From'] = Header('512song@sina.com')\n", " msg['To'] = Header('songyi@yeah.net','utf-8')\n", "\n", " from_addr = '512song@sina.com' #发件邮箱\n", " password = '409fe5d8471da663' #邮箱密码\n", " to_addr = 'songyi@yeah.net' #收件邮箱\n", " smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n", " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit() \n", " \n", " \n", "\n", "pattern = re.compile(r'\\\"(.*)\\\"')\n", "url = 'http://hq.sinajs.cn/list=USDCAD'\n", "strhtml = requests.get(url)\n", "data = strhtml.text\n", "if pattern.findall(data):\n", " for data1 in pattern.findall(data):\n", " data2 = data1.split(',')\n", "#print(data2)\n", "with open('price.txt','r') as fl:\n", " for line in fl:\n", " p_high = line.split(',')[0]\n", " p_low = line.split(',')[1]\n", "m_message = '当前美元加元买入价:{}'.format(data2[1])\n", "while time.strftime(\"%w\", time.localtime()) in '12345':\n", " \n", " print(p_high,p_low)\n", " time.sleep(10)\n", " strhtml = requests.get(url)\n", " data = strhtml.text\n", " if pattern.findall(data):\n", " for data1 in pattern.findall(data):\n", " data2 = data1.split(',')\n", " if float(data2[1]) > float(p_high):\n", " m_message = '当前美元加元买入价:{}'.format(data2[1])\n", " sendmail(m_message)\n", " p_high = str(float(p_high) + 0.04) \n", " if float(data2[1]) > float(p_high):\n", " m_message = '当前美元加元卖出价:{}'.format(data2[2])\n", " p_low = str(float(p_low) - 0.04)\n", " sendmail(m_message)\n", " time.sleep(900)\n", " " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "# AWS应用" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## AWS获取sns信息" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Create an SNS client\n", "sns = boto3.client('sns')\n", "\n", "# Call SNS to list topics\n", "response = sns.list_topics()\n", "\n", "# Get a list of all topic ARNs from the response\n", "topics = [topic['TopicArn'] for topic in response['Topics']]\n", "\n", "# Print out the topic list\n", "print(\"Topic List: %s\" % topics)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## AWS操作DynamoDB" ] }, { "cell_type": "code", "execution_count": 23, "metadata": { "execution": { "iopub.execute_input": "2020-11-17T01:38:23.108908Z", "iopub.status.busy": "2020-11-17T01:38:23.107956Z", "iopub.status.idle": "2020-11-17T01:38:44.538417Z", "shell.execute_reply": "2020-11-17T01:38:44.535587Z", "shell.execute_reply.started": "2020-11-17T01:38:23.108799Z" } }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "0\n" ] } ], "source": [ "import boto3\n", "\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "# Create the DynamoDB table.\n", "table = dynamodb.create_table(\n", " TableName='waihui',\n", " \n", " AttributeDefinitions=[ \n", " {\n", " 'AttributeName': 'code',\n", " 'AttributeType': 'S'\n", " }\n", " \n", " \n", " ],\n", " KeySchema=[\n", " {\n", " 'AttributeName': 'code',\n", " 'KeyType': 'HASH'\n", " }\n", " \n", " ],\n", " ProvisionedThroughput={\n", " 'ReadCapacityUnits': 5,\n", " 'WriteCapacityUnits': 5\n", " }\n", " \n", ")\n", "\n", "# Wait until the table exists.\n", "table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n", "\n", "# Print out some data about the table.\n", "print(table.item_count)" ] }, { "cell_type": "code", "execution_count": 28, "metadata": { "execution": { "iopub.execute_input": "2020-11-17T01:43:20.350295Z", "iopub.status.busy": "2020-11-17T01:43:20.349393Z", "iopub.status.idle": "2020-11-17T01:43:23.386976Z", "shell.execute_reply": "2020-11-17T01:43:23.384304Z", "shell.execute_reply.started": "2020-11-17T01:43:20.350192Z" } }, "outputs": [ { "data": { "text/plain": [ "{'ResponseMetadata': {'RequestId': 'C9B2079636H46D3BCM9K0IJJPVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", " 'HTTPStatusCode': 200,\n", " 'HTTPHeaders': {'server': 'Server',\n", " 'date': 'Tue, 17 Nov 2020 01:43:23 GMT',\n", " 'content-type': 'application/x-amz-json-1.0',\n", " 'content-length': '2',\n", " 'connection': 'keep-alive',\n", " 'x-amzn-requestid': 'C9B2079636H46D3BCM9K0IJJPVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", " 'x-amz-crc32': '2745614147'},\n", " 'RetryAttempts': 0}}" ] }, "execution_count": 28, "metadata": {}, "output_type": "execute_result" } ], "source": [ "import boto3\n", "import decimal\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "table.put_item(\n", " Item={\n", " 'code': 'USDCAD',\n", " 'high': Decimal('1.3200'),\n", " 'low': Decimal('1.3000'),\n", " }\n", ")" ] }, { "cell_type": "code", "execution_count": 31, "metadata": { "execution": { "iopub.execute_input": "2020-11-17T01:58:10.720722Z", "iopub.status.busy": "2020-11-17T01:58:10.719781Z", "iopub.status.idle": "2020-11-17T01:58:13.750170Z", "shell.execute_reply": "2020-11-17T01:58:13.747156Z", "shell.execute_reply.started": "2020-11-17T01:58:10.720618Z" } }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "{'high': Decimal('1.32'), 'low': Decimal('1.298'), 'code': 'USDCAD'}\n" ] } ], "source": [ "import boto3\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "response = table.get_item(\n", " Key={\n", " 'code': 'USDCAD' \n", " }\n", ")\n", "item = response['Item']\n", "print(item)" ] }, { "cell_type": "code", "execution_count": 27, "metadata": { "execution": { "iopub.execute_input": "2020-11-17T01:43:05.153025Z", "iopub.status.busy": "2020-11-17T01:43:05.152079Z", "iopub.status.idle": "2020-11-17T01:43:06.138506Z", "shell.execute_reply": "2020-11-17T01:43:06.135790Z", "shell.execute_reply.started": "2020-11-17T01:43:05.152919Z" } }, "outputs": [ { "data": { "text/plain": [ "{'ResponseMetadata': {'RequestId': 'HJ7KVGIA4B36S7EAIGBNI8OGMVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", " 'HTTPStatusCode': 200,\n", " 'HTTPHeaders': {'server': 'Server',\n", " 'date': 'Tue, 17 Nov 2020 01:43:06 GMT',\n", " 'content-type': 'application/x-amz-json-1.0',\n", " 'content-length': '2',\n", " 'connection': 'keep-alive',\n", " 'x-amzn-requestid': 'HJ7KVGIA4B36S7EAIGBNI8OGMVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", " 'x-amz-crc32': '2745614147'},\n", " 'RetryAttempts': 0}}" ] }, "execution_count": 27, "metadata": {}, "output_type": "execute_result" } ], "source": [ "import boto3\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "table.delete_item(\n", " Key={\n", " 'code': 'USDCAD' \n", " }\n", ")\n" ] }, { "cell_type": "code", "execution_count": 30, "metadata": { "execution": { "iopub.execute_input": "2020-11-17T01:58:00.482537Z", "iopub.status.busy": "2020-11-17T01:58:00.481642Z", "iopub.status.idle": "2020-11-17T01:58:01.617227Z", "shell.execute_reply": "2020-11-17T01:58:01.614695Z", "shell.execute_reply.started": "2020-11-17T01:58:00.482434Z" } }, "outputs": [ { "data": { "text/plain": [ "{'ResponseMetadata': {'RequestId': 'F9DR02KDF2PKEQ7R1MSAUCKVIBVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", " 'HTTPStatusCode': 200,\n", " 'HTTPHeaders': {'server': 'Server',\n", " 'date': 'Tue, 17 Nov 2020 01:58:01 GMT',\n", " 'content-type': 'application/x-amz-json-1.0',\n", " 'content-length': '2',\n", " 'connection': 'keep-alive',\n", " 'x-amzn-requestid': 'F9DR02KDF2PKEQ7R1MSAUCKVIBVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", " 'x-amz-crc32': '2745614147'},\n", " 'RetryAttempts': 0}}" ] }, "execution_count": 30, "metadata": {}, "output_type": "execute_result" } ], "source": [ "import boto3\n", "import decimal\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "table.update_item(\n", " Key={\n", " 'code': 'USDCAD'\n", " },\n", " UpdateExpression='SET low = :val1',\n", " ExpressionAttributeValues={\n", " ':val1': Decimal('1.2980')\n", " }\n", ")\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# MongoDB系统GridFS文件管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 文件上传" ] }, { "cell_type": "code", "execution_count": 17, "metadata": { "execution": { "iopub.execute_input": "2020-11-26T05:30:36.433098Z", "iopub.status.busy": "2020-11-26T05:30:36.432192Z", "iopub.status.idle": "2020-11-26T05:30:37.438663Z", "shell.execute_reply": "2020-11-26T05:30:37.436233Z", "shell.execute_reply.started": "2020-11-26T05:30:36.432996Z" } }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "5fbf3d7c62452a56d7d16630\n", "5fbf3d7d62452a56d7d16635\n", "5fbf3d7d62452a56d7d1663a\n", "5fbf3d7d62452a56d7d1663f\n", "5fbf3d7d62452a56d7d16643\n", "5fbf3d7d62452a56d7d16648\n", "5fbf3d7d62452a56d7d1664d\n" ] } ], "source": [ "import pymongo\n", "from gridfs import GridFS\n", "from bson.objectid import ObjectId\n", "import os\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "\n", "UploadCache = \"uploadcache\"\n", "dbURL = \"mongodb://localhost:27017\"\n", "\n", "#上传文件\n", "def upLoadFile(file_coll,file_name,data_link):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n", " gridfs_col = GridFS(db, collection=file_coll)\n", " file_ = \"0\"\n", " query = {\"filename\":\"\"}\n", " query[\"filename\"] = file_name\n", "\n", " if gridfs_col.exists(query):\n", " print('已经存在该文件')\n", " else:\n", "\n", " with open(file_name, 'rb') as file_r:\n", " file_data = file_r.read()\n", " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", "\n", " print(file_)\n", "\n", "\n", " return file_ \n", "# 按文件名获取文档\n", "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", "\n", " with open(out_name, 'wb') as file_w:\n", " file_w.write(file_data)\n", "\n", "# 按文件_Id获取文档 \n", "def downLoadFilebyID(self,file_coll,_id,out_name):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " O_Id = ObjectId(_id)\n", "\n", " gf = gridfs_col.get(file_id=O_Id)\n", " file_data = gf.read()\n", " with open(out_name, 'wb') as file_w:\n", "\n", " file_w.write(file_data) \n", "\n", "\n", " return gf.filename \n", "m_dir = './data/tmp'\n", "fls=os.listdir(m_dir)\n", "n = 0\n", "for fl in fls:\n", " #oldname=fl[n]\n", " name, suffix = os.path.splitext(fl)\n", " #if name in old:\n", " # new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " # old_name = fi_path+ os.sep + fl[n]\n", " # os.rename(old_name,new_name)\n", " #print(os.path.basename(fl))\n", " #print(fl,suffix[1:])\n", " full_path = m_dir+ '/' + fl\n", " upLoadFile(\"document\",full_path,\"\")\n", "#a = MongoGridFS(\"\")\n", "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n", "#print (ll)" ] }, { "cell_type": "code", "execution_count": 12, "metadata": { "execution": { "iopub.execute_input": "2020-11-26T05:10:44.453509Z", "iopub.status.busy": "2020-11-26T05:10:44.452576Z", "iopub.status.idle": "2020-11-26T05:10:44.517014Z", "shell.execute_reply": "2020-11-26T05:10:44.515170Z", "shell.execute_reply.started": "2020-11-26T05:10:44.453402Z" } }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "./data/tmp/山东大学强基计划招生专业培养方案(2020版)+-+物理学.pdf\n" ] } ], "source": [ "import pymongo\n", "from gridfs import GridFS\n", "from bson.objectid import ObjectId\n", "import os\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "\n", "UploadCache = \"uploadcache\"\n", "dbURL = \"mongodb://localhost:27017\"\n", "\n", "#上传文件\n", "def upLoadFile(file_coll,file_name,data_link):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " filter_condition = {\"filename\": file_name, \"url\": data_link}\n", " gridfs_col = GridFS(db, collection=file_coll)\n", " file_ = \"0\"\n", " query = {\"filename\":\"\"}\n", " query[\"filename\"] = file_name\n", "\n", " if gridfs_col.exists(query):\n", " print('已经存在该文件')\n", " else:\n", "\n", " with open(file_name, 'rb') as file_r:\n", " file_data = file_r.read()\n", " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", "\n", " print(file_)\n", "\n", "\n", " return file_ \n", "# 按文件名获取文档\n", "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", "\n", " with open(out_name, 'wb') as file_w:\n", " file_w.write(file_data)\n", "\n", "# 按文件_Id获取文档 \n", "def downLoadFilebyID(file_coll,_id,out_name):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " O_Id = ObjectId(_id)\n", "\n", " gf = gridfs_col.get(file_id=O_Id)\n", " file_data = gf.read()\n", " with open(out_name, 'wb') as file_w:\n", "\n", " file_w.write(file_data) \n", "\n", "\n", " return gf.filename \n", "ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n", "print (ll)\n", "#a = MongoGridFS(\"\")\n", "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n", "#print (ll)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.8.5" } }, "nbformat": 4, "nbformat_minor": 4 }