Jupyterlab

Jupyterlab
This commit is contained in:
512song@sina.com committed 2021-10-09 15:12:24 +08:00
commit 4e8c1b3254
12 files changed
+6769

No files matched your search

+880
View File
@@ -0,0 +1,880 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 数字文件名转换为文本文件名"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 数字文件名转换为文本文件名"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import openpyxl\n",
"import math\n",
"\n",
"fi_xls = os.getcwd()+'/file/中国石油化工股份有限公司安庆炼化分公司员工在职人员名单.xlsx'\n",
"fi_name = {}\n",
"fi_path = os.getcwd()+'/file/210924'\n",
"old = []\n",
"new = []\n",
"dict1 = {}\n",
"\n",
"wb = openpyxl.load_workbook(fi_xls)\n",
"sheet = wb.active\n",
"depart = []\n",
"for n in range(2,sheet.max_row+1):\n",
" \n",
" m_name = sheet.cell(n,1).value.strip()\n",
" m_depart = sheet.cell(n,5).value\n",
" if m_depart not in depart:\n",
" depart.append(sheet.cell(n,5).value)\n",
" dict1[int(m_name.split('.')[0])] = [sheet.cell(n,2).value.strip(),sheet.cell(n,5).value.strip()]\n",
" #print()\n",
"# 创建部门办公室 \n",
"#m_path = fi_path = os.getcwd()+'/file/210924/new'\n",
"#for pn in depart:\n",
"# if not os.path.exists(m_path + '/' + pn):\n",
"# os.mkdir(m_path + '/' + pn)\n",
"#print(dict1)\n",
"fl=os.listdir(fi_path)\n",
"for fn in fl:\n",
" if os.path.isfile(fi_path + '/' + fn):\n",
" ofn = int(fn.split('.')[0])\n",
" old.append(ofn)\n",
" #print(fn)\n",
"old.sort()\n",
" #print(str(nfn)+'.pdf')\n",
"'''\n",
"for n in old:\n",
" \n",
" o_name = f'{fi_path}/{n}.pdf'\n",
" n_name = f'{fi_path}/new/{dict1[n][1]}/{dict1[n][0]}.pdf'\n",
" if not os.path.exists(n_name):\n",
" shutil.copyfile(o_name,n_name)\n",
" print(n_name)\n",
"#print(old)\n",
"'''\n",
"print(dict1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 目录文件按照文件名排序"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"\n",
"import os,sys\n",
"\n",
"\n",
"fi_xls = 'test1.xlsx'\n",
"fi_name = {}\n",
"#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n",
"fi_path = os.getcwd()+'/data'\n",
"old = []\n",
"new = []\n",
"\n",
"fl=os.listdir(fi_path)\n",
"fl.sort()\n",
"n = 0\n",
"for i in fl:\n",
" oldname=fl[n]\n",
" name, suffix = os.path.splitext(oldname)\n",
" if name in old:\n",
" new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" old_name = fi_path+ os.sep + fl[n]\n",
" os.rename(old_name,new_name)\n",
" n+= 1\n",
"fl"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 将pdf文件转为图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n",
"print(path)\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 图像文件夹打包"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import zipfile\n",
"from pdf2image import convert_from_path, convert_from_bytes\n",
"import os,sys\n",
"import tempfile\n",
"import shutil\n",
"import time\n",
"\n",
"from pdf2image.exceptions import (\n",
" PDFInfoNotInstalledError,\n",
" PDFPageCountError,\n",
" PDFSyntaxError\n",
")\n",
"def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'w') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"def addfile(zipfilename, dirname):\n",
" if os.path.isfile(dirname):\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" z.write(dirname)\n",
" else:\n",
" with zipfile.ZipFile(zipfilename, 'a') as z:\n",
" for root, dirs, files in os.walk(dirname):\n",
" for single_file in files:\n",
" if single_file != zipfilename:\n",
" filepath = os.path.join(root, single_file)\n",
" z.write(filepath)\n",
"\n",
"#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n",
"def make_path(p):\n",
" if os.path.exists(p): # 判断文件夹是否存在\n",
" shutil.rmtree(p) # 删除文件夹\n",
" os.mkdir(p) \n",
"pdf_file = '2.pdf'\n",
"output_folder='./pic1'\n",
"zip_file = 'ribenweiqishihua.zip'\n",
"make_path(output_folder)\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n",
"with tempfile.TemporaryDirectory() as path:\n",
" images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n",
"compress_file(zip_file, output_folder) # 执行函数\n",
"print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"## 文本文件操作"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 基本读取"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import re\n",
"file_name = 'data/2012.txt'\n",
"with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n",
" for l in fl:\n",
" l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n",
" if l.split():\n",
" print(l)\n",
" fl1.write(l)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### 读取分隔符分割文件,导入MongoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"import re\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"city\"]\n",
"m_mongo = {}\n",
"m_xx = []\n",
"fl_name = 'china-city-list.txt'\n",
"n = 0\n",
"with open(fl_name,'r') as fl:\n",
" for l in fl:\n",
" n += 1\n",
" if n >6:\n",
" m_mongo = {}\n",
" m_xx = re.sub('[ ]{1,}', '', l).split('|')\n",
" #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n",
" m_mongo['name'] = m_xx[3]\n",
" m_mongo['code'] = m_xx[1]\n",
" m_mongo['sheng'] = m_xx[8]\n",
" m_mongo['shi'] = m_xx[10]\n",
" m_mongo['jing'] = m_xx[11]\n",
" m_mongo['wei'] = m_xx[12]\n",
" mycol.insert_one(m_mongo) \n",
" #print(m_mongo)\n",
"print('ok!')\n",
"\n",
"\n",
"\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# Twilio使用"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"from twilio.rest import Client\n",
"\n",
"\n",
"# Your Account Sid and Auth Token from twilio.com/console\n",
"# and set the environment variables. See http://twil.io/secure\n",
"account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n",
"auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n",
"client = Client(account_sid, auth_token)\n",
"\n",
"message = client.messages \\\n",
" .create(\n",
" body=\"I'm back.\",\n",
" from_='+12056066931',\n",
" to='+8613793180751'\n",
" )\n",
"\n",
"print(message.sid)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import time\n",
"\n",
"localtime = time.localtime(time.time())\n",
"#type(localtime)\n",
"print (\"本地时间为 :\", localtime)\n",
"jyr = '12345'\n",
"if time.strftime(\"%w\", time.localtime()) in jyr:\n",
" print('ok')\n",
"else:\n",
" print('今日不是交易日!')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from email.mime.text import MIMEText\n",
"from email.header import Header\n",
"import smtplib\n",
"import requests\n",
"import time\n",
"import re\n",
"\n",
"def sendmail(message):\n",
" msg = MIMEText(message,'plain','utf-8')\n",
" msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n",
" msg['From'] = Header('512song@sina.com')\n",
" msg['To'] = Header('songyi@yeah.net','utf-8')\n",
"\n",
" from_addr = '512song@sina.com' #发件邮箱\n",
" password = '409fe5d8471da663' #邮箱密码\n",
" to_addr = 'songyi@yeah.net' #收件邮箱\n",
" smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n",
" server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n",
" server.login(from_addr,password) #登录邮箱\n",
" server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n",
" server.quit() \n",
" \n",
" \n",
"\n",
"pattern = re.compile(r'\\\"(.*)\\\"')\n",
"url = 'http://hq.sinajs.cn/list=USDCAD'\n",
"strhtml = requests.get(url)\n",
"data = strhtml.text\n",
"if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
"#print(data2)\n",
"with open('price.txt','r') as fl:\n",
" for line in fl:\n",
" p_high = line.split(',')[0]\n",
" p_low = line.split(',')[1]\n",
"m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
"while time.strftime(\"%w\", time.localtime()) in '12345':\n",
" \n",
" print(p_high,p_low)\n",
" time.sleep(10)\n",
" strhtml = requests.get(url)\n",
" data = strhtml.text\n",
" if pattern.findall(data):\n",
" for data1 in pattern.findall(data):\n",
" data2 = data1.split(',')\n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元买入价:{}'.format(data2[1])\n",
" sendmail(m_message)\n",
" p_high = str(float(p_high) + 0.04) \n",
" if float(data2[1]) > float(p_high):\n",
" m_message = '当前美元加元卖出价:{}'.format(data2[2])\n",
" p_low = str(float(p_low) - 0.04)\n",
" sendmail(m_message)\n",
" time.sleep(900)\n",
" "
]
},
{
"cell_type": "markdown",
"metadata": {
"toc-hr-collapsed": true,
"toc-nb-collapsed": true
},
"source": [
"# AWS应用"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS获取sns信息"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Topic List: ['arn:aws:sns:us-east-1:915521803346:MyTopic', 'arn:aws:sns:us-east-1:915521803346:dynamodb']\n"
]
}
],
"source": [
"import boto3\n",
"\n",
"# Create an SNS client\n",
"sns = boto3.client('sns')\n",
"\n",
"# Call SNS to list topics\n",
"response = sns.list_topics()\n",
"\n",
"# Get a list of all topic ARNs from the response\n",
"topics = [topic['TopicArn'] for topic in response['Topics']]\n",
"\n",
"# Print out the topic list\n",
"print(\"Topic List: %s\" % topics)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## AWS操作DynamoDB"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"# Create the DynamoDB table.\n",
"table = dynamodb.create_table(\n",
" TableName='waihui',\n",
" \n",
" AttributeDefinitions=[ \n",
" {\n",
" 'AttributeName': 'code',\n",
" 'AttributeType': 'S'\n",
" }\n",
" \n",
" \n",
" ],\n",
" KeySchema=[\n",
" {\n",
" 'AttributeName': 'code',\n",
" 'KeyType': 'HASH'\n",
" }\n",
" \n",
" ],\n",
" ProvisionedThroughput={\n",
" 'ReadCapacityUnits': 5,\n",
" 'WriteCapacityUnits': 5\n",
" }\n",
" \n",
")\n",
"\n",
"# Wait until the table exists.\n",
"table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n",
"\n",
"# Print out some data about the table.\n",
"print(table.item_count)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.put_item(\n",
" Item={\n",
" 'code': 'USDCAD',\n",
" 'high': Decimal('1.3200'),\n",
" 'low': Decimal('1.3000'),\n",
" }\n",
")"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"{'low': Decimal('1.29'), 'code': 'USDCAD', 'high': Decimal('1.32')}\n"
]
}
],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"response = table.get_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n",
"item = response['Item']\n",
"print(item)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import boto3\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"\n",
"table.delete_item(\n",
" Key={\n",
" 'code': 'USDCAD' \n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'ResponseMetadata': {'RequestId': 'LCHMG55BG6A89FF2KBSDM7JRHNVV4KQNSO5AEMVJF66Q9ASUAAJG',\n",
" 'HTTPStatusCode': 200,\n",
" 'HTTPHeaders': {'server': 'Server',\n",
" 'date': 'Thu, 30 Sep 2021 05:04:36 GMT',\n",
" 'content-type': 'application/x-amz-json-1.0',\n",
" 'content-length': '2',\n",
" 'connection': 'keep-alive',\n",
" 'x-amzn-requestid': 'LCHMG55BG6A89FF2KBSDM7JRHNVV4KQNSO5AEMVJF66Q9ASUAAJG',\n",
" 'x-amz-crc32': '2745614147'},\n",
" 'RetryAttempts': 0}}"
]
},
"execution_count": 6,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"import boto3\n",
"import decimal\n",
"# Get the service resource.\n",
"dynamodb = boto3.resource('dynamodb')\n",
"\n",
"table = dynamodb.Table('waihui')\n",
"table.update_item(\n",
" Key={\n",
" 'code': 'USDCAD'\n",
" },\n",
" UpdateExpression='SET low = :val1',\n",
" ExpressionAttributeValues={\n",
" ':val1': decimal.Decimal('1.2900')\n",
" }\n",
")\n"
]
},
{
"cell_type": "code",
"execution_count": 15,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Received and deleted message: {'MessageId': 'ecd656af-72e2-4bfc-9deb-3043dbe1bb80', 'ReceiptHandle': 'AQEBTx6u3PRfCgUy0SfLyOZcsD4fa5ge7AZvDTeQ/P4sOslg+sDMRYDZelJQOvxDK2XsLvy24DvD/n6GxBa2RwN55b/bnN7PxeYvv42d8GgkIy1dOPEBTytVKNIf+Z7t9+jMZMT2kq2ypXSa7jLFPFN00ArUkd0GI7nZ5SIS133s8exP46gRUGjgaS04CzAwPocTpWNt9GZpA114bvlicx2gNMYTIBsF65K2MzcyBoNVjLzmSJ1Pi0OCcQrAYaHvwzfnkriOBJLUX8BU4ksjWGJg0m8wh8hlPVMTLX+qxMneLzLawbf3cM/ricQR1XtooXrhsAuwAXA+VXaq5TzlpFJZaDZu5BvebF0Yj3rst7j54L/O3DwY0WfVmIrgv2yjOqL7', 'MD5OfBody': '9794e060a0eae4992dbf298549c42dc5', 'Body': 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1', 'Attributes': {'SentTimestamp': '1632980620260'}}\n"
]
}
],
"source": [
"import boto3\n",
"\n",
"# Create SQS client\n",
"sqs = boto3.client('sqs')\n",
"\n",
"queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n",
"\n",
"# Receive message from SQS queue\n",
"response = sqs.receive_message(\n",
" QueueUrl=queue_url,\n",
" AttributeNames=[\n",
" 'SentTimestamp'\n",
" ],\n",
" MaxNumberOfMessages=1,\n",
" MessageAttributeNames=[\n",
" 'All'\n",
" ],\n",
" VisibilityTimeout=0,\n",
" WaitTimeSeconds=0\n",
")\n",
"\n",
"message = response['Messages'][0]\n",
"receipt_handle = message['ReceiptHandle']\n",
"\n",
"# Delete received message from queue\n",
"sqs.delete_message(\n",
" QueueUrl=queue_url,\n",
" ReceiptHandle=receipt_handle\n",
")\n",
"print('Received and deleted message: %s' % message)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# MongoDB系统GridFS文件管理"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 文件上传"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(self,file_coll,_id,out_name):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"m_dir = './data/tmp'\n",
"fls=os.listdir(m_dir)\n",
"n = 0\n",
"for fl in fls:\n",
" #oldname=fl[n]\n",
" name, suffix = os.path.splitext(fl)\n",
" #if name in old:\n",
" # new_name = fi_path+ os.sep + fi_name[name]+suffix\n",
" # old_name = fi_path+ os.sep + fl[n]\n",
" # os.rename(old_name,new_name)\n",
" #print(os.path.basename(fl))\n",
" #print(fl,suffix[1:])\n",
" full_path = m_dir+ '/' + fl\n",
" upLoadFile(\"document\",full_path,\"\")\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import pymongo\n",
"from gridfs import GridFS\n",
"from bson.objectid import ObjectId\n",
"import os\n",
"\n",
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"mydb = myclient[\"gaokao\"]\n",
"mycol = mydb[\"college\"]\n",
"\n",
"UploadCache = \"uploadcache\"\n",
"dbURL = \"mongodb://localhost:27017\"\n",
"\n",
"#上传文件\n",
"def upLoadFile(file_coll,file_name,data_link):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" filter_condition = {\"filename\": file_name, \"url\": data_link}\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
" file_ = \"0\"\n",
" query = {\"filename\":\"\"}\n",
" query[\"filename\"] = file_name\n",
"\n",
" if gridfs_col.exists(query):\n",
" print('已经存在该文件')\n",
" else:\n",
"\n",
" with open(file_name, 'rb') as file_r:\n",
" file_data = file_r.read()\n",
" file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n",
"\n",
" print(file_)\n",
"\n",
"\n",
" return file_ \n",
"# 按文件名获取文档\n",
"def downLoadFile(self,file_coll,file_name,out_name,ver):\n",
" client = pymongo.MongoClient(self.dbURL)\n",
"\n",
" db = client[\"store\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n",
"\n",
" with open(out_name, 'wb') as file_w:\n",
" file_w.write(file_data)\n",
"\n",
"# 按文件_Id获取文档 \n",
"def downLoadFilebyID(file_coll,_id,out_name):\n",
" client = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"\n",
" db = client[\"gaokao\"]\n",
"\n",
" gridfs_col = GridFS(db, collection=file_coll)\n",
"\n",
" O_Id = ObjectId(_id)\n",
"\n",
" gf = gridfs_col.get(file_id=O_Id)\n",
" file_data = gf.read()\n",
" with open(out_name, 'wb') as file_w:\n",
"\n",
" file_w.write(file_data) \n",
"\n",
"\n",
" return gf.filename \n",
"ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n",
"print (ll)\n",
"#a = MongoGridFS(\"\")\n",
"#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n",
"#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n",
"#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n",
"#print (ll)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.10"
}
},
"nbformat": 4,
"nbformat_minor": 4
}