Files
jupyter/文件操作.ipynb
T
2022-08-05 14:19:59 +08:00

28 KiB

数字文件名转换为文本文件名

数字文件名转换为文本文件名

In [ ]:
import os,sys,shutil
import openpyxl
import math

fi_xls = os.getcwd()+'/file/高新区.xlsx'
fi_name = {}
fi_path = os.getcwd()+'/file/220720'
old = []
new = []
dict1 = {}

wb = openpyxl.load_workbook(fi_xls)
sheet = wb.active
depart = []
for n in range(2,sheet.max_row+1):
    if sheet.cell(n,3).value is not None: 
        m_name = sheet.cell(n,6).value.strip()
        m_depart = sheet.cell(n,3).value.strip()           
        depart.append(m_depart)    
        dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]
    #print()
# 创建部门办公室    
m_path = os.getcwd()+'/file/220720/new'
for pn in depart:
    if not os.path.exists(m_path + '/' + pn):
        os.mkdir(m_path + '/' + pn)
#print(dict1)


fl=os.listdir(fi_path)
for fn in fl:
    if os.path.isfile(fi_path + '/' + fn):
        ofn = int(fn.split('.')[0])
        old.append(ofn)
    #print(fn)
old.sort()
    #print(str(nfn)+'.pdf')

for n in old:
    
    o_name = f'{fi_path}/{n}.pdf'
    n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,"0")}-{dict1[n][0]}.pdf'
    if not os.path.exists(n_name):
        shutil.copyfile(o_name,n_name)
        print(n_name)
#print(old)

#print(dict1)

目录文件按照文件名排序

In [ ]:

import os,sys


fi_xls = 'test1.xlsx'
fi_name = {}
#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'
fi_path = os.getcwd()+'/data'
old = []
new = []

fl=os.listdir(fi_path)
fl.sort()
n = 0
for i in fl:
    oldname=fl[n]
    name, suffix = os.path.splitext(oldname)
    if name in old:
        new_name = fi_path+ os.sep + fi_name[name]+suffix
        old_name = fi_path+ os.sep + fl[n]
        os.rename(old_name,new_name)
    n+= 1
fl

将pdf文件转为图片

In [ ]:
from pdf2image import convert_from_path, convert_from_bytes
import os,sys
import tempfile
from pdf2image.exceptions import (
    PDFInfoNotInstalledError,
    PDFPageCountError,
    PDFSyntaxError
)
#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')
with tempfile.TemporaryDirectory() as path:
    images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')
print(path)

图像文件夹打包

In [ ]:
import zipfile
from pdf2image import convert_from_path, convert_from_bytes
import os,sys
import tempfile
import shutil
import time

from pdf2image.exceptions import (
    PDFInfoNotInstalledError,
    PDFPageCountError,
    PDFSyntaxError
)
def compress_file(zipfilename, dirname):      # zipfilename是压缩包名字,dirname是要打包的目录
    if os.path.isfile(dirname):
        with zipfile.ZipFile(zipfilename, 'w') as z:
            z.write(dirname)
    else:
        with zipfile.ZipFile(zipfilename, 'w') as z:
            for root, dirs, files in os.walk(dirname):
                for single_file in files:
                    if single_file != zipfilename:
                        filepath = os.path.join(root, single_file)
                        z.write(filepath)

def addfile(zipfilename, dirname):
    if os.path.isfile(dirname):
        with zipfile.ZipFile(zipfilename, 'a') as z:
            z.write(dirname)
    else:
        with zipfile.ZipFile(zipfilename, 'a') as z:
            for root, dirs, files in os.walk(dirname):
                for single_file in files:
                    if single_file != zipfilename:
                        filepath = os.path.join(root, single_file)
                        z.write(filepath)

#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')
def make_path(p):
    if os.path.exists(p):       # 判断文件夹是否存在
        shutil.rmtree(p)        # 删除文件夹
    os.mkdir(p)    
pdf_file = '2.pdf'
output_folder='./pic1'
zip_file = 'ribenweiqishihua.zip'
make_path(output_folder)
print (time.strftime("%a %b %d %H:%M:%S %Y", time.localtime()))
with tempfile.TemporaryDirectory() as path:
    images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)
compress_file(zip_file, output_folder)      # 执行函数
print (time.strftime("%a %b %d %H:%M:%S %Y", time.localtime()))

文本文件操作

基本读取

In [ ]:
import re
file_name = 'data/2012.txt'
with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:
    for l in fl:
        l = re.sub('[\r\n\f ]{1,}', '', l)
        if l.split():
            print(l)
            fl1.write(l)

读取分隔符分割文件,导入MongoDB

In [ ]:
import pymongo
import re
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["city"]
m_mongo = {}
m_xx = []
fl_name = 'china-city-list.txt'
n = 0
with open(fl_name,'r') as fl:
    for l in fl:
        n += 1
        if n >6:
            m_mongo = {}
            m_xx = re.sub('[ ]{1,}', '', l).split('|')
            #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])
            m_mongo['name'] = m_xx[3]
            m_mongo['code'] = m_xx[1]
            m_mongo['sheng'] = m_xx[8]
            m_mongo['shi'] = m_xx[10]
            m_mongo['jing'] = m_xx[11]
            m_mongo['wei'] = m_xx[12]
            mycol.insert_one(m_mongo) 
            #print(m_mongo)
print('ok!')




邮件管理

使用网易邮箱群发邮件

In [ ]:
from email.mime.text import MIMEText
from email.mime.multipart import MIMEMultipart
from email.mime.application import MIMEApplication
from email.header import Header
import smtplib
import requests
import time
import re
import json


      

fl_name = 'data/低碳院报告.json'

with open(fl_name,'r') as fl:
     m_xx = json.load(fl)
for k, v in m_xx.items():
    m_bh = str(k).rjust(5,"0")
    fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'
    fl = f'{m_bh}-{v[0]}.pdf'
    m_rec = v[2]+'@ceic.com'
    
    from_addr = 'kmingedu@163.com' #发件邮箱
    password = 'GKWUZXMVYKSSUSSL' #邮箱密码
    smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例
    server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口
    msg = MIMEMultipart()
    msg['Subject'] = Header("低碳清洁能源研究院体质监测报告",'utf-8')
    msg['From'] = Header('北京坤铭体质监测评估中心')
    msg['To'] = Header(m_rec)

    from_addr = 'kmingedu@163.com' #发件邮箱
    password = 'GKWUZXMVYKSSUSSL' #邮箱密码
    to_addr = m_rec #收件邮箱
    att1 =MIMEApplication(open(fl_name, 'rb').read())
    #att1["Content-Type"] = 'application/octet-stream'
    # 这里的filename可以任意写,写什么名字,邮件中显示什么名字
    att1.add_header('Content-Disposition','attachment',filename=fl)
    msg.attach(att1)
    try:
        server.login(from_addr,password) #登录邮箱
        server.sendmail(from_addr,to_addr,msg.as_string())  #将msg转化成string发出
        server.quit
        time.sleep(5)
        print(f'{fl}邮件发送成功!')
    except smtplib.SMTPException:
        print ("Error: 无法发送邮件")
    
In [ ]:
from email.mime.text import MIMEText
from email.mime.multipart import MIMEMultipart
from email.mime.application import MIMEApplication
from email.header import Header
import smtplib
import requests
import time
import re
import json



from_addr = 'kmingedu@163.com' #发件邮箱
password = 'GKWUZXMVYKSSUSSL' #邮箱密码
smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例
server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口
      

fl_name = 'data/低碳院报告.json'

with open(fl_name,'r') as fl:
     m_xx = json.load(fl)
for k, v in m_xx.items():
    m_bh = str(k).rjust(5,"0")
    fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'
    fl = f'{m_bh}-{v[0]}.pdf'
    m_rec = v[2]+'@ceic.com'
    msg = MIMEMultipart()
    msg['Subject'] = Header("低碳清洁能源研究院体质监测报告",'utf-8')
    msg['From'] = Header('北京坤铭体质监测评估中心')
    msg['To'] = Header(m_rec)

    from_addr = 'kmingedu@163.com' #发件邮箱
    password = 'GKWUZXMVYKSSUSSL' #邮箱密码
    to_addr = v[2] #收件邮箱
    att1 =MIMEApplication(open(fl_name, 'rb').read())
    #att1["Content-Type"] = 'application/octet-stream'
    # 这里的filename可以任意写,写什么名字,邮件中显示什么名字
    att1.add_header('Content-Disposition','attachment',filename=fl)
    print(m_bh,v[0],m_rec,fl)

Twilio使用

In [ ]:
import os
from twilio.rest import Client


# Your Account Sid and Auth Token from twilio.com/console
# and set the environment variables. See http://twil.io/secure
account_sid = 'AC1aac8c18078bf371992fda0f924860c8'
auth_token = '956199d0f1b724d00ef8bb934fcaefe9'
client = Client(account_sid, auth_token)

message = client.messages \
                .create(
                     body="I'm back.",
                     from_='+12056066931',
                     to='+8613793180751'
                 )

print(message.sid)
In [ ]:
import time

localtime = time.localtime(time.time())
#type(localtime)
print ("本地时间为 :", localtime)
jyr = '12345'
if time.strftime("%w", time.localtime()) in jyr:
    print('ok')
else:
    print('今日不是交易日!')
In [ ]:
from email.mime.text import MIMEText
from email.header import Header
import smtplib
import requests
import time
import re

def sendmail(message):
    msg = MIMEText(message,'plain','utf-8')
    msg['Subject'] = Header("外汇价格已经到达预期价位!",'utf-8')
    msg['From'] = Header('512song@sina.com')
    msg['To'] = Header('songyi@yeah.net','utf-8')

    from_addr = '512song@sina.com' #发件邮箱
    password = '409fe5d8471da663' #邮箱密码
    to_addr = 'songyi@yeah.net' #收件邮箱
    smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例
    server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口
    server.login(from_addr,password) #登录邮箱
    server.sendmail(from_addr,to_addr,msg.as_string())  #将msg转化成string发出
    server.quit()    
    
       

pattern = re.compile(r'\"(.*)\"')
url = 'http://hq.sinajs.cn/list=USDCAD'
strhtml = requests.get(url)
data = strhtml.text
if pattern.findall(data):
    for data1 in pattern.findall(data):
        data2 = data1.split(',')
#print(data2)
with open('price.txt','r') as fl:
    for line in fl:
        p_high = line.split(',')[0]
        p_low = line.split(',')[1]
m_message = '当前美元加元买入价:{}'.format(data2[1])
while time.strftime("%w", time.localtime()) in '12345':
    
    print(p_high,p_low)
    time.sleep(10)
    strhtml = requests.get(url)
    data = strhtml.text
    if pattern.findall(data):
        for data1 in pattern.findall(data):
            data2 = data1.split(',')
        if float(data2[1]) > float(p_high):
            m_message = '当前美元加元买入价:{}'.format(data2[1])
            sendmail(m_message)
            p_high = str(float(p_high) + 0.04) 
        if float(data2[1]) > float(p_high):
            m_message = '当前美元加元卖出价:{}'.format(data2[2])
            p_low = str(float(p_low) - 0.04)
            sendmail(m_message)
    time.sleep(900)
            

AWS应用

AWS获取sns信息

In [ ]:
import boto3

# Create an SNS client
sns = boto3.client('sns')

# Call SNS to list topics
response = sns.list_topics()

# Get a list of all topic ARNs from the response
topics = [topic['TopicArn'] for topic in response['Topics']]

# Print out the topic list
print("Topic List: %s" % topics)

AWS操作DynamoDB

In [ ]:
import boto3

# Get the service resource.
dynamodb = boto3.resource('dynamodb')

# Create the DynamoDB table.
table = dynamodb.create_table(
    TableName='waihui',
    
    AttributeDefinitions=[        
        {
            'AttributeName': 'code',
            'AttributeType': 'S'
        }
        
        
    ],
    KeySchema=[
        {
            'AttributeName': 'code',
            'KeyType': 'HASH'
        }
        
    ],
    ProvisionedThroughput={
        'ReadCapacityUnits': 5,
        'WriteCapacityUnits': 5
    }
    
)

# Wait until the table exists.
table.meta.client.get_waiter('table_exists').wait(TableName='waihui')

# Print out some data about the table.
print(table.item_count)
In [ ]:
import boto3
import decimal
# Get the service resource.
dynamodb = boto3.resource('dynamodb')

table = dynamodb.Table('waihui')

table.put_item(
   Item={
        'code': 'USDCAD',
        'high': Decimal('1.3200'),
        'low': Decimal('1.3000'),
    }
)
In [ ]:
import boto3
# Get the service resource.
dynamodb = boto3.resource('dynamodb')

table = dynamodb.Table('waihui')

response = table.get_item(
   Key={
        'code': 'USDCAD'        
    }
)
item = response['Item']
print(item)
In [ ]:
import boto3
# Get the service resource.
dynamodb = boto3.resource('dynamodb')

table = dynamodb.Table('waihui')

table.delete_item(
   Key={
        'code': 'USDCAD'        
    }
)
In [ ]:
import boto3
import decimal
# Get the service resource.
dynamodb = boto3.resource('dynamodb')

table = dynamodb.Table('waihui')
table.update_item(
    Key={
        'code': 'USDCAD'
    },
    UpdateExpression='SET low = :val1',
    ExpressionAttributeValues={
        ':val1': decimal.Decimal('1.2900')
    }
)
In [ ]:
import boto3

# Create SQS client
sqs = boto3.client('sqs')

queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'

# Receive message from SQS queue
response = sqs.receive_message(
    QueueUrl=queue_url,
    AttributeNames=[
        'SentTimestamp'
    ],
    MaxNumberOfMessages=1,
    MessageAttributeNames=[
        'All'
    ],
    VisibilityTimeout=0,
    WaitTimeSeconds=0
)

message = response['Messages'][0]
receipt_handle = message['ReceiptHandle']

# Delete received message from queue
sqs.delete_message(
    QueueUrl=queue_url,
    ReceiptHandle=receipt_handle
)
print('Received and deleted message: %s' % message)

MongoDB系统GridFS文件管理

文件上传

In [ ]:
import pymongo
from gridfs import GridFS
from bson.objectid import ObjectId
import os

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college"]

UploadCache = "uploadcache"
dbURL = "mongodb://localhost:27017"

#上传文件
def upLoadFile(file_coll,file_name,data_link):
    client = pymongo.MongoClient('mongodb://localhost:27017/')

    db = client["gaokao"]

    filter_condition = {"filename": os.path.basename(file_name), "url": data_link}
    gridfs_col = GridFS(db, collection=file_coll)
    file_ = "0"
    query = {"filename":""}
    query["filename"] = file_name

    if gridfs_col.exists(query):
        print('已经存在该文件')
    else:

        with open(file_name, 'rb') as file_r:
            file_data = file_r.read()
            file_ = gridfs_col.put(data=file_data, **filter_condition)  # 上传到gridfs

            print(file_)


    return file_   
# 按文件名获取文档
def downLoadFile(self,file_coll,file_name,out_name,ver):
    client = pymongo.MongoClient(self.dbURL)

    db = client["store"]

    gridfs_col = GridFS(db, collection=file_coll)

    file_data = gridfs_col.get_version(filename=file_name, version=ver).read()

    with open(out_name, 'wb') as file_w:
        file_w.write(file_data)

# 按文件_Id获取文档       
def downLoadFilebyID(self,file_coll,_id,out_name):
    client = pymongo.MongoClient(self.dbURL)

    db = client["store"]

    gridfs_col = GridFS(db, collection=file_coll)

    O_Id = ObjectId(_id)

    gf = gridfs_col.get(file_id=O_Id)
    file_data = gf.read()
    with open(out_name, 'wb') as file_w:

        file_w.write(file_data)     


    return gf.filename    
m_dir = './data/tmp'
fls=os.listdir(m_dir)
n = 0
for fl in fls:
    #oldname=fl[n]
    name, suffix = os.path.splitext(fl)
    #if name in old:
    #    new_name = fi_path+ os.sep + fi_name[name]+suffix
    #    old_name = fi_path+ os.sep + fl[n]
     #   os.rename(old_name,new_name)
    #print(os.path.basename(fl))
    #print(fl,suffix[1:])
    full_path = m_dir+ '/' + fl
    upLoadFile("document",full_path,"")
#a = MongoGridFS("")
#a.upLoadFile("pdf","MongoGridFS.py","")
#a.downLoadFile("pdf","MongoGridFS.py","out2.p",2)
#ll = a.downLoadFilebyID("pdf","5d70a5b283a3c5104cd39346","out3.p")
#print (ll)
In [ ]:
import pymongo
from gridfs import GridFS
from bson.objectid import ObjectId
import os

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college"]

UploadCache = "uploadcache"
dbURL = "mongodb://localhost:27017"

#上传文件
def upLoadFile(file_coll,file_name,data_link):
    client = pymongo.MongoClient('mongodb://localhost:27017/')

    db = client["gaokao"]

    filter_condition = {"filename": file_name, "url": data_link}
    gridfs_col = GridFS(db, collection=file_coll)
    file_ = "0"
    query = {"filename":""}
    query["filename"] = file_name

    if gridfs_col.exists(query):
        print('已经存在该文件')
    else:

        with open(file_name, 'rb') as file_r:
            file_data = file_r.read()
            file_ = gridfs_col.put(data=file_data, **filter_condition)  # 上传到gridfs

            print(file_)


    return file_   
# 按文件名获取文档
def downLoadFile(self,file_coll,file_name,out_name,ver):
    client = pymongo.MongoClient(self.dbURL)

    db = client["store"]

    gridfs_col = GridFS(db, collection=file_coll)

    file_data = gridfs_col.get_version(filename=file_name, version=ver).read()

    with open(out_name, 'wb') as file_w:
        file_w.write(file_data)

# 按文件_Id获取文档       
def downLoadFilebyID(file_coll,_id,out_name):
    client = pymongo.MongoClient('mongodb://localhost:27017/')

    db = client["gaokao"]

    gridfs_col = GridFS(db, collection=file_coll)

    O_Id = ObjectId(_id)

    gf = gridfs_col.get(file_id=O_Id)
    file_data = gf.read()
    with open(out_name, 'wb') as file_w:

        file_w.write(file_data)     


    return gf.filename    
ll = downLoadFilebyID("pdf","5fbf351b62452a56d7d16603","out3.pdf")
print (ll)
#a = MongoGridFS("")
#a.upLoadFile("pdf","MongoGridFS.py","")
#a.downLoadFile("pdf","MongoGridFS.py","out2.p",2)
#ll = a.downLoadFilebyID("pdf","5d70a5b283a3c5104cd39346","out3.pdf")
#print (ll)
In [ ]: