{ "cells": [ { "cell_type": "markdown", "metadata": {}, "source": [ "# 数字文件名转换为文本文件名" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 数字文件名转换为文本文件名" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import os,sys,shutil\n", "import openpyxl\n", "import math\n", "\n", "fi_xls = os.getcwd()+'/file/北海石化监测花名册2022 .xlsx'\n", "fi_name = {}\n", "fi_path = os.getcwd()+'/file/221103'\n", "old = []\n", "new = []\n", "dict1 = {}\n", "\n", "wb = openpyxl.load_workbook(fi_xls)\n", "sheet = wb.active\n", "depart = []\n", "for n in range(2,sheet.max_row+1):\n", " if sheet.cell(n,3).value is not None: \n", " m_name = sheet.cell(n,6).value.strip()\n", " m_depart = sheet.cell(n,3).value.strip() \n", " depart.append(m_depart) \n", " dict1[int(sheet.cell(n,4).value)] = [m_name,m_depart]\n", " #print()\n", "# 创建部门办公室 \n", "m_path = os.getcwd()+'/file/221103/new'\n", "for pn in depart:\n", " if not os.path.exists(m_path + '/' + pn):\n", " os.mkdir(m_path + '/' + pn)\n", "#print(dict1)\n", "\n", "\n", "fl=os.listdir(fi_path)\n", "for fn in fl:\n", " if os.path.isfile(fi_path + '/' + fn):\n", " ofn = int(fn.split('.')[0])\n", " old.append(ofn)\n", " #print(fn)\n", "old.sort()\n", " #print(str(nfn)+'.pdf')\n", "\n", "for n in old:\n", " \n", " o_name = f'{fi_path}/{n}.pdf'\n", " n_name = f'{fi_path}/new/{dict1[n][1]}/{str(n).rjust(5,\"0\")}-{dict1[n][0]}.pdf'\n", " if not os.path.exists(n_name):\n", " shutil.copyfile(o_name,n_name)\n", " print(n_name)\n", "#print(old)\n", "\n", "#print(dict1)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 目录文件按照文件名排序" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "\n", "import os,sys\n", "\n", "\n", "fi_xls = 'test1.xlsx'\n", "fi_name = {}\n", "#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n", "fi_path = os.getcwd()+'/data'\n", "old = []\n", "new = []\n", "\n", "fl=os.listdir(fi_path)\n", "fl.sort()\n", "n = 0\n", "for i in fl:\n", " oldname=fl[n]\n", " name, suffix = os.path.splitext(oldname)\n", " if name in old:\n", " new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " old_name = fi_path+ os.sep + fl[n]\n", " os.rename(old_name,new_name)\n", " n+= 1\n", "fl" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 将pdf文件转为图片" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", "import tempfile\n", "from pdf2image.exceptions import (\n", " PDFInfoNotInstalledError,\n", " PDFPageCountError,\n", " PDFSyntaxError\n", ")\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "with tempfile.TemporaryDirectory() as path:\n", " images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n", "print(path)\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 新建二维码图片pdf文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "from fpdf import FPDF\n", "import warnings\n", "\n", "\n", "warnings.filterwarnings(\"ignore\")\n", "pdf = FPDF()\n", "# compression is not yet supported in py3k version\n", "pdf.compress = False\n", "pdf.add_page()\n", "# Unicode is not yet supported in the py3k version; use windows-1252 standard font\n", "pdf.add_font('ali-regular', '', r\"fonts/AlibabaPuHuiTi-2-55-Regular.ttf\", uni=True)\n", "pdf.set_font('ali-regular', '', 14) \n", "#pdf.ln(10)\n", "#pdf.write(5, '欢迎')\n", "pdf.text(20,280,'欢迎来到中国,中国欢迎您!')\n", "pdf.image(\"data/water/平和体质.png\", 50, 250,20)\n", "\n", "pdf.output('py3k.pdf', 'F')" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 图像文件夹打包" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import zipfile\n", "from pdf2image import convert_from_path, convert_from_bytes\n", "import os,sys\n", "import tempfile\n", "import shutil\n", "import time\n", "\n", "from pdf2image.exceptions import (\n", " PDFInfoNotInstalledError,\n", " PDFPageCountError,\n", " PDFSyntaxError\n", ")\n", "def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n", " if os.path.isfile(dirname):\n", " with zipfile.ZipFile(zipfilename, 'w') as z:\n", " z.write(dirname)\n", " else:\n", " with zipfile.ZipFile(zipfilename, 'w') as z:\n", " for root, dirs, files in os.walk(dirname):\n", " for single_file in files:\n", " if single_file != zipfilename:\n", " filepath = os.path.join(root, single_file)\n", " z.write(filepath)\n", "\n", "def addfile(zipfilename, dirname):\n", " if os.path.isfile(dirname):\n", " with zipfile.ZipFile(zipfilename, 'a') as z:\n", " z.write(dirname)\n", " else:\n", " with zipfile.ZipFile(zipfilename, 'a') as z:\n", " for root, dirs, files in os.walk(dirname):\n", " for single_file in files:\n", " if single_file != zipfilename:\n", " filepath = os.path.join(root, single_file)\n", " z.write(filepath)\n", "\n", "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", "def make_path(p):\n", " if os.path.exists(p): # 判断文件夹是否存在\n", " shutil.rmtree(p) # 删除文件夹\n", " os.mkdir(p) \n", "pdf_file = '2.pdf'\n", "output_folder='./pic1'\n", "zip_file = 'ribenweiqishihua.zip'\n", "make_path(output_folder)\n", "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n", "with tempfile.TemporaryDirectory() as path:\n", " images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n", "compress_file(zip_file, output_folder) # 执行函数\n", "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 文本文件操作" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 基本读取" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import re\n", "file_name = 'data/2012.txt'\n", "with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n", " for l in fl:\n", " l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n", " if l.split():\n", " print(l)\n", " fl1.write(l)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 读取csv文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import csv\n", "import re\n", "\n", "filename = 'data/130.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", " for line in fl:\n", " #line = re.sub('[\\r\\n\\f ]{1,}', '', line)\n", " print(line)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 读取分隔符分割文件,导入MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"city\"]\n", "m_mongo = {}\n", "m_xx = []\n", "fl_name = 'china-city-list.txt'\n", "n = 0\n", "with open(fl_name,'r') as fl:\n", " for l in fl:\n", " n += 1\n", " if n >6:\n", " m_mongo = {}\n", " m_xx = re.sub('[ ]{1,}', '', l).split('|')\n", " #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n", " m_mongo['name'] = m_xx[3]\n", " m_mongo['code'] = m_xx[1]\n", " m_mongo['sheng'] = m_xx[8]\n", " m_mongo['shi'] = m_xx[10]\n", " m_mongo['jing'] = m_xx[11]\n", " m_mongo['wei'] = m_xx[12]\n", " mycol.insert_one(m_mongo) \n", " #print(m_mongo)\n", "print('ok!')\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 整理文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import re\n", "file_name = 'file/三国演义.txt'\n", "list1 = []\n", "with open(file_name,'r') as fl:\n", " for l in fl:\n", " if len(l.strip())>0:\n", " list1.append(l.strip()+'\\n')\n", "with open('file/三国演义_new.txt','w') as fl1:\n", " fl1.writelines(list1)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## Excel文件修改" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import openpyxl\n", "import json\n", "\n", "filename = 'data/122.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "dict2 = {}\n", "for k, v in dict1.items():\n", " for item in v:\n", " bh = item['avatar_id']\n", " dict2[bh] = item\n", "print(dict2) \n", " \n", "wb = openpyxl.load_workbook('data/济南炼化职工分组表.xlsx')\n", "sheet = wb.active\n", "person = {}\n", "for n in range(2, sheet.max_row+1):\n", " if sheet.cell(n, 14).value is not None and sheet.cell(n, 14).value in dict2.keys():\n", " code = sheet.cell(n, 14).value\n", " sheet[f'F{n}']=dict2[code]['birth']\n", "wb.save('data/济南炼化职工分组表1.xlsx') \n", "print('ok!') \n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### Excel文件内容生成markdown格式文本" ] }, { "cell_type": "code", "execution_count": 23, "metadata": { "execution": { "iopub.execute_input": "2024-06-25T13:17:07.778238Z", "iopub.status.busy": "2024-06-25T13:17:07.777435Z", "iopub.status.idle": "2024-06-25T13:17:07.878156Z", "shell.execute_reply": "2024-06-25T13:17:07.877616Z", "shell.execute_reply.started": "2024-06-25T13:17:07.778142Z" } }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "## 第一局 \n", "林门入 \n", "道硕 \n", "共百卅二着 \n", "对子 \n", "棋经连珠 \n", "\n", "\n", "## 第二局 \n", "道硕 \n", "林扑入 先 \n", "共百六十着 \n", "棋经连珠 \n", "\n", "\n", "## 第三局 \n", "道硕 \n", "算哲 先 \n", "共百八十八着 \n", "棋经连珠 \n", "\n", "\n", "## 第四局 \n", "道硕 \n", "算哲 \n", "共百六十四着 \n", "棋经连珠 \n", "\n", "\n", "## 第五局 \n", "算智 \n", "因策 \n", "共百六十六着 \n", "棋经连珠 \n", "\n", "\n", "## 第六局 \n", "算智 \n", "哲斋 \n", "共百三十三着 \n", "棋经连珠 \n", "\n", "\n", "## 第七局 \n", "算智 \n", "哲斋 \n", "共百卅九着 \n", "棋经连珠 \n", "\n", "\n", "## 第八局 \n", "道悦 先 \n", "板垣 \n", "共百四十九着 \n", "棋经连珠 \n", "\n", "\n", "## 第九局 \n", "道悦 先 \n", "板垣 \n", "共百五十三着 \n", "棋经连珠 \n", "\n", "\n", "## 第十局 \n", "道悦 \n", "山甲 \n", "共百九十七着 \n", "棋经连珠 \n", "\n", "\n", "## 第十一局 \n", "道悦 \n", "愚硕 先 \n", "共百七十六着 \n", "棋经连珠 \n", "\n", "\n", "## 第十二局 \n", "道悦 \n", "知哲 先 \n", "共百廿九着 \n", "棋经连珠 \n", "\n", "\n", "## 第十三局 \n", "知哲 先 \n", "算哲 \n", "共百九十六着 \n", "棋经连珠 \n", "\n", "\n", "## 第十四局 \n", "道策 先 \n", "愚硕 \n", "共百四十一着 \n", "棋经连珠 \n", "\n", "\n", "## 第十五局 \n", "道策 先 \n", "南里 \n", "共百七十七着 \n", "棋经连珠 \n", "\n", "\n", "## 第十六局 \n", "道策 \n", "朝寻 先 \n", "共百七十八着 \n", "棋经连珠 \n", "\n", "\n", "## 第十七局 \n", "道策 先 \n", "算哲 \n", "共百三十七着 \n", "棋经连珠 \n", "\n", "\n", "## 第十八局 \n", "道策 先 \n", "算哲 \n", "共百五十五着 \n", "棋经连珠 \n", "\n", "\n", "## 第十九局 \n", "俊知 先 \n", "道砂 \n", "共百七十六着 \n", "棋经连珠 \n", "\n", "\n", "## 第二十局 \n", "道砂 先 \n", "俊知 \n", "共三百十五着 \n", "棋经连珠 \n", "\n", "\n", "## 第二十一局 \n", "道的 \n", "道节 先 \n", "共百廿八着 \n", "棋经连珠 \n", "\n", "\n", "## 第二十二局 \n", "道节 先 \n", "八硕 \n", "共百五十六着 \n", "棋经连珠 \n", "\n", "\n", "## 第二十三局 \n", "道智 \n", "可儿 先 \n", "共百零二着 \n", "棋经连珠 \n", "\n", "\n", "## 第二十四局 \n", "道智 \n", "因入 先 \n", "共百七十着 \n", "棋经连珠 \n", "\n", "\n", "## 第二十五局 \n", "井上安节 \n", "丈和 \n", "共二百十八着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第二十六局 \n", "安节 \n", "丈和 \n", "共二百九十六着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第二十七局 \n", "立彻 \n", "丈和 \n", "共一百四十三着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第二十八局 \n", "立彻 \n", "丈和 \n", "共一百六十一着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第二十九局 \n", "立彻 \n", "丈和 \n", "共三百五十四着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十局 \n", "丈和 \n", "奥贯智策 \n", "共二百零四着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十一局 \n", "智策 \n", "丈和 \n", "共九十一着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十二局 \n", "丈和 \n", "智策 \n", "共一百六十二着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十三局 \n", "丈和 \n", "智策 \n", "共一百四十四着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十四局 \n", "松原南冈 \n", "丈和 \n", "共一百四十着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十五局 \n", "丈和 \n", "舟桥元美 \n", "共一百五十七着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十六局 \n", "水谷琢顺 \n", "丈和 \n", "共二百十着 \n", "对局 \n", "东洋棋谱 \n", "\n", "\n", "## 第三十七局 \n", "津波 \n", "元才 \n", "共一百八十着 \n", "对子 \n", "中山弈谱 \n", "\n", "\n", "## 第三十八局 \n", "津波 \n", "元才 \n", "共二百五十五着 \n", "对子 \n", "中山弈谱 \n", "\n", "\n", "## 第三十九局 \n", "津波 \n", "元才 \n", "共一百五十一着 \n", "对子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十局 \n", "津波 \n", "元才 \n", "共一百零五着 \n", "对子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十一局 \n", "宫城 \n", "祝岭 \n", "共二百卅三着 \n", "对子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十二局 \n", "宫城 \n", "祝岭 \n", "共二百五十八着 \n", "对子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十三局 \n", "孙思忠 \n", "向盛保 \n", "共一百九十七着 \n", "受三子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十四局 \n", "孙思忠 \n", "向盛保 \n", "共二百零八着 \n", "受三子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十五局 \n", "孙思忠 \n", "杨光裕 \n", "共一百七十四着 \n", "受三子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十六局 \n", "孙思忠 \n", "杨光裕 \n", "共一百八十一着 \n", "受三子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十七局 \n", "孙思忠 \n", "武邦瑞 \n", "共一百六十五着 \n", "受三子 \n", "中山弈谱 \n", "\n", "\n", "## 第四十八局 \n", "孙思忠 \n", "武邦瑞 \n", "共一百三十二着 \n", "受三子 \n", "中山弈谱 \n", "\n", "\n" ] } ], "source": [ "import openpyxl\n", "\n", "fi_xls = 'data/对局目录.xlsx'\n", "\n", "wb = openpyxl.load_workbook(fi_xls)\n", "sheet = wb['卷23'] \n", "\n", "\n", "for n in range(2,sheet.max_row+1): \n", " print('## 第'+sheet.cell(n,1).value+'局 ')\n", " print(sheet.cell(n,2).value+' ')\n", " print(sheet.cell(n,3).value+' ')\n", " print('共'+sheet.cell(n,4).value+'着 ')\n", " if sheet.cell(n,5).value is not None:\n", " print(sheet.cell(n,5).value+' ')\n", " print(sheet.cell(n,6).value+' ')\n", " print('\\n')" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import openpyxl\n", "\n", "fi_xls = 'data/对局目录.xlsx'\n", "\n", "wb = openpyxl.load_workbook(fi_xls)\n", "sheet = wb['卷22'] \n", "\n", "\n", "for n in range(2,sheet.max_row+1): \n", " print('## 第'+sheet.cell(n,1).value+'局 ')\n", " print('第'+sheet.cell(n,2).value+'局 ')\n", " print('共'+sheet.cell(n,3).value+'着 ')\n", " print(sheet.cell(n,4).value+' ')\n", " print(sheet.cell(n,5).value+' ')\n", " print('\\n')" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## YALM文件操作" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### YALM文件读取转存JSON文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import yaml\n", "import json\n", "\n", "filename = 'data/全维健康中医体质检测问卷.yaml'\n", "with open(filename, 'r', encoding='utf-8') as f:\n", " result = yaml.load(f.read(), Loader=yaml.FullLoader)\n", "i = 1\n", "#print(result)\n", "filename = 'data/全维健康中医体质检测问卷.json'\n", "with open(filename, 'w') as fl:\n", " json.dump(result, fl, ensure_ascii=False)\n", "print('ok')" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 按照年龄分组生成心理问卷" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import json\n", "import os,sys,shutil\n", "import openpyxl\n", "\n", "list1 = [\n", " [1,'完全不正确','有点正确','多数正确','完全正确'],\n", " [2,'从不','极少','偶尔','经常','频繁','非常频繁','每天'],\n", " [3,'不符合','有时符合','常常符合','总是符合'],\n", " [4,'从不','很少','有时','经常','总是'],\n", " [5,'非常不满意','不满意','一般','满意','非常满意'],\n", " [6,'没有或很少时间(少于1天)','一部分时间(1-2天)','相当多时间(3-4天)','绝大部分时间(5-7天)'],\n", " [7,'从来没有这种感觉','很少有这种感觉','有时有这种感觉','经常有这种感觉'],\n", " [8,'完全不同意','不同意','同意','完全同意'], \n", " [10,'晚多了','晚一点','差不多','早一点','早多了']\n", "]\n", "dict1 = {}\n", "for item in list1:\n", " #dict1.setdefault(item[0],[])\n", " list2 =[]\n", " for i in range(1,len(item)):\n", " list2.append(item[i])\n", " dict1[item[0]] = list2\n", "dict_choiced = {}\n", "print(dict1)\n", "wj = {}\n", "wj[\"login\"] = \"phone\"\n", "wj[\"locale\"] = \"zh-cn\"\n", "wj[\"title\"] = '社区心理测评问卷'\n", "list2 =[]\n", "dict2 = {\n", " \"type\": \"radiogroup\",\n", " \"name\": \"Gender\",\n", " \"title\": {\n", " \"zh-cn\": \"您的性别\"\n", " },\n", " \"isRequired\": True,\n", " \"choices\": [\n", " {\n", " \"value\": \"m\",\n", " \"text\": \"男性\"\n", " },\n", " {\n", " \"value\": \"f\",\n", " \"text\": \"女性\"\n", " }\n", " ] \n", "}\n", "list2.append(dict2)\n", "dict2 = {}\n", "dict3 = {}\n", "dict2 = {\n", " \"type\": \"radiogroup\",\n", " \"name\": \"Age\",\n", " \"title\": {\n", " \"zh-cn\": \"您的年龄\"\n", " },\n", " \"isRequired\": True,\n", " \"choices\": [\n", " {\n", " \"value\": \"Y\",\n", " \"text\": \"8-17岁\"\n", " },\n", " {\n", " \"value\": \"M\",\n", " \"text\": \"15-59岁\"\n", " },\n", " {\n", " \"value\": \"O\",\n", " \"text\": \"60岁及以上\"\n", " }\n", " ] \n", "}\n", "list2.append(dict2)\n", "\n", "wb = openpyxl.load_workbook('data/心理问卷.xlsx')\n", "sheet = wb.active\n", "data1 =list(sheet.values)\n", "del data1[0]\n", "for item in data1:\n", " \n", " dict2 = {} \n", " dict2['type'] = item[4] \n", " dict2['name'] = 'q'+str(item[0])+item[3]\n", " dict2['visibleIf'] = '{Age}=\"'+item[3]+'\"'\n", " dict3 = {}\n", " dict3[\"zh-cn\"] = item[1]\n", " dict2['title'] =dict3 \n", " dict2['isRequired'] = True\n", " if item[4] == 'rating':\n", " dict2['rateMin'] = 0\n", " dict2['rateMax'] = 7\n", " dict2['minRateDescription'] = { 'zh-cn':'没有'}\n", " dict2['maxRateDescription'] = { 'zh-cn':'非常同意'}\n", " else:\n", " if int(item[2]) in dict_choiced.keys():\n", " dict2['choicesFromQuestion'] = dict_choiced[int(item[2])]\n", " else:\n", " list3 = []\n", " i = 1\n", " for xm in dict1[int(item[2])]:\n", " dict3 = {}\n", " dict3 = {'value': i,'text' : xm}\n", " list3.append(dict3)\n", " i+=1\n", " dict2['choices'] = list3\n", " dict_choiced[int(item[2])] = dict2['name']\n", " list2.append(dict2)\n", "dict3 = {}\n", "dict3[\"elements\"] = list2\n", "wj[\"pages\"] =[]\n", "\n", "wj[\"pages\"].append(dict3)\n", "#filename = 'data/心理问卷1.json'\n", "#with open(filename, 'w') as fl:\n", "# json.dump(wj, fl, ensure_ascii=False)\n", "#print('ok')\n", "with open('心理问卷.yml', 'w', encoding='utf-8') as f:\n", " yaml.dump(data=wj, stream=f, allow_unicode=True)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import yaml\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# 邮件管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 使用网易邮箱群发邮件" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.mime.multipart import MIMEMultipart\n", "from email.mime.application import MIMEApplication\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "import json\n", "\n", "\n", " \n", "\n", "fl_name = 'data/低碳院报告.json'\n", "\n", "with open(fl_name,'r') as fl:\n", " m_xx = json.load(fl)\n", "for k, v in m_xx.items():\n", " m_bh = str(k).rjust(5,\"0\")\n", " fl_name = f'file/220724/new/{m_bh}-{v[0]}.pdf'\n", " fl = f'{m_bh}-{v[0]}.pdf'\n", " m_rec = v[2]+'@ceic.com'\n", " \n", " from_addr = 'kmingedu@163.com' #发件邮箱\n", " password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", " smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n", " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " msg = MIMEMultipart()\n", " msg['Subject'] = Header(\"低碳清洁能源研究院体质监测报告\",'utf-8')\n", " msg['From'] = Header('北京坤铭体质监测评估中心')\n", " msg['To'] = Header(m_rec)\n", "\n", " from_addr = 'kmingedu@163.com' #发件邮箱\n", " password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", " to_addr = m_rec #收件邮箱\n", " att1 =MIMEApplication(open(fl_name, 'rb').read())\n", " #att1[\"Content-Type\"] = 'application/octet-stream'\n", " # 这里的filename可以任意写,写什么名字,邮件中显示什么名字\n", " att1.add_header('Content-Disposition','attachment',filename=fl)\n", " msg.attach(att1)\n", " try:\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit\n", " time.sleep(5)\n", " print(f'{fl}邮件发送成功!')\n", " except smtplib.SMTPException:\n", " print (\"Error: 无法发送邮件\")\n", " \n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 使用网易邮箱群发测试链接" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.mime.multipart import MIMEMultipart\n", "from email.mime.application import MIMEApplication\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "import json\n", "import jwt\n", "\n", "key = \"Xixi1234\"\n", "encoded = jwt.encode({\"email\": \"eygsoft@gmail.com\"}, key, algorithm='HS256')\n", "#print(encoded)\n", "\n", "\n", " \n", "list1 = [\n", " {'name':'lillianwong','E_mail':'lillianwong73@gmail.com'},\n", " {'name':'杨玉冰','E_mail':'lkyyb@126.com'}]\n", "\n", "\n", "for item in list1:\n", " m_mail = item['E_mail']\n", " m_name = item['name']\n", " encoded = jwt.encode({\"email\": m_mail}, key, algorithm='HS256')\n", " lianjie = f'https://www.forraid.com/survey/token/{encoded}/tcm0'\n", " mail_msg = f\"\"\"\n", "
欢迎参加北京坤铭体质监测评估中心中医体质检测
\n", "\n", "\"\"\"\n", " from_addr = 'kmingedu@163.com' #发件邮箱\n", " password = 'GKWUZXMVYKSSUSSL' #邮箱密码\n", " smtp_server = 'smtp.163.com' #SMTP服务器,以新浪为例\n", " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " msg = MIMEText(mail_msg, 'html', 'utf-8')\n", " msg['Subject'] = Header(\"中医体质检测测试链接\",'utf-8')\n", " msg['From'] = Header('北京坤铭体质监测评估中心')\n", " msg['To'] = Header(m_mail)\n", "\n", " \n", " to_addr = m_mail #收件邮箱\n", " try:\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit\n", " time.sleep(5)\n", " print('邮件发送成功!')\n", " except smtplib.SMTPException:\n", " print (\"Error: 无法发送邮件\")\n", " " ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "try:\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit\n", " time.sleep(5)\n", " print(f'{fl}邮件发送成功!')\n", " except smtplib.SMTPException:\n", " print (\"Error: 无法发送邮件\")\n", " " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "# Twilio使用" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import os\n", "from twilio.rest import Client\n", "\n", "\n", "# Your Account Sid and Auth Token from twilio.com/console\n", "# and set the environment variables. See http://twil.io/secure\n", "account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n", "auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n", "client = Client(account_sid, auth_token)\n", "\n", "message = client.messages \\\n", " .create(\n", " body=\"I'm back.\",\n", " from_='+12056066931',\n", " to='+8613793180751'\n", " )\n", "\n", "print(message.sid)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import time\n", "\n", "localtime = time.localtime(time.time())\n", "#type(localtime)\n", "print (\"本地时间为 :\", localtime)\n", "jyr = '12345'\n", "if time.strftime(\"%w\", time.localtime()) in jyr:\n", " print('ok')\n", "else:\n", " print('今日不是交易日!')\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from email.mime.text import MIMEText\n", "from email.header import Header\n", "import smtplib\n", "import requests\n", "import time\n", "import re\n", "\n", "def sendmail(message):\n", " msg = MIMEText(message,'plain','utf-8')\n", " msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n", " msg['From'] = Header('512song@sina.com')\n", " msg['To'] = Header('songyi@yeah.net','utf-8')\n", "\n", " from_addr = '512song@sina.com' #发件邮箱\n", " password = '409fe5d8471da663' #邮箱密码\n", " to_addr = 'songyi@yeah.net' #收件邮箱\n", " smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n", " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", " server.login(from_addr,password) #登录邮箱\n", " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", " server.quit() \n", " \n", " \n", "\n", "pattern = re.compile(r'\\\"(.*)\\\"')\n", "url = 'http://hq.sinajs.cn/list=USDCAD'\n", "strhtml = requests.get(url)\n", "data = strhtml.text\n", "if pattern.findall(data):\n", " for data1 in pattern.findall(data):\n", " data2 = data1.split(',')\n", "#print(data2)\n", "with open('price.txt','r') as fl:\n", " for line in fl:\n", " p_high = line.split(',')[0]\n", " p_low = line.split(',')[1]\n", "m_message = '当前美元加元买入价:{}'.format(data2[1])\n", "while time.strftime(\"%w\", time.localtime()) in '12345':\n", " \n", " print(p_high,p_low)\n", " time.sleep(10)\n", " strhtml = requests.get(url)\n", " data = strhtml.text\n", " if pattern.findall(data):\n", " for data1 in pattern.findall(data):\n", " data2 = data1.split(',')\n", " if float(data2[1]) > float(p_high):\n", " m_message = '当前美元加元买入价:{}'.format(data2[1])\n", " sendmail(m_message)\n", " p_high = str(float(p_high) + 0.04) \n", " if float(data2[1]) > float(p_high):\n", " m_message = '当前美元加元卖出价:{}'.format(data2[2])\n", " p_low = str(float(p_low) - 0.04)\n", " sendmail(m_message)\n", " time.sleep(900)\n", " " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "# AWS应用" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## AWS获取sns信息" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Create an SNS client\n", "sns = boto3.client('sns')\n", "\n", "# Call SNS to list topics\n", "response = sns.list_topics()\n", "\n", "# Get a list of all topic ARNs from the response\n", "topics = [topic['TopicArn'] for topic in response['Topics']]\n", "\n", "# Print out the topic list\n", "print(\"Topic List: %s\" % topics)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## AWS操作DynamoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "# Create the DynamoDB table.\n", "table = dynamodb.create_table(\n", " TableName='waihui',\n", " \n", " AttributeDefinitions=[ \n", " {\n", " 'AttributeName': 'code',\n", " 'AttributeType': 'S'\n", " }\n", " \n", " \n", " ],\n", " KeySchema=[\n", " {\n", " 'AttributeName': 'code',\n", " 'KeyType': 'HASH'\n", " }\n", " \n", " ],\n", " ProvisionedThroughput={\n", " 'ReadCapacityUnits': 5,\n", " 'WriteCapacityUnits': 5\n", " }\n", " \n", ")\n", "\n", "# Wait until the table exists.\n", "table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n", "\n", "# Print out some data about the table.\n", "print(table.item_count)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "import decimal\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "table.put_item(\n", " Item={\n", " 'code': 'USDCAD',\n", " 'high': Decimal('1.3200'),\n", " 'low': Decimal('1.3000'),\n", " }\n", ")" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "response = table.get_item(\n", " Key={\n", " 'code': 'USDCAD' \n", " }\n", ")\n", "item = response['Item']\n", "print(item)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "\n", "table.delete_item(\n", " Key={\n", " 'code': 'USDCAD' \n", " }\n", ")\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "import decimal\n", "# Get the service resource.\n", "dynamodb = boto3.resource('dynamodb')\n", "\n", "table = dynamodb.Table('waihui')\n", "table.update_item(\n", " Key={\n", " 'code': 'USDCAD'\n", " },\n", " UpdateExpression='SET low = :val1',\n", " ExpressionAttributeValues={\n", " ':val1': decimal.Decimal('1.2900')\n", " }\n", ")\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import boto3\n", "\n", "# Create SQS client\n", "sqs = boto3.client('sqs')\n", "\n", "queue_url = 'https://sqs.us-east-1.amazonaws.com/915521803346/MySqs1'\n", "\n", "# Receive message from SQS queue\n", "response = sqs.receive_message(\n", " QueueUrl=queue_url,\n", " AttributeNames=[\n", " 'SentTimestamp'\n", " ],\n", " MaxNumberOfMessages=1,\n", " MessageAttributeNames=[\n", " 'All'\n", " ],\n", " VisibilityTimeout=0,\n", " WaitTimeSeconds=0\n", ")\n", "\n", "message = response['Messages'][0]\n", "receipt_handle = message['ReceiptHandle']\n", "\n", "# Delete received message from queue\n", "sqs.delete_message(\n", " QueueUrl=queue_url,\n", " ReceiptHandle=receipt_handle\n", ")\n", "print('Received and deleted message: %s' % message)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# MongoDB系统GridFS文件管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 文件上传" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import pymongo\n", "from gridfs import GridFS\n", "from bson.objectid import ObjectId\n", "import os\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "\n", "UploadCache = \"uploadcache\"\n", "dbURL = \"mongodb://localhost:27017\"\n", "\n", "#上传文件\n", "def upLoadFile(file_coll,file_name,data_link):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n", " gridfs_col = GridFS(db, collection=file_coll)\n", " file_ = \"0\"\n", " query = {\"filename\":\"\"}\n", " query[\"filename\"] = file_name\n", "\n", " if gridfs_col.exists(query):\n", " print('已经存在该文件')\n", " else:\n", "\n", " with open(file_name, 'rb') as file_r:\n", " file_data = file_r.read()\n", " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", "\n", " print(file_)\n", "\n", "\n", " return file_ \n", "# 按文件名获取文档\n", "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", "\n", " with open(out_name, 'wb') as file_w:\n", " file_w.write(file_data)\n", "\n", "# 按文件_Id获取文档 \n", "def downLoadFilebyID(self,file_coll,_id,out_name):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " O_Id = ObjectId(_id)\n", "\n", " gf = gridfs_col.get(file_id=O_Id)\n", " file_data = gf.read()\n", " with open(out_name, 'wb') as file_w:\n", "\n", " file_w.write(file_data) \n", "\n", "\n", " return gf.filename \n", "m_dir = './data/tmp'\n", "fls=os.listdir(m_dir)\n", "n = 0\n", "for fl in fls:\n", " #oldname=fl[n]\n", " name, suffix = os.path.splitext(fl)\n", " #if name in old:\n", " # new_name = fi_path+ os.sep + fi_name[name]+suffix\n", " # old_name = fi_path+ os.sep + fl[n]\n", " # os.rename(old_name,new_name)\n", " #print(os.path.basename(fl))\n", " #print(fl,suffix[1:])\n", " full_path = m_dir+ '/' + fl\n", " upLoadFile(\"document\",full_path,\"\")\n", "#a = MongoGridFS(\"\")\n", "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n", "#print (ll)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "from gridfs import GridFS\n", "from bson.objectid import ObjectId\n", "import os\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "\n", "UploadCache = \"uploadcache\"\n", "dbURL = \"mongodb://localhost:27017\"\n", "\n", "#上传文件\n", "def upLoadFile(file_coll,file_name,data_link):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " filter_condition = {\"filename\": file_name, \"url\": data_link}\n", " gridfs_col = GridFS(db, collection=file_coll)\n", " file_ = \"0\"\n", " query = {\"filename\":\"\"}\n", " query[\"filename\"] = file_name\n", "\n", " if gridfs_col.exists(query):\n", " print('已经存在该文件')\n", " else:\n", "\n", " with open(file_name, 'rb') as file_r:\n", " file_data = file_r.read()\n", " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", "\n", " print(file_)\n", "\n", "\n", " return file_ \n", "# 按文件名获取文档\n", "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", " client = pymongo.MongoClient(self.dbURL)\n", "\n", " db = client[\"store\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", "\n", " with open(out_name, 'wb') as file_w:\n", " file_w.write(file_data)\n", "\n", "# 按文件_Id获取文档 \n", "def downLoadFilebyID(file_coll,_id,out_name):\n", " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", "\n", " db = client[\"gaokao\"]\n", "\n", " gridfs_col = GridFS(db, collection=file_coll)\n", "\n", " O_Id = ObjectId(_id)\n", "\n", " gf = gridfs_col.get(file_id=O_Id)\n", " file_data = gf.read()\n", " with open(out_name, 'wb') as file_w:\n", "\n", " file_w.write(file_data) \n", "\n", "\n", " return gf.filename \n", "ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n", "print (ll)\n", "#a = MongoGridFS(\"\")\n", "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n", "#print (ll)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3 (ipykernel)", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.12.3" } }, "nbformat": 4, "nbformat_minor": 4 }