commit ad0d0a5396f949d78a25575c5bfd07c1e3b6726c Author: 512song <512song@sina.com> Date: Sun Mar 7 19:16:41 2021 +0800 jupyterlab jupyterlab项目 diff --git a/excel文件操作.ipynb b/excel文件操作.ipynb new file mode 100644 index 0000000..d05695b --- /dev/null +++ b/excel文件操作.ipynb @@ -0,0 +1,323 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import csv\n", + "import pymysql\n", + "\n", + "def read_data(filename,re_id):\n", + " detail = {}\n", + " with open(filename) as f:\n", + " reader = csv.reader(f)\n", + " header_row =next(reader)\n", + " for row in reader:\n", + " detail.setdefault(row[0],)\n", + " detail[row[0]].append((re_id,row[0],row[2]))\n", + " return detail\n", + "\n", + "per_id = 1\n", + "item_id = 1\n", + "re_date = '2020-07-07'\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "filename = '户外跑步数据.csv'\n", + "sql = \"select id from sports_record where re_date=%s and item_id =%s and person_id =%s\"\n", + "cursor.execute(sql, (re_date,item_id,per_id))\n", + "result = cursor.fetchone()\n", + "if result:\n", + " print('记录已经存在!')\n", + "else:\n", + " sql = 'insert into sports_record (re_date,item_id,person_id) values(%s,%s,%s)'\n", + " cursor.execute(sql,(re_date,item_id,per_id))\n", + " db.commit()\n", + " re_id = cursor.lastrowid;\n", + " print(re_id)\n", + " detail = read_data(filename,re_id)\n", + " sql = \"insert into sports_detail (rec_id_id,target_id_id,value) values(%s,%s,%s)\"\n", + " try:\n", + " cursor.executemany(sql,detail)\n", + " db.commit()\n", + " print(\"ok!\")\n", + " except:\n", + " # 如果发生错误则回滚\n", + " db.rollback() \n", + "db.close()\n", + "\n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "choice = input('记录已存在,是否覆盖?(y/n)')\n", + "if choice.upper() == \"Y\":\n", + " print('记录已更新')\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import csv\n", + "import pymysql\n", + "def read_data(filename,re_id):\n", + " detail = []\n", + " with open(filename) as f:\n", + " reader = csv.reader(f)\n", + " header_row =next(reader)\n", + " for row in reader:\n", + " detail.append((re_id,row[0],row[2]))\n", + " return detail\n", + " \n", + " \n", + "per_id = 1\n", + "item_id = 1\n", + "re_date = '2020-06-28'\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "filename = '户外跑步数据.csv'\n", + "re_id = 19;\n", + "detail = read_data(filename,re_id)\n", + "print(detail)\n", + " \n", + "db.close()\n" + ] + }, + { + "cell_type": "code", + "execution_count": 32, + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import openpyxl\n", + "import pymysql\n", + "import json\n", + "\n", + "db = pymysql.connect(\"81.68.135.145\",\"colab\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'select code from college';\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "college = []\n", + "for result in results:\n", + " college.append(result[0])\n", + "wb = openpyxl.load_workbook('./data/2017-2019.xlsx')\n", + "#sheet = wb.active\n", + "sheets = wb.sheetnames\n", + "new_col = []\n", + "dict1 = {}\n", + "new_code = []\n", + "for m in sheets:\n", + " sheet = wb[m]\n", + " \n", + " \n", + " for n in range(4,sheet.max_row):\n", + " col_code = sheet.cell(n,1).value\n", + " \n", + " if col_code not in college and col_code not in new_code: \n", + " m_year = []\n", + " dict2 = {}\n", + " #dict2.setdefault('nian',[])\n", + " new_code.append(col_code) \n", + " dict2['name'] = sheet.cell(n,2).value\n", + " dict2['nian'] = []\n", + " #m_year.append(m[0:4])\n", + " #dict2['nian'][] = (m[0:4])\n", + " dict1[col_code] = dict2\n", + " for m_code in dict1.keys():\n", + " if m[0:4] not in dict1[m_code]['nian']:\n", + " dict1[m_code]['nian'].append(m[0:4])\n", + " \n", + "filename = './data/2020年未招生学校.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl,ensure_ascii=False)\n", + " \n", + " \n", + " \n", + "# new_col.append((col_code,sheet.cell(n,2).value,m[0:4]))\n", + " # print(col_code)\n", + "\n", + "print('ok!')" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "### 导入2020年投档录取信息\n", + "import pymysql\n", + "import json\n", + "db = pymysql.connect(\"81.68.135.145\",\"colab\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = \"select * from digao_2020\"\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "college = []\n", + "m_adm = []\n", + "for result in results:\n", + " bm_col = result[1][0:4]\n", + " bm_adm = result[2][0:2]\n", + " name_adm = result[2][2:]\n", + " if bm_col not in college:\n", + " college.append(bm_col) \n", + " \n", + " m_adm.append((bm_col,bm_adm,result[3],result[4],result[5],result[6],result[7],'2020')) \n", + "sql = \"insert into admission (college,speciality,plan,plan_dispense,num_dispense,num_min,rank_min,nian) values(%s,%s,%s,%s,%s,%s,%s,%s)\"\n", + "try:\n", + " cursor.executemany(sql,m_adm)\n", + " db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "\n", + "db.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "### 导入2017-2019年投档录取信息\n", + "import pymysql\n", + "import json\n", + "db = pymysql.connect(\"81.68.135.145\",\"colab\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = \"select * from digao_1719\"\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "college = []\n", + "m_adm = []\n", + "for result in results:\n", + " bm_col = result[1][0:4]\n", + " bm_adm = result[2][0:2]\n", + " name_adm = result[2][2:]\n", + " if bm_col not in college:\n", + " college.append(bm_col) \n", + " \n", + " m_adm.append((bm_col,bm_adm,result[3],result[4],result[5],result[6],result[7],'2020')) \n", + "sql = \"insert into admission (college,speciality,plan,plan_dispense,num_dispense,num_min,rank_min,nian) values(%s,%s,%s,%s,%s,%s,%s,%s)\"\n", + "try:\n", + " cursor.executemany(sql,m_adm)\n", + " db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "\n", + "db.close()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 合并excel文件" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import openpyxl\n", + "import json\n", + "\n", + "for i in range(1,23):\n", + " fl_name = '识别结果'\n", + "wb = openpyxl.load_workbook('./data/2017-2019.xlsx')\n", + "#sheet = wb.active\n", + "sheets = wb.sheetnames\n", + "new_col = []\n", + "dict1 = {}\n", + "new_code = []\n", + "for m in sheets:\n", + " sheet = wb[m]\n", + " \n", + " \n", + " for n in range(4,sheet.max_row):\n", + " col_code = sheet.cell(n,1).value\n", + " \n", + " if col_code not in college and col_code not in new_code: \n", + " m_year = []\n", + " dict2 = {}\n", + " #dict2.setdefault('nian',[])\n", + " new_code.append(col_code) \n", + " dict2['name'] = sheet.cell(n,2).value\n", + " dict2['nian'] = []\n", + " #m_year.append(m[0:4])\n", + " #dict2['nian'][] = (m[0:4])\n", + " dict1[col_code] = dict2\n", + " for m_code in dict1.keys():\n", + " if m[0:4] not in dict1[m_code]['nian']:\n", + " dict1[m_code]['nian'].append(m[0:4])\n", + " \n", + "filename = './data/2020年未招生学校.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl,ensure_ascii=False)\n", + " \n", + " \n", + " \n", + "# new_col.append((col_code,sheet.cell(n,2).value,m[0:4]))\n", + " # print(col_code)\n", + "\n", + "print('ok!')" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + }, + "toc-autonumbering": false, + "toc-showmarkdowntxt": false, + "toc-showtags": false + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/pyecharts图表.ipynb b/pyecharts图表.ipynb new file mode 100644 index 0000000..d22fbf1 --- /dev/null +++ b/pyecharts图表.ipynb @@ -0,0 +1,578 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "scrolled": true + }, + "outputs": [], + "source": [ + "import pymysql\n", + "import pyecharts\n", + "from datetime import datetime\n", + "\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "item_id =1\n", + "per_id = 1\n", + "list_date = []\n", + "dict_tar = {}\n", + "dict_det = {}\n", + "dict_date = {}\n", + "## 获取项目信息\n", + "sql = \"select a.id,a.name,a.unit from sports_target as a,sports_item_target as b where b.item_id =%s and a.id =b.target_id\"\n", + "cursor.execute(sql, (item_id))\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " dict_tar[result[0]] = [result[1],result[2]]\n", + "## 获取运动记录信息\n", + "sql = \"select id,re_date from sports_record where item_id =%s and person_id =%s\"\n", + "cursor.execute(sql, (item_id,per_id))\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " list_date.append(result[0])\n", + " dict_date[result[0]] = result[1].strftime(\"%Y-%m-%d\")\n", + "s_date = '' \n", + "sql = \"select rec_id_id,target_id_id,value from sports_detail where rec_id_id in {}\".format(tuple(list_date))\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for s in list_date:\n", + " dict_det[dict_date[s]] = {}\n", + "for result in results:\n", + " s = result[0]\n", + " dict_det[dict_date[s]][result[1]] = result[2]\n", + "#print(dict_det)\n", + "\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "#柱状、曲线组合\n", + "from pyecharts.globals import CurrentConfig, NotebookType\n", + "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", + "from pyecharts import options as opts\n", + "from pyecharts.charts import Bar,Line\n", + "import pyecharts.options as opts\n", + "x_data = []\n", + "y_data1 = []\n", + "y_data2 = []\n", + "y_data3 = []\n", + "for key,value in dict_det.items():\n", + " x_data.append(key)\n", + " y_data1.append(value[8])\n", + " y_data2.append(value[9])\n", + " y_data3.append(value[2])\n", + "bar = (\n", + " Bar()\n", + " .add_xaxis(x_data)\n", + " .add_yaxis(\"平均心率\", y_data1,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\")\n", + " .add_yaxis(\"最大心率\", y_data2,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\") \n", + " .extend_axis(\n", + " yaxis=opts.AxisOpts(\n", + " name=\"运动时间\",\n", + " type_=\"value\",\n", + " min_=min(y_data3),\n", + " max_=max(y_data3),\n", + " interval=int((max(y_data3)-min(y_data3))/4),\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 分\"),\n", + " )\n", + " )\n", + " .set_global_opts(\n", + " tooltip_opts=opts.TooltipOpts(\n", + " is_show=True, trigger=\"axis\", axis_pointer_type=\"cross\"\n", + " ),\n", + " xaxis_opts=opts.AxisOpts(\n", + " type_=\"category\",\n", + " axispointer_opts=opts.AxisPointerOpts(is_show=True, type_=\"shadow\"),\n", + " ),\n", + " yaxis_opts=opts.AxisOpts(\n", + " name=\"心率\",\n", + " type_=\"value\",\n", + " min_=100,\n", + " max_=180,\n", + " interval=20,\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 次/分钟\"),\n", + " axistick_opts=opts.AxisTickOpts(is_show=True),\n", + " splitline_opts=opts.SplitLineOpts(is_show=True),\n", + " ),\n", + " )\n", + " .set_global_opts(title_opts=opts.TitleOpts(title=\"运动心率\", subtitle=\"户外运动\"),)\n", + ")\n", + "line = (\n", + " Line()\n", + " .add_xaxis(xaxis_data=x_data)\n", + " .add_yaxis(\n", + " series_name=\"运动时间\",\n", + " yaxis_index=1,\n", + " y_axis=y_data3,\n", + " label_opts=opts.LabelOpts(is_show=False),\n", + " )\n", + ")\n", + "\n", + "bar.overlap(line).load_javascript()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "bar.overlap(line).render_notebook()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "#Grid - Grid_vertical多图组合\n", + "from pyecharts.globals import CurrentConfig, NotebookType\n", + "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", + "from pyecharts import options as opts\n", + "from pyecharts.charts import Bar, Grid\n", + "import pyecharts.options as opts\n", + "\n", + "x_data = []\n", + "y_data1 = []\n", + "y_data2 = []\n", + "y_data3 = []\n", + "y_data4 = []\n", + "for key,value in dict_det.items():\n", + " x_data.append(key)\n", + " y_data1.append(value[8])\n", + " y_data2.append(value[9])\n", + " y_data3.append(value[6])\n", + " y_data4.append(value[5])\n", + "bar1 = (\n", + " Bar()\n", + " .add_xaxis(x_data)\n", + " .add_yaxis(\"平均心率\", y_data1,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\")\n", + " .add_yaxis(\"最大心率\", y_data2,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\") \n", + " .set_global_opts(\n", + " tooltip_opts=opts.TooltipOpts(\n", + " is_show=True, trigger=\"axis\", axis_pointer_type=\"cross\"\n", + " ),\n", + " xaxis_opts=opts.AxisOpts(\n", + " type_=\"category\",\n", + " axispointer_opts=opts.AxisPointerOpts(is_show=True, type_=\"shadow\"),\n", + " ),\n", + " yaxis_opts=opts.AxisOpts(\n", + " name=\"心率\",\n", + " type_=\"value\",\n", + " min_=100,\n", + " max_=180,\n", + " interval=20,\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 次/分钟\"),\n", + " axistick_opts=opts.AxisTickOpts(is_show=True),\n", + " splitline_opts=opts.SplitLineOpts(is_show=True),\n", + " ),\n", + " )\n", + " .set_global_opts(title_opts=opts.TitleOpts(title=\"运动心率\", subtitle=\"户外运动\"),)\n", + ")\n", + "bar2 = (\n", + " Bar()\n", + " .add_xaxis(x_data)\n", + " .add_yaxis(\"平均步幅\", y_data3,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\")\n", + " .add_yaxis(\"平均步频\", y_data4,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\") \n", + " .set_global_opts(\n", + " tooltip_opts=opts.TooltipOpts(\n", + " is_show=True, trigger=\"axis\", axis_pointer_type=\"cross\"\n", + " ),\n", + " xaxis_opts=opts.AxisOpts(\n", + " type_=\"category\",\n", + " axispointer_opts=opts.AxisPointerOpts(is_show=True, type_=\"shadow\"),\n", + " ),\n", + " yaxis_opts=opts.AxisOpts(\n", + " name=\"心率\",\n", + " type_=\"value\",\n", + " min_=40,\n", + " max_=180,\n", + " interval=20,\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} \"),\n", + " axistick_opts=opts.AxisTickOpts(is_show=True),\n", + " splitline_opts=opts.SplitLineOpts(is_show=True),\n", + " ),\n", + " )\n", + " .set_global_opts(\n", + " title_opts=opts.TitleOpts(title=\"运动步频\", pos_top=\"48%\"),\n", + " legend_opts=opts.LegendOpts(pos_top=\"48%\"),\n", + " )\n", + ")\n", + "grid = (\n", + " Grid()\n", + " .add(bar1, grid_opts=opts.GridOpts(pos_bottom=\"60%\"))\n", + " .add(bar2, grid_opts=opts.GridOpts(pos_top=\"60%\"))\n", + " \n", + ")\n", + "grid.load_javascript()\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grid.render_notebook()" + ] + }, + { + "cell_type": "code", + "execution_count": 45, + "metadata": {}, + "outputs": [ + { + "data": { + "application/javascript": [ + "new Promise(function(resolve, reject) {\n", + " var script = document.createElement(\"script\");\n", + " script.onload = resolve;\n", + " script.onerror = reject;\n", + " script.src = \"https://assets.pyecharts.org/assets/echarts.min.js\";\n", + " document.head.appendChild(script);\n", + "}).then(() => {\n", + "\n", + "});" + ], + "text/plain": [ + "" + ] + }, + "execution_count": 45, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from pyecharts import options as opts\n", + "from pyecharts.globals import CurrentConfig, NotebookType\n", + "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", + "from pyecharts.charts import Bar\n", + "from pyecharts.globals import ThemeType\n", + "from pyecharts.faker import Faker\n", + "c = (\n", + " Bar(init_opts=opts.InitOpts(theme=ThemeType.VINTAGE))\n", + " # 等价于 Bar(init_opts=opts.InitOpts(theme=ThemeType.WHITE))\n", + " .add_xaxis(Faker.choose())\n", + " .add_yaxis(\"商家A\", Faker.values())\n", + " .add_yaxis(\"商家B\", Faker.values())\n", + " .add_yaxis(\"商家C\", Faker.values())\n", + " .add_yaxis(\"商家D\", Faker.values())\n", + " .set_global_opts(title_opts=opts.TitleOpts(\"Theme-default\"))\n", + " )\n", + "c.load_javascript()" + ] + }, + { + "cell_type": "code", + "execution_count": 46, + "metadata": {}, + "outputs": [ + { + "data": { + "text/html": [ + "\n", + "\n", + "\n", + " \n", + "\n", + "\n", + "
\n", + " \n", + "\n", + "\n" + ], + "text/plain": [ + "" + ] + }, + "execution_count": 46, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "c.render_notebook()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/户外运动.ipynb b/户外运动.ipynb new file mode 100644 index 0000000..4c1f775 --- /dev/null +++ b/户外运动.ipynb @@ -0,0 +1,1354 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": 7, + "metadata": { + "scrolled": true + }, + "outputs": [], + "source": [ + "import pymysql\n", + "import pyecharts\n", + "from datetime import datetime\n", + "\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "item_id =1\n", + "per_id = 1\n", + "list_date = []\n", + "dict_tar = {}\n", + "dict_det = {}\n", + "dict_date = {}\n", + "## 获取项目信息\n", + "sql = \"select a.id,a.name,a.unit from sports_target as a,sports_item_target as b where b.item_id =%s and a.id =b.target_id\"\n", + "cursor.execute(sql, (item_id))\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " dict_tar[result[0]] = [result[1],result[2]]\n", + "## 获取运动记录信息\n", + "sql = \"select id,re_date from sports_record where item_id =%s and person_id =%s order by re_date\"\n", + "cursor.execute(sql, (item_id,per_id))\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " list_date.append(result[0])\n", + " dict_date[result[0]] = result[1].strftime(\"%Y-%m-%d\")\n", + "s_date = '' \n", + "sql = \"select rec_id_id,target_id_id,value from sports_detail where rec_id_id in {}\".format(tuple(list_date))\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for s in list_date:\n", + " dict_det[dict_date[s]] = {}\n", + "for result in results:\n", + " s = result[0]\n", + " dict_det[dict_date[s]][result[1]] = result[2]\n", + "#print(dict_det)\n", + "\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": 8, + "metadata": {}, + "outputs": [ + { + "data": { + "application/javascript": [ + "new Promise(function(resolve, reject) {\n", + " var script = document.createElement(\"script\");\n", + " script.onload = resolve;\n", + " script.onerror = reject;\n", + " script.src = \"https://assets.pyecharts.org/assets/echarts.min.js\";\n", + " document.head.appendChild(script);\n", + "}).then(() => {\n", + "\n", + "});" + ], + "text/plain": [ + "" + ] + }, + "execution_count": 8, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from pyecharts.globals import CurrentConfig, NotebookType\n", + "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", + "from pyecharts import options as opts\n", + "from pyecharts.charts import Bar,Line\n", + "import pyecharts.options as opts\n", + "x_data = []\n", + "y_data1 = []\n", + "y_data2 = []\n", + "y_data3 = []\n", + "y_data4 = []\n", + "for key,value in dict_det.items():\n", + " x_data.append(key)\n", + " y_data1.append(value[8])\n", + " y_data2.append(value[9])\n", + " y_data3.append(value[4])\n", + " y_data4.append(value[5])\n", + "bar = (\n", + " Bar()\n", + " .add_xaxis(x_data)\n", + " .add_yaxis(\"平均心率\", y_data1,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\")\n", + " .add_yaxis(\"最大心率\", y_data2,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\") \n", + " .add_yaxis(\"平均步频\", y_data4,label_opts=opts.LabelOpts(is_show=False),gap=\"0%\")\n", + " .extend_axis(\n", + " yaxis=opts.AxisOpts(\n", + " name=\"平均速度\",\n", + " type_=\"value\",\n", + " min_=min(y_data3),\n", + " max_=max(y_data3),\n", + " interval=(max(y_data3)-min(y_data3))/5,\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} km/h\"),\n", + " )\n", + " )\n", + " .set_global_opts(\n", + " tooltip_opts=opts.TooltipOpts(\n", + " is_show=True, trigger=\"axis\", axis_pointer_type=\"cross\"\n", + " ),\n", + " xaxis_opts=opts.AxisOpts(\n", + " type_=\"category\",\n", + " axispointer_opts=opts.AxisPointerOpts(is_show=True, type_=\"shadow\"),\n", + " ),\n", + " yaxis_opts=opts.AxisOpts(\n", + " name=\"心率\",\n", + " type_=\"value\",\n", + " min_=0,\n", + " max_=180,\n", + " interval=20,\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 次/分钟\"),\n", + " axistick_opts=opts.AxisTickOpts(is_show=True),\n", + " splitline_opts=opts.SplitLineOpts(is_show=True),\n", + " ),\n", + " )\n", + " .set_global_opts(title_opts=opts.TitleOpts(title=\"运动心率\", subtitle=\"户外运动\"),)\n", + ")\n", + "line = (\n", + " Line()\n", + " .add_xaxis(xaxis_data=x_data)\n", + " .add_yaxis(\n", + " series_name=\"平均速度\",\n", + " yaxis_index=1,\n", + " y_axis=y_data3,\n", + " label_opts=opts.LabelOpts(is_show=False),\n", + " )\n", + ")\n", + "\n", + "bar.overlap(line).load_javascript()" + ] + }, + { + "cell_type": "code", + "execution_count": 9, + "metadata": {}, + "outputs": [ + { + "data": { + "text/html": [ + "\n", + "\n", + "\n", + " \n", + "\n", + "\n", + "
\n", + " \n", + "\n", + "\n" + ], + "text/plain": [ + "" + ] + }, + "execution_count": 9, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "bar.overlap(line).render_notebook()" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "metadata": {}, + "outputs": [ + { + "data": { + "application/javascript": [ + "new Promise(function(resolve, reject) {\n", + " var script = document.createElement(\"script\");\n", + " script.onload = resolve;\n", + " script.onerror = reject;\n", + " script.src = \"https://assets.pyecharts.org/assets/echarts.min.js\";\n", + " document.head.appendChild(script);\n", + "}).then(() => {\n", + "\n", + "});" + ], + "text/plain": [ + "" + ] + }, + "execution_count": 5, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from pyecharts.globals import CurrentConfig, NotebookType\n", + "CurrentConfig.NOTEBOOK_TYPE = NotebookType.JUPYTER_LAB\n", + "from pyecharts import options as opts\n", + "from pyecharts.charts import Bar,Line\n", + "import pyecharts.options as opts\n", + "\n", + "colors = [\"#5793f3\", \"#d14a61\", \"#675bba\"]\n", + "x_data = []\n", + "y_data1 = []\n", + "y_data2 = []\n", + "y_data3 = []\n", + "for key,value in dict_det.items():\n", + " x_data.append(key)\n", + " y_data1.append(value[6])\n", + " y_data2.append(value[5])\n", + " y_data3.append(value[2])\n", + "bar = (\n", + " Bar()\n", + " .add_xaxis(x_data)\n", + " .add_yaxis(\n", + " series_name=\"平均步幅\", \n", + " yaxis_data=y_data1,\n", + " yaxis_index=0,\n", + " color=colors[1],\n", + " gap=\"0%\"\n", + " )\n", + " .add_yaxis(\"平均步频\", yaxis_data=y_data2, yaxis_index=1, color=colors[0],gap=\"0%\") \n", + " .extend_axis(\n", + " yaxis=opts.AxisOpts(\n", + " name=\"平均步幅\",\n", + " type_=\"value\",\n", + " min_=0,\n", + " max_=200,\n", + " position=\"right\",\n", + " axisline_opts=opts.AxisLineOpts(\n", + " linestyle_opts=opts.LineStyleOpts(color=colors[1])\n", + " ),\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 厘米\"),\n", + " )\n", + " )\n", + " .extend_axis(\n", + " yaxis=opts.AxisOpts(\n", + " name=\"运动时间\",\n", + " type_=\"value\",\n", + " min_=min(y_data3),\n", + " max_=max(y_data3),\n", + " interval=int((max(y_data3)-min(y_data3))/4),\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 分\"),\n", + " position=\"left\",\n", + " )\n", + " )\n", + " .set_global_opts(\n", + " tooltip_opts=opts.TooltipOpts(\n", + " is_show=True, trigger=\"axis\", axis_pointer_type=\"cross\"\n", + " ),\n", + " xaxis_opts=opts.AxisOpts(\n", + " type_=\"category\",\n", + " axispointer_opts=opts.AxisPointerOpts(is_show=True, type_=\"shadow\"),\n", + " ),\n", + " yaxis_opts=opts.AxisOpts(\n", + " name=\"步频\",\n", + " type_=\"value\",\n", + " min_=0,\n", + " max_=200,\n", + " interval=50,\n", + " position=\"right\",\n", + " offset=40,\n", + " axislabel_opts=opts.LabelOpts(formatter=\"{value} 步/分钟\"),\n", + " axistick_opts=opts.AxisTickOpts(is_show=True),\n", + " splitline_opts=opts.SplitLineOpts(is_show=True),\n", + " ),\n", + " )\n", + " .set_global_opts(title_opts=opts.TitleOpts(title=\"运动步幅及步频\", subtitle=\"户外运动\"),)\n", + ")\n", + "line = (\n", + " Line()\n", + " .add_xaxis(xaxis_data=x_data)\n", + " .add_yaxis(\n", + " series_name=\"运动时间\",\n", + " yaxis_index=2,\n", + " y_axis=y_data3,\n", + " label_opts=opts.LabelOpts(is_show=False),\n", + " )\n", + ")\n", + "\n", + "bar.overlap(line).load_javascript()" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "metadata": {}, + "outputs": [ + { + "data": { + "text/html": [ + "\n", + "\n", + "\n", + " \n", + "\n", + "\n", + "
\n", + " \n", + "\n", + "\n" + ], + "text/plain": [ + "" + ] + }, + "execution_count": 6, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "bar.overlap(line).render_notebook()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.2" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/数据库操作.ipynb b/数据库操作.ipynb new file mode 100644 index 0000000..51a7481 --- /dev/null +++ b/数据库操作.ipynb @@ -0,0 +1,130 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "cursor.execute(\"SELECT VERSION()\")\n", + "data = cursor.fetchone()\n", + "print (\"Database version : %s \" % data)\n", + "db.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "sql = \"select id,name from sports_person where name= %s\"\n", + "cursor.execute(sql, ('张联红',))\n", + "result = cursor.fetchone()\n", + "print(result)\n", + "db.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "sql = \"select id,name from sports_person\"\n", + "cursor.execute(sql)\n", + "result = cursor.fetchone()\n", + "print(\"人员信息:\")\n", + "print(result)\n", + "sql = \"select id,name from sports_item\"\n", + "cursor.execute(sql)\n", + "result = cursor.fetchone()\n", + "print(\"\\n运动项目信息:\")\n", + "print(result)\n", + "db.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "sql = \"select a.target_id,b.name from sports_item_target as a,sports_target as b where a.target_id=b.id\"\n", + "cursor.execute(sql)\n", + "re_ta = {}\n", + "print(\"\\n运动项目信息:\")\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + "# print (\"%d--%s\" %(result[0],result[1]))\n", + " re_ta[result[0]] = result[1]\n", + " print(\"re_ta[%d]= #%s\" %(result[0],result[1]))\n", + "print(re_ta)\n", + "db.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "db = pymysql.connect(\"localhost\",\"songyi\",\"yylzs\",\"mydata\" )\n", + "cursor = db.cursor()\n", + "person_id = \n", + "record_date = ''\n", + "item_id =\n", + "re_ta = {}\n", + "re_ta[1]= #运动距离\n", + "re_ta[2]= #运动时间\n", + "re_ta[3]= #平均配速\n", + "re_ta[4]= #平均速度\n", + "re_ta[5]= #平均步频\n", + "re_ta[6]= #平均步幅\n", + "re_ta[7]= #步数\n", + "re_ta[8]= #平均心率\n", + "re_ta[9]= #最大心率\n", + "re_ta[10]= #无氧耐力\n", + "re_ta[11]= #有氧耐力\n", + "re_ta[12]= #最大步频\n", + "re_ta[13]= #最快配速\n", + "sql = \"select * from sports_person\"\n", + "db.close()" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/文件操作.ipynb b/文件操作.ipynb new file mode 100644 index 0000000..aed42b1 --- /dev/null +++ b/文件操作.ipynb @@ -0,0 +1,933 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# 数字文件名转换为文本文件名" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 数字文件名转换为文本文件名" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import os,sys\n", + "import xlrd\n", + "import math\n", + "\n", + "fi_xls = 'test1.xlsx'\n", + "fi_name = {}\n", + "fi_path = os.getcwd()+'/pdf'\n", + "old = []\n", + "new = []\n", + "wb = xlrd.open_workbook(fi_xls)\n", + "sheet1 = wb.sheet_by_index(0)\n", + "for r in range(sheet1.nrows):\n", + " col = []\n", + " m1 = str(sheet1.cell(r,0).value)\n", + " m1 = str(math.floor(eval(m1)))\n", + " m2 = str(sheet1.cell(r,1).value)\n", + " fi_name[m1] = m2\n", + "old = fi_name.keys()\n", + "new = fi_name.values()\n", + "fl=os.listdir(fi_path)\n", + "n = 0\n", + "for i in fl:\n", + " oldname=fl[n]\n", + " name, suffix = os.path.splitext(oldname)\n", + " if name in old:\n", + " new_name = fi_path+ os.sep + fi_name[name]+suffix\n", + " old_name = fi_path+ os.sep + fl[n]\n", + " os.rename(old_name,new_name)\n", + " n+= 1\n", + "print(n)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 目录文件按照文件名排序" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "\n", + "import os,sys\n", + "\n", + "\n", + "fi_xls = 'test1.xlsx'\n", + "fi_name = {}\n", + "#fi_path = 'drive/My Drive/Colab Notebooks'+'/data'\n", + "fi_path = os.getcwd()+'/data'\n", + "old = []\n", + "new = []\n", + "\n", + "fl=os.listdir(fi_path)\n", + "fl.sort()\n", + "n = 0\n", + "for i in fl:\n", + " oldname=fl[n]\n", + " name, suffix = os.path.splitext(oldname)\n", + " if name in old:\n", + " new_name = fi_path+ os.sep + fi_name[name]+suffix\n", + " old_name = fi_path+ os.sep + fl[n]\n", + " os.rename(old_name,new_name)\n", + " n+= 1\n", + "fl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 将pdf文件转为图片" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from pdf2image import convert_from_path, convert_from_bytes\n", + "import os,sys\n", + "import tempfile\n", + "from pdf2image.exceptions import (\n", + " PDFInfoNotInstalledError,\n", + " PDFPageCountError,\n", + " PDFSyntaxError\n", + ")\n", + "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", + "with tempfile.TemporaryDirectory() as path:\n", + " images_from_path = convert_from_path('./data/普通高等学校本科专业目录.pdf', dpi=300,fmt='jpg', output_folder='./data/pic')\n", + "print(path)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 图像文件夹打包" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import zipfile\n", + "from pdf2image import convert_from_path, convert_from_bytes\n", + "import os,sys\n", + "import tempfile\n", + "import shutil\n", + "import time\n", + "\n", + "from pdf2image.exceptions import (\n", + " PDFInfoNotInstalledError,\n", + " PDFPageCountError,\n", + " PDFSyntaxError\n", + ")\n", + "def compress_file(zipfilename, dirname): # zipfilename是压缩包名字,dirname是要打包的目录\n", + " if os.path.isfile(dirname):\n", + " with zipfile.ZipFile(zipfilename, 'w') as z:\n", + " z.write(dirname)\n", + " else:\n", + " with zipfile.ZipFile(zipfilename, 'w') as z:\n", + " for root, dirs, files in os.walk(dirname):\n", + " for single_file in files:\n", + " if single_file != zipfilename:\n", + " filepath = os.path.join(root, single_file)\n", + " z.write(filepath)\n", + "\n", + "def addfile(zipfilename, dirname):\n", + " if os.path.isfile(dirname):\n", + " with zipfile.ZipFile(zipfilename, 'a') as z:\n", + " z.write(dirname)\n", + " else:\n", + " with zipfile.ZipFile(zipfilename, 'a') as z:\n", + " for root, dirs, files in os.walk(dirname):\n", + " for single_file in files:\n", + " if single_file != zipfilename:\n", + " filepath = os.path.join(root, single_file)\n", + " z.write(filepath)\n", + "\n", + "#images = convert_from_path('1.pdf',dpi=200,fmt='jpg', output_folder='./sample_data')\n", + "def make_path(p):\n", + " if os.path.exists(p): # 判断文件夹是否存在\n", + " shutil.rmtree(p) # 删除文件夹\n", + " os.mkdir(p) \n", + "pdf_file = '2.pdf'\n", + "output_folder='./pic1'\n", + "zip_file = 'ribenweiqishihua.zip'\n", + "make_path(output_folder)\n", + "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))\n", + "with tempfile.TemporaryDirectory() as path:\n", + " images_from_path = convert_from_path(pdf_file, dpi=300,fmt='jpg', output_folder=output_folder)\n", + "compress_file(zip_file, output_folder) # 执行函数\n", + "print (time.strftime(\"%a %b %d %H:%M:%S %Y\", time.localtime()))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 文本文件操作" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 基本读取" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import re\n", + "file_name = 'data/2012.txt'\n", + "with open(file_name,'r') as fl,open('new_2012_1.txt','w') as fl1:\n", + " for l in fl:\n", + " l = re.sub('[\\r\\n\\f ]{1,}', '', l)\n", + " if l.split():\n", + " print(l)\n", + " fl1.write(l)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 读取分隔符分割文件,导入MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymongo\n", + "import re\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"city\"]\n", + "m_mongo = {}\n", + "m_xx = []\n", + "fl_name = 'china-city-list.txt'\n", + "n = 0\n", + "with open(fl_name,'r') as fl:\n", + " for l in fl:\n", + " n += 1\n", + " if n >6:\n", + " m_mongo = {}\n", + " m_xx = re.sub('[ ]{1,}', '', l).split('|')\n", + " #print(m_xx[1],m_xx[3],m_xx[8],m_xx[10])\n", + " m_mongo['name'] = m_xx[3]\n", + " m_mongo['code'] = m_xx[1]\n", + " m_mongo['sheng'] = m_xx[8]\n", + " m_mongo['shi'] = m_xx[10]\n", + " m_mongo['jing'] = m_xx[11]\n", + " m_mongo['wei'] = m_xx[12]\n", + " mycol.insert_one(m_mongo) \n", + " #print(m_mongo)\n", + "print('ok!')\n", + "\n", + "\n", + "\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "# Twilio使用" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import os\n", + "from twilio.rest import Client\n", + "\n", + "\n", + "# Your Account Sid and Auth Token from twilio.com/console\n", + "# and set the environment variables. See http://twil.io/secure\n", + "account_sid = 'AC1aac8c18078bf371992fda0f924860c8'\n", + "auth_token = '956199d0f1b724d00ef8bb934fcaefe9'\n", + "client = Client(account_sid, auth_token)\n", + "\n", + "message = client.messages \\\n", + " .create(\n", + " body=\"I'm back.\",\n", + " from_='+12056066931',\n", + " to='+8613793180751'\n", + " )\n", + "\n", + "print(message.sid)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import time\n", + "\n", + "localtime = time.localtime(time.time())\n", + "#type(localtime)\n", + "print (\"本地时间为 :\", localtime)\n", + "jyr = '12345'\n", + "if time.strftime(\"%w\", time.localtime()) in jyr:\n", + " print('ok')\n", + "else:\n", + " print('今日不是交易日!')\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from email.mime.text import MIMEText\n", + "from email.header import Header\n", + "import smtplib\n", + "import requests\n", + "import time\n", + "import re\n", + "\n", + "def sendmail(message):\n", + " msg = MIMEText(message,'plain','utf-8')\n", + " msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n", + " msg['From'] = Header('512song@sina.com')\n", + " msg['To'] = Header('songyi@yeah.net','utf-8')\n", + "\n", + " from_addr = '512song@sina.com' #发件邮箱\n", + " password = '409fe5d8471da663' #邮箱密码\n", + " to_addr = 'songyi@yeah.net' #收件邮箱\n", + " smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n", + " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", + " server.login(from_addr,password) #登录邮箱\n", + " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", + " server.quit() \n", + " \n", + " \n", + "\n", + "pattern = re.compile(r'\\\"(.*)\\\"')\n", + "url = 'http://hq.sinajs.cn/list=USDCAD'\n", + "strhtml = requests.get(url)\n", + "data = strhtml.text\n", + "if pattern.findall(data):\n", + " for data1 in pattern.findall(data):\n", + " data2 = data1.split(',')\n", + "#print(data2)\n", + "with open('price.txt','r') as fl:\n", + " for line in fl:\n", + " p_high = line.split(',')[0]\n", + " p_low = line.split(',')[1]\n", + "m_message = '当前美元加元买入价:{}'.format(data2[1])\n", + "while time.strftime(\"%w\", time.localtime()) in '12345':\n", + " \n", + " print(p_high,p_low)\n", + " time.sleep(10)\n", + " strhtml = requests.get(url)\n", + " data = strhtml.text\n", + " if pattern.findall(data):\n", + " for data1 in pattern.findall(data):\n", + " data2 = data1.split(',')\n", + " if float(data2[1]) > float(p_high):\n", + " m_message = '当前美元加元买入价:{}'.format(data2[1])\n", + " sendmail(m_message)\n", + " p_high = str(float(p_high) + 0.04) \n", + " if float(data2[1]) > float(p_high):\n", + " m_message = '当前美元加元卖出价:{}'.format(data2[2])\n", + " p_low = str(float(p_low) - 0.04)\n", + " sendmail(m_message)\n", + " time.sleep(900)\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "# AWS应用" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## AWS获取sns信息" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import boto3\n", + "\n", + "# Create an SNS client\n", + "sns = boto3.client('sns')\n", + "\n", + "# Call SNS to list topics\n", + "response = sns.list_topics()\n", + "\n", + "# Get a list of all topic ARNs from the response\n", + "topics = [topic['TopicArn'] for topic in response['Topics']]\n", + "\n", + "# Print out the topic list\n", + "print(\"Topic List: %s\" % topics)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## AWS操作DynamoDB" + ] + }, + { + "cell_type": "code", + "execution_count": 23, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-17T01:38:23.108908Z", + "iopub.status.busy": "2020-11-17T01:38:23.107956Z", + "iopub.status.idle": "2020-11-17T01:38:44.538417Z", + "shell.execute_reply": "2020-11-17T01:38:44.535587Z", + "shell.execute_reply.started": "2020-11-17T01:38:23.108799Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "0\n" + ] + } + ], + "source": [ + "import boto3\n", + "\n", + "# Get the service resource.\n", + "dynamodb = boto3.resource('dynamodb')\n", + "\n", + "# Create the DynamoDB table.\n", + "table = dynamodb.create_table(\n", + " TableName='waihui',\n", + " \n", + " AttributeDefinitions=[ \n", + " {\n", + " 'AttributeName': 'code',\n", + " 'AttributeType': 'S'\n", + " }\n", + " \n", + " \n", + " ],\n", + " KeySchema=[\n", + " {\n", + " 'AttributeName': 'code',\n", + " 'KeyType': 'HASH'\n", + " }\n", + " \n", + " ],\n", + " ProvisionedThroughput={\n", + " 'ReadCapacityUnits': 5,\n", + " 'WriteCapacityUnits': 5\n", + " }\n", + " \n", + ")\n", + "\n", + "# Wait until the table exists.\n", + "table.meta.client.get_waiter('table_exists').wait(TableName='waihui')\n", + "\n", + "# Print out some data about the table.\n", + "print(table.item_count)" + ] + }, + { + "cell_type": "code", + "execution_count": 28, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-17T01:43:20.350295Z", + "iopub.status.busy": "2020-11-17T01:43:20.349393Z", + "iopub.status.idle": "2020-11-17T01:43:23.386976Z", + "shell.execute_reply": "2020-11-17T01:43:23.384304Z", + "shell.execute_reply.started": "2020-11-17T01:43:20.350192Z" + } + }, + "outputs": [ + { + "data": { + "text/plain": [ + "{'ResponseMetadata': {'RequestId': 'C9B2079636H46D3BCM9K0IJJPVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", + " 'HTTPStatusCode': 200,\n", + " 'HTTPHeaders': {'server': 'Server',\n", + " 'date': 'Tue, 17 Nov 2020 01:43:23 GMT',\n", + " 'content-type': 'application/x-amz-json-1.0',\n", + " 'content-length': '2',\n", + " 'connection': 'keep-alive',\n", + " 'x-amzn-requestid': 'C9B2079636H46D3BCM9K0IJJPVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", + " 'x-amz-crc32': '2745614147'},\n", + " 'RetryAttempts': 0}}" + ] + }, + "execution_count": 28, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import boto3\n", + "import decimal\n", + "# Get the service resource.\n", + "dynamodb = boto3.resource('dynamodb')\n", + "\n", + "table = dynamodb.Table('waihui')\n", + "\n", + "table.put_item(\n", + " Item={\n", + " 'code': 'USDCAD',\n", + " 'high': Decimal('1.3200'),\n", + " 'low': Decimal('1.3000'),\n", + " }\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": 31, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-17T01:58:10.720722Z", + "iopub.status.busy": "2020-11-17T01:58:10.719781Z", + "iopub.status.idle": "2020-11-17T01:58:13.750170Z", + "shell.execute_reply": "2020-11-17T01:58:13.747156Z", + "shell.execute_reply.started": "2020-11-17T01:58:10.720618Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "{'high': Decimal('1.32'), 'low': Decimal('1.298'), 'code': 'USDCAD'}\n" + ] + } + ], + "source": [ + "import boto3\n", + "# Get the service resource.\n", + "dynamodb = boto3.resource('dynamodb')\n", + "\n", + "table = dynamodb.Table('waihui')\n", + "\n", + "response = table.get_item(\n", + " Key={\n", + " 'code': 'USDCAD' \n", + " }\n", + ")\n", + "item = response['Item']\n", + "print(item)" + ] + }, + { + "cell_type": "code", + "execution_count": 27, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-17T01:43:05.153025Z", + "iopub.status.busy": "2020-11-17T01:43:05.152079Z", + "iopub.status.idle": "2020-11-17T01:43:06.138506Z", + "shell.execute_reply": "2020-11-17T01:43:06.135790Z", + "shell.execute_reply.started": "2020-11-17T01:43:05.152919Z" + } + }, + "outputs": [ + { + "data": { + "text/plain": [ + "{'ResponseMetadata': {'RequestId': 'HJ7KVGIA4B36S7EAIGBNI8OGMVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", + " 'HTTPStatusCode': 200,\n", + " 'HTTPHeaders': {'server': 'Server',\n", + " 'date': 'Tue, 17 Nov 2020 01:43:06 GMT',\n", + " 'content-type': 'application/x-amz-json-1.0',\n", + " 'content-length': '2',\n", + " 'connection': 'keep-alive',\n", + " 'x-amzn-requestid': 'HJ7KVGIA4B36S7EAIGBNI8OGMVVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", + " 'x-amz-crc32': '2745614147'},\n", + " 'RetryAttempts': 0}}" + ] + }, + "execution_count": 27, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import boto3\n", + "# Get the service resource.\n", + "dynamodb = boto3.resource('dynamodb')\n", + "\n", + "table = dynamodb.Table('waihui')\n", + "\n", + "table.delete_item(\n", + " Key={\n", + " 'code': 'USDCAD' \n", + " }\n", + ")\n" + ] + }, + { + "cell_type": "code", + "execution_count": 30, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-17T01:58:00.482537Z", + "iopub.status.busy": "2020-11-17T01:58:00.481642Z", + "iopub.status.idle": "2020-11-17T01:58:01.617227Z", + "shell.execute_reply": "2020-11-17T01:58:01.614695Z", + "shell.execute_reply.started": "2020-11-17T01:58:00.482434Z" + } + }, + "outputs": [ + { + "data": { + "text/plain": [ + "{'ResponseMetadata': {'RequestId': 'F9DR02KDF2PKEQ7R1MSAUCKVIBVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", + " 'HTTPStatusCode': 200,\n", + " 'HTTPHeaders': {'server': 'Server',\n", + " 'date': 'Tue, 17 Nov 2020 01:58:01 GMT',\n", + " 'content-type': 'application/x-amz-json-1.0',\n", + " 'content-length': '2',\n", + " 'connection': 'keep-alive',\n", + " 'x-amzn-requestid': 'F9DR02KDF2PKEQ7R1MSAUCKVIBVV4KQNSO5AEMVJF66Q9ASUAAJG',\n", + " 'x-amz-crc32': '2745614147'},\n", + " 'RetryAttempts': 0}}" + ] + }, + "execution_count": 30, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import boto3\n", + "import decimal\n", + "# Get the service resource.\n", + "dynamodb = boto3.resource('dynamodb')\n", + "\n", + "table = dynamodb.Table('waihui')\n", + "table.update_item(\n", + " Key={\n", + " 'code': 'USDCAD'\n", + " },\n", + " UpdateExpression='SET low = :val1',\n", + " ExpressionAttributeValues={\n", + " ':val1': Decimal('1.2980')\n", + " }\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# MongoDB系统GridFS文件管理" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 文件上传" + ] + }, + { + "cell_type": "code", + "execution_count": 17, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-26T05:30:36.433098Z", + "iopub.status.busy": "2020-11-26T05:30:36.432192Z", + "iopub.status.idle": "2020-11-26T05:30:37.438663Z", + "shell.execute_reply": "2020-11-26T05:30:37.436233Z", + "shell.execute_reply.started": "2020-11-26T05:30:36.432996Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "5fbf3d7c62452a56d7d16630\n", + "5fbf3d7d62452a56d7d16635\n", + "5fbf3d7d62452a56d7d1663a\n", + "5fbf3d7d62452a56d7d1663f\n", + "5fbf3d7d62452a56d7d16643\n", + "5fbf3d7d62452a56d7d16648\n", + "5fbf3d7d62452a56d7d1664d\n" + ] + } + ], + "source": [ + "import pymongo\n", + "from gridfs import GridFS\n", + "from bson.objectid import ObjectId\n", + "import os\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "\n", + "UploadCache = \"uploadcache\"\n", + "dbURL = \"mongodb://localhost:27017\"\n", + "\n", + "#上传文件\n", + "def upLoadFile(file_coll,file_name,data_link):\n", + " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "\n", + " db = client[\"gaokao\"]\n", + "\n", + " filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n", + " gridfs_col = GridFS(db, collection=file_coll)\n", + " file_ = \"0\"\n", + " query = {\"filename\":\"\"}\n", + " query[\"filename\"] = file_name\n", + "\n", + " if gridfs_col.exists(query):\n", + " print('已经存在该文件')\n", + " else:\n", + "\n", + " with open(file_name, 'rb') as file_r:\n", + " file_data = file_r.read()\n", + " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", + "\n", + " print(file_)\n", + "\n", + "\n", + " return file_ \n", + "# 按文件名获取文档\n", + "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", + " client = pymongo.MongoClient(self.dbURL)\n", + "\n", + " db = client[\"store\"]\n", + "\n", + " gridfs_col = GridFS(db, collection=file_coll)\n", + "\n", + " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", + "\n", + " with open(out_name, 'wb') as file_w:\n", + " file_w.write(file_data)\n", + "\n", + "# 按文件_Id获取文档 \n", + "def downLoadFilebyID(self,file_coll,_id,out_name):\n", + " client = pymongo.MongoClient(self.dbURL)\n", + "\n", + " db = client[\"store\"]\n", + "\n", + " gridfs_col = GridFS(db, collection=file_coll)\n", + "\n", + " O_Id = ObjectId(_id)\n", + "\n", + " gf = gridfs_col.get(file_id=O_Id)\n", + " file_data = gf.read()\n", + " with open(out_name, 'wb') as file_w:\n", + "\n", + " file_w.write(file_data) \n", + "\n", + "\n", + " return gf.filename \n", + "m_dir = './data/tmp'\n", + "fls=os.listdir(m_dir)\n", + "n = 0\n", + "for fl in fls:\n", + " #oldname=fl[n]\n", + " name, suffix = os.path.splitext(fl)\n", + " #if name in old:\n", + " # new_name = fi_path+ os.sep + fi_name[name]+suffix\n", + " # old_name = fi_path+ os.sep + fl[n]\n", + " # os.rename(old_name,new_name)\n", + " #print(os.path.basename(fl))\n", + " #print(fl,suffix[1:])\n", + " full_path = m_dir+ '/' + fl\n", + " upLoadFile(\"document\",full_path,\"\")\n", + "#a = MongoGridFS(\"\")\n", + "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", + "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", + "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.p\")\n", + "#print (ll)" + ] + }, + { + "cell_type": "code", + "execution_count": 12, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-26T05:10:44.453509Z", + "iopub.status.busy": "2020-11-26T05:10:44.452576Z", + "iopub.status.idle": "2020-11-26T05:10:44.517014Z", + "shell.execute_reply": "2020-11-26T05:10:44.515170Z", + "shell.execute_reply.started": "2020-11-26T05:10:44.453402Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "./data/tmp/山东大学强基计划招生专业培养方案(2020版)+-+物理学.pdf\n" + ] + } + ], + "source": [ + "import pymongo\n", + "from gridfs import GridFS\n", + "from bson.objectid import ObjectId\n", + "import os\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "\n", + "UploadCache = \"uploadcache\"\n", + "dbURL = \"mongodb://localhost:27017\"\n", + "\n", + "#上传文件\n", + "def upLoadFile(file_coll,file_name,data_link):\n", + " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "\n", + " db = client[\"gaokao\"]\n", + "\n", + " filter_condition = {\"filename\": file_name, \"url\": data_link}\n", + " gridfs_col = GridFS(db, collection=file_coll)\n", + " file_ = \"0\"\n", + " query = {\"filename\":\"\"}\n", + " query[\"filename\"] = file_name\n", + "\n", + " if gridfs_col.exists(query):\n", + " print('已经存在该文件')\n", + " else:\n", + "\n", + " with open(file_name, 'rb') as file_r:\n", + " file_data = file_r.read()\n", + " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", + "\n", + " print(file_)\n", + "\n", + "\n", + " return file_ \n", + "# 按文件名获取文档\n", + "def downLoadFile(self,file_coll,file_name,out_name,ver):\n", + " client = pymongo.MongoClient(self.dbURL)\n", + "\n", + " db = client[\"store\"]\n", + "\n", + " gridfs_col = GridFS(db, collection=file_coll)\n", + "\n", + " file_data = gridfs_col.get_version(filename=file_name, version=ver).read()\n", + "\n", + " with open(out_name, 'wb') as file_w:\n", + " file_w.write(file_data)\n", + "\n", + "# 按文件_Id获取文档 \n", + "def downLoadFilebyID(file_coll,_id,out_name):\n", + " client = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "\n", + " db = client[\"gaokao\"]\n", + "\n", + " gridfs_col = GridFS(db, collection=file_coll)\n", + "\n", + " O_Id = ObjectId(_id)\n", + "\n", + " gf = gridfs_col.get(file_id=O_Id)\n", + " file_data = gf.read()\n", + " with open(out_name, 'wb') as file_w:\n", + "\n", + " file_w.write(file_data) \n", + "\n", + "\n", + " return gf.filename \n", + "ll = downLoadFilebyID(\"pdf\",\"5fbf351b62452a56d7d16603\",\"out3.pdf\")\n", + "print (ll)\n", + "#a = MongoGridFS(\"\")\n", + "#a.upLoadFile(\"pdf\",\"MongoGridFS.py\",\"\")\n", + "#a.downLoadFile(\"pdf\",\"MongoGridFS.py\",\"out2.p\",2)\n", + "#ll = a.downLoadFilebyID(\"pdf\",\"5d70a5b283a3c5104cd39346\",\"out3.pdf\")\n", + "#print (ll)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/百度OCR.ipynb b/百度OCR.ipynb new file mode 100644 index 0000000..80284cf --- /dev/null +++ b/百度OCR.ipynb @@ -0,0 +1,377 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from aip import AipOcr\n", + "\n", + "\"\"\" 你的 APPID AK SK \"\"\"\n", + "APP_ID = '17553946'\n", + "API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", + "SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", + "\n", + "client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n", + "def get_file_content(filePath):\n", + " with open(filePath, 'rb') as fp:\n", + " return fp.read()\n", + "\n", + "image = get_file_content('1.jpg')\n", + "\n", + "\"\"\" 调用通用文字识别, 图片参数为本地图片 \"\"\"\n", + "#client.basicGeneral(image);\n", + "\n", + "\"\"\" 如果有可选参数 \"\"\"\n", + "options = {}\n", + "options[\"language_type\"] = \"CHN_ENG\"\n", + "options[\"detect_direction\"] = \"true\"\n", + "options[\"detect_language\"] = \"true\"\n", + "options[\"probability\"] = \"true\"\n", + "\n", + "\"\"\" 带参数调用通用文字识别, 图片参数为本地图片 \"\"\"\n", + "result= client.basicGeneral(image, options)\n", + "if 'words_result' in result:\n", + " print('\\n'.join([w['words'] for w in result['words_result']]))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "#将指定目录下图片文件进行文字识别\n", + "import os,sys\n", + "from aip import AipOcr\n", + "\n", + "\"\"\" 你的 APPID AK SK \"\"\"\n", + "APP_ID = '17553946'\n", + "API_KEY = 'IMC1ss3Tyo3vEAdVH6jgcdv2'\n", + "SECRET_KEY = 'Iua6KdhuDtz159rjzGeFGpqgjhnoU0zZ'\n", + "\n", + "client = AipOcr(APP_ID, API_KEY, SECRET_KEY)\n", + "def get_file_content(filePath):\n", + " with open(filePath, 'rb') as fp:\n", + " return fp.read()\n", + "\n", + "options = {}\n", + "options[\"language_type\"] = \"CHN_ENG\"\n", + "options[\"detect_direction\"] = \"true\"\n", + "options[\"detect_language\"] = \"true\"\n", + "options[\"probability\"] = \"true\"\n", + "\n", + "fi_path = os.getcwd()+'/data'\n", + "fl = os.listdir(fi_path)\n", + "fl.sort()\n", + "for fl1 in fl:\n", + " file_name = fi_path+'/' + fl1\n", + " image = get_file_content(file_name)\n", + " result= client.basicGeneral(image, options)\n", + " if 'words_result' in result:\n", + " print('\\n'.join([w['words'] for w in result['words_result']]))\n", + " print('\\n')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 识别图片中表格" + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-26T08:59:12.967127Z", + "iopub.status.busy": "2020-11-26T08:59:12.966172Z", + "iopub.status.idle": "2020-11-26T08:59:28.010488Z", + "shell.execute_reply": "2020-11-26T08:59:28.006947Z", + "shell.execute_reply.started": "2020-11-26T08:59:12.967019Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "22917135_2274033\n", + "已完成\n" + ] + } + ], + "source": [ + "import requests\n", + "import json\n", + "import base64\n", + "import time\n", + "\n", + "def get_access_token():\n", + " client_id = 'KwXkGawxh0sjOQdF9Ae9LeLb'\n", + " client_secret = 'siprEKMp5UcRTOAngEfIOOe9x6xkqGXq' \n", + " # client_id 为官网获取的AK, client_secret 为官网获取的SK\n", + " host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id={}&client_secret={}'.format(\n", + " client_id, client_secret)\n", + " response = requests.get(host).text\n", + " data = json.loads(response)\n", + " access_token = data['access_token']\n", + " return access_token\n", + "\n", + "def get_excel(requests_id, access_token):\n", + " headers = {'content-type': 'application/x-www-form-urlencoded'}\n", + " pargams = {\n", + " 'request_id': requests_id,\n", + " 'result_type': 'excel'\n", + " }\n", + " url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", + " url_all = url + \"?access_token=\" + access_token\n", + " res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", + " info_1 = res.json()['result']['ret_msg']\n", + " excel_url=res.json()['result']['result_data']\n", + " excel_1=requests.get(excel_url).content\n", + " with open('识别结果11.xls','wb+') as f:\n", + " f.write(excel_1)\n", + " print(info_1)\n", + "\n", + "\n", + "request_url = \"https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request\"\n", + "# 二进制方式打开图片文件\n", + "f = open('山东大学强基计划(2020).jpg', 'rb')\n", + "img = base64.b64encode(f.read())\n", + "\n", + "params = {\"image\":img}\n", + "access_token = get_access_token()\n", + "request_url = request_url + \"?access_token=\" + access_token\n", + "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", + "response = requests.post(request_url, data=params, headers=headers)\n", + "if response:\n", + " m_xx = response.json()\n", + "requests_id = m_xx['result'][0]['request_id'] \n", + "print(requests_id)\n", + "time.sleep(10)\n", + "get_excel(requests_id, access_token)" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-03T02:06:23.901979Z", + "iopub.status.busy": "2020-11-03T02:06:23.900925Z", + "iopub.status.idle": "2020-11-03T02:06:24.312339Z", + "shell.execute_reply": "2020-11-03T02:06:24.309116Z", + "shell.execute_reply.started": "2020-11-03T02:06:23.901714Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "{'refresh_token': '25.1dfb7a14cd15d03051976c8db246fa92.315360000.1919729184.282335-22917135', 'expires_in': 2592000, 'session_key': '9mzdCSFczT8Mv7ZN07SjsPo0dZr0AvxAeDt6Kjn1Z9jiiotA6kW2TZYzslnuYOFd1ZCx75mmzW0TRF+nxYAHzPOb5fmjlw==', 'access_token': '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135', 'scope': 'public vis-ocr_ocr brain_ocr_scope brain_ocr_general brain_ocr_general_basic vis-ocr_business_license brain_ocr_webimage brain_all_scope brain_ocr_idcard brain_ocr_driving_license brain_ocr_vehicle_license vis-ocr_plate_number brain_solution brain_ocr_plate_number brain_ocr_accurate brain_ocr_accurate_basic brain_ocr_receipt brain_ocr_business_license brain_solution_iocr brain_qrcode brain_ocr_handwriting brain_ocr_passport brain_ocr_vat_invoice brain_numbers brain_ocr_business_card brain_ocr_train_ticket brain_ocr_taxi_receipt vis-ocr_household_register vis-ocr_vis-classify_birth_certificate vis-ocr_台湾通行证 vis-ocr_港澳通行证 vis-ocr_机动车购车发票识别 vis-ocr_机动车检验合格证识别 vis-ocr_车辆vin码识别 vis-ocr_定额发票识别 vis-ocr_保单识别 vis-ocr_机打发票识别 vis-ocr_行程单识别 brain_ocr_vin brain_ocr_quota_invoice brain_ocr_birth_certificate brain_ocr_household_register brain_ocr_HK_Macau_pass brain_ocr_taiwan_pass brain_ocr_vehicle_invoice brain_ocr_vehicle_certificate brain_ocr_air_ticket brain_ocr_invoice brain_ocr_insurance_doc brain_formula brain_ocr_meter brain_doc_analysis brain_ocr_webimage_loc wise_adapt lebo_resource_base lightservice_public hetu_basic lightcms_map_poi kaidian_kaidian ApsMisTest_Test权限 vis-classify_flower lpq_开放 cop_helloScope ApsMis_fangdi_permission smartapp_snsapi_base smartapp_mapp_dev_manage iop_autocar oauth_tp_app smartapp_smart_game_openapi oauth_sessionkey smartapp_swanid_verify smartapp_opensource_openapi smartapp_opensource_recapi fake_face_detect_开放Scope vis-ocr_虚拟人物助理 idl-video_虚拟人物助理 smartapp_component', 'session_secret': '6ac230d6705697808e228241c519f303'}\n" + ] + } + ], + "source": [ + "import requests \n", + "\n", + "# client_id 为官网获取的AK, client_secret 为官网获取的SK\n", + "host = 'https://aip.baidubce.com/oauth/2.0/token?grant_type=client_credentials&client_id=KwXkGawxh0sjOQdF9Ae9LeLb&client_secret=siprEKMp5UcRTOAngEfIOOe9x6xkqGXq'\n", + "response = requests.get(host)\n", + "if response:\n", + " print(response.json())" + ] + }, + { + "cell_type": "code", + "execution_count": 14, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-03T03:24:08.412670Z", + "iopub.status.busy": "2020-11-03T03:24:08.411768Z", + "iopub.status.idle": "2020-11-03T03:24:08.813580Z", + "shell.execute_reply": "2020-11-03T03:24:08.810985Z", + "shell.execute_reply.started": "2020-11-03T03:24:08.412567Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "已完成\n" + ] + } + ], + "source": [ + "import requests\n", + "import json\n", + "import base64\n", + "access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n", + "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", + "pargams = {\n", + " 'request_id': '22917135_2227436',\n", + " 'result_type': 'excel'\n", + "}\n", + "\n", + "\n", + "url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", + "url_all = url + \"?access_token=\" + access_token\n", + "res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", + "info_1 = res.json()['result']['ret_msg']\n", + "excel_url=res.json()['result']['result_data']\n", + "excel_1=requests.get(excel_url).content\n", + "with open('识别结果12.xls','wb+') as f:\n", + " f.write(excel_1)\n", + "print(info_1)" + ] + }, + { + "cell_type": "code", + "execution_count": 80, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-03T08:43:37.180693Z", + "iopub.status.busy": "2020-11-03T08:43:37.179800Z", + "iopub.status.idle": "2020-11-03T08:43:37.656055Z", + "shell.execute_reply": "2020-11-03T08:43:37.653668Z", + "shell.execute_reply.started": "2020-11-03T08:43:37.180589Z" + } + }, + "outputs": [ + { + "data": { + "text/plain": [ + "list" + ] + }, + "execution_count": 80, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import requests\n", + "import json\n", + "import base64\n", + "import demjson\n", + "access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n", + "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", + "pargams = {\n", + " 'request_id': '22917135_2227436',\n", + " 'result_type': 'json'\n", + "}\n", + "\n", + "\n", + "url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", + "url_all = url + \"?access_token=\" + access_token\n", + "res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", + "#info_1 = res.json()['result']['ret_msg']\n", + "excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n", + "type(excel_1)\n", + "#excel_new = demjson.decode(excel_1)\n", + "#for m_col in excel_new['forms'][0]['body']:\n", + "# print(m_col)\n", + "m_xx =json.loads(excel_1)\n", + "\n", + "#with open('识别结果12.json','w') as fl:\n", + "# json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n", + "#print(info_1)\n", + "#print(json.dumps(m_xx['forms'][0],ensure_ascii=False))\n", + "print(m_xx['forms'][0]['body'])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 识别保存为json文件" + ] + }, + { + "cell_type": "code", + "execution_count": 81, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-03T08:48:26.082167Z", + "iopub.status.busy": "2020-11-03T08:48:26.081266Z", + "iopub.status.idle": "2020-11-03T08:48:26.430039Z", + "shell.execute_reply": "2020-11-03T08:48:26.428236Z", + "shell.execute_reply.started": "2020-11-03T08:48:26.082060Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "已完成\n" + ] + } + ], + "source": [ + "import requests\n", + "import json\n", + "import base64\n", + "import demjson\n", + "access_token = '24.277af0e5357bc3958c5b3cac49662a80.2592000.1606961184.282335-22917135'\n", + "headers = {'content-type': 'application/x-www-form-urlencoded'}\n", + "pargams = {\n", + " 'request_id': '22917135_2227436',\n", + " 'result_type': 'json'\n", + "}\n", + "\n", + "\n", + "url = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result'\n", + "url_all = url + \"?access_token=\" + access_token\n", + "res = requests.post(url_all, headers=headers, params=pargams)#访问链接获取excel下载页\n", + "#info_1 = res.json()['result']['ret_msg']\n", + "excel_1=res.json()['result']['result_data']#['forms'][0]['body']\n", + "type(excel_1)\n", + "#excel_new = demjson.decode(excel_1)\n", + "#for m_col in excel_new['forms'][0]['body']:\n", + "# print(m_col)\n", + "m_xx =json.loads(excel_1)\n", + "\n", + "with open('识别结果12.json','w') as fl:\n", + " json.dump(m_xx['forms'][0],fl,ensure_ascii=False)\n", + "print(info_1)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + }, + "toc-autonumbering": true, + "toc-showmarkdowntxt": false, + "toc-showtags": false + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/股票管理.ipynb b/股票管理.ipynb new file mode 100644 index 0000000..16b2c79 --- /dev/null +++ b/股票管理.ipynb @@ -0,0 +1,499 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# 股票管理" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 股票信息导入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import baostock as bs\n", + "import pandas as pd\n", + "\n", + "#### 登陆系统 ####\n", + "lg = bs.login()\n", + "# 显示登陆返回信息\n", + "print('login respond error_code:'+lg.error_code)\n", + "print('login respond error_msg:'+lg.error_msg)\n", + "\n", + "#### 获取证券信息 ####\n", + "rs = bs.query_all_stock(day=\"2020-10-20\")\n", + "print('query_all_stock respond error_code:'+rs.error_code)\n", + "print('query_all_stock respond error_msg:'+rs.error_msg)\n", + "\n", + "#### 打印结果集 ####\n", + "data_list = []\n", + "while (rs.error_code == '0') & rs.next():\n", + " # 获取一条记录,将记录合并在一起\n", + " data_list.append(rs.get_row_data())\n", + "#results = pd.DataFrame(data_list, columns=rs.fields)\n", + "\n", + "#### 结果集输出到csv文件 #### \n", + "#result.to_csv(\"all_stock.csv\", encoding=\"utf-8\", index=False)\n", + "for result in data_list:\n", + " print(result)\n", + "\n", + "#### 登出系统 ####\n", + "bs.logout()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import baostock as bs\n", + "import pandas as pd\n", + "import pymysql\n", + "\n", + "# 登陆系统\n", + "lg = bs.login()\n", + "# 显示登陆返回信息\n", + "print('login respond error_code:'+lg.error_code)\n", + "print('login respond error_msg:'+lg.error_msg)\n", + "\n", + "# 获取证券基本资料\n", + "rs = bs.query_stock_basic(code=\"\")\n", + "# rs = bs.query_stock_basic(code_name=\"浦发银行\") # 支持模糊查询\n", + "print('query_stock_basic respond error_code:'+rs.error_code)\n", + "print('query_stock_basic respond error_msg:'+rs.error_msg)\n", + "\n", + "# 打印结果集\n", + "data_list = []\n", + "while (rs.error_code == '0') & rs.next():\n", + " # 获取一条记录,将记录合并在一起\n", + " data_list.append(rs.get_row_data())\n", + "#result = pd.DataFrame(data_list, columns=rs.fields)\n", + "# 结果集输出到csv文件\n", + "#result.to_csv(\"D:/stock_basic.csv\", encoding=\"gbk\", index=False)\n", + "\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n", + "cursor = db.cursor()\n", + "\n", + "sql = \"insert into stock_info (code,name,ipoDate,outDate,type,status) values(%s,%s,%s,%s,%s,%s)\"\n", + "try:\n", + " cursor.executemany(sql,tuple(data_list))\n", + " db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "db.close()\n", + "# 登出系统\n", + "bs.logout()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 每日持有标的信息导入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import csv\n", + "import pymysql\n", + "\n", + "def read_data(filename):\n", + " detail = {}\n", + " with open(filename) as f:\n", + " reader = csv.reader(f)\n", + " #header_row =next(reader)\n", + " for row in reader:\n", + " detail.setdefault(row[0],[])\n", + " detail[row[0]].append(row[1])\n", + " return detail\n", + "m_xx = []\n", + "m_stock = {}\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n", + "cursor = db.cursor()\n", + "sql = 'select code,name from stock_info where type=\"1\"'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_stock[result[0]] = result[1]\n", + "#print(m_stock)\n", + "filename = '每日标的信息.csv'\n", + "detail = read_data(filename)\n", + "for k,v in detail.items():\n", + " m_rq = '2020-' + k[0:2] + '-' + k[2:]\n", + " i = 1\n", + " for m_dm in v:\n", + " if m_dm[0:1] == '6':\n", + " m_dm = 'sh.' + m_dm\n", + " else:\n", + " m_dm = 'sz.' + m_dm\n", + " print('\\t'+m_dm + '\\t'+m_stock[m_dm])\n", + " m_xx.append((m_rq,m_dm,i))\n", + " i += 1\n", + "choice = input('以上为本日数据,是否导入?(y/n)')\n", + "if choice.upper() == \"Y\":\n", + " sql = \"insert into daily_item (rq,code,ord) values(%s,%s,%s)\"\n", + " try:\n", + " cursor.executemany(sql,m_xx)\n", + " db.commit()\n", + " print(\"ok!\")\n", + " except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "db.close()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 每日调仓信息导入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import csv\n", + "import pymysql\n", + "\n", + "def read_data(filename):\n", + " detail = {}\n", + " \n", + " with open(filename) as f:\n", + " reader = csv.reader(f)\n", + "# header_row =next(reader)\n", + " for row in reader:\n", + " detail1 = {}\n", + " detail.setdefault(row[3],[])\n", + " detail1 = {'dm':row[0],'zj':row[1],'ly':row[2],'cb':row[4]}\n", + " detail[row[3]].append(detail1)\n", + " return detail\n", + "m_xx = []\n", + "m_add = []\n", + "m_sub = []\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n", + "cursor = db.cursor()\n", + "filename = '调仓明细.csv'\n", + "detail = read_data(filename)\n", + "#print(detail)\n", + "for rq in sorted(detail.keys()):\n", + " m_rq = '2020-' + rq[0:2] + '-' + rq[2:]\n", + " for xx_move in detail[rq]:\n", + " m_dm = xx_move['dm']\n", + " if m_dm[0:1] == '6':\n", + " m_dm = 'sh.' + m_dm\n", + " else:\n", + " m_dm = 'sz.' + m_dm\n", + " m_xx.append((m_rq,m_dm,int(xx_move['zj']),xx_move['ly']))\n", + " if xx_move['zj'] == '1':\n", + " m_add.append((m_dm,xx_move['ly'],float(xx_move['cb'])))\n", + " else:\n", + " m_sub.append((m_dm))\n", + "#print(m_xx)\n", + "sql = \"insert into change_item (rq,code,pos,reason) values(%s,%s,%s,%s)\"\n", + "try:\n", + " cursor.executemany(sql,m_xx)\n", + " if len(m_add) > 0:\n", + " sql_add = \"insert into stock_item (code,reason,cost) values(%s,%s,%s)\"\n", + " cursor.executemany(sql_add,m_add)\n", + " if len(m_sub) > 0:\n", + " sql_sub = \"update stock_item set status=0 where code=%s\"\n", + " cursor.executemany(sql_sub,m_sub) \n", + " db.commit()\n", + " print(\"导入成功!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "\n", + "db.close()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 持仓股票信息导入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"stock\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT a.unit,a.CODE,b.cost FROM daily_item AS a,daily_cost as b WHERE a.rq=(select max(rq) FROM daily_item) AND a.code=b.code'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "sql = \"insert into stock_item(item,code,cost) values(%s,%s,%s)\"\n", + "try:\n", + " cursor.executemany(sql,list(results))\n", + " #db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "sql = 'SELECT a.reason,a.code from change_item AS a WHERE a.pos=1'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "sql = \"update stock_item set reason=%s where code=%s \"\n", + "try:\n", + " cursor.executemany(sql,list(results))\n", + " #db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "db.close()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 外汇实时数据采集" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "import re\n", + "\n", + "#pattern = re.compile('(?<=\\\").*(?=\\\")')\n", + "pattern = re.compile(r'\\\"(.*)\\\"')\n", + "#myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "#mydb = myclient[\"gaokao\"]\n", + "#mycol = mydb[\"news\"]\n", + "url = 'http://hq.sinajs.cn/list=USDCAD'\n", + "strhtml = requests.get(url)\n", + "#strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "data = strhtml.text\n", + "#data1 = data.split(\"\\r\")\n", + "#data = soup.select('schoolList')\n", + "#for item1 in data:\n", + "# print(item1.get_text())\n", + "if pattern.findall(data):\n", + " for data1 in pattern.findall(data):\n", + " data2 = data1.split(',')\n", + " print(data2)\n", + " print('当前买入价:',data2[1])\n", + " print('当前卖出价:',data2[2])\n", + " print('昨收价:',data2[3])\n", + " print('今开价:',data2[5])\n", + " print('最高价:',data2[6])\n", + " print('最低价:',data2[7])\n", + " print(float(data2[7]))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from email.mime.text import MIMEText\n", + "from email.header import Header\n", + "import smtplib\n", + "import requests\n", + "import time\n", + "import re\n", + "\n", + "pattern = re.compile(r'\\\"(.*)\\\"')\n", + "url = 'http://hq.sinajs.cn/list=USDCAD'\n", + "strhtml = requests.get(url)\n", + "data = strhtml.text\n", + "if pattern.findall(data):\n", + " for data1 in pattern.findall(data):\n", + " data2 = data1.split(',')\n", + "#print(data2)\n", + "with open('price.txt','r') as fl:\n", + " for line in fi:\n", + " p_high = line.split(',')[0]\n", + " p_low = line.split(',')[1]\n", + "\n", + "message ='当前美元加元买入价:{},卖出价:{}'.format(data2[1],data2[2])\n", + "msg = MIMEText(message,'plain','utf-8')\n", + "msg['Subject'] = Header(\"外汇价格已经到达预期价位!\",'utf-8')\n", + "msg['From'] = Header('512song@sina.com')\n", + "msg['To'] = Header('491525765@qq.com','utf-8')\n", + "\n", + "from_addr = '512song@sina.com' #发件邮箱\n", + "password = '409fe5d8471da663' #邮箱密码\n", + "to_addr = 'songyi@yeah.net' #收件邮箱\n", + "smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n", + "try:\n", + " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", + " print('开始登录')\n", + " server.set_debuglevel(1) \n", + " server.login(from_addr,password) #登录邮箱\n", + " print('登录成功')\n", + " print(\"邮件开始发送\")\n", + " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", + " server.quit()\n", + " print(\"邮件发送成功\")\n", + "except smtplib.SMTPException as e:\n", + " print(\"邮件发送失败\",e)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from email.mime.text import MIMEText\n", + "from email.header import Header\n", + "import smtplib\n", + "import requests\n", + "import time\n", + "import re\n", + "\n", + "pattern = re.compile(r'\\\"(.*)\\\"')\n", + "url = 'http://hq.sinajs.cn/list=USDCAD'\n", + "strhtml = requests.get(url)\n", + "data = strhtml.text\n", + "if pattern.findall(data):\n", + " for data1 in pattern.findall(data):\n", + " data2 = data1.split(',')\n", + "print(data2)\n", + "\n", + "message ='当前美元加元最高价:{},最低价:{}'.formtat()\n", + "msg = MIMEText(message,'plain','utf-8')\n", + "\n", + "msg['Subject'] = Header(\"测试smtp邮件\",'utf-8')\n", + "msg['From'] = Header('512song@sina.com')\n", + "msg['To'] = Header('491525765@qq.com','utf-8')\n", + "\n", + "from_addr = '512song@sina.com' #发件邮箱\n", + "password = '409fe5d8471da663' #邮箱密码\n", + "to_addr = '491525765@qq.com' #收件邮箱\n", + "\n", + "smtp_server = 'smtp.sina.com' #SMTP服务器,以新浪为例\n", + "try:\n", + " server = smtplib.SMTP(smtp_server,25) #第二个参数为默认端口为25,有些邮件有特殊端口\n", + " print('开始登录')\n", + " server.set_debuglevel(1) \n", + " server.login(from_addr,password) #登录邮箱\n", + " print('登录成功')\n", + " print(\"邮件开始发送\")\n", + " server.sendmail(from_addr,to_addr,msg.as_string()) #将msg转化成string发出\n", + " server.quit()\n", + " print(\"邮件发送成功\")\n", + "except smtplib.SMTPException as e:\n", + " print(\"邮件发送失败\",e)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "# pygal图表" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pygal\n", + "bar_chart = pygal.Bar(height=300)\n", + "bar_chart.add('Fibonacci', [0, 1, 1, 2, 3, 5, 8, 13, 21, 34, 55])\n", + "bar_chart.add('Padovan', [1, 1, 1, 2, 2, 3, 4, 5, 7, 9, 12])\n", + "svg = bar_chart.render()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from IPython.display import SVG\n", + "SVG(svg)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + }, + "toc-autonumbering": true, + "toc-showcode": true, + "toc-showmarkdowntxt": true + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/语音合成.ipynb b/语音合成.ipynb new file mode 100644 index 0000000..78aa99e --- /dev/null +++ b/语音合成.ipynb @@ -0,0 +1,74 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "#gTTS语音\n", + "from gtts import gTTS\n", + "#engine = pyttsx3.init('espeak')\n", + "tts = gTTS(text=\"军士上前,将英玉兰架起,两个抓着脚踝,两个托住肩头,一起用力,英玉兰无奈分开两条大浪腿,露出骚屄,被举下\",lang='zh-cn')\n", + "#engine.save_to_file(\"欢迎使用百度语音合成,本次测试为Python接口\",'./test')\n", + "tts.save(\"./test.mp3\")\n", + "print('ok')" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "#百度语音在线\n", + "from aip import AipSpeech\n", + "\n", + "\"\"\" 你的 APPID AK SK \"\"\"\n", + "APP_ID = '17553946'\n", + "API_KEY = 'i5LBXalkBn2KHTqdifesA1EB'\n", + "SECRET_KEY = '1ktP6qDH7nMFHjotRdKpnI13w41Bvk59'\n", + "\n", + "client = AipSpeech(APP_ID, API_KEY, SECRET_KEY)\n", + "\n", + "result = client.synthesis('军士上前,将英玉兰架起,两个抓着脚踝,两个托住肩头,一起用力,英玉兰无奈分开两条大浪腿,露出骚屄,被举下。', 'zh', 5, {\n", + " 'vol': 5,'per': 4,\n", + "})\n", + "\n", + "# 识别正确返回语音二进制 错误则返回dict 参照下面错误码\n", + "if not isinstance(result, dict):\n", + " with open('./test3.mp3', 'wb') as f:\n", + " f.write(result)\n", + "else:print(result)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/高校资料管理.ipynb b/高校资料管理.ipynb new file mode 100644 index 0000000..c639eaf --- /dev/null +++ b/高校资料管理.ipynb @@ -0,0 +1,235 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# 重点高校信息管理" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 强基计划信息管理" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 录入强基计划信息" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-27T07:02:30.612429Z", + "iopub.status.busy": "2020-11-27T07:02:30.611524Z", + "iopub.status.idle": "2020-11-27T07:02:31.128055Z", + "shell.execute_reply": "2020-11-27T07:02:31.124461Z", + "shell.execute_reply.started": "2020-11-27T07:02:30.612326Z" + } + }, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"qiangji\"]\n", + "m_content = ''\n", + "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", + "url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n", + "strhtml = requests.get(url,headers = headers)\n", + "strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "#data = strhtml.text\n", + "#data1 = data.split(\"\\r\")\n", + "data = soup.select('body > div.container.content > div > div > div.col-md-9.col-sm-8 > div')\n", + "for item1 in data:\n", + " m_content +=item1.get_text()\n", + "#print(m_content)\n", + "m_title = soup.select('body > div.y_tit_box > div > div > div > h1')\n", + "m_name = '中国人民大学'\n", + "m_code = 'A002'\n", + "m_year = '2020'\n", + "myquery = {'code':m_code}\n", + "x = mycol.count_documents(myquery)\n", + "#print(x)\n", + "m_id = time.strftime(\"%Y%m%d%H%M%S\", time.localtime())\n", + "if x > 0: \n", + " m_mg = {} \n", + " m_mg.setdefault(m_id,{})\n", + " m_mg[m_id]['title'] = m_title[0].text\n", + " m_mg[m_id]['content'] = m_content\n", + " m_mg[m_id]['url'] = url\n", + " mycol.update_one(myquery,{'$push':{m_year:m_mg}})\n", + " print(x,s)\n", + "else:\n", + " m_mg = {}\n", + " m_mg['name'] = m_name\n", + " m_mg['code'] = m_code\n", + " m_mg.setdefault(m_year,[])\n", + " m_mg1 = {}\n", + " m_mg1.setdefault(m_id,{})\n", + " m_mg1[m_id]['title'] = m_title[0].text\n", + " m_mg1[m_id]['content'] = m_content\n", + " m_mg1[m_id]['url'] = url\n", + " m_mg[m_year].append(m_mg1)\n", + " mycol.insert_one(m_mg) \n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 录入强基计划附件" + ] + }, + { + "cell_type": "code", + "execution_count": 73, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-26T10:35:49.756505Z", + "iopub.status.busy": "2020-11-26T10:35:49.755575Z", + "iopub.status.idle": "2020-11-26T10:35:50.001279Z", + "shell.execute_reply": "2020-11-26T10:35:49.998646Z", + "shell.execute_reply.started": "2020-11-26T10:35:49.756395Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "1 gaokao.qiangji.2020\n" + ] + } + ], + "source": [ + "import requests\n", + "import time\n", + "import pymongo\n", + "import os\n", + "from gridfs import GridFS\n", + "\n", + "def upLoadFile(file_coll,file_name,data_link): \n", + " filter_condition = {\"filename\": os.path.basename(file_name), \"url\": data_link}\n", + " gridfs_col = GridFS(mydb, collection=file_coll)\n", + " file_ = \"0\"\n", + " query = {\"filename\":\"\"}\n", + " query[\"filename\"] = file_name\n", + " if gridfs_col.exists(query):\n", + " print('已经存在该文件')\n", + " else:\n", + " with open(file_name, 'rb') as file_r:\n", + " file_data = file_r.read()\n", + " file_ = gridfs_col.put(data=file_data, **filter_condition) # 上传到gridfs\n", + "\n", + " #print(file_)\n", + " return file_ \n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"qiangji\"]\n", + "m_content = ''\n", + "url = 'https://www.bkzs.sdu.edu.cn/info/1036/1635.htm'\n", + "m_title = ''\n", + "m_name = '山东大学'\n", + "m_code = 'A422'\n", + "m_year = '2020'\n", + "myquery = {'code':m_code}\n", + "x = mycol.count_documents(myquery)\n", + "m_id = time.strftime(\"%Y%m%d%H%M%S\", time.localtime())\n", + "m_dir = './data/tmp'\n", + "fls=os.listdir(m_dir)\n", + "l_fls =[]\n", + "for fl in fls: \n", + " full_path = m_dir+ '/' + fl\n", + " l_fls.append(upLoadFile(\"document\",full_path,\"\"))\n", + "m_mg = {}\n", + "if x > 0: \n", + " m_mg.setdefault(m_id,{})\n", + " m_mg[m_id]['title'] = m_title\n", + " m_mg[m_id]['files'] = l_fls\n", + " m_mg[m_id]['url'] = url\n", + " mycol.update_one(myquery,{'$push':{m_year:m_mg}})\n", + "else:\n", + " m_mg['name'] = m_name\n", + " m_mg['code'] = m_code\n", + " m_mg.setdefault(m_year,[])\n", + " m_mg1 = {}\n", + " m_mg1.setdefault(m_id,{})\n", + " m_mg1[m_id]['title'] = m_title\n", + " m_mg[m_id]['files'] = l_fls\n", + " m_mg1[m_id]['url'] = url\n", + " m_mg[m_year].append(m_mg1)\n", + " mycol.insert_one(m_mg) \n" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "execution": { + "iopub.execute_input": "2021-02-24T08:13:56.382398Z", + "iopub.status.busy": "2021-02-24T08:13:56.381289Z", + "iopub.status.idle": "2021-02-24T08:13:56.397601Z", + "shell.execute_reply": "2021-02-24T08:13:56.395607Z", + "shell.execute_reply.started": "2021-02-24T08:13:56.382135Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "20210224161356\n" + ] + } + ], + "source": [ + "import time\n", + "\n", + "print(time.strftime(\"%Y%m%d%H%M%S\", time.localtime()))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + }, + "toc-autonumbering": false + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/高考志愿管理.ipynb b/高考志愿管理.ipynb new file mode 100644 index 0000000..c9d437f --- /dev/null +++ b/高考志愿管理.ipynb @@ -0,0 +1,212 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# 高考志愿管理" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2020年高考录取信息导入MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "metadata": { + "tags": [] + }, + "outputs": [], + "source": [ + "import pymysql\n", + "import pymongo\n", + "import decimal\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "mycol1 = mydb[\"admission_2020\"]\n", + "\n", + "m_col = {}\n", + "m_spe = {}\n", + "m_xx = {}\n", + "for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n", + " m_col[x['code']] = x['name']\n", + "\n", + "\n", + "db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2020\"'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_spe.setdefault(result[1],{}) \n", + " m_spe[result[1]][result[0]] = result[2]\n", + "sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "i = 0\n", + "m_min = 0\n", + "ii = 0\n", + "for result in results:\n", + " m_xx.clear()\n", + " if result[6] == m_min:\n", + " ii = ii\n", + " i = i+1\n", + " else:\n", + " i = i+1\n", + " ii = i\n", + " m_min = result[6]\n", + " m_xx['pos'] = ii\n", + " m_xx['col_code'] = result[1]\n", + " m_xx['col_name'] = m_col[result[1]]\n", + " m_xx['spe_code'] = result[2]\n", + " m_xx['spe_name'] = m_spe[result[1]][result[2]]\n", + " m_xx['plan'] = result[3]\n", + " m_xx['dispense'] = result[5]\n", + " m_xx['num_min'] = result[6]\n", + " m_xx['num_avg'] = int(result[7])\n", + " m_xx['rank_min'] = result[8]\n", + " m_xx['nian'] = '2020' \n", + " mycol1.insert_one(m_xx)\n", + "#print(m_spe)\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 计算志愿分数概率" + ] + }, + { + "cell_type": "code", + "execution_count": 17, + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "633 619\n", + "634 556\n", + "635 450\n", + "636 397\n", + "637 328\n", + "638 275\n", + "639 351\n" + ] + } + ], + "source": [ + "import random\n", + "\n", + "array2 = []\n", + "for i in range(10000):\n", + " array1 = []\n", + " s = 0\n", + " for ii in range(12):\n", + " m1 = random.randint(633,639)\n", + " array1.append(m1)\n", + " s = s + m1 \n", + " m_avg = round(s/12,2)\n", + " if m_avg == 635.5 and (633 in array1) and (639 in array1):\n", + " #print(array1)\n", + " array2.extend(array1)\n", + " \n", + "#print(array2)\n", + "m_set = set(array2)\n", + "for m in m_set:\n", + " print(m,array2.count(m))" + ] + }, + { + "cell_type": "code", + "execution_count": 18, + "metadata": {}, + "outputs": [ + { + "data": { + "text/plain": [ + "" + ] + }, + "execution_count": 18, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import openpyxl\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"admission_college\"]\n", + "wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')\n", + "sheet = wb.active\n", + "#sheets = wb.sheetnames\n", + "code = 'A422'\n", + "name = '山东大学'\n", + "new_col = []\n", + "dict1 = {}\n", + "\n", + "new_code = []\n", + "dict1['code'] = code\n", + "dict1['name'] = name\n", + "for n in range(1,sheet.max_row+1):\n", + " nian = str(sheet.cell(n,1).value)\n", + " dict2 = {}\n", + " \n", + " dict1.setdefault(nian,[])\n", + " if sheet.cell(n,2).value =='理工':\n", + " m_lb = 'l'\n", + " elif sheet.cell(n,2).value =='文史':\n", + " m_lb = 'w'\n", + " else:\n", + " m_lb = 'z' \n", + " dict2['type'] = m_lb\n", + " dict2['spe_name'] = sheet.cell(n,4).value\n", + " dict2['max_score'] = sheet.cell(n,5).value\n", + " dict2['min_score'] = sheet.cell(n,6).value\n", + " dict2['avg_score'] = sheet.cell(n,7).value\n", + " dict2['dispense'] = sheet.cell(n,8).value\n", + " dict1[nian].append(dict2)\n", + "mycol.insert_one(dict1)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/高考数据导入.ipynb b/高考数据导入.ipynb new file mode 100644 index 0000000..bfb949d --- /dev/null +++ b/高考数据导入.ipynb @@ -0,0 +1,1472 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# 高考数据导入" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 导入2012年高校本科专业目录" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "m_xx = dict()\n", + "fl_name = 'teshe_2012.txt'\n", + "with open(fl_name,'r') as fl:\n", + " for line in fl:\n", + " if line.split():\n", + " l = line.split('\\t',1)\n", + " #print(l[0]+'--'+re.sub('\\n', '', l[1]))\n", + " m_len = len(l[0])\n", + " m_name = re.sub('\\n', '', l[1])\n", + " if m_len == 2:\n", + " m_xx.setdefault(l[0],{})\n", + " ll = m_name.split(':',1)\n", + " m_xx[l[0]]['name'] = ll[1]\n", + " elif m_len == 4:\n", + " m_xx[l[0][:2]].setdefault(l[0],{})\n", + " m_xx[l[0][:2]][l[0]]['name'] = m_name\n", + " else:\n", + " m_xx[l[0][:2]][l[0][:4]].setdefault(l[0],{})\n", + " m_xx[l[0][:2]][l[0][:4]][l[0]]['code'] = l[0]\n", + " m_xx[l[0][:2]][l[0][:4]][l[0]]['name'] = m_name\n", + "#print(m_xx)\n", + "with open('teshe_2012.json','w') as fl1:\n", + " json.dump(m_xx,fl1,ensure_ascii=False) " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 合并2012年目录基本与特设专业" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import re\n", + "m_teshe = dict()\n", + "m_new = dict()\n", + "fl_name = 'jiben_2012.json'\n", + "fl_name1 = 'teshe_2012.json'\n", + "\n", + "with open(fl_name,'r') as fl:\n", + " m_new = json.load(fl)\n", + "for k,v in m_new.items():\n", + " for k1,v1 in v.items():\n", + " if k1 !='name':\n", + " for k2,v2 in m_new[k][k1].items():\n", + " if k2 !='name':\n", + " m_new[k][k1][k2]['lb'] = '基本'\n", + "\n", + "with open(fl_name1,'r') as fl1:\n", + " m_teshe = json.load(fl1)\n", + "for k,v in m_teshe.items():\n", + " for k1,v1 in v.items():\n", + " if k1 !='name':\n", + " for k2,v2 in m_teshe[k][k1].items():\n", + " if k2 !='name':\n", + " m_teshe[k][k1][k2]['lb'] = '特设'\n", + " m_new[k][k1][k2] = m_teshe[k][k1][k2]\n", + "with open('hebing_2012.json','w') as fl2:\n", + " json.dump(m_new,fl2,ensure_ascii=False) " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 导入2020年高校本科专业目录" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"zhuanyemulu\"]\n", + "m_xx = dict()\n", + "fl_name = '普通高等学校本科专业目录.xlsx'\n", + "wb = openpyxl.load_workbook(fl_name)\n", + "sheet = wb.active\n", + "for n in range(2,sheet.max_row+1):\n", + " m_code = sheet.cell(n,4).value\n", + " m_name = sheet.cell(n,5).value\n", + " m_k1 = m_code[:2]\n", + " m_k2 = m_code[:4]\n", + " m_xx.setdefault(m_k1,{})\n", + " m_xx[m_k1]['name'] = sheet.cell(n,2).value\n", + " m_xx[m_k1]['level'] = '门类'\n", + " m_xx[m_k1].setdefault(m_k2,{})\n", + " m_xx[m_k1][m_k2]['name'] = sheet.cell(n,3).value\n", + " m_xx[m_k1][m_k2]['level'] ='专业类'\n", + " m_xx[m_k1][m_k2].setdefault(m_code,{})\n", + " m_xx[m_k1][m_k2][m_code]['name'] = sheet.cell(n,5).value\n", + " m_xx[m_k1][m_k2][m_code]['shouyumenlei'] = sheet.cell(n,6).value\n", + " m_xx[m_k1][m_k2][m_code]['xiuyenianxian'] = sheet.cell(n,7).value\n", + " m_nf = sheet.cell(n,8).value\n", + " if m_nf.strip() != '':\n", + " m_xx[m_k1][m_k2][m_code]['zengshenianfen'] = sheet.cell(n,8).value\n", + " \n", + "#print(m_xx)\n", + "'''\n", + "# 导入MongoDB数据库\n", + "for k,v in m_xx.items():\n", + " m_mongo = {} \n", + " m_mongo['code'] = k\n", + " for k1,v1 in v.items():\n", + " m_mongo[k1] = v1\n", + " mycol.insert_one(m_mongo) \n", + "'''\n", + "with open('2020.json','w') as fl2:\n", + " json.dump(m_xx,fl2,ensure_ascii=False) \n", + " \n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 汇总学校录取分数及位次" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 汇总2020年学校录取情况" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-24T02:07:03.946677Z", + "iopub.status.busy": "2020-11-24T02:07:03.945594Z", + "iopub.status.idle": "2020-11-24T02:07:10.521837Z", + "shell.execute_reply": "2020-11-24T02:07:10.519835Z", + "shell.execute_reply.started": "2020-11-24T02:07:03.946417Z" + } + }, + "outputs": [], + "source": [ + "import pymysql\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "m_xx = {}\n", + "m_mongo = {}\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_xx.clear()\n", + " m_code = result[0]\n", + " m_nian ='2020'\n", + " m_lb = 'z'\n", + " m_xx.setdefault(m_nian,{})\n", + " m_xx[m_nian].setdefault(m_lb,{})\n", + " m_xx[m_nian][m_lb] ['name']= '综合'\n", + " m_xx[m_nian][m_lb] ['dispense']= int(result[1])\n", + " m_xx[m_nian][m_lb] ['num_min']= result[2]\n", + " m_xx[m_nian][m_lb] ['rank_min']= result[3]\n", + " #print(m_code,m_xx)\n", + " m_mongo.clear()\n", + " #m_mongo['admission'] = m_xx\n", + " #m_mongo['discipline'] = x['discipline']\n", + " myquery = {'code':m_code}\n", + " m_new = {\"$set\":{'admission':m_xx}}\n", + " mycol.update_one(myquery,m_new)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 汇总2017-2019学校录取情况" + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-24T02:07:21.187718Z", + "iopub.status.busy": "2020-11-24T02:07:21.186750Z", + "iopub.status.idle": "2020-11-24T02:07:29.186812Z", + "shell.execute_reply": "2020-11-24T02:07:29.184357Z", + "shell.execute_reply.started": "2020-11-24T02:07:21.187611Z" + } + }, + "outputs": [], + "source": [ + "import pymysql\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "m_lqrs = {} #录取人数\n", + "m_xx = {}\n", + "dict_lb ={'z':'综合','l':'理科','w':'文科'}\n", + "m_mongo = {}\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT code,year,category,sum(num_act) FROM draft group BY code,year,category ORDER BY code,year,category'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "#生成录取人数字典\n", + "for result in results:\n", + " m_lqrs.setdefault(result[0],{})\n", + " m_lqrs[result[0]].setdefault(result[1],{})\n", + " \n", + " m_lqrs[result[0]][result[1]][result[2]] = result[3]\n", + "#print(m_lqrs)\n", + "sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results: \n", + " m_code = result[0]\n", + " m_nian ='2020'\n", + " m_lb = 'z'\n", + " m_xx.setdefault(m_code,{})\n", + " m_xx[m_code].setdefault(m_nian,{})\n", + " m_xx[m_code][m_nian].setdefault(m_lb,{})\n", + " m_xx[m_code][m_nian][m_lb] ['name']= '综合'\n", + " m_xx[m_code][m_nian][m_lb] ['dispense']= int(result[1])\n", + " m_xx[m_code][m_nian][m_lb] ['num_min']= result[2]\n", + " m_xx[m_code][m_nian][m_lb] ['rank_min']= result[3]\n", + "sql = 'SELECT college,nian,lb,MIN(num_min),max(rank_min) FROM admission where nian not IN (\"2020\") group BY college,nian,lb ORDER BY college,nian desc,lb'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " #m_xx.clear()\n", + " m_code = result[0]\n", + " m_nian = result[1]\n", + " m_lb = result[2]\n", + " m_xx.setdefault(m_code,{})\n", + " m_xx[m_code].setdefault(m_nian,{})\n", + " m_xx[m_code][m_nian].setdefault(m_lb,{})\n", + " m_xx[m_code][m_nian][m_lb] ['name']= dict_lb[m_lb]\n", + " if m_nian in m_lqrs[m_code]:\n", + " m_xx[m_code][m_nian][m_lb] ['dispense']= int(m_lqrs[m_code][m_nian][m_lb])\n", + " m_xx[m_code][m_nian][m_lb] ['num_min']= result[3]\n", + " m_xx[m_code][m_nian][m_lb] ['rank_min']= result[4]\n", + " \n", + "#print(m_xx)\n", + "for k,v in m_xx.items():\n", + " #m_item ='admission.'+m_nian\n", + " m_mongo.clear() \n", + " myquery = {'code':k}\n", + " m_new = {\"$set\":{'admission':v}}\n", + " mycol.update_one(myquery,m_new)\n", + " #print(k,v)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 导入全国普通高等学校名单(截至2020年6月)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "### 以文件形式导入至MongoDB中" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"document\"]\n", + "m_mongo = {}\n", + "fl_name = '全国普通高等学校名单.xlsx'\n", + "m_title = fl_name.split('.')[0]\n", + "m_type = fl_name.split('.')[1]\n", + "#print(m_title,m_type)\n", + "wb = openpyxl.load_workbook(fl_name)\n", + "sheet = wb.active\n", + "m_head = []\n", + "m_body = []\n", + "for cell in sheet[2]:\n", + " m_head.append(cell.value)\n", + "#for n in range(2,sheet.max_row+1):\n", + "#print(m_head) \n", + "for n in range(3,sheet.max_row+1):\n", + " m_cell = []\n", + " for i in range(1,len(m_head)+1):\n", + " m_cell.append(sheet.cell(n,i).value)\n", + " m_body.append(m_cell)\n", + "m_mongo['title'] = m_title\n", + "m_mongo['head'] = m_head\n", + "m_mongo['body'] = m_body\n", + "m_mongo['beizhu'] = '截至2020年6月30日'\n", + "mycol.insert_one(m_mongo) \n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "### 以文档形式导入至MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"gaoxiaomingdan\"]\n", + "m_mongo = {}\n", + "fl_name = '全国普通高等学校名单.xlsx'\n", + "wb = openpyxl.load_workbook(fl_name)\n", + "sheet = wb.active\n", + "for n in range(3,sheet.max_row+1):\n", + " m_mongo ={}\n", + " m_mongo['name'] = sheet.cell(n,2).value \n", + " m_mongo['code'] = str(sheet.cell(n,3).value) \n", + " m_mongo['charge'] = sheet.cell(n,4).value\n", + " m_mongo['city'] = sheet.cell(n,5).value\n", + " m_mongo['grade'] = sheet.cell(n,6).value\n", + " m_mongo['note'] = sheet.cell(n,7).value \n", + " mycol.insert_one(m_mongo) \n", + "print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "### 合并导入高校信息" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "import pymongo\n", + "import openpyxl\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "m_college = {}\n", + "sql = \"select code,name from college\"\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_college[result[1]] =result[0]\n", + "fl_name = '全国普通高等学校名单.xlsx'\n", + "wb = openpyxl.load_workbook(fl_name)\n", + "sheet = wb.active\n", + "for n in range(3,sheet.max_row+1):\n", + " m_mongo ={}\n", + " if sheet.cell(n,2).value in m_college.keys():\n", + " m_mongo['code'] = m_college[sheet.cell(n,2).value]\n", + " m_mongo['name'] = sheet.cell(n,2).value \n", + " m_mongo['id_code'] = str(sheet.cell(n,3).value) \n", + " m_mongo['charge'] = sheet.cell(n,4).value\n", + " m_mongo['city'] = sheet.cell(n,5).value\n", + " m_mongo['grade'] = sheet.cell(n,6).value\n", + " m_mongo['note'] = sheet.cell(n,7).value \n", + " mycol.insert_one(m_mongo)\n", + " #print(m_mongo)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 根据招生编码整理高校信息" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "### 获取mongodb中不存在的学校" + ] + }, + { + "cell_type": "code", + "execution_count": 45, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-22T12:52:46.525924Z", + "iopub.status.busy": "2020-11-22T12:52:46.524992Z", + "iopub.status.idle": "2020-11-22T12:52:52.304174Z", + "shell.execute_reply": "2020-11-22T12:52:52.301880Z", + "shell.execute_reply.started": "2020-11-22T12:52:46.525819Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Y007 山东财经大学\n", + "D687 中国传媒大学南广学院\n", + "D682 西安工业大学北方信息工程学院\n", + "Y002 山东科技大学\n", + "Y060 山东大学威海分校(走读)\n", + "Y061 济南大学(走读)\n", + "Y063 山东财经大学(走读)\n" + ] + } + ], + "source": [ + "import pymysql\n", + "import pymongo\n", + "import openpyxl\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "m_college = {}\n", + "sql = \"select code,name from college\"\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "l_col = ['哈尔滨理工大学','山东大学','山东科技大学','青岛科技大学','济南大学','青岛理工大学','山东师范大学','山东财经大学','青岛大学']\n", + "for result in results:\n", + " m_code = result[0]\n", + " m_name = result[1]\n", + " myquery = {'code':m_code}\n", + " #x = mycol.find_one(myquery)\n", + " if not mycol.find_one(myquery) :\n", + " print(m_code,m_name)\n", + " #if mydoc.count ==0:\n", + " # print(m_name=' 不存在!')\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 单一学校信息录入MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": 43, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-22T12:51:58.582586Z", + "iopub.status.busy": "2020-11-22T12:51:58.581686Z", + "iopub.status.idle": "2020-11-22T12:51:58.614642Z", + "shell.execute_reply": "2020-11-22T12:51:58.612513Z", + "shell.execute_reply.started": "2020-11-22T12:51:58.582483Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "山东科技大学导入成功!\n" + ] + } + ], + "source": [ + "import pymysql\n", + "import pymongo\n", + "import openpyxl\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "m_mongo = {\"code\":'Y022','name':'山东科技大学','charge':'山东省','city':'青岛市','grade':'本科','note':'校企合作,与青软实训教育科技股份有限公司'}\n", + "mycol.insert_one(m_mongo) \n", + "print(m_mongo['name'] + '导入成功!')\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 导入2017-2019年录取数据" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import pymysql\n", + "\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = \"select code,name from college\"\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_xx = []\n", + " m_code = result[0]\n", + " sql1 = 'select xx,zy_dm,zdf,zdwc,pjf,nian,lb from digao_1719 where xx=%s order by nian'\n", + " cursor.execute(sql1,m_code)\n", + " ad_results = cursor.fetchall()\n", + " for ad_result in ad_results:\n", + " m_xx.append(tuple(ad_result))\n", + " sql = \"insert into admission (college,speciality,num_min,rank_min,num_avg,nian,lb) values(%s,%s,%s,%s,%s,%s,%s)\"\n", + " try:\n", + " cursor.executemany(sql,m_xx)\n", + " db.commit()\n", + " print(result[1]+\"已添加!\")\n", + " except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "db.close()\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 从json文件导入2020年未招生学校名称" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import pymysql\n", + "m_col = []\n", + "filename = './data/2020年未招生学校.json'\n", + "with open(filename,'r') as fl:\n", + " m_xx = json.load(fl)\n", + "for k,v in m_xx.items():\n", + " m_col.append((k,v['name']))\n", + "\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "\n", + "sql = \"insert into college (code,name) values(%s,%s)\"\n", + "try:\n", + " cursor.executemany(sql,m_col)\n", + " #db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "\n", + "db.close()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import pymysql\n", + "\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'select xx, zy_dm,zdf from digao_1719 where xx=\"A001\"'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "print(list(results))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 一分一段表导入" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 导入2020年一分一段表" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import openpyxl\n", + "import pymysql\n", + "import json\n", + "\n", + "m_xx = []\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "wb = openpyxl.load_workbook('./data/2020一分一段表.xlsx')\n", + "sheet = wb.active\n", + "i = 1\n", + "for n in range(4,sheet.max_row):\n", + " m_score = sheet.cell(n,1).value\n", + " m_num = sheet.cell(n,14).value\n", + " m_sum = sheet.cell(n,15).value\n", + " m_max = m_sum - m_num + 1\n", + " m_xx.append((m_score,m_num,m_max,m_sum,'z','2020'))\n", + "#print(m_xx)\n", + "sql = 'insert into fenduan(score,per_num,max_rank,min_rank,category,nian) values (%s,%s,%s,%s,%s,%s)'\n", + "try:\n", + " cursor.executemany(sql,m_xx)\n", + " db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "\n", + "db.close()\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 导出一分一段表至JSON文件" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "import json\n", + "\n", + "m_xx = {}\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_year = result[6]\n", + " m_xx.setdefault(m_year,{})\n", + " m_xx[m_year].setdefault(result[5],{})\n", + " m_xx[m_year][result[5]].setdefault(result[1],{})\n", + " m_xx[m_year][result[5]][result[1]]['num_person'] = result[2]\n", + " m_xx[m_year][result[5]][result[1]]['max_rank'] = result[3]\n", + " m_xx[m_year][result[5]][result[1]]['min_rank'] = result[4]\n", + "fl_name = 'data/17-20年一分一段表.json'\n", + "with open(fl_name,'w') as fl:\n", + " json.dump(m_xx,fl,ensure_ascii=False)\n", + "print('ok!')\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 导出一分一段表至MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"fenduan\"]\n", + "m_xx = {}\n", + "\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_year = result[6]\n", + " m_xx.setdefault(m_year,{})\n", + " m_xx[m_year].setdefault(result[5],{})\n", + " m_fenshu = str(result[1])\n", + " m_xx[m_year][result[5]].setdefault(m_fenshu,{})\n", + " m_xx[m_year][result[5]][m_fenshu]['num_person'] = result[2]\n", + " m_xx[m_year][result[5]][m_fenshu]['max_rank'] = result[3]\n", + " m_xx[m_year][result[5]][m_fenshu]['min_rank'] = result[4]\n", + "for k,v in m_xx.items():\n", + " m_mongo = {} \n", + " m_mongo.setdefault(k,{})\n", + " m_mongo[k] = v\n", + " x = mycol.insert_one(m_mongo) \n", + " \n", + "#print('ok!')" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 汇总2020年大学招生录取情况" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "import json\n", + "\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT a.college,sum(a.plan),SUM(a.num_dispense),MIN(a.num_min),max(a.rank_min) FROM admission_2020 AS a GROUP BY a.college'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "#print(results)\n", + "sql = \"insert into admission_coll (college,num_plan,num_act,min_score,min_rank) values(%s,%s,%s,%s,%s)\"\n", + "\n", + "try:\n", + " cursor.executemany(sql,results)\n", + " db.commit()\n", + " print(\"ok!\")\n", + "except:\n", + " # 如果发生错误则回滚\n", + " print(\"error!\")\n", + " db.rollback() \n", + "\n", + "db.close()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 整理以往年度学校录取情况" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "### 数据导入至MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"hz_luqu\"]\n", + "m_xx = {}\n", + "m_mongo = {}\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n", + "left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_code = result[0]\n", + " m_xx.setdefault(m_code,{})\n", + " m_xx[m_code]['code'] = result[0]\n", + " m_xx[m_code]['name'] = result[1]\n", + " m_xx[m_code].setdefault(result[2],{})\n", + " m_xx[m_code][result[2]].setdefault(result[7],{})\n", + " m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n", + " m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n", + " m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n", + " m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n", + " \n", + "for k,v in m_xx.items():\n", + " m_mongo[k] = v\n", + " x = mycol.insert_one(m_mongo[k]) \n", + " print(m_mongo[k])\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "### 数据导入至JSON文件" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymysql\n", + "import json\n", + "\n", + "m_xx = {}\n", + "m_mongo = {}\n", + "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", + "cursor = db.cursor()\n", + "sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n", + "left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n", + "cursor.execute(sql)\n", + "results = cursor.fetchall()\n", + "for result in results:\n", + " m_code = result[0]\n", + " m_xx.setdefault(m_code,{})\n", + " m_xx[m_code]['code'] = result[0]\n", + " m_xx[m_code]['name'] = result[1]\n", + " m_xx[m_code].setdefault(result[2],{})\n", + " m_xx[m_code][result[2]].setdefault(result[7],{})\n", + " m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n", + " m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n", + " m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n", + " m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n", + " \n", + "fl_name = 'data/15-19年高校录取汇总情况.json'\n", + "with open(fl_name,'w') as fl:\n", + " json.dump(m_xx,fl,ensure_ascii=False)\n", + "#print(json.dumps(m_xx,ensure_ascii=False))\n", + "#print(m_xx)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# 高考数据采集" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 新浪高考热讯采集导入MongoDB" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"news\"]\n", + "#web_xx = {}\n", + "n = 0\n", + "for i in range(1,11): \n", + " url = 'http://edu.sina.com.cn/other/roll.d.html?cat=80459&page={}&page_size=30'.format(str(i))\n", + " strhtml = requests.get(url)\n", + " strhtml.encoding = 'utf8'\n", + " soup = BeautifulSoup(strhtml.text,'lxml')\n", + " #data = strhtml.text\n", + " #data1 = data.split(\"\\r\")\n", + " data = soup.select('#Main > div.listBlk > ul > li > a')\n", + " #data = soup.select('#artibody')\n", + " for item in data:\n", + " web_xx.clear()\n", + " n += 1\n", + " m_num = 'sina'+str(n).rjust(6, '0')\n", + " web_xx.setdefault(m_num,{})\n", + " web_xx[m_num]['title'] = item.get_text()\n", + " c_url = item.get('href')\n", + " web_xx[m_num]['url'] = c_url\n", + " content = requests.get(c_url)\n", + " content.encoding = 'utf8'\n", + " soup_content = BeautifulSoup(content.text)\n", + " data1 = soup_content.select('#artibody')\n", + " for item1 in data1:\n", + " web_xx[m_num]['content'] = item1.get_text()\n", + " time.sleep(3)\n", + " mycol.insert_one(web_xx) \n", + " print(m_num+'入库成功!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 中国教育在线信息采集" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 双一流学科录入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "import re\n", + "\n", + "m_mongo = {}\n", + "requests.packages.urllib3.disable_warnings()\n", + "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"syl_jianshexueke\"]\n", + "url = 'http://daxue.eol.cn/syl.shtml'\n", + "strhtml = requests.get(url)\n", + "strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "data = soup.select('body > div.container.box > div.con > div:nth-child(4) > table > tbody > tr')\n", + "i = 0\n", + "for item in data:\n", + " m_mongo.clear()\n", + " data1 = item.select('td')\n", + " ss = re.sub('\\n', '', data1[1].text)\n", + " #print(ss.split('、'))\n", + " m_mongo['college'] = data1[0].text\n", + " m_mongo['discipline'] = ss.split('、')\n", + " mycol.insert_one(m_mongo) \n", + " print(data1[0].text + '导入成功!')\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 双一流学校录入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "import re\n", + "\n", + "m_mongo = {}\n", + "requests.packages.urllib3.disable_warnings()\n", + "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"syl_college\"]\n", + "url = 'http://daxue.eol.cn/syl.shtml'\n", + "strhtml = requests.get(url)\n", + "strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody:nth-child(3) > tr')\n", + "m_name = []\n", + "#for item in data:\n", + "# m_mongo.clear()\n", + "for n in range(1,len(data)):\n", + " item = data[n].select('td')\n", + " ss = re.sub('\\n', ' ', data[n].text)\n", + " m_name = m_name + ss.strip().split(' ')\n", + " \n", + "for col in m_name:\n", + " m_mongo.clear()\n", + " m_mongo['name'] = col\n", + " m_mongo['level'] = 'B'\n", + " mycol.insert_one(m_mongo) " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 双一流学校、学科合并" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymongo\n", + "import re\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college_syl\"]\n", + "mycol1 = mydb[\"syl_college\"]\n", + "mycol2 = mydb[\"syl_jianshexueke\"]\n", + "m_mongo = {}\n", + "for x in mycol1.find({},{'_id':0}):\n", + " m_mongo.clear()\n", + " m_mongo['college'] = x['name']\n", + " m_mongo['level'] = x['level']\n", + " myquery = {'college':x['name']}\n", + " for x1 in mycol2.find(myquery,{'_id':0}):\n", + " m_mongo['discipline'] = x1['discipline']\n", + " mycol.insert_one(m_mongo) \n", + " \n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 双一流学校信息录入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "import re\n", + "\n", + "pattern = re.compile(r'\\d+')\n", + "m_mongo = {}\n", + "requests.packages.urllib3.disable_warnings()\n", + "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"syl_jianshexueke\"]\n", + "url = 'http://daxue.eol.cn/syl.shtml'\n", + "strhtml = requests.get(url)\n", + "strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody > tr > td > a')\n", + "i = 0\n", + "for item in data:\n", + " m_url = item.get('href')\n", + " mm =re.search('\\d+',m_url).group(0)\n", + " m_name = item.text\n", + " url1 = 'http://192.168.3.100:8050/render.html?https://gkcx.eol.cn/school/%s/introDetails' %mm\n", + " \n", + " strhtml = requests.get(url1)\n", + " time.sleep(10)\n", + " soup = BeautifulSoup(strhtml.text,'lxml')\n", + " data1 = soup.select('#root > div > div > div > div > div > div > div.main > div > div.tuiji_left > div.intro_details_content')\n", + " print(url1,m_name)\n", + " print(data1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 985、211学校及学科录入" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import openpyxl\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college_211\"]\n", + "m_mon = {}\n", + "fl_name = './data/985.xlsx'\n", + "wb = openpyxl.load_workbook(fl_name)\n", + "sheet = wb['211']\n", + "for n in range(1,sheet.max_row+1):\n", + " m_xx = []\n", + " m_mon.clear()\n", + " m_mon['college'] = sheet.cell(n,1).value\n", + " m_zy = sheet.cell(n,2).value\n", + " m_mon['discipline'] = m_zy.split('、')\n", + " mycol.insert_one(m_mon) " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 985信息合并入大学信息" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymongo\n", + "import re\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "mycol1 = mydb[\"college_985\"]\n", + "mycol2 = mydb[\"college_syl\"]\n", + "m_mongo = {}\n", + "for x in mycol2.find({},{'_id':0}):\n", + " m_mongo.clear()\n", + " m_mongo['level'] = x['level']\n", + " m_mongo['discipline'] = x['discipline']\n", + " myquery = {'name':x['college']}\n", + " m_new = {\"$set\":{'syl':m_mongo}}\n", + " mycol.update_one(myquery,m_new)\n", + " \n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymongo\n", + "import re\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"college\"]\n", + "myquery = {'name':{\"$regex\":\"\\(\"}}\n", + "for x in mycol.find(myquery,{'_id':0}):\n", + " #s = x['name'].replace('(','(').replace(')',')')\n", + " #myquery1 = {'id_code': x['id_code']}\n", + " #m_new = {\"$set\":{'name':s}}\n", + " #mycol.update_one(myquery1,m_new)\n", + " \n", + " \n", + " print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 山东省考试院新闻采集" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "import re\n", + "\n", + "m_mongo = {}\n", + "requests.packages.urllib3.disable_warnings()\n", + "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"news\"]\n", + "url = 'http://www.sdzk.cn/NewsList.aspx?BCID=2'\n", + "a_url = 'http://www.sdzk.cn/'\n", + "strhtml = requests.get(url)\n", + "strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "data = soup.select('#ctl00_ContentPlaceHolder1_RadAjaxPanel1 > ul > li > a')\n", + "m_name = []\n", + "for item in data:\n", + " m_mongo.clear()\n", + " news_url = a_url + item.get('href')\n", + " content = requests.get(news_url)\n", + " content.encoding = 'utf8'\n", + " soup_content = BeautifulSoup(content.text)\n", + " data1 = soup_content.select('#form1 > div.contain.laylist > div.fl.laylist-r > div > p')\n", + " print(item.text)\n", + " m_txt = ''\n", + " for item1 in data1: \n", + " m_txt += item1.text\n", + " m_txt += '\\n'\n", + " m_mongo['url'] = item.get('href')\n", + " m_mongo['source'] = '山东省教育招生考试院'\n", + " m_mongo['title'] = item.text[:-10]\n", + " m_mongo['date'] = item.text[-10:]\n", + " m_mongo['content'] = m_txt\n", + " mycol.insert_one(m_mongo) " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "toc-hr-collapsed": true, + "toc-nb-collapsed": true + }, + "source": [ + "## 单网页数据采集" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import requests\n", + "url = 'http://localhost:8050/render.html?url=http://gkcx.eol.cn/schoolhtm/schoolTemple/school31.htm&wait=5'\n", + "response = requests.get(url)\n", + "print(response.text)" + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "metadata": { + "execution": { + "iopub.execute_input": "2020-11-27T07:00:38.964152Z", + "iopub.status.busy": "2020-11-27T07:00:38.963212Z", + "iopub.status.idle": "2020-11-27T07:00:39.734632Z", + "shell.execute_reply": "2020-11-27T07:00:39.731755Z", + "shell.execute_reply.started": "2020-11-27T07:00:38.964048Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[

中国人民大学2020年强基计划热点问题解答

]\n" + ] + } + ], + "source": [ + "import requests\n", + "import time\n", + "from bs4 import BeautifulSoup\n", + "import pymongo\n", + "\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"news\"]\n", + "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", + "url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n", + "strhtml = requests.get(url,headers = headers)\n", + "strhtml.encoding = 'utf8'\n", + "soup = BeautifulSoup(strhtml.text,'lxml')\n", + "data = strhtml.text\n", + "#data1 = data.split(\"\\r\")\n", + "data = soup.select('body > div.y_tit_box > div > div > div > h1')\n", + "#for item1 in data:\n", + "# print(item1.get_text())\n", + "print(data)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymongo\n", + "import re\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"news\"]\n", + "mycol_new = mydb[\"news_new\"]\n", + "m_mongo = {}\n", + "for x in mycol.find({},{'_id':0}):\n", + " for k,v in x.items():\n", + " if k[0:4] == 'sina':\n", + " m_mongo.clear()\n", + " m_mongo['url'] = v['url']\n", + " m_mongo['source'] = '新浪高考热讯'\n", + " m_mongo['title'] = v['title']\n", + " m_mongo['date'] = v['url'].split('/')[4]\n", + " m_mongo['content'] = re.sub('\\n\\n','\\n',v['content']) \n", + " mycol_new.insert_one(m_mongo)\n", + " \n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import pymongo\n", + "import re\n", + "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", + "mydb = myclient[\"gaokao\"]\n", + "mycol = mydb[\"news\"]\n", + "mycol_new = mydb[\"news_new\"]\n", + "myquery = {'source':'山东省教育招生考试院'}\n", + "m_mongo = {}\n", + "for x in mycol.find(myquery):\n", + " m_mongo.clear()\n", + " m_mongo['url'] = 'http://www.sdzk.cn/' + x['url']\n", + " m_mongo['source'] = '山东省教育招生考试院'\n", + " m_mongo['title'] = x['title']\n", + " m_mongo['date'] = x['date']\n", + " m_mongo['content'] = re.sub('\\n\\n','\\n',x['content']) \n", + " mycol_new.insert_one(m_mongo)\n", + " #print(m_mongo)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + }, + "toc-autonumbering": true, + "toc-showcode": false, + "toc-showmarkdowntxt": true, + "toc-showtags": false + }, + "nbformat": 4, + "nbformat_minor": 4 +}