{ "cells": [ { "cell_type": "markdown", "metadata": { "tags": [] }, "source": [ "# 高考数据导入" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 导入2012年高校本科专业目录" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import re\n", "m_xx = dict()\n", "fl_name = 'teshe_2012.txt'\n", "with open(fl_name,'r') as fl:\n", " for line in fl:\n", " if line.split():\n", " l = line.split('\\t',1)\n", " #print(l[0]+'--'+re.sub('\\n', '', l[1]))\n", " m_len = len(l[0])\n", " m_name = re.sub('\\n', '', l[1])\n", " if m_len == 2:\n", " m_xx.setdefault(l[0],{})\n", " ll = m_name.split(':',1)\n", " m_xx[l[0]]['name'] = ll[1]\n", " elif m_len == 4:\n", " m_xx[l[0][:2]].setdefault(l[0],{})\n", " m_xx[l[0][:2]][l[0]]['name'] = m_name\n", " else:\n", " m_xx[l[0][:2]][l[0][:4]].setdefault(l[0],{})\n", " m_xx[l[0][:2]][l[0][:4]][l[0]]['code'] = l[0]\n", " m_xx[l[0][:2]][l[0][:4]][l[0]]['name'] = m_name\n", "#print(m_xx)\n", "with open('teshe_2012.json','w') as fl1:\n", " json.dump(m_xx,fl1,ensure_ascii=False) " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 合并2012年目录基本与特设专业" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import re\n", "m_teshe = dict()\n", "m_new = dict()\n", "fl_name = 'jiben_2012.json'\n", "fl_name1 = 'teshe_2012.json'\n", "\n", "with open(fl_name,'r') as fl:\n", " m_new = json.load(fl)\n", "for k,v in m_new.items():\n", " for k1,v1 in v.items():\n", " if k1 !='name':\n", " for k2,v2 in m_new[k][k1].items():\n", " if k2 !='name':\n", " m_new[k][k1][k2]['lb'] = '基本'\n", "\n", "with open(fl_name1,'r') as fl1:\n", " m_teshe = json.load(fl1)\n", "for k,v in m_teshe.items():\n", " for k1,v1 in v.items():\n", " if k1 !='name':\n", " for k2,v2 in m_teshe[k][k1].items():\n", " if k2 !='name':\n", " m_teshe[k][k1][k2]['lb'] = '特设'\n", " m_new[k][k1][k2] = m_teshe[k][k1][k2]\n", "with open('hebing_2012.json','w') as fl2:\n", " json.dump(m_new,fl2,ensure_ascii=False) " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 导入2020年高校本科专业目录" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"zhuanyemulu\"]\n", "m_xx = dict()\n", "fl_name = '普通高等学校本科专业目录.xlsx'\n", "wb = openpyxl.load_workbook(fl_name)\n", "sheet = wb.active\n", "for n in range(2,sheet.max_row+1):\n", " m_code = sheet.cell(n,4).value\n", " m_name = sheet.cell(n,5).value\n", " m_k1 = m_code[:2]\n", " m_k2 = m_code[:4]\n", " m_xx.setdefault(m_k1,{})\n", " m_xx[m_k1]['name'] = sheet.cell(n,2).value\n", " m_xx[m_k1]['level'] = '门类'\n", " m_xx[m_k1].setdefault(m_k2,{})\n", " m_xx[m_k1][m_k2]['name'] = sheet.cell(n,3).value\n", " m_xx[m_k1][m_k2]['level'] ='专业类'\n", " m_xx[m_k1][m_k2].setdefault(m_code,{})\n", " m_xx[m_k1][m_k2][m_code]['name'] = sheet.cell(n,5).value\n", " m_xx[m_k1][m_k2][m_code]['shouyumenlei'] = sheet.cell(n,6).value\n", " m_xx[m_k1][m_k2][m_code]['xiuyenianxian'] = sheet.cell(n,7).value\n", " m_nf = sheet.cell(n,8).value\n", " if m_nf.strip() != '':\n", " m_xx[m_k1][m_k2][m_code]['zengshenianfen'] = sheet.cell(n,8).value\n", " \n", "#print(m_xx)\n", "'''\n", "# 导入MongoDB数据库\n", "for k,v in m_xx.items():\n", " m_mongo = {} \n", " m_mongo['code'] = k\n", " for k1,v1 in v.items():\n", " m_mongo[k1] = v1\n", " mycol.insert_one(m_mongo) \n", "'''\n", "with open('2020.json','w') as fl2:\n", " json.dump(m_xx,fl2,ensure_ascii=False) \n", " \n", " " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 汇总学校录取分数及位次" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 汇总2020年学校录取情况" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "m_xx = {}\n", "m_mongo = {}\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_xx.clear()\n", " m_code = result[0]\n", " m_nian ='2020'\n", " m_lb = 'z'\n", " m_xx.setdefault(m_nian,{})\n", " m_xx[m_nian].setdefault(m_lb,{})\n", " m_xx[m_nian][m_lb] ['name']= '综合'\n", " m_xx[m_nian][m_lb] ['dispense']= int(result[1])\n", " m_xx[m_nian][m_lb] ['num_min']= result[2]\n", " m_xx[m_nian][m_lb] ['rank_min']= result[3]\n", " #print(m_code,m_xx)\n", " m_mongo.clear()\n", " #m_mongo['admission'] = m_xx\n", " #m_mongo['discipline'] = x['discipline']\n", " myquery = {'code':m_code}\n", " m_new = {\"$set\":{'admission':m_xx}}\n", " mycol.update_one(myquery,m_new)\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 汇总2017-2019学校录取情况" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "m_lqrs = {} #录取人数\n", "m_xx = {}\n", "dict_lb ={'z':'综合','l':'理科','w':'文科'}\n", "m_mongo = {}\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT code,year,category,sum(num_act) FROM draft group BY code,year,category ORDER BY code,year,category'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "#生成录取人数字典\n", "for result in results:\n", " m_lqrs.setdefault(result[0],{})\n", " m_lqrs[result[0]].setdefault(result[1],{})\n", " \n", " m_lqrs[result[0]][result[1]][result[2]] = result[3]\n", "#print(m_lqrs)\n", "sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results: \n", " m_code = result[0]\n", " m_nian ='2020'\n", " m_lb = 'z'\n", " m_xx.setdefault(m_code,{})\n", " m_xx[m_code].setdefault(m_nian,{})\n", " m_xx[m_code][m_nian].setdefault(m_lb,{})\n", " m_xx[m_code][m_nian][m_lb] ['name']= '综合'\n", " m_xx[m_code][m_nian][m_lb] ['dispense']= int(result[1])\n", " m_xx[m_code][m_nian][m_lb] ['num_min']= result[2]\n", " m_xx[m_code][m_nian][m_lb] ['rank_min']= result[3]\n", "sql = 'SELECT college,nian,lb,MIN(num_min),max(rank_min) FROM admission where nian not IN (\"2020\") group BY college,nian,lb ORDER BY college,nian desc,lb'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " #m_xx.clear()\n", " m_code = result[0]\n", " m_nian = result[1]\n", " m_lb = result[2]\n", " m_xx.setdefault(m_code,{})\n", " m_xx[m_code].setdefault(m_nian,{})\n", " m_xx[m_code][m_nian].setdefault(m_lb,{})\n", " m_xx[m_code][m_nian][m_lb] ['name']= dict_lb[m_lb]\n", " if m_nian in m_lqrs[m_code]:\n", " m_xx[m_code][m_nian][m_lb] ['dispense']= int(m_lqrs[m_code][m_nian][m_lb])\n", " m_xx[m_code][m_nian][m_lb] ['num_min']= result[3]\n", " m_xx[m_code][m_nian][m_lb] ['rank_min']= result[4]\n", " \n", "#print(m_xx)\n", "for k,v in m_xx.items():\n", " #m_item ='admission.'+m_nian\n", " m_mongo.clear() \n", " myquery = {'code':k}\n", " m_new = {\"$set\":{'admission':v}}\n", " mycol.update_one(myquery,m_new)\n", " #print(k,v)" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 导入全国普通高等学校名单(截至2020年6月)" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "### 以文件形式导入至MongoDB中" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"document\"]\n", "m_mongo = {}\n", "fl_name = '全国普通高等学校名单.xlsx'\n", "m_title = fl_name.split('.')[0]\n", "m_type = fl_name.split('.')[1]\n", "#print(m_title,m_type)\n", "wb = openpyxl.load_workbook(fl_name)\n", "sheet = wb.active\n", "m_head = []\n", "m_body = []\n", "for cell in sheet[2]:\n", " m_head.append(cell.value)\n", "#for n in range(2,sheet.max_row+1):\n", "#print(m_head) \n", "for n in range(3,sheet.max_row+1):\n", " m_cell = []\n", " for i in range(1,len(m_head)+1):\n", " m_cell.append(sheet.cell(n,i).value)\n", " m_body.append(m_cell)\n", "m_mongo['title'] = m_title\n", "m_mongo['head'] = m_head\n", "m_mongo['body'] = m_body\n", "m_mongo['beizhu'] = '截至2020年6月30日'\n", "mycol.insert_one(m_mongo) \n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "### 以文档形式导入至MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"gaoxiaomingdan\"]\n", "m_mongo = {}\n", "fl_name = '全国普通高等学校名单.xlsx'\n", "wb = openpyxl.load_workbook(fl_name)\n", "sheet = wb.active\n", "for n in range(3,sheet.max_row+1):\n", " m_mongo ={}\n", " m_mongo['name'] = sheet.cell(n,2).value \n", " m_mongo['code'] = str(sheet.cell(n,3).value) \n", " m_mongo['charge'] = sheet.cell(n,4).value\n", " m_mongo['city'] = sheet.cell(n,5).value\n", " m_mongo['grade'] = sheet.cell(n,6).value\n", " m_mongo['note'] = sheet.cell(n,7).value \n", " mycol.insert_one(m_mongo) \n", "print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "### 合并导入高校信息" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "import openpyxl\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "m_college = {}\n", "sql = \"select code,name from college\"\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_college[result[1]] =result[0]\n", "fl_name = '全国普通高等学校名单.xlsx'\n", "wb = openpyxl.load_workbook(fl_name)\n", "sheet = wb.active\n", "for n in range(3,sheet.max_row+1):\n", " m_mongo ={}\n", " if sheet.cell(n,2).value in m_college.keys():\n", " m_mongo['code'] = m_college[sheet.cell(n,2).value]\n", " m_mongo['name'] = sheet.cell(n,2).value \n", " m_mongo['id_code'] = str(sheet.cell(n,3).value) \n", " m_mongo['charge'] = sheet.cell(n,4).value\n", " m_mongo['city'] = sheet.cell(n,5).value\n", " m_mongo['grade'] = sheet.cell(n,6).value\n", " m_mongo['note'] = sheet.cell(n,7).value \n", " mycol.insert_one(m_mongo)\n", " #print(m_mongo)" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 根据招生编码整理高校信息" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "### 获取mongodb中不存在的学校" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "import openpyxl\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "m_college = {}\n", "sql = \"select code,name from college\"\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "l_col = ['哈尔滨理工大学','山东大学','山东科技大学','青岛科技大学','济南大学','青岛理工大学','山东师范大学','山东财经大学','青岛大学']\n", "for result in results:\n", " m_code = result[0]\n", " m_name = result[1]\n", " myquery = {'code':m_code}\n", " #x = mycol.find_one(myquery)\n", " if not mycol.find_one(myquery) :\n", " print(m_code,m_name)\n", " #if mydoc.count ==0:\n", " # print(m_name=' 不存在!')\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 单一学校信息录入MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "import openpyxl\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "m_mongo = {\"code\":'Y022','name':'山东科技大学','charge':'山东省','city':'青岛市','grade':'本科','note':'校企合作,与青软实训教育科技股份有限公司'}\n", "mycol.insert_one(m_mongo) \n", "print(m_mongo['name'] + '导入成功!')\n" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 导入2017-2019年录取数据" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import pymysql\n", "\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = \"select code,name from college\"\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_xx = []\n", " m_code = result[0]\n", " sql1 = 'select xx,zy_dm,zdf,zdwc,pjf,nian,lb from digao_1719 where xx=%s order by nian'\n", " cursor.execute(sql1,m_code)\n", " ad_results = cursor.fetchall()\n", " for ad_result in ad_results:\n", " m_xx.append(tuple(ad_result))\n", " sql = \"insert into admission (college,speciality,num_min,rank_min,num_avg,nian,lb) values(%s,%s,%s,%s,%s,%s,%s)\"\n", " try:\n", " cursor.executemany(sql,m_xx)\n", " db.commit()\n", " print(result[1]+\"已添加!\")\n", " except:\n", " # 如果发生错误则回滚\n", " print(\"error!\")\n", " db.rollback() \n", "db.close()\n" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 从json文件导入2020年未招生学校名称" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import pymysql\n", "m_col = []\n", "filename = './data/2020年未招生学校.json'\n", "with open(filename,'r') as fl:\n", " m_xx = json.load(fl)\n", "for k,v in m_xx.items():\n", " m_col.append((k,v['name']))\n", "\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "\n", "sql = \"insert into college (code,name) values(%s,%s)\"\n", "try:\n", " cursor.executemany(sql,m_col)\n", " #db.commit()\n", " print(\"ok!\")\n", "except:\n", " # 如果发生错误则回滚\n", " print(\"error!\")\n", " db.rollback() \n", "\n", "db.close()" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import pymysql\n", "\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'select xx, zy_dm,zdf from digao_1719 where xx=\"A001\"'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "print(list(results))" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 一分一段表导入" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 导入2020年一分一段表" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import openpyxl\n", "import pymysql\n", "import json\n", "\n", "m_xx = []\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "wb = openpyxl.load_workbook('./data/2020一分一段表.xlsx')\n", "sheet = wb.active\n", "i = 1\n", "for n in range(4,sheet.max_row):\n", " m_score = sheet.cell(n,1).value\n", " m_num = sheet.cell(n,14).value\n", " m_sum = sheet.cell(n,15).value\n", " m_max = m_sum - m_num + 1\n", " m_xx.append((m_score,m_num,m_max,m_sum,'z','2020'))\n", "#print(m_xx)\n", "sql = 'insert into fenduan(score,per_num,max_rank,min_rank,category,nian) values (%s,%s,%s,%s,%s,%s)'\n", "try:\n", " cursor.executemany(sql,m_xx)\n", " db.commit()\n", " print(\"ok!\")\n", "except:\n", " # 如果发生错误则回滚\n", " print(\"error!\")\n", " db.rollback() \n", "\n", "db.close()\n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 导出一分一段表至JSON文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import json\n", "\n", "m_xx = {}\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_year = result[6]\n", " m_xx.setdefault(m_year,{})\n", " m_xx[m_year].setdefault(result[5],{})\n", " m_xx[m_year][result[5]].setdefault(result[1],{})\n", " m_xx[m_year][result[5]][result[1]]['num_person'] = result[2]\n", " m_xx[m_year][result[5]][result[1]]['max_rank'] = result[3]\n", " m_xx[m_year][result[5]][result[1]]['min_rank'] = result[4]\n", "fl_name = 'data/17-20年一分一段表.json'\n", "with open(fl_name,'w') as fl:\n", " json.dump(m_xx,fl,ensure_ascii=False)\n", "print('ok!')\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 导出一分一段表至MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"fenduan\"]\n", "m_xx = {}\n", "\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_year = result[6]\n", " m_xx.setdefault(m_year,{})\n", " m_xx[m_year].setdefault(result[5],{})\n", " m_fenshu = str(result[1])\n", " m_xx[m_year][result[5]].setdefault(m_fenshu,{})\n", " m_xx[m_year][result[5]][m_fenshu]['num_person'] = result[2]\n", " m_xx[m_year][result[5]][m_fenshu]['max_rank'] = result[3]\n", " m_xx[m_year][result[5]][m_fenshu]['min_rank'] = result[4]\n", "for k,v in m_xx.items():\n", " m_mongo = {} \n", " m_mongo.setdefault(k,{})\n", " m_mongo[k] = v\n", " x = mycol.insert_one(m_mongo) \n", " \n", "#print('ok!')" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 汇总2020年大学招生录取情况" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import json\n", "\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT a.college,sum(a.plan),SUM(a.num_dispense),MIN(a.num_min),max(a.rank_min) FROM admission_2020 AS a GROUP BY a.college'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "#print(results)\n", "sql = \"insert into admission_coll (college,num_plan,num_act,min_score,min_rank) values(%s,%s,%s,%s,%s)\"\n", "\n", "try:\n", " cursor.executemany(sql,results)\n", " db.commit()\n", " print(\"ok!\")\n", "except:\n", " # 如果发生错误则回滚\n", " print(\"error!\")\n", " db.rollback() \n", "\n", "db.close()" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 整理以往年度学校录取情况" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "### 数据导入至MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"hz_luqu\"]\n", "m_xx = {}\n", "m_mongo = {}\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n", "left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_code = result[0]\n", " m_xx.setdefault(m_code,{})\n", " m_xx[m_code]['code'] = result[0]\n", " m_xx[m_code]['name'] = result[1]\n", " m_xx[m_code].setdefault(result[2],{})\n", " m_xx[m_code][result[2]].setdefault(result[7],{})\n", " m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n", " m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n", " m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n", " m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n", " \n", "for k,v in m_xx.items():\n", " m_mongo[k] = v\n", " x = mycol.insert_one(m_mongo[k]) \n", " print(m_mongo[k])\n", "\n" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "### 数据导入至JSON文件" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymysql\n", "import json\n", "\n", "m_xx = {}\n", "m_mongo = {}\n", "db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n", "left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_code = result[0]\n", " m_xx.setdefault(m_code,{})\n", " m_xx[m_code]['code'] = result[0]\n", " m_xx[m_code]['name'] = result[1]\n", " m_xx[m_code].setdefault(result[2],{})\n", " m_xx[m_code][result[2]].setdefault(result[7],{})\n", " m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n", " m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n", " m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n", " m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n", " \n", "fl_name = 'data/15-19年高校录取汇总情况.json'\n", "with open(fl_name,'w') as fl:\n", " json.dump(m_xx,fl,ensure_ascii=False)\n", "#print(json.dumps(m_xx,ensure_ascii=False))\n", "#print(m_xx)\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# 高考数据采集" ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 新浪高考热讯采集导入MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"news\"]\n", "#web_xx = {}\n", "n = 0\n", "for i in range(1,11): \n", " url = 'http://edu.sina.com.cn/other/roll.d.html?cat=80459&page={}&page_size=30'.format(str(i))\n", " strhtml = requests.get(url)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " #data = strhtml.text\n", " #data1 = data.split(\"\\r\")\n", " data = soup.select('#Main > div.listBlk > ul > li > a')\n", " #data = soup.select('#artibody')\n", " for item in data:\n", " web_xx.clear()\n", " n += 1\n", " m_num = 'sina'+str(n).rjust(6, '0')\n", " web_xx.setdefault(m_num,{})\n", " web_xx[m_num]['title'] = item.get_text()\n", " c_url = item.get('href')\n", " web_xx[m_num]['url'] = c_url\n", " content = requests.get(c_url)\n", " content.encoding = 'utf8'\n", " soup_content = BeautifulSoup(content.text)\n", " data1 = soup_content.select('#artibody')\n", " for item1 in data1:\n", " web_xx[m_num]['content'] = item1.get_text()\n", " time.sleep(3)\n", " mycol.insert_one(web_xx) \n", " print(m_num+'入库成功!')" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 中国教育在线信息采集" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 双一流学科录入" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import re\n", "\n", "m_mongo = {}\n", "requests.packages.urllib3.disable_warnings()\n", "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"syl_jianshexueke\"]\n", "url = 'http://daxue.eol.cn/syl.shtml'\n", "strhtml = requests.get(url)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "data = soup.select('body > div.container.box > div.con > div:nth-child(4) > table > tbody > tr')\n", "i = 0\n", "for item in data:\n", " m_mongo.clear()\n", " data1 = item.select('td')\n", " ss = re.sub('\\n', '', data1[1].text)\n", " #print(ss.split('、'))\n", " m_mongo['college'] = data1[0].text\n", " m_mongo['discipline'] = ss.split('、')\n", " mycol.insert_one(m_mongo) \n", " print(data1[0].text + '导入成功!')\n", " " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 双一流学校录入" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import re\n", "\n", "m_mongo = {}\n", "requests.packages.urllib3.disable_warnings()\n", "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"syl_college\"]\n", "url = 'http://daxue.eol.cn/syl.shtml'\n", "strhtml = requests.get(url)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody:nth-child(3) > tr')\n", "m_name = []\n", "#for item in data:\n", "# m_mongo.clear()\n", "for n in range(1,len(data)):\n", " item = data[n].select('td')\n", " ss = re.sub('\\n', ' ', data[n].text)\n", " m_name = m_name + ss.strip().split(' ')\n", " \n", "for col in m_name:\n", " m_mongo.clear()\n", " m_mongo['name'] = col\n", " m_mongo['level'] = 'B'\n", " mycol.insert_one(m_mongo) " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 双一流学校、学科合并" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college_syl\"]\n", "mycol1 = mydb[\"syl_college\"]\n", "mycol2 = mydb[\"syl_jianshexueke\"]\n", "m_mongo = {}\n", "for x in mycol1.find({},{'_id':0}):\n", " m_mongo.clear()\n", " m_mongo['college'] = x['name']\n", " m_mongo['level'] = x['level']\n", " myquery = {'college':x['name']}\n", " for x1 in mycol2.find(myquery,{'_id':0}):\n", " m_mongo['discipline'] = x1['discipline']\n", " mycol.insert_one(m_mongo) \n", " \n", " " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 双一流学校信息录入" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import re\n", "\n", "pattern = re.compile(r'\\d+')\n", "m_mongo = {}\n", "requests.packages.urllib3.disable_warnings()\n", "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"syl_jianshexueke\"]\n", "url = 'http://daxue.eol.cn/syl.shtml'\n", "strhtml = requests.get(url)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody > tr > td > a')\n", "i = 0\n", "for item in data:\n", " m_url = item.get('href')\n", " mm =re.search('\\d+',m_url).group(0)\n", " m_name = item.text\n", " url1 = 'http://192.168.3.100:8050/render.html?https://gkcx.eol.cn/school/%s/introDetails' %mm\n", " \n", " strhtml = requests.get(url1)\n", " time.sleep(10)\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data1 = soup.select('#root > div > div > div > div > div > div > div.main > div > div.tuiji_left > div.intro_details_content')\n", " print(url1,m_name)\n", " print(data1)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 985、211学校及学科录入" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import json\n", "import openpyxl\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college_211\"]\n", "m_mon = {}\n", "fl_name = './data/985.xlsx'\n", "wb = openpyxl.load_workbook(fl_name)\n", "sheet = wb['211']\n", "for n in range(1,sheet.max_row+1):\n", " m_xx = []\n", " m_mon.clear()\n", " m_mon['college'] = sheet.cell(n,1).value\n", " m_zy = sheet.cell(n,2).value\n", " m_mon['discipline'] = m_zy.split('、')\n", " mycol.insert_one(m_mon) " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 985信息合并入大学信息" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "mycol1 = mydb[\"college_985\"]\n", "mycol2 = mydb[\"college_syl\"]\n", "m_mongo = {}\n", "for x in mycol2.find({},{'_id':0}):\n", " m_mongo.clear()\n", " m_mongo['level'] = x['level']\n", " m_mongo['discipline'] = x['discipline']\n", " myquery = {'name':x['college']}\n", " m_new = {\"$set\":{'syl':m_mongo}}\n", " mycol.update_one(myquery,m_new)\n", " \n", " " ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "myquery = {'name':{\"$regex\":\"\\(\"}}\n", "for x in mycol.find(myquery,{'_id':0}):\n", " #s = x['name'].replace('(','(').replace(')',')')\n", " #myquery1 = {'id_code': x['id_code']}\n", " #m_new = {\"$set\":{'name':s}}\n", " #mycol.update_one(myquery1,m_new)\n", " \n", " \n", " print(x)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 山东省考试院新闻采集" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import re\n", "\n", "m_mongo = {}\n", "requests.packages.urllib3.disable_warnings()\n", "requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"news\"]\n", "url = 'http://www.sdzk.cn/NewsList.aspx?BCID=2'\n", "a_url = 'http://www.sdzk.cn/'\n", "strhtml = requests.get(url)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "data = soup.select('#ctl00_ContentPlaceHolder1_RadAjaxPanel1 > ul > li > a')\n", "m_name = []\n", "for item in data:\n", " m_mongo.clear()\n", " news_url = a_url + item.get('href')\n", " content = requests.get(news_url)\n", " content.encoding = 'utf8'\n", " soup_content = BeautifulSoup(content.text)\n", " data1 = soup_content.select('#form1 > div.contain.laylist > div.fl.laylist-r > div > p')\n", " print(item.text)\n", " m_txt = ''\n", " for item1 in data1: \n", " m_txt += item1.text\n", " m_txt += '\\n'\n", " m_mongo['url'] = item.get('href')\n", " m_mongo['source'] = '山东省教育招生考试院'\n", " m_mongo['title'] = item.text[:-10]\n", " m_mongo['date'] = item.text[-10:]\n", " m_mongo['content'] = m_txt\n", " mycol.insert_one(m_mongo) " ] }, { "cell_type": "markdown", "metadata": { "toc-hr-collapsed": true, "toc-nb-collapsed": true }, "source": [ "## 单网页数据采集" ] }, { "cell_type": "code", "execution_count": 1, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", "\n", " \n", " \n", "\n", " \n", " \n", " \n", "\n", "\n", " \n", " \n", " \n", "\n", "\n", "\n", " \n", "\n", " \n", " \n", " \n", " \n", "\n", "\n", "\n", " \n", " \n", "\n", " \n", " \n", "\n", " \n", " \n", " \n", "\n", " \n", "\n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", "\n", " \n", " \n", "\n", "\n", "\n", "\n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "\n", " \n", "\n", " \n", " \n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "\n", " \n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "\n", " \n", " \n", " \n", "\n", " \n", "\n", " \n", "\n", " \n", "\n", " \n", " \n", " \n", " \n", "\n", "\n", "\n", "\n", " \n", "\n", "\n", "\n", " \n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "ABOUT US | MOVE Sports Physio\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "
 
\"\"
\"Intro
\"Girl
\"Screen
\"trail

MOVE is a contemporary physiotherapy and rehabilitation

\n", "\n", "

practice situated in the heart of Ballito overlooking the rolling cane

\n", "\n", "

farms of The Dolphin Coast.

\n", "\n", "

 

\n", "\n", "

We provide a focussed and comprehensive service for people of all ages to have their aches, pains, sports injuries and post-operative rehabilitation performed under one roof. Our clients of all ages and abilities can’t help but feel comfortable and relaxed, whilst being professionally managed by qualified and experienced HPCSA-registered physiotherapists and biokineticists throughout the course of their recovery and rehabilitation.

\n", "\n", "

 

\n", "\n", "

Our mission is to develop our clients' understanding and greater ownership of their musculoskeletal condition in order to take back control of their health and wellbeing. All stakeholders can be assured of receiving evidence-based treatment and rehabilitation treatments and programmes.

\n", "\n", "

 

\n", "\n", "

We know how important it is for you to regain your health as quickly

\n", "\n", "

as possible, and in line with your own goals. Our extensive

\n", "\n", "

international experience and our own personal sports

\n", "\n", "

background make us the ideal place to get you

\n", "\n", "

‘back on track’ as quickly as possible.

Our Practice

\"Mountain
\"Screen

Meet The Team

\"Andrew

Andrew Leipus

Sports & Musculoskeletal Physiotherapist

Physiotherapist

\"Mieke

Kaitlin Holgate

Mieke Potgieter

Biokineticist / Sports & Therapeutic Massage

\"India

Musculoskeletal Physiotherapy

\n", "\n", "

Sports & Exercise Physiotherapy

\n", "\n", "

Orthopaedic Physiotherapy & Rehabilitation

\n", "\n", "

Injury Prevention & Screening

\n", "\n", "

Biomechanical & Movement Analysis

\n", "\n", "

Manual Therapies

\n", "\n", "

Exercise Rehabilitation

\n", "\n", "

Shockwave Therapy

\n", "\n", "

Hydrotherapy

\n", "\n", "

Sports Massage 

\n", "\n", "

Private Rehabilitation Gym

\n", "\n", "

Taping

\n", "\n", "

Needling

\n", "\n", "

Clinical Pilates

Our Services

\"Kait
\"HPCSA
\"SASMA
\"SPSA
\"SASP
\"APA_M_H_POS_CMYK.jpg\"
\"SASMA

COPYRIGHT ​© 2019 MOVE Sports Physio & Rehab, Ballito, South Africa

 
\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n", "\n" ] } ], "source": [ "import requests\n", "url = 'https://www.movephysio.co.za/'\n", "response = requests.get(url)\n", "print(response.text)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"news\"]\n", "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", "url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n", "strhtml = requests.get(url,headers = headers)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "data = strhtml.text\n", "#data1 = data.split(\"\\r\")\n", "data = soup.select('body > div.y_tit_box > div > div > div > h1')\n", "#for item1 in data:\n", "# print(item1.get_text())\n", "print(data)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"news\"]\n", "mycol_new = mydb[\"news_new\"]\n", "m_mongo = {}\n", "for x in mycol.find({},{'_id':0}):\n", " for k,v in x.items():\n", " if k[0:4] == 'sina':\n", " m_mongo.clear()\n", " m_mongo['url'] = v['url']\n", " m_mongo['source'] = '新浪高考热讯'\n", " m_mongo['title'] = v['title']\n", " m_mongo['date'] = v['url'].split('/')[4]\n", " m_mongo['content'] = re.sub('\\n\\n','\\n',v['content']) \n", " mycol_new.insert_one(m_mongo)\n", " \n" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "import re\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"news\"]\n", "mycol_new = mydb[\"news_new\"]\n", "myquery = {'source':'山东省教育招生考试院'}\n", "m_mongo = {}\n", "for x in mycol.find(myquery):\n", " m_mongo.clear()\n", " m_mongo['url'] = 'http://www.sdzk.cn/' + x['url']\n", " m_mongo['source'] = '山东省教育招生考试院'\n", " m_mongo['title'] = x['title']\n", " m_mongo['date'] = x['date']\n", " m_mongo['content'] = re.sub('\\n\\n','\\n',x['content']) \n", " mycol_new.insert_one(m_mongo)\n", " #print(m_mongo)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.8.10" }, "toc-autonumbering": true, "toc-showcode": false, "toc-showmarkdowntxt": true, "toc-showtags": false }, "nbformat": 4, "nbformat_minor": 4 }