1473 lines
44 KiB
Plaintext
1473 lines
44 KiB
Plaintext
{
|
||
"cells": [
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"# 高考数据导入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入2012年高校本科专业目录"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import re\n",
|
||
"m_xx = dict()\n",
|
||
"fl_name = 'teshe_2012.txt'\n",
|
||
"with open(fl_name,'r') as fl:\n",
|
||
" for line in fl:\n",
|
||
" if line.split():\n",
|
||
" l = line.split('\\t',1)\n",
|
||
" #print(l[0]+'--'+re.sub('\\n', '', l[1]))\n",
|
||
" m_len = len(l[0])\n",
|
||
" m_name = re.sub('\\n', '', l[1])\n",
|
||
" if m_len == 2:\n",
|
||
" m_xx.setdefault(l[0],{})\n",
|
||
" ll = m_name.split(':',1)\n",
|
||
" m_xx[l[0]]['name'] = ll[1]\n",
|
||
" elif m_len == 4:\n",
|
||
" m_xx[l[0][:2]].setdefault(l[0],{})\n",
|
||
" m_xx[l[0][:2]][l[0]]['name'] = m_name\n",
|
||
" else:\n",
|
||
" m_xx[l[0][:2]][l[0][:4]].setdefault(l[0],{})\n",
|
||
" m_xx[l[0][:2]][l[0][:4]][l[0]]['code'] = l[0]\n",
|
||
" m_xx[l[0][:2]][l[0][:4]][l[0]]['name'] = m_name\n",
|
||
"#print(m_xx)\n",
|
||
"with open('teshe_2012.json','w') as fl1:\n",
|
||
" json.dump(m_xx,fl1,ensure_ascii=False) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 合并2012年目录基本与特设专业"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import re\n",
|
||
"m_teshe = dict()\n",
|
||
"m_new = dict()\n",
|
||
"fl_name = 'jiben_2012.json'\n",
|
||
"fl_name1 = 'teshe_2012.json'\n",
|
||
"\n",
|
||
"with open(fl_name,'r') as fl:\n",
|
||
" m_new = json.load(fl)\n",
|
||
"for k,v in m_new.items():\n",
|
||
" for k1,v1 in v.items():\n",
|
||
" if k1 !='name':\n",
|
||
" for k2,v2 in m_new[k][k1].items():\n",
|
||
" if k2 !='name':\n",
|
||
" m_new[k][k1][k2]['lb'] = '基本'\n",
|
||
"\n",
|
||
"with open(fl_name1,'r') as fl1:\n",
|
||
" m_teshe = json.load(fl1)\n",
|
||
"for k,v in m_teshe.items():\n",
|
||
" for k1,v1 in v.items():\n",
|
||
" if k1 !='name':\n",
|
||
" for k2,v2 in m_teshe[k][k1].items():\n",
|
||
" if k2 !='name':\n",
|
||
" m_teshe[k][k1][k2]['lb'] = '特设'\n",
|
||
" m_new[k][k1][k2] = m_teshe[k][k1][k2]\n",
|
||
"with open('hebing_2012.json','w') as fl2:\n",
|
||
" json.dump(m_new,fl2,ensure_ascii=False) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入2020年高校本科专业目录"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"zhuanyemulu\"]\n",
|
||
"m_xx = dict()\n",
|
||
"fl_name = '普通高等学校本科专业目录.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(2,sheet.max_row+1):\n",
|
||
" m_code = sheet.cell(n,4).value\n",
|
||
" m_name = sheet.cell(n,5).value\n",
|
||
" m_k1 = m_code[:2]\n",
|
||
" m_k2 = m_code[:4]\n",
|
||
" m_xx.setdefault(m_k1,{})\n",
|
||
" m_xx[m_k1]['name'] = sheet.cell(n,2).value\n",
|
||
" m_xx[m_k1]['level'] = '门类'\n",
|
||
" m_xx[m_k1].setdefault(m_k2,{})\n",
|
||
" m_xx[m_k1][m_k2]['name'] = sheet.cell(n,3).value\n",
|
||
" m_xx[m_k1][m_k2]['level'] ='专业类'\n",
|
||
" m_xx[m_k1][m_k2].setdefault(m_code,{})\n",
|
||
" m_xx[m_k1][m_k2][m_code]['name'] = sheet.cell(n,5).value\n",
|
||
" m_xx[m_k1][m_k2][m_code]['shouyumenlei'] = sheet.cell(n,6).value\n",
|
||
" m_xx[m_k1][m_k2][m_code]['xiuyenianxian'] = sheet.cell(n,7).value\n",
|
||
" m_nf = sheet.cell(n,8).value\n",
|
||
" if m_nf.strip() != '':\n",
|
||
" m_xx[m_k1][m_k2][m_code]['zengshenianfen'] = sheet.cell(n,8).value\n",
|
||
" \n",
|
||
"#print(m_xx)\n",
|
||
"'''\n",
|
||
"# 导入MongoDB数据库\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_mongo = {} \n",
|
||
" m_mongo['code'] = k\n",
|
||
" for k1,v1 in v.items():\n",
|
||
" m_mongo[k1] = v1\n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
"'''\n",
|
||
"with open('2020.json','w') as fl2:\n",
|
||
" json.dump(m_xx,fl2,ensure_ascii=False) \n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 汇总学校录取分数及位次"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 汇总2020年学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 1,
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2020-11-24T02:07:03.946677Z",
|
||
"iopub.status.busy": "2020-11-24T02:07:03.945594Z",
|
||
"iopub.status.idle": "2020-11-24T02:07:10.521837Z",
|
||
"shell.execute_reply": "2020-11-24T02:07:10.519835Z",
|
||
"shell.execute_reply.started": "2020-11-24T02:07:03.946417Z"
|
||
}
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"m_xx = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_xx.clear()\n",
|
||
" m_code = result[0]\n",
|
||
" m_nian ='2020'\n",
|
||
" m_lb = 'z'\n",
|
||
" m_xx.setdefault(m_nian,{})\n",
|
||
" m_xx[m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_nian][m_lb] ['name']= '综合'\n",
|
||
" m_xx[m_nian][m_lb] ['dispense']= int(result[1])\n",
|
||
" m_xx[m_nian][m_lb] ['num_min']= result[2]\n",
|
||
" m_xx[m_nian][m_lb] ['rank_min']= result[3]\n",
|
||
" #print(m_code,m_xx)\n",
|
||
" m_mongo.clear()\n",
|
||
" #m_mongo['admission'] = m_xx\n",
|
||
" #m_mongo['discipline'] = x['discipline']\n",
|
||
" myquery = {'code':m_code}\n",
|
||
" m_new = {\"$set\":{'admission':m_xx}}\n",
|
||
" mycol.update_one(myquery,m_new)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 汇总2017-2019学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 2,
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2020-11-24T02:07:21.187718Z",
|
||
"iopub.status.busy": "2020-11-24T02:07:21.186750Z",
|
||
"iopub.status.idle": "2020-11-24T02:07:29.186812Z",
|
||
"shell.execute_reply": "2020-11-24T02:07:29.184357Z",
|
||
"shell.execute_reply.started": "2020-11-24T02:07:21.187611Z"
|
||
}
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"m_lqrs = {} #录取人数\n",
|
||
"m_xx = {}\n",
|
||
"dict_lb ={'z':'综合','l':'理科','w':'文科'}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT code,year,category,sum(num_act) FROM draft group BY code,year,category ORDER BY code,year,category'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"#生成录取人数字典\n",
|
||
"for result in results:\n",
|
||
" m_lqrs.setdefault(result[0],{})\n",
|
||
" m_lqrs[result[0]].setdefault(result[1],{})\n",
|
||
" \n",
|
||
" m_lqrs[result[0]][result[1]][result[2]] = result[3]\n",
|
||
"#print(m_lqrs)\n",
|
||
"sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results: \n",
|
||
" m_code = result[0]\n",
|
||
" m_nian ='2020'\n",
|
||
" m_lb = 'z'\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code].setdefault(m_nian,{})\n",
|
||
" m_xx[m_code][m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['name']= '综合'\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['dispense']= int(result[1])\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['num_min']= result[2]\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['rank_min']= result[3]\n",
|
||
"sql = 'SELECT college,nian,lb,MIN(num_min),max(rank_min) FROM admission where nian not IN (\"2020\") group BY college,nian,lb ORDER BY college,nian desc,lb'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" #m_xx.clear()\n",
|
||
" m_code = result[0]\n",
|
||
" m_nian = result[1]\n",
|
||
" m_lb = result[2]\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code].setdefault(m_nian,{})\n",
|
||
" m_xx[m_code][m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['name']= dict_lb[m_lb]\n",
|
||
" if m_nian in m_lqrs[m_code]:\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['dispense']= int(m_lqrs[m_code][m_nian][m_lb])\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['num_min']= result[3]\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['rank_min']= result[4]\n",
|
||
" \n",
|
||
"#print(m_xx)\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" #m_item ='admission.'+m_nian\n",
|
||
" m_mongo.clear() \n",
|
||
" myquery = {'code':k}\n",
|
||
" m_new = {\"$set\":{'admission':v}}\n",
|
||
" mycol.update_one(myquery,m_new)\n",
|
||
" #print(k,v)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入全国普通高等学校名单(截至2020年6月)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 以文件形式导入至MongoDB中"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"document\"]\n",
|
||
"m_mongo = {}\n",
|
||
"fl_name = '全国普通高等学校名单.xlsx'\n",
|
||
"m_title = fl_name.split('.')[0]\n",
|
||
"m_type = fl_name.split('.')[1]\n",
|
||
"#print(m_title,m_type)\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"m_head = []\n",
|
||
"m_body = []\n",
|
||
"for cell in sheet[2]:\n",
|
||
" m_head.append(cell.value)\n",
|
||
"#for n in range(2,sheet.max_row+1):\n",
|
||
"#print(m_head) \n",
|
||
"for n in range(3,sheet.max_row+1):\n",
|
||
" m_cell = []\n",
|
||
" for i in range(1,len(m_head)+1):\n",
|
||
" m_cell.append(sheet.cell(n,i).value)\n",
|
||
" m_body.append(m_cell)\n",
|
||
"m_mongo['title'] = m_title\n",
|
||
"m_mongo['head'] = m_head\n",
|
||
"m_mongo['body'] = m_body\n",
|
||
"m_mongo['beizhu'] = '截至2020年6月30日'\n",
|
||
"mycol.insert_one(m_mongo) \n",
|
||
"print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 以文档形式导入至MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"gaoxiaomingdan\"]\n",
|
||
"m_mongo = {}\n",
|
||
"fl_name = '全国普通高等学校名单.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(3,sheet.max_row+1):\n",
|
||
" m_mongo ={}\n",
|
||
" m_mongo['name'] = sheet.cell(n,2).value \n",
|
||
" m_mongo['code'] = str(sheet.cell(n,3).value) \n",
|
||
" m_mongo['charge'] = sheet.cell(n,4).value\n",
|
||
" m_mongo['city'] = sheet.cell(n,5).value\n",
|
||
" m_mongo['grade'] = sheet.cell(n,6).value\n",
|
||
" m_mongo['note'] = sheet.cell(n,7).value \n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
"print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 合并导入高校信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import openpyxl\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"m_college = {}\n",
|
||
"sql = \"select code,name from college\"\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_college[result[1]] =result[0]\n",
|
||
"fl_name = '全国普通高等学校名单.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(3,sheet.max_row+1):\n",
|
||
" m_mongo ={}\n",
|
||
" if sheet.cell(n,2).value in m_college.keys():\n",
|
||
" m_mongo['code'] = m_college[sheet.cell(n,2).value]\n",
|
||
" m_mongo['name'] = sheet.cell(n,2).value \n",
|
||
" m_mongo['id_code'] = str(sheet.cell(n,3).value) \n",
|
||
" m_mongo['charge'] = sheet.cell(n,4).value\n",
|
||
" m_mongo['city'] = sheet.cell(n,5).value\n",
|
||
" m_mongo['grade'] = sheet.cell(n,6).value\n",
|
||
" m_mongo['note'] = sheet.cell(n,7).value \n",
|
||
" mycol.insert_one(m_mongo)\n",
|
||
" #print(m_mongo)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 根据招生编码整理高校信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 获取mongodb中不存在的学校"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 45,
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2020-11-22T12:52:46.525924Z",
|
||
"iopub.status.busy": "2020-11-22T12:52:46.524992Z",
|
||
"iopub.status.idle": "2020-11-22T12:52:52.304174Z",
|
||
"shell.execute_reply": "2020-11-22T12:52:52.301880Z",
|
||
"shell.execute_reply.started": "2020-11-22T12:52:46.525819Z"
|
||
}
|
||
},
|
||
"outputs": [
|
||
{
|
||
"name": "stdout",
|
||
"output_type": "stream",
|
||
"text": [
|
||
"Y007 山东财经大学\n",
|
||
"D687 中国传媒大学南广学院\n",
|
||
"D682 西安工业大学北方信息工程学院\n",
|
||
"Y002 山东科技大学\n",
|
||
"Y060 山东大学威海分校(走读)\n",
|
||
"Y061 济南大学(走读)\n",
|
||
"Y063 山东财经大学(走读)\n"
|
||
]
|
||
}
|
||
],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import openpyxl\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"m_college = {}\n",
|
||
"sql = \"select code,name from college\"\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"l_col = ['哈尔滨理工大学','山东大学','山东科技大学','青岛科技大学','济南大学','青岛理工大学','山东师范大学','山东财经大学','青岛大学']\n",
|
||
"for result in results:\n",
|
||
" m_code = result[0]\n",
|
||
" m_name = result[1]\n",
|
||
" myquery = {'code':m_code}\n",
|
||
" #x = mycol.find_one(myquery)\n",
|
||
" if not mycol.find_one(myquery) :\n",
|
||
" print(m_code,m_name)\n",
|
||
" #if mydoc.count ==0:\n",
|
||
" # print(m_name=' 不存在!')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 单一学校信息录入MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 43,
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2020-11-22T12:51:58.582586Z",
|
||
"iopub.status.busy": "2020-11-22T12:51:58.581686Z",
|
||
"iopub.status.idle": "2020-11-22T12:51:58.614642Z",
|
||
"shell.execute_reply": "2020-11-22T12:51:58.612513Z",
|
||
"shell.execute_reply.started": "2020-11-22T12:51:58.582483Z"
|
||
}
|
||
},
|
||
"outputs": [
|
||
{
|
||
"name": "stdout",
|
||
"output_type": "stream",
|
||
"text": [
|
||
"山东科技大学导入成功!\n"
|
||
]
|
||
}
|
||
],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import openpyxl\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"m_mongo = {\"code\":'Y022','name':'山东科技大学','charge':'山东省','city':'青岛市','grade':'本科','note':'校企合作,与青软实训教育科技股份有限公司'}\n",
|
||
"mycol.insert_one(m_mongo) \n",
|
||
"print(m_mongo['name'] + '导入成功!')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入2017-2019年录取数据"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import pymysql\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = \"select code,name from college\"\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_xx = []\n",
|
||
" m_code = result[0]\n",
|
||
" sql1 = 'select xx,zy_dm,zdf,zdwc,pjf,nian,lb from digao_1719 where xx=%s order by nian'\n",
|
||
" cursor.execute(sql1,m_code)\n",
|
||
" ad_results = cursor.fetchall()\n",
|
||
" for ad_result in ad_results:\n",
|
||
" m_xx.append(tuple(ad_result))\n",
|
||
" sql = \"insert into admission (college,speciality,num_min,rank_min,num_avg,nian,lb) values(%s,%s,%s,%s,%s,%s,%s)\"\n",
|
||
" try:\n",
|
||
" cursor.executemany(sql,m_xx)\n",
|
||
" db.commit()\n",
|
||
" print(result[1]+\"已添加!\")\n",
|
||
" except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"db.close()\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 从json文件导入2020年未招生学校名称"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import pymysql\n",
|
||
"m_col = []\n",
|
||
"filename = './data/2020年未招生学校.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_col.append((k,v['name']))\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"\n",
|
||
"sql = \"insert into college (code,name) values(%s,%s)\"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,m_col)\n",
|
||
" #db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import pymysql\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'select xx, zy_dm,zdf from digao_1719 where xx=\"A001\"'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"print(list(results))"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 一分一段表导入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导入2020年一分一段表"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import openpyxl\n",
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = []\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"wb = openpyxl.load_workbook('./data/2020一分一段表.xlsx')\n",
|
||
"sheet = wb.active\n",
|
||
"i = 1\n",
|
||
"for n in range(4,sheet.max_row):\n",
|
||
" m_score = sheet.cell(n,1).value\n",
|
||
" m_num = sheet.cell(n,14).value\n",
|
||
" m_sum = sheet.cell(n,15).value\n",
|
||
" m_max = m_sum - m_num + 1\n",
|
||
" m_xx.append((m_score,m_num,m_max,m_sum,'z','2020'))\n",
|
||
"#print(m_xx)\n",
|
||
"sql = 'insert into fenduan(score,per_num,max_rank,min_rank,category,nian) values (%s,%s,%s,%s,%s,%s)'\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,m_xx)\n",
|
||
" db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()\n",
|
||
"\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导出一分一段表至JSON文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_year = result[6]\n",
|
||
" m_xx.setdefault(m_year,{})\n",
|
||
" m_xx[m_year].setdefault(result[5],{})\n",
|
||
" m_xx[m_year][result[5]].setdefault(result[1],{})\n",
|
||
" m_xx[m_year][result[5]][result[1]]['num_person'] = result[2]\n",
|
||
" m_xx[m_year][result[5]][result[1]]['max_rank'] = result[3]\n",
|
||
" m_xx[m_year][result[5]][result[1]]['min_rank'] = result[4]\n",
|
||
"fl_name = 'data/17-20年一分一段表.json'\n",
|
||
"with open(fl_name,'w') as fl:\n",
|
||
" json.dump(m_xx,fl,ensure_ascii=False)\n",
|
||
"print('ok!')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导出一分一段表至MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"fenduan\"]\n",
|
||
"m_xx = {}\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_year = result[6]\n",
|
||
" m_xx.setdefault(m_year,{})\n",
|
||
" m_xx[m_year].setdefault(result[5],{})\n",
|
||
" m_fenshu = str(result[1])\n",
|
||
" m_xx[m_year][result[5]].setdefault(m_fenshu,{})\n",
|
||
" m_xx[m_year][result[5]][m_fenshu]['num_person'] = result[2]\n",
|
||
" m_xx[m_year][result[5]][m_fenshu]['max_rank'] = result[3]\n",
|
||
" m_xx[m_year][result[5]][m_fenshu]['min_rank'] = result[4]\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_mongo = {} \n",
|
||
" m_mongo.setdefault(k,{})\n",
|
||
" m_mongo[k] = v\n",
|
||
" x = mycol.insert_one(m_mongo) \n",
|
||
" \n",
|
||
"#print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 汇总2020年大学招生录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.college,sum(a.plan),SUM(a.num_dispense),MIN(a.num_min),max(a.rank_min) FROM admission_2020 AS a GROUP BY a.college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"#print(results)\n",
|
||
"sql = \"insert into admission_coll (college,num_plan,num_act,min_score,min_rank) values(%s,%s,%s,%s,%s)\"\n",
|
||
"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,results)\n",
|
||
" db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 整理以往年度学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 数据导入至MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"hz_luqu\"]\n",
|
||
"m_xx = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n",
|
||
"left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_code = result[0]\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code]['code'] = result[0]\n",
|
||
" m_xx[m_code]['name'] = result[1]\n",
|
||
" m_xx[m_code].setdefault(result[2],{})\n",
|
||
" m_xx[m_code][result[2]].setdefault(result[7],{})\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n",
|
||
" \n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_mongo[k] = v\n",
|
||
" x = mycol.insert_one(m_mongo[k]) \n",
|
||
" print(m_mongo[k])\n",
|
||
"\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 数据导入至JSON文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n",
|
||
"left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_code = result[0]\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code]['code'] = result[0]\n",
|
||
" m_xx[m_code]['name'] = result[1]\n",
|
||
" m_xx[m_code].setdefault(result[2],{})\n",
|
||
" m_xx[m_code][result[2]].setdefault(result[7],{})\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n",
|
||
" \n",
|
||
"fl_name = 'data/15-19年高校录取汇总情况.json'\n",
|
||
"with open(fl_name,'w') as fl:\n",
|
||
" json.dump(m_xx,fl,ensure_ascii=False)\n",
|
||
"#print(json.dumps(m_xx,ensure_ascii=False))\n",
|
||
"#print(m_xx)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"# 高考数据采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 新浪高考热讯采集导入MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"#web_xx = {}\n",
|
||
"n = 0\n",
|
||
"for i in range(1,11): \n",
|
||
" url = 'http://edu.sina.com.cn/other/roll.d.html?cat=80459&page={}&page_size=30'.format(str(i))\n",
|
||
" strhtml = requests.get(url)\n",
|
||
" strhtml.encoding = 'utf8'\n",
|
||
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
" #data = strhtml.text\n",
|
||
" #data1 = data.split(\"\\r\")\n",
|
||
" data = soup.select('#Main > div.listBlk > ul > li > a')\n",
|
||
" #data = soup.select('#artibody')\n",
|
||
" for item in data:\n",
|
||
" web_xx.clear()\n",
|
||
" n += 1\n",
|
||
" m_num = 'sina'+str(n).rjust(6, '0')\n",
|
||
" web_xx.setdefault(m_num,{})\n",
|
||
" web_xx[m_num]['title'] = item.get_text()\n",
|
||
" c_url = item.get('href')\n",
|
||
" web_xx[m_num]['url'] = c_url\n",
|
||
" content = requests.get(c_url)\n",
|
||
" content.encoding = 'utf8'\n",
|
||
" soup_content = BeautifulSoup(content.text)\n",
|
||
" data1 = soup_content.select('#artibody')\n",
|
||
" for item1 in data1:\n",
|
||
" web_xx[m_num]['content'] = item1.get_text()\n",
|
||
" time.sleep(3)\n",
|
||
" mycol.insert_one(web_xx) \n",
|
||
" print(m_num+'入库成功!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 中国教育在线信息采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学科录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"syl_jianshexueke\"]\n",
|
||
"url = 'http://daxue.eol.cn/syl.shtml'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('body > div.container.box > div.con > div:nth-child(4) > table > tbody > tr')\n",
|
||
"i = 0\n",
|
||
"for item in data:\n",
|
||
" m_mongo.clear()\n",
|
||
" data1 = item.select('td')\n",
|
||
" ss = re.sub('\\n', '', data1[1].text)\n",
|
||
" #print(ss.split('、'))\n",
|
||
" m_mongo['college'] = data1[0].text\n",
|
||
" m_mongo['discipline'] = ss.split('、')\n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
" print(data1[0].text + '导入成功!')\n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学校录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"syl_college\"]\n",
|
||
"url = 'http://daxue.eol.cn/syl.shtml'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody:nth-child(3) > tr')\n",
|
||
"m_name = []\n",
|
||
"#for item in data:\n",
|
||
"# m_mongo.clear()\n",
|
||
"for n in range(1,len(data)):\n",
|
||
" item = data[n].select('td')\n",
|
||
" ss = re.sub('\\n', ' ', data[n].text)\n",
|
||
" m_name = m_name + ss.strip().split(' ')\n",
|
||
" \n",
|
||
"for col in m_name:\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['name'] = col\n",
|
||
" m_mongo['level'] = 'B'\n",
|
||
" mycol.insert_one(m_mongo) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学校、学科合并"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_syl\"]\n",
|
||
"mycol1 = mydb[\"syl_college\"]\n",
|
||
"mycol2 = mydb[\"syl_jianshexueke\"]\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol1.find({},{'_id':0}):\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['college'] = x['name']\n",
|
||
" m_mongo['level'] = x['level']\n",
|
||
" myquery = {'college':x['name']}\n",
|
||
" for x1 in mycol2.find(myquery,{'_id':0}):\n",
|
||
" m_mongo['discipline'] = x1['discipline']\n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学校信息录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"pattern = re.compile(r'\\d+')\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"syl_jianshexueke\"]\n",
|
||
"url = 'http://daxue.eol.cn/syl.shtml'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody > tr > td > a')\n",
|
||
"i = 0\n",
|
||
"for item in data:\n",
|
||
" m_url = item.get('href')\n",
|
||
" mm =re.search('\\d+',m_url).group(0)\n",
|
||
" m_name = item.text\n",
|
||
" url1 = 'http://192.168.3.100:8050/render.html?https://gkcx.eol.cn/school/%s/introDetails' %mm\n",
|
||
" \n",
|
||
" strhtml = requests.get(url1)\n",
|
||
" time.sleep(10)\n",
|
||
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
" data1 = soup.select('#root > div > div > div > div > div > div > div.main > div > div.tuiji_left > div.intro_details_content')\n",
|
||
" print(url1,m_name)\n",
|
||
" print(data1)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 985、211学校及学科录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_211\"]\n",
|
||
"m_mon = {}\n",
|
||
"fl_name = './data/985.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb['211']\n",
|
||
"for n in range(1,sheet.max_row+1):\n",
|
||
" m_xx = []\n",
|
||
" m_mon.clear()\n",
|
||
" m_mon['college'] = sheet.cell(n,1).value\n",
|
||
" m_zy = sheet.cell(n,2).value\n",
|
||
" m_mon['discipline'] = m_zy.split('、')\n",
|
||
" mycol.insert_one(m_mon) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 985信息合并入大学信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"mycol1 = mydb[\"college_985\"]\n",
|
||
"mycol2 = mydb[\"college_syl\"]\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol2.find({},{'_id':0}):\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['level'] = x['level']\n",
|
||
" m_mongo['discipline'] = x['discipline']\n",
|
||
" myquery = {'name':x['college']}\n",
|
||
" m_new = {\"$set\":{'syl':m_mongo}}\n",
|
||
" mycol.update_one(myquery,m_new)\n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"myquery = {'name':{\"$regex\":\"\\(\"}}\n",
|
||
"for x in mycol.find(myquery,{'_id':0}):\n",
|
||
" #s = x['name'].replace('(','(').replace(')',')')\n",
|
||
" #myquery1 = {'id_code': x['id_code']}\n",
|
||
" #m_new = {\"$set\":{'name':s}}\n",
|
||
" #mycol.update_one(myquery1,m_new)\n",
|
||
" \n",
|
||
" \n",
|
||
" print(x)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 山东省考试院新闻采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"url = 'http://www.sdzk.cn/NewsList.aspx?BCID=2'\n",
|
||
"a_url = 'http://www.sdzk.cn/'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('#ctl00_ContentPlaceHolder1_RadAjaxPanel1 > ul > li > a')\n",
|
||
"m_name = []\n",
|
||
"for item in data:\n",
|
||
" m_mongo.clear()\n",
|
||
" news_url = a_url + item.get('href')\n",
|
||
" content = requests.get(news_url)\n",
|
||
" content.encoding = 'utf8'\n",
|
||
" soup_content = BeautifulSoup(content.text)\n",
|
||
" data1 = soup_content.select('#form1 > div.contain.laylist > div.fl.laylist-r > div > p')\n",
|
||
" print(item.text)\n",
|
||
" m_txt = ''\n",
|
||
" for item1 in data1: \n",
|
||
" m_txt += item1.text\n",
|
||
" m_txt += '\\n'\n",
|
||
" m_mongo['url'] = item.get('href')\n",
|
||
" m_mongo['source'] = '山东省教育招生考试院'\n",
|
||
" m_mongo['title'] = item.text[:-10]\n",
|
||
" m_mongo['date'] = item.text[-10:]\n",
|
||
" m_mongo['content'] = m_txt\n",
|
||
" mycol.insert_one(m_mongo) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 单网页数据采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"url = 'http://localhost:8050/render.html?url=http://gkcx.eol.cn/schoolhtm/schoolTemple/school31.htm&wait=5'\n",
|
||
"response = requests.get(url)\n",
|
||
"print(response.text)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 10,
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2020-11-27T07:00:38.964152Z",
|
||
"iopub.status.busy": "2020-11-27T07:00:38.963212Z",
|
||
"iopub.status.idle": "2020-11-27T07:00:39.734632Z",
|
||
"shell.execute_reply": "2020-11-27T07:00:39.731755Z",
|
||
"shell.execute_reply.started": "2020-11-27T07:00:38.964048Z"
|
||
}
|
||
},
|
||
"outputs": [
|
||
{
|
||
"name": "stdout",
|
||
"output_type": "stream",
|
||
"text": [
|
||
"[<h1 class=\"y_tit\">中国人民大学2020年强基计划热点问题解答</h1>]\n"
|
||
]
|
||
}
|
||
],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||
"url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n",
|
||
"strhtml = requests.get(url,headers = headers)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = strhtml.text\n",
|
||
"#data1 = data.split(\"\\r\")\n",
|
||
"data = soup.select('body > div.y_tit_box > div > div > div > h1')\n",
|
||
"#for item1 in data:\n",
|
||
"# print(item1.get_text())\n",
|
||
"print(data)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"mycol_new = mydb[\"news_new\"]\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol.find({},{'_id':0}):\n",
|
||
" for k,v in x.items():\n",
|
||
" if k[0:4] == 'sina':\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['url'] = v['url']\n",
|
||
" m_mongo['source'] = '新浪高考热讯'\n",
|
||
" m_mongo['title'] = v['title']\n",
|
||
" m_mongo['date'] = v['url'].split('/')[4]\n",
|
||
" m_mongo['content'] = re.sub('\\n\\n','\\n',v['content']) \n",
|
||
" mycol_new.insert_one(m_mongo)\n",
|
||
" \n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"mycol_new = mydb[\"news_new\"]\n",
|
||
"myquery = {'source':'山东省教育招生考试院'}\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol.find(myquery):\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['url'] = 'http://www.sdzk.cn/' + x['url']\n",
|
||
" m_mongo['source'] = '山东省教育招生考试院'\n",
|
||
" m_mongo['title'] = x['title']\n",
|
||
" m_mongo['date'] = x['date']\n",
|
||
" m_mongo['content'] = re.sub('\\n\\n','\\n',x['content']) \n",
|
||
" mycol_new.insert_one(m_mongo)\n",
|
||
" #print(m_mongo)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
}
|
||
],
|
||
"metadata": {
|
||
"kernelspec": {
|
||
"display_name": "Python 3",
|
||
"language": "python",
|
||
"name": "python3"
|
||
},
|
||
"language_info": {
|
||
"codemirror_mode": {
|
||
"name": "ipython",
|
||
"version": 3
|
||
},
|
||
"file_extension": ".py",
|
||
"mimetype": "text/x-python",
|
||
"name": "python",
|
||
"nbconvert_exporter": "python",
|
||
"pygments_lexer": "ipython3",
|
||
"version": "3.8.5"
|
||
},
|
||
"toc-autonumbering": true,
|
||
"toc-showcode": false,
|
||
"toc-showmarkdowntxt": true,
|
||
"toc-showtags": false
|
||
},
|
||
"nbformat": 4,
|
||
"nbformat_minor": 4
|
||
}
|