1900 lines
56 KiB
Plaintext
1900 lines
56 KiB
Plaintext
{
|
||
"cells": [
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"source": [
|
||
"# 高考数据导入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入2012年高校本科专业目录"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import re\n",
|
||
"m_xx = dict()\n",
|
||
"fl_name = 'teshe_2012.txt'\n",
|
||
"with open(fl_name,'r') as fl:\n",
|
||
" for line in fl:\n",
|
||
" if line.split():\n",
|
||
" l = line.split('\\t',1)\n",
|
||
" #print(l[0]+'--'+re.sub('\\n', '', l[1]))\n",
|
||
" m_len = len(l[0])\n",
|
||
" m_name = re.sub('\\n', '', l[1])\n",
|
||
" if m_len == 2:\n",
|
||
" m_xx.setdefault(l[0],{})\n",
|
||
" ll = m_name.split(':',1)\n",
|
||
" m_xx[l[0]]['name'] = ll[1]\n",
|
||
" elif m_len == 4:\n",
|
||
" m_xx[l[0][:2]].setdefault(l[0],{})\n",
|
||
" m_xx[l[0][:2]][l[0]]['name'] = m_name\n",
|
||
" else:\n",
|
||
" m_xx[l[0][:2]][l[0][:4]].setdefault(l[0],{})\n",
|
||
" m_xx[l[0][:2]][l[0][:4]][l[0]]['code'] = l[0]\n",
|
||
" m_xx[l[0][:2]][l[0][:4]][l[0]]['name'] = m_name\n",
|
||
"#print(m_xx)\n",
|
||
"with open('teshe_2012.json','w') as fl1:\n",
|
||
" json.dump(m_xx,fl1,ensure_ascii=False) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 合并2012年目录基本与特设专业"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import re\n",
|
||
"m_teshe = dict()\n",
|
||
"m_new = dict()\n",
|
||
"fl_name = 'jiben_2012.json'\n",
|
||
"fl_name1 = 'teshe_2012.json'\n",
|
||
"\n",
|
||
"with open(fl_name,'r') as fl:\n",
|
||
" m_new = json.load(fl)\n",
|
||
"for k,v in m_new.items():\n",
|
||
" for k1,v1 in v.items():\n",
|
||
" if k1 !='name':\n",
|
||
" for k2,v2 in m_new[k][k1].items():\n",
|
||
" if k2 !='name':\n",
|
||
" m_new[k][k1][k2]['lb'] = '基本'\n",
|
||
"\n",
|
||
"with open(fl_name1,'r') as fl1:\n",
|
||
" m_teshe = json.load(fl1)\n",
|
||
"for k,v in m_teshe.items():\n",
|
||
" for k1,v1 in v.items():\n",
|
||
" if k1 !='name':\n",
|
||
" for k2,v2 in m_teshe[k][k1].items():\n",
|
||
" if k2 !='name':\n",
|
||
" m_teshe[k][k1][k2]['lb'] = '特设'\n",
|
||
" m_new[k][k1][k2] = m_teshe[k][k1][k2]\n",
|
||
"with open('hebing_2012.json','w') as fl2:\n",
|
||
" json.dump(m_new,fl2,ensure_ascii=False) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入2020年高校本科专业目录"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"zhuanyemulu\"]\n",
|
||
"m_xx = dict()\n",
|
||
"fl_name = '普通高等学校本科专业目录.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(2,sheet.max_row+1):\n",
|
||
" m_code = sheet.cell(n,4).value\n",
|
||
" m_name = sheet.cell(n,5).value\n",
|
||
" m_k1 = m_code[:2]\n",
|
||
" m_k2 = m_code[:4]\n",
|
||
" m_xx.setdefault(m_k1,{})\n",
|
||
" m_xx[m_k1]['name'] = sheet.cell(n,2).value\n",
|
||
" m_xx[m_k1]['level'] = '门类'\n",
|
||
" m_xx[m_k1].setdefault(m_k2,{})\n",
|
||
" m_xx[m_k1][m_k2]['name'] = sheet.cell(n,3).value\n",
|
||
" m_xx[m_k1][m_k2]['level'] ='专业类'\n",
|
||
" m_xx[m_k1][m_k2].setdefault(m_code,{})\n",
|
||
" m_xx[m_k1][m_k2][m_code]['name'] = sheet.cell(n,5).value\n",
|
||
" m_xx[m_k1][m_k2][m_code]['shouyumenlei'] = sheet.cell(n,6).value\n",
|
||
" m_xx[m_k1][m_k2][m_code]['xiuyenianxian'] = sheet.cell(n,7).value\n",
|
||
" m_nf = sheet.cell(n,8).value\n",
|
||
" if m_nf.strip() != '':\n",
|
||
" m_xx[m_k1][m_k2][m_code]['zengshenianfen'] = sheet.cell(n,8).value\n",
|
||
" \n",
|
||
"#print(m_xx)\n",
|
||
"'''\n",
|
||
"# 导入MongoDB数据库\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_mongo = {} \n",
|
||
" m_mongo['code'] = k\n",
|
||
" for k1,v1 in v.items():\n",
|
||
" m_mongo[k1] = v1\n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
"'''\n",
|
||
"with open('2020.json','w') as fl2:\n",
|
||
" json.dump(m_xx,fl2,ensure_ascii=False) \n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 汇总学校录取分数及位次"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 汇总2021年学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_2021\"]\n",
|
||
"m_xx = {}\n",
|
||
"m_data = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT college,SUM(plan),min(num_min),max(RANK_min) FROM admission_2021 GROUP BY college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"i = 1\n",
|
||
"for result in results:\n",
|
||
" #m_xx.clear()\n",
|
||
" m_code = result[0]\n",
|
||
" m_nian ='2021'\n",
|
||
" m_lb = 'z'\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code].setdefault(m_nian,{})\n",
|
||
" m_xx[m_code][m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['name']= '综合'\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['dispense']= int(result[1])\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['num_min']= result[2]\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['rank_min']= result[3] \n",
|
||
" myquery = {'code':m_code}\n",
|
||
" colleges = mycol.find(myquery,{ \"_id\": 0, \"admission\": 1 })\n",
|
||
" for x in colleges:\n",
|
||
" for k, v in x.items():\n",
|
||
" for k1, v1 in v.items():\n",
|
||
" m_xx[m_code][k1] = v1\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" #m_item ='admission.'+m_nian\n",
|
||
" m_mongo.clear() \n",
|
||
" myquery = {'code':k}\n",
|
||
" m_new = {\"$set\":{'admission':v}}\n",
|
||
" mycol.update_one(myquery,m_new)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 生成历年录取信息json文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_2021\"]\n",
|
||
"m_xx = {}\n",
|
||
"m_data = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT college,SUM(plan),min(num_min),max(RANK_min) FROM admission_2021 GROUP BY college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"i = 1\n",
|
||
"for result in results:\n",
|
||
" #m_xx.clear()\n",
|
||
" m_code = result[0]\n",
|
||
" m_nian ='2021'\n",
|
||
" m_lb = 'z'\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code].setdefault(m_nian,{})\n",
|
||
" m_xx[m_code][m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['name']= '综合'\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['dispense']= int(result[1])\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['num_min']= result[2]\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['rank_min']= result[3] \n",
|
||
" myquery = {'code':m_code}\n",
|
||
" colleges = mycol.find(myquery,{ \"_id\": 0, \"admission\": 1 })\n",
|
||
" for x in colleges:\n",
|
||
" for k, v in x.items():\n",
|
||
" for k1, v1 in v.items():\n",
|
||
" m_xx[m_code][k1] = v1\n",
|
||
"fl_name = 'data/admission17-21.json'\n",
|
||
"with open(fl_name,'w') as fl:\n",
|
||
" json.dump(m_xx,fl)\n",
|
||
"print('ok!')\n",
|
||
"\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 2021年专业录取分数及位次"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": 112,
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2022-06-15T09:08:30.396155Z",
|
||
"iopub.status.busy": "2022-06-15T09:08:30.395636Z",
|
||
"iopub.status.idle": "2022-06-15T09:08:39.230104Z",
|
||
"shell.execute_reply": "2022-06-15T09:08:39.228838Z",
|
||
"shell.execute_reply.started": "2022-06-15T09:08:30.396108Z"
|
||
},
|
||
"tags": []
|
||
},
|
||
"outputs": [
|
||
{
|
||
"name": "stdout",
|
||
"output_type": "stream",
|
||
"text": [
|
||
"ok\n"
|
||
]
|
||
}
|
||
],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import decimal\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_2021\"]\n",
|
||
"mycol1 = mydb[\"admission_2021\"]\n",
|
||
"\n",
|
||
"m_col = {}\n",
|
||
"m_spe = {}\n",
|
||
"m_xx = {}\n",
|
||
"for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n",
|
||
" m_col[x['code']] = x['name']\n",
|
||
"\n",
|
||
"\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2021\"'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_spe.setdefault(result[1],{}) \n",
|
||
" m_spe[result[1]][result[0]] = result[2]\n",
|
||
"sql = 'SELECT * FROM admission_2021 AS a where college not in (\"D628\",\"D440\",\"Y010\",\"D245\") ORDER BY a.rank_min'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"i = 0\n",
|
||
"m_min = 0\n",
|
||
"ii = 0\n",
|
||
"for result in results:\n",
|
||
" m_xx.clear()\n",
|
||
" if result[4] == m_min:\n",
|
||
" ii = ii\n",
|
||
" i = i+1\n",
|
||
" else:\n",
|
||
" i = i+1\n",
|
||
" ii = i\n",
|
||
" m_min = result[6]\n",
|
||
" \n",
|
||
" m_xx['pos'] = ii\n",
|
||
" m_xx['col_code'] = result[1]\n",
|
||
" m_xx['col_name'] = m_col[result[1]]\n",
|
||
" m_xx['spe_code'] = result[2]\n",
|
||
" m_xx['spe_name'] = m_spe[result[1]][result[2]]\n",
|
||
" m_xx['plan'] = result[3]\n",
|
||
" m_xx['num_min'] = result[4] \n",
|
||
" m_xx['rank_min'] = result[5]\n",
|
||
" m_xx['nian'] = '2021' \n",
|
||
" mycol1.insert_one(m_xx)\n",
|
||
" #print(m_xx)\n",
|
||
"print('ok')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 汇总2020年学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"m_xx = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_xx.clear()\n",
|
||
" m_code = result[0]\n",
|
||
" m_nian ='2020'\n",
|
||
" m_lb = 'z'\n",
|
||
" m_xx.setdefault(m_nian,{})\n",
|
||
" m_xx[m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_nian][m_lb] ['name']= '综合'\n",
|
||
" m_xx[m_nian][m_lb] ['dispense']= int(result[1])\n",
|
||
" m_xx[m_nian][m_lb] ['num_min']= result[2]\n",
|
||
" m_xx[m_nian][m_lb] ['rank_min']= result[3]\n",
|
||
" #print(m_code,m_xx)\n",
|
||
" m_mongo.clear()\n",
|
||
" #m_mongo['admission'] = m_xx\n",
|
||
" #m_mongo['discipline'] = x['discipline']\n",
|
||
" myquery = {'code':m_code}\n",
|
||
" m_new = {\"$set\":{'admission':m_xx}}\n",
|
||
" mycol.update_one(myquery,m_new)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 汇总2017-2019学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"m_lqrs = {} #录取人数\n",
|
||
"m_xx = {}\n",
|
||
"dict_lb ={'z':'综合','l':'理科','w':'文科'}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT code,year,category,sum(num_act) FROM draft group BY code,year,category ORDER BY code,year,category'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"#生成录取人数字典\n",
|
||
"for result in results:\n",
|
||
" m_lqrs.setdefault(result[0],{})\n",
|
||
" m_lqrs[result[0]].setdefault(result[1],{})\n",
|
||
" \n",
|
||
" m_lqrs[result[0]][result[1]][result[2]] = result[3]\n",
|
||
"#print(m_lqrs)\n",
|
||
"sql = 'SELECT college,SUM(num_dispense),min(num_min),max(RANK_min) FROM admission_2020 GROUP BY college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results: \n",
|
||
" m_code = result[0]\n",
|
||
" m_nian ='2020'\n",
|
||
" m_lb = 'z'\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code].setdefault(m_nian,{})\n",
|
||
" m_xx[m_code][m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['name']= '综合'\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['dispense']= int(result[1])\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['num_min']= result[2]\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['rank_min']= result[3]\n",
|
||
"sql = 'SELECT college,nian,lb,MIN(num_min),max(rank_min) FROM admission where nian not IN (\"2020\") group BY college,nian,lb ORDER BY college,nian desc,lb'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" #m_xx.clear()\n",
|
||
" m_code = result[0]\n",
|
||
" m_nian = result[1]\n",
|
||
" m_lb = result[2]\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code].setdefault(m_nian,{})\n",
|
||
" m_xx[m_code][m_nian].setdefault(m_lb,{})\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['name']= dict_lb[m_lb]\n",
|
||
" if m_nian in m_lqrs[m_code]:\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['dispense']= int(m_lqrs[m_code][m_nian][m_lb])\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['num_min']= result[3]\n",
|
||
" m_xx[m_code][m_nian][m_lb] ['rank_min']= result[4]\n",
|
||
" \n",
|
||
"#print(m_xx)\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" #m_item ='admission.'+m_nian\n",
|
||
" m_mongo.clear() \n",
|
||
" myquery = {'code':k}\n",
|
||
" m_new = {\"$set\":{'admission':v}}\n",
|
||
" mycol.update_one(myquery,m_new)\n",
|
||
" #print(k,v)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入全国普通高等学校名单(截至2020年6月)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 以文件形式导入至MongoDB中"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"document\"]\n",
|
||
"m_mongo = {}\n",
|
||
"fl_name = '全国普通高等学校名单.xlsx'\n",
|
||
"m_title = fl_name.split('.')[0]\n",
|
||
"m_type = fl_name.split('.')[1]\n",
|
||
"#print(m_title,m_type)\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"m_head = []\n",
|
||
"m_body = []\n",
|
||
"for cell in sheet[2]:\n",
|
||
" m_head.append(cell.value)\n",
|
||
"#for n in range(2,sheet.max_row+1):\n",
|
||
"#print(m_head) \n",
|
||
"for n in range(3,sheet.max_row+1):\n",
|
||
" m_cell = []\n",
|
||
" for i in range(1,len(m_head)+1):\n",
|
||
" m_cell.append(sheet.cell(n,i).value)\n",
|
||
" m_body.append(m_cell)\n",
|
||
"m_mongo['title'] = m_title\n",
|
||
"m_mongo['head'] = m_head\n",
|
||
"m_mongo['body'] = m_body\n",
|
||
"m_mongo['beizhu'] = '截至2020年6月30日'\n",
|
||
"mycol.insert_one(m_mongo) \n",
|
||
"print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 以文档形式导入至MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"gaoxiaomingdan\"]\n",
|
||
"m_mongo = {}\n",
|
||
"fl_name = '全国普通高等学校名单.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(3,sheet.max_row+1):\n",
|
||
" m_mongo ={}\n",
|
||
" m_mongo['name'] = sheet.cell(n,2).value \n",
|
||
" m_mongo['code'] = str(sheet.cell(n,3).value) \n",
|
||
" m_mongo['charge'] = sheet.cell(n,4).value\n",
|
||
" m_mongo['city'] = sheet.cell(n,5).value\n",
|
||
" m_mongo['grade'] = sheet.cell(n,6).value\n",
|
||
" m_mongo['note'] = sheet.cell(n,7).value \n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
"print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 合并导入高校信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import openpyxl\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"m_college = {}\n",
|
||
"sql = \"select code,name from college\"\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_college[result[1]] =result[0]\n",
|
||
"fl_name = '全国普通高等学校名单.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb.active\n",
|
||
"for n in range(3,sheet.max_row+1):\n",
|
||
" m_mongo ={}\n",
|
||
" if sheet.cell(n,2).value in m_college.keys():\n",
|
||
" m_mongo['code'] = m_college[sheet.cell(n,2).value]\n",
|
||
" m_mongo['name'] = sheet.cell(n,2).value \n",
|
||
" m_mongo['id_code'] = str(sheet.cell(n,3).value) \n",
|
||
" m_mongo['charge'] = sheet.cell(n,4).value\n",
|
||
" m_mongo['city'] = sheet.cell(n,5).value\n",
|
||
" m_mongo['grade'] = sheet.cell(n,6).value\n",
|
||
" m_mongo['note'] = sheet.cell(n,7).value \n",
|
||
" mycol.insert_one(m_mongo)\n",
|
||
" #print(m_mongo)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"tags": [],
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 根据招生编码整理高校信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 获取mongodb中不存在的学校"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import openpyxl\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"m_college = {}\n",
|
||
"sql = \"select code,name from college\"\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"l_col = ['哈尔滨理工大学','山东大学','山东科技大学','青岛科技大学','济南大学','青岛理工大学','山东师范大学','山东财经大学','青岛大学']\n",
|
||
"for result in results:\n",
|
||
" m_code = result[0]\n",
|
||
" m_name = result[1]\n",
|
||
" myquery = {'code':m_code}\n",
|
||
" #x = mycol.find_one(myquery)\n",
|
||
" if not mycol.find_one(myquery) :\n",
|
||
" print(m_code,m_name)\n",
|
||
" #if mydoc.count ==0:\n",
|
||
" # print(m_name=' 不存在!')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 单一学校信息录入MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"import openpyxl\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_2021\"]\n",
|
||
"m_mongo = {\"code\":'H027','name':'北京师范大学(珠海校区)','charge':'教育部','city':'珠海市','grade':'本科','note':''}\n",
|
||
"mycol.insert_one(m_mongo) \n",
|
||
"print(m_mongo['name'] + '导入成功!')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 导入2017-2019年录取数据"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import pymysql\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = \"select code,name from college\"\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_xx = []\n",
|
||
" m_code = result[0]\n",
|
||
" sql1 = 'select xx,zy_dm,zdf,zdwc,pjf,nian,lb from digao_1719 where xx=%s order by nian'\n",
|
||
" cursor.execute(sql1,m_code)\n",
|
||
" ad_results = cursor.fetchall()\n",
|
||
" for ad_result in ad_results:\n",
|
||
" m_xx.append(tuple(ad_result))\n",
|
||
" sql = \"insert into admission (college,speciality,num_min,rank_min,num_avg,nian,lb) values(%s,%s,%s,%s,%s,%s,%s)\"\n",
|
||
" try:\n",
|
||
" cursor.executemany(sql,m_xx)\n",
|
||
" db.commit()\n",
|
||
" print(result[1]+\"已添加!\")\n",
|
||
" except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"db.close()\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 导入2021年录取数据"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导出2021年录取数据并生成分数"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import openpyxl\n",
|
||
"import json\n",
|
||
"\n",
|
||
"filename = 'data/17-21年一分一段表.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"data = m_xx['2021']['z']\n",
|
||
"yfyd = {}\n",
|
||
"for k, v in data.items():\n",
|
||
" list1 = []\n",
|
||
" list1 = (v['min_rank'],v['max_rank'])\n",
|
||
" yfyd[k] = list1\n",
|
||
"dict1 = {}\n",
|
||
"filename = 'data/山东省2021年普通类常规批第1次志愿投档情况表.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(filename)\n",
|
||
"sheet = wb.active\n",
|
||
"data1 =list(sheet.values)\n",
|
||
"del data1[0:2]\n",
|
||
"m_num = 1\n",
|
||
"i = 1\n",
|
||
"for items in data1:\n",
|
||
" m_rank = items[3]\n",
|
||
" list1 = []\n",
|
||
" for k, v in yfyd.items():\n",
|
||
" if m_rank >=v[1] and m_rank <=v[0]:\n",
|
||
" m_num = int(k)\n",
|
||
" break\n",
|
||
" list1 = (items[0],items[1],items[2],items[3],m_num)\n",
|
||
" dict1[i] = list1\n",
|
||
" i += 1 \n",
|
||
"with open('data/2021.json','w') as fl2:\n",
|
||
" json.dump(dict1,fl2) \n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 2021年录取数据导入mysql数据库"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"l_code = []\n",
|
||
"l_new = []\n",
|
||
"l_data = []\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"\n",
|
||
"sql = 'select code from college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" l_code.append(result[0])\n",
|
||
"\n",
|
||
"with open('data/2021.json','r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k, v in m_xx.items():\n",
|
||
" #code = v[1][:4]\n",
|
||
" l_data.append([v[1][:4],v[0][:2],v[2],v[4],v[3]])\n",
|
||
"\n",
|
||
"sql = \"insert into admission_2021 (college,speciality,plan,num_min,rank_min,nian) values(%s,%s,%s,%s,%s,'2021')\"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,l_data)\n",
|
||
" db.commit()\n",
|
||
" print(\"已添加!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"db.close() "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 整理2021年招生学校"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"l_code = []\n",
|
||
"l_new = []\n",
|
||
"l_data = []\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"\n",
|
||
"sql = 'select code from college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" l_code.append(result[0])\n",
|
||
"\n",
|
||
"with open('data/2021.json','r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k, v in m_xx.items():\n",
|
||
" code = v[1][:4]\n",
|
||
" if code not in l_code:\n",
|
||
" l_code.append(code)\n",
|
||
" l_new.append([v[1][0:4],v[1][4:]])\n",
|
||
" #l_data.append([v[1][:4],v[0][:2],v[2],v[4],v[3]])\n",
|
||
"sql = \"insert into college (code,name) values(%s,%s)\"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,l_new)\n",
|
||
" db.commit()\n",
|
||
" print(\"已添加!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"db.close() "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"execution": {
|
||
"iopub.execute_input": "2022-06-15T05:33:15.229612Z",
|
||
"iopub.status.busy": "2022-06-15T05:33:15.229087Z",
|
||
"iopub.status.idle": "2022-06-15T05:33:15.235612Z",
|
||
"shell.execute_reply": "2022-06-15T05:33:15.233803Z",
|
||
"shell.execute_reply.started": "2022-06-15T05:33:15.229565Z"
|
||
}
|
||
},
|
||
"source": [
|
||
"### 整理2021年新增招生学校"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import decimal\n",
|
||
"import json\n",
|
||
"\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_2021\"]\n",
|
||
"mycol1 = mydb[\"admission_2020\"]\n",
|
||
"\n",
|
||
"m_col = {}\n",
|
||
"m_xx = {}\n",
|
||
"dict1 = {}\n",
|
||
"list1 = []\n",
|
||
"for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n",
|
||
" m_col[x['code']] = x['name']\n",
|
||
"\n",
|
||
"with open('data/2021.json','r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k, v in m_xx.items():\n",
|
||
" #code = v[1][:4]\n",
|
||
" dict1[v[1][:4]]= v[1][4:]\n",
|
||
"for code in dict1.keys():\n",
|
||
" if code not in m_col.keys():\n",
|
||
" #print(dict1[code])\n",
|
||
" \n",
|
||
" if dict1[code] not in m_col.values():\n",
|
||
" list1.append(code)\n",
|
||
"print(list1)\n",
|
||
"for x in mycol.find({\"id_code\":{'$exists': 'true'}},{\"_id\": 0, \"id_code\": 1, \"name\": 1}):\n",
|
||
" m_col[x['id_code']] = x['name']\n",
|
||
"for code in list1:\n",
|
||
" if dict1[code] in m_col.values():\n",
|
||
" myquery = {'name':dict1[code]}\n",
|
||
" m_new = {\"$set\":{'code':code}}\n",
|
||
" mycol.update_one(myquery,m_new)\n",
|
||
" print(dict1[code])\n",
|
||
" #print(dict1[code])\n",
|
||
" \n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 整理2021年新增专业"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"l_data = []\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"\n",
|
||
"with open('data/2021.json','r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k, v in m_xx.items():\n",
|
||
" l_data.append([v[0][:2],v[0][2:],v[1][0:4],'2021'])\n",
|
||
"sql = \"insert into speciality (code,name,college,nian) values(%s,%s,%s,%s)\"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,l_data)\n",
|
||
" db.commit()\n",
|
||
" print(\"已添加!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"db.close() "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 从json文件导入2020年未招生学校名称"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import pymysql\n",
|
||
"m_col = []\n",
|
||
"filename = './data/2020年未招生学校.json'\n",
|
||
"with open(filename,'r') as fl:\n",
|
||
" m_xx = json.load(fl)\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_col.append((k,v['name']))\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"\n",
|
||
"sql = \"insert into college (code,name) values(%s,%s)\"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,m_col)\n",
|
||
" #db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import pymysql\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'select xx, zy_dm,zdf from digao_1719 where xx=\"A001\"'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"print(list(results))"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 一分一段表导入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导入2020年一分一段表"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import openpyxl\n",
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = []\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"wb = openpyxl.load_workbook('./data/2020一分一段表.xlsx')\n",
|
||
"sheet = wb.active\n",
|
||
"i = 1\n",
|
||
"for n in range(4,sheet.max_row):\n",
|
||
" m_score = sheet.cell(n,1).value\n",
|
||
" m_num = sheet.cell(n,14).value\n",
|
||
" m_sum = sheet.cell(n,15).value\n",
|
||
" m_max = m_sum - m_num + 1\n",
|
||
" m_xx.append((m_score,m_num,m_max,m_sum,'z','2020'))\n",
|
||
"#print(m_xx)\n",
|
||
"sql = 'insert into fenduan(score,per_num,max_rank,min_rank,category,nian) values (%s,%s,%s,%s,%s,%s)'\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,m_xx)\n",
|
||
" db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()\n",
|
||
"\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导入2021年一分一段表"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {
|
||
"tags": []
|
||
},
|
||
"outputs": [],
|
||
"source": [
|
||
"import openpyxl\n",
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = []\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"wb = openpyxl.load_workbook('./data/2021年一分一段表.xlsx')\n",
|
||
"sheet = wb.active\n",
|
||
"i = 1\n",
|
||
"for n in range(4,sheet.max_row):\n",
|
||
" m_score = sheet.cell(n,1).value\n",
|
||
" m_num = sheet.cell(n,2).value\n",
|
||
" m_sum = sheet.cell(n,3).value\n",
|
||
" m_max = m_sum - m_num + 1\n",
|
||
" m_xx.append((m_score,m_num,m_max,m_sum,'z','2021'))\n",
|
||
"#print(m_xx)\n",
|
||
"sql = 'insert into fenduan(score,per_num,max_rank,min_rank,category,nian) values (%s,%s,%s,%s,%s,%s)'\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,m_xx)\n",
|
||
" db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导出一分一段表至JSON文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = {}\n",
|
||
"db = pymysql.connect(host = \"localhost\",user = \"root\",password = \"songyi\",database = \"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_year = result[6]\n",
|
||
" m_xx.setdefault(m_year,{})\n",
|
||
" m_xx[m_year].setdefault(result[5],{})\n",
|
||
" m_xx[m_year][result[5]].setdefault(result[1],{})\n",
|
||
" m_xx[m_year][result[5]][result[1]]['num_person'] = result[2]\n",
|
||
" m_xx[m_year][result[5]][result[1]]['max_rank'] = result[3]\n",
|
||
" m_xx[m_year][result[5]][result[1]]['min_rank'] = result[4]\n",
|
||
"fl_name = 'data/17-21年一分一段表.json'\n",
|
||
"with open(fl_name,'w') as fl:\n",
|
||
" json.dump(m_xx,fl,ensure_ascii=False)\n",
|
||
"print('ok!')\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 导出一分一段表至MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"fenduan\"]\n",
|
||
"m_xx = {}\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT * FROM fenduan AS a ORDER BY nian,category'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_year = result[6]\n",
|
||
" m_xx.setdefault(m_year,{})\n",
|
||
" m_xx[m_year].setdefault(result[5],{})\n",
|
||
" m_fenshu = str(result[1])\n",
|
||
" m_xx[m_year][result[5]].setdefault(m_fenshu,{})\n",
|
||
" m_xx[m_year][result[5]][m_fenshu]['num_person'] = result[2]\n",
|
||
" m_xx[m_year][result[5]][m_fenshu]['max_rank'] = result[3]\n",
|
||
" m_xx[m_year][result[5]][m_fenshu]['min_rank'] = result[4]\n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_mongo = {} \n",
|
||
" m_mongo.setdefault(k,{})\n",
|
||
" m_mongo[k] = v\n",
|
||
" x = mycol.insert_one(m_mongo) \n",
|
||
" \n",
|
||
"#print('ok!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 汇总2020年大学招生录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.college,sum(a.plan),SUM(a.num_dispense),MIN(a.num_min),max(a.rank_min) FROM admission_2020 AS a GROUP BY a.college'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"#print(results)\n",
|
||
"sql = \"insert into admission_coll (college,num_plan,num_act,min_score,min_rank) values(%s,%s,%s,%s,%s)\"\n",
|
||
"\n",
|
||
"try:\n",
|
||
" cursor.executemany(sql,results)\n",
|
||
" db.commit()\n",
|
||
" print(\"ok!\")\n",
|
||
"except:\n",
|
||
" # 如果发生错误则回滚\n",
|
||
" print(\"error!\")\n",
|
||
" db.rollback() \n",
|
||
"\n",
|
||
"db.close()"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 整理以往年度学校录取情况"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 数据导入至MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"hz_luqu\"]\n",
|
||
"m_xx = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n",
|
||
"left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_code = result[0]\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code]['code'] = result[0]\n",
|
||
" m_xx[m_code]['name'] = result[1]\n",
|
||
" m_xx[m_code].setdefault(result[2],{})\n",
|
||
" m_xx[m_code][result[2]].setdefault(result[7],{})\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n",
|
||
" \n",
|
||
"for k,v in m_xx.items():\n",
|
||
" m_mongo[k] = v\n",
|
||
" x = mycol.insert_one(m_mongo[k]) \n",
|
||
" print(m_mongo[k])\n",
|
||
"\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"### 数据导入至JSON文件"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymysql\n",
|
||
"import json\n",
|
||
"\n",
|
||
"m_xx = {}\n",
|
||
"m_mongo = {}\n",
|
||
"db = pymysql.connect(\"localhost\",\"root\",\"songyi\",\"gaokao\" )\n",
|
||
"cursor = db.cursor()\n",
|
||
"sql = 'SELECT a.code,b.name,a.year,a.num_plan,a.num_act,a.max_score,a.min_score,a.category FROM draft AS a \\\n",
|
||
"left join college AS b ON a.code=b.code AND b.name is NOT NULL order BY a.code,a.year'\n",
|
||
"cursor.execute(sql)\n",
|
||
"results = cursor.fetchall()\n",
|
||
"for result in results:\n",
|
||
" m_code = result[0]\n",
|
||
" m_xx.setdefault(m_code,{})\n",
|
||
" m_xx[m_code]['code'] = result[0]\n",
|
||
" m_xx[m_code]['name'] = result[1]\n",
|
||
" m_xx[m_code].setdefault(result[2],{})\n",
|
||
" m_xx[m_code][result[2]].setdefault(result[7],{})\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_plan'] = result[3]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['num_act'] = result[4]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['max_score'] = result[5]\n",
|
||
" m_xx[m_code][result[2]][result[7]]['min_score'] = result[6]\n",
|
||
" \n",
|
||
"fl_name = 'data/15-19年高校录取汇总情况.json'\n",
|
||
"with open(fl_name,'w') as fl:\n",
|
||
" json.dump(m_xx,fl,ensure_ascii=False)\n",
|
||
"#print(json.dumps(m_xx,ensure_ascii=False))\n",
|
||
"#print(m_xx)\n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"# 高考数据采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 新浪高考热讯采集导入MongoDB"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"#web_xx = {}\n",
|
||
"n = 0\n",
|
||
"for i in range(1,11): \n",
|
||
" url = 'http://edu.sina.com.cn/other/roll.d.html?cat=80459&page={}&page_size=30'.format(str(i))\n",
|
||
" strhtml = requests.get(url)\n",
|
||
" strhtml.encoding = 'utf8'\n",
|
||
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
" #data = strhtml.text\n",
|
||
" #data1 = data.split(\"\\r\")\n",
|
||
" data = soup.select('#Main > div.listBlk > ul > li > a')\n",
|
||
" #data = soup.select('#artibody')\n",
|
||
" for item in data:\n",
|
||
" web_xx.clear()\n",
|
||
" n += 1\n",
|
||
" m_num = 'sina'+str(n).rjust(6, '0')\n",
|
||
" web_xx.setdefault(m_num,{})\n",
|
||
" web_xx[m_num]['title'] = item.get_text()\n",
|
||
" c_url = item.get('href')\n",
|
||
" web_xx[m_num]['url'] = c_url\n",
|
||
" content = requests.get(c_url)\n",
|
||
" content.encoding = 'utf8'\n",
|
||
" soup_content = BeautifulSoup(content.text)\n",
|
||
" data1 = soup_content.select('#artibody')\n",
|
||
" for item1 in data1:\n",
|
||
" web_xx[m_num]['content'] = item1.get_text()\n",
|
||
" time.sleep(3)\n",
|
||
" mycol.insert_one(web_xx) \n",
|
||
" print(m_num+'入库成功!')"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 中国教育在线信息采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学科录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"syl_jianshexueke\"]\n",
|
||
"url = 'http://daxue.eol.cn/syl.shtml'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('body > div.container.box > div.con > div:nth-child(4) > table > tbody > tr')\n",
|
||
"i = 0\n",
|
||
"for item in data:\n",
|
||
" m_mongo.clear()\n",
|
||
" data1 = item.select('td')\n",
|
||
" ss = re.sub('\\n', '', data1[1].text)\n",
|
||
" #print(ss.split('、'))\n",
|
||
" m_mongo['college'] = data1[0].text\n",
|
||
" m_mongo['discipline'] = ss.split('、')\n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
" print(data1[0].text + '导入成功!')\n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学校录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"syl_college\"]\n",
|
||
"url = 'http://daxue.eol.cn/syl.shtml'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody:nth-child(3) > tr')\n",
|
||
"m_name = []\n",
|
||
"#for item in data:\n",
|
||
"# m_mongo.clear()\n",
|
||
"for n in range(1,len(data)):\n",
|
||
" item = data[n].select('td')\n",
|
||
" ss = re.sub('\\n', ' ', data[n].text)\n",
|
||
" m_name = m_name + ss.strip().split(' ')\n",
|
||
" \n",
|
||
"for col in m_name:\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['name'] = col\n",
|
||
" m_mongo['level'] = 'B'\n",
|
||
" mycol.insert_one(m_mongo) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学校、学科合并"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_syl\"]\n",
|
||
"mycol1 = mydb[\"syl_college\"]\n",
|
||
"mycol2 = mydb[\"syl_jianshexueke\"]\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol1.find({},{'_id':0}):\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['college'] = x['name']\n",
|
||
" m_mongo['level'] = x['level']\n",
|
||
" myquery = {'college':x['name']}\n",
|
||
" for x1 in mycol2.find(myquery,{'_id':0}):\n",
|
||
" m_mongo['discipline'] = x1['discipline']\n",
|
||
" mycol.insert_one(m_mongo) \n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 双一流学校信息录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"pattern = re.compile(r'\\d+')\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"syl_jianshexueke\"]\n",
|
||
"url = 'http://daxue.eol.cn/syl.shtml'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('body > div.container.box > div.con > div:nth-child(2) > table > tbody > tr > td > a')\n",
|
||
"i = 0\n",
|
||
"for item in data:\n",
|
||
" m_url = item.get('href')\n",
|
||
" mm =re.search('\\d+',m_url).group(0)\n",
|
||
" m_name = item.text\n",
|
||
" url1 = 'http://192.168.3.100:8050/render.html?https://gkcx.eol.cn/school/%s/introDetails' %mm\n",
|
||
" \n",
|
||
" strhtml = requests.get(url1)\n",
|
||
" time.sleep(10)\n",
|
||
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
" data1 = soup.select('#root > div > div > div > div > div > div > div.main > div > div.tuiji_left > div.intro_details_content')\n",
|
||
" print(url1,m_name)\n",
|
||
" print(data1)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 985、211学校及学科录入"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import json\n",
|
||
"import openpyxl\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college_211\"]\n",
|
||
"m_mon = {}\n",
|
||
"fl_name = './data/985.xlsx'\n",
|
||
"wb = openpyxl.load_workbook(fl_name)\n",
|
||
"sheet = wb['211']\n",
|
||
"for n in range(1,sheet.max_row+1):\n",
|
||
" m_xx = []\n",
|
||
" m_mon.clear()\n",
|
||
" m_mon['college'] = sheet.cell(n,1).value\n",
|
||
" m_zy = sheet.cell(n,2).value\n",
|
||
" m_mon['discipline'] = m_zy.split('、')\n",
|
||
" mycol.insert_one(m_mon) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"### 985信息合并入大学信息"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"mycol1 = mydb[\"college_985\"]\n",
|
||
"mycol2 = mydb[\"college_syl\"]\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol2.find({},{'_id':0}):\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['level'] = x['level']\n",
|
||
" m_mongo['discipline'] = x['discipline']\n",
|
||
" myquery = {'name':x['college']}\n",
|
||
" m_new = {\"$set\":{'syl':m_mongo}}\n",
|
||
" mycol.update_one(myquery,m_new)\n",
|
||
" \n",
|
||
" "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"college\"]\n",
|
||
"myquery = {'name':{\"$regex\":\"\\(\"}}\n",
|
||
"for x in mycol.find(myquery,{'_id':0}):\n",
|
||
" #s = x['name'].replace('(','(').replace(')',')')\n",
|
||
" #myquery1 = {'id_code': x['id_code']}\n",
|
||
" #m_new = {\"$set\":{'name':s}}\n",
|
||
" #mycol.update_one(myquery1,m_new)\n",
|
||
" \n",
|
||
" \n",
|
||
" print(x)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {},
|
||
"source": [
|
||
"## 山东省考试院新闻采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"\n",
|
||
"m_mongo = {}\n",
|
||
"requests.packages.urllib3.disable_warnings()\n",
|
||
"requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"url = 'http://www.sdzk.cn/NewsList.aspx?BCID=2'\n",
|
||
"a_url = 'http://www.sdzk.cn/'\n",
|
||
"strhtml = requests.get(url)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = soup.select('#ctl00_ContentPlaceHolder1_RadAjaxPanel1 > ul > li > a')\n",
|
||
"m_name = []\n",
|
||
"for item in data:\n",
|
||
" m_mongo.clear()\n",
|
||
" news_url = a_url + item.get('href')\n",
|
||
" content = requests.get(news_url)\n",
|
||
" content.encoding = 'utf8'\n",
|
||
" soup_content = BeautifulSoup(content.text)\n",
|
||
" data1 = soup_content.select('#form1 > div.contain.laylist > div.fl.laylist-r > div > p')\n",
|
||
" print(item.text)\n",
|
||
" m_txt = ''\n",
|
||
" for item1 in data1: \n",
|
||
" m_txt += item1.text\n",
|
||
" m_txt += '\\n'\n",
|
||
" m_mongo['url'] = item.get('href')\n",
|
||
" m_mongo['source'] = '山东省教育招生考试院'\n",
|
||
" m_mongo['title'] = item.text[:-10]\n",
|
||
" m_mongo['date'] = item.text[-10:]\n",
|
||
" m_mongo['content'] = m_txt\n",
|
||
" mycol.insert_one(m_mongo) "
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "markdown",
|
||
"metadata": {
|
||
"toc-hr-collapsed": true,
|
||
"toc-nb-collapsed": true
|
||
},
|
||
"source": [
|
||
"## 单网页数据采集"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"url = 'https://www.movephysio.co.za/'\n",
|
||
"response = requests.get(url)\n",
|
||
"print(response.text)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import requests\n",
|
||
"import time\n",
|
||
"from bs4 import BeautifulSoup\n",
|
||
"import pymongo\n",
|
||
"\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
||
"url = 'https://rdzs.ruc.edu.cn/cms/item/1642.html'\n",
|
||
"strhtml = requests.get(url,headers = headers)\n",
|
||
"strhtml.encoding = 'utf8'\n",
|
||
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
||
"data = strhtml.text\n",
|
||
"#data1 = data.split(\"\\r\")\n",
|
||
"data = soup.select('body > div.y_tit_box > div > div > div > h1')\n",
|
||
"#for item1 in data:\n",
|
||
"# print(item1.get_text())\n",
|
||
"print(data)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"mycol_new = mydb[\"news_new\"]\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol.find({},{'_id':0}):\n",
|
||
" for k,v in x.items():\n",
|
||
" if k[0:4] == 'sina':\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['url'] = v['url']\n",
|
||
" m_mongo['source'] = '新浪高考热讯'\n",
|
||
" m_mongo['title'] = v['title']\n",
|
||
" m_mongo['date'] = v['url'].split('/')[4]\n",
|
||
" m_mongo['content'] = re.sub('\\n\\n','\\n',v['content']) \n",
|
||
" mycol_new.insert_one(m_mongo)\n",
|
||
" \n"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": [
|
||
"import pymongo\n",
|
||
"import re\n",
|
||
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
||
"mydb = myclient[\"gaokao\"]\n",
|
||
"mycol = mydb[\"news\"]\n",
|
||
"mycol_new = mydb[\"news_new\"]\n",
|
||
"myquery = {'source':'山东省教育招生考试院'}\n",
|
||
"m_mongo = {}\n",
|
||
"for x in mycol.find(myquery):\n",
|
||
" m_mongo.clear()\n",
|
||
" m_mongo['url'] = 'http://www.sdzk.cn/' + x['url']\n",
|
||
" m_mongo['source'] = '山东省教育招生考试院'\n",
|
||
" m_mongo['title'] = x['title']\n",
|
||
" m_mongo['date'] = x['date']\n",
|
||
" m_mongo['content'] = re.sub('\\n\\n','\\n',x['content']) \n",
|
||
" mycol_new.insert_one(m_mongo)\n",
|
||
" #print(m_mongo)"
|
||
]
|
||
},
|
||
{
|
||
"cell_type": "code",
|
||
"execution_count": null,
|
||
"metadata": {},
|
||
"outputs": [],
|
||
"source": []
|
||
}
|
||
],
|
||
"metadata": {
|
||
"kernelspec": {
|
||
"display_name": "Python 3",
|
||
"language": "python",
|
||
"name": "python3"
|
||
},
|
||
"language_info": {
|
||
"codemirror_mode": {
|
||
"name": "ipython",
|
||
"version": 3
|
||
},
|
||
"file_extension": ".py",
|
||
"mimetype": "text/x-python",
|
||
"name": "python",
|
||
"nbconvert_exporter": "python",
|
||
"pygments_lexer": "ipython3",
|
||
"version": "3.8.10"
|
||
},
|
||
"toc-autonumbering": true,
|
||
"toc-showcode": false,
|
||
"toc-showmarkdowntxt": true,
|
||
"toc-showtags": false
|
||
},
|
||
"nbformat": 4,
|
||
"nbformat_minor": 4
|
||
}
|