845 lines
27 KiB
Plaintext
845 lines
27 KiB
Plaintext
{
|
|
"cells": [
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {
|
|
"tags": []
|
|
},
|
|
"source": [
|
|
"# 高考志愿管理"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 2020年高考录取信息导入MongoDB"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {
|
|
"tags": []
|
|
},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymysql\n",
|
|
"import pymongo\n",
|
|
"import decimal\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"college\"]\n",
|
|
"mycol1 = mydb[\"admission_2020\"]\n",
|
|
"\n",
|
|
"m_col = {}\n",
|
|
"m_spe = {}\n",
|
|
"m_xx = {}\n",
|
|
"for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n",
|
|
" m_col[x['code']] = x['name']\n",
|
|
"\n",
|
|
"\n",
|
|
"db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n",
|
|
"cursor = db.cursor()\n",
|
|
"sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2020\"'\n",
|
|
"cursor.execute(sql)\n",
|
|
"results = cursor.fetchall()\n",
|
|
"for result in results:\n",
|
|
" m_spe.setdefault(result[1],{}) \n",
|
|
" m_spe[result[1]][result[0]] = result[2]\n",
|
|
"sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'\n",
|
|
"cursor.execute(sql)\n",
|
|
"results = cursor.fetchall()\n",
|
|
"i = 0\n",
|
|
"m_min = 0\n",
|
|
"ii = 0\n",
|
|
"for result in results:\n",
|
|
" m_xx.clear()\n",
|
|
" if result[6] == m_min:\n",
|
|
" ii = ii\n",
|
|
" i = i+1\n",
|
|
" else:\n",
|
|
" i = i+1\n",
|
|
" ii = i\n",
|
|
" m_min = result[6]\n",
|
|
" m_xx['pos'] = ii\n",
|
|
" m_xx['col_code'] = result[1]\n",
|
|
" m_xx['col_name'] = m_col[result[1]]\n",
|
|
" m_xx['spe_code'] = result[2]\n",
|
|
" m_xx['spe_name'] = m_spe[result[1]][result[2]]\n",
|
|
" m_xx['plan'] = result[3]\n",
|
|
" m_xx['dispense'] = result[5]\n",
|
|
" m_xx['num_min'] = result[6]\n",
|
|
" m_xx['num_avg'] = int(result[7])\n",
|
|
" m_xx['rank_min'] = result[8]\n",
|
|
" m_xx['nian'] = '2020' \n",
|
|
" mycol1.insert_one(m_xx)\n",
|
|
"#print(m_spe)\n",
|
|
"\n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 2021年全国高等学校信息更新"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"### 2021年全国高等学校信息导出至json文件"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": 21,
|
|
"metadata": {
|
|
"execution": {
|
|
"iopub.execute_input": "2022-06-15T07:00:27.890983Z",
|
|
"iopub.status.busy": "2022-06-15T07:00:27.890444Z",
|
|
"iopub.status.idle": "2022-06-15T07:00:28.430955Z",
|
|
"shell.execute_reply": "2022-06-15T07:00:28.429949Z",
|
|
"shell.execute_reply.started": "2022-06-15T07:00:27.890936Z"
|
|
},
|
|
"tags": []
|
|
},
|
|
"outputs": [],
|
|
"source": [
|
|
"import openpyxl\n",
|
|
"import json\n",
|
|
"\n",
|
|
"filename = 'data/2021年全国普通高等学校名单.xlsx'\n",
|
|
"wb = openpyxl.load_workbook(filename)\n",
|
|
"sheet = wb.active\n",
|
|
"data1 =list(sheet.values)\n",
|
|
"del data1[0:3]\n",
|
|
"i = 1\n",
|
|
"dict1 = {}\n",
|
|
"for items in data1:\n",
|
|
" list1 = (items[1],str(items[2]),items[3],items[4],items[5],items[6])\n",
|
|
" dict1[i] = list1\n",
|
|
" i += 1 \n",
|
|
"with open('data/2021年全国普通高等学校名单.json','w') as fl2:\n",
|
|
" json.dump(dict1,fl2) \n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"### 整理2021年全国高等学校信息"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": 25,
|
|
"metadata": {
|
|
"execution": {
|
|
"iopub.execute_input": "2022-06-15T08:26:50.903511Z",
|
|
"iopub.status.busy": "2022-06-15T08:26:50.902972Z",
|
|
"iopub.status.idle": "2022-06-15T08:26:50.972239Z",
|
|
"shell.execute_reply": "2022-06-15T08:26:50.971224Z",
|
|
"shell.execute_reply.started": "2022-06-15T08:26:50.903463Z"
|
|
},
|
|
"tags": []
|
|
},
|
|
"outputs": [
|
|
{
|
|
"name": "stdout",
|
|
"output_type": "stream",
|
|
"text": [
|
|
"河北工业职业技术大学导入成功!\n",
|
|
"河北科技工程职业技术大学导入成功!\n",
|
|
"河北石油职业技术大学导入成功!\n",
|
|
"邢台应用技术职业学院导入成功!\n",
|
|
"山西工程科技职业大学导入成功!\n",
|
|
"吉林通用航空职业技术学院导入成功!\n",
|
|
"通化医药健康职业学院导入成功!\n",
|
|
"上海南湖职业技术学院导入成功!\n",
|
|
"浙江药科职业大学导入成功!\n",
|
|
"浙江金华科贸职业技术学院导入成功!\n",
|
|
"宿州航空职业学院导入成功!\n",
|
|
"和君职业学院导入成功!\n",
|
|
"滨州科技职业学院导入成功!\n",
|
|
"洛阳文化旅游职业学院导入成功!\n",
|
|
"周口文理职业学院导入成功!\n",
|
|
"信阳艺术职业学院导入成功!\n",
|
|
"郑州城建职业学院导入成功!\n",
|
|
"郑州医药健康职业学院导入成功!\n",
|
|
"湖北孝感美珈职业学院导入成功!\n",
|
|
"广州幼儿师范高等专科学校导入成功!\n",
|
|
"广东汕头幼儿师范高等专科学校导入成功!\n",
|
|
"广东梅州职业技术学院导入成功!\n",
|
|
"广东潮州卫生健康职业学院导入成功!\n",
|
|
"广东云浮中医药职业学院导入成功!\n",
|
|
"广东肇庆航空职业学院导入成功!\n",
|
|
"广西农业职业技术大学导入成功!\n",
|
|
"防城港职业技术学院导入成功!\n",
|
|
"广西信息职业技术学院导入成功!\n",
|
|
"广西农业工程职业技术学院导入成功!\n",
|
|
"北海康养职业学院导入成功!\n",
|
|
"重庆工信职业学院导入成功!\n",
|
|
"甘孜职业学院导入成功!\n",
|
|
"自贡职业技术学院导入成功!\n",
|
|
"贵阳康养职业大学导入成功!\n",
|
|
"贵州文化旅游职业学院导入成功!\n",
|
|
"宝鸡中北职业学院导入成功!\n",
|
|
"兰州石化职业技术大学导入成功!\n",
|
|
"兰州资源环境职业技术大学导入成功!\n",
|
|
"兰州航空职业技术学院导入成功!\n",
|
|
"白银希望职业技术学院导入成功!\n"
|
|
]
|
|
}
|
|
],
|
|
"source": [
|
|
"import openpyxl\n",
|
|
"import json\n",
|
|
"import pymongo\n",
|
|
"\n",
|
|
"\n",
|
|
"filename = 'data/2021年全国普通高等学校名单.json'\n",
|
|
"with open(filename,'r') as fl:\n",
|
|
" m_xx = json.load(fl)\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"college_2021\"]\n",
|
|
"\n",
|
|
"m_col = {}\n",
|
|
"\n",
|
|
"\n",
|
|
"for x in mycol.find({\"id_code\":{'$exists': 'true'}},{\"_id\": 0, \"id_code\": 1, \"name\": 1}):\n",
|
|
" m_col[x['id_code']] = x['name']\n",
|
|
"#print(m_col.keys())\n",
|
|
"for k, v in m_xx.items():\n",
|
|
" m_code = v[1]\n",
|
|
" #print(m_code)\n",
|
|
" if m_code not in m_col.keys():\n",
|
|
" dict1 = {}\n",
|
|
" dict1['id_code'] = v[1]\n",
|
|
" dict1['name'] = v[0]\n",
|
|
" dict1['charge'] = v[2]\n",
|
|
" dict1['city'] = v[3]\n",
|
|
" dict1['grade'] = v[4]\n",
|
|
" dict1['note'] = v[5]\n",
|
|
" mycol.insert_one(dict1) \n",
|
|
" print(dict1['name'] + '导入成功!')\n",
|
|
" #print(v[0],v[1])\n",
|
|
" \n",
|
|
"\n",
|
|
"\n",
|
|
" "
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 计算志愿分数概率"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import random\n",
|
|
"\n",
|
|
"array2 = []\n",
|
|
"for i in range(300000):\n",
|
|
" array1 = []\n",
|
|
" s = 0\n",
|
|
" for ii in range(60):\n",
|
|
" m1 = random.randint(580,595)\n",
|
|
" array1.append(m1)\n",
|
|
" s = s + m1 \n",
|
|
" m_avg = round(s/60,2)\n",
|
|
" if m_avg == 582.8 and (580 in array1) and (595 in array1):\n",
|
|
" #print(array1)\n",
|
|
" array2.extend(array1)\n",
|
|
" \n",
|
|
"#print(array2)\n",
|
|
"m_set = set(array2)\n",
|
|
"for m in m_set:\n",
|
|
" print(m,array2.count(m))"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 导入山东大学录取明细"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import openpyxl\n",
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"admission_college\"]\n",
|
|
"wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')\n",
|
|
"sheet = wb.active\n",
|
|
"#sheets = wb.sheetnames\n",
|
|
"code = 'A422'\n",
|
|
"name = '山东大学'\n",
|
|
"new_col = []\n",
|
|
"dict1 = {}\n",
|
|
"\n",
|
|
"new_code = []\n",
|
|
"dict1['code'] = code\n",
|
|
"dict1['name'] = name\n",
|
|
"for n in range(1,sheet.max_row+1):\n",
|
|
" nian = str(sheet.cell(n,1).value)\n",
|
|
" dict2 = {}\n",
|
|
" \n",
|
|
" dict1.setdefault(nian,[])\n",
|
|
" if sheet.cell(n,2).value =='理工':\n",
|
|
" m_lb = 'l'\n",
|
|
" elif sheet.cell(n,2).value =='文史':\n",
|
|
" m_lb = 'w'\n",
|
|
" else:\n",
|
|
" m_lb = 'z' \n",
|
|
" dict2['type'] = m_lb\n",
|
|
" dict2['spe_name'] = sheet.cell(n,4).value\n",
|
|
" dict2['max_score'] = sheet.cell(n,5).value\n",
|
|
" dict2['min_score'] = sheet.cell(n,6).value\n",
|
|
" dict2['avg_score'] = sheet.cell(n,7).value\n",
|
|
" dict2['dispense'] = sheet.cell(n,8).value\n",
|
|
" dict1[nian].append(dict2)\n",
|
|
"mycol.insert_one(dict1)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 导入山东师范大学录取明细"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import openpyxl\n",
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"admission_college\"]\n",
|
|
"wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')\n",
|
|
"sheet = wb.active\n",
|
|
"#sheets = wb.sheetnames\n",
|
|
"code = 'A423'\n",
|
|
"name = '中国海洋大学'\n",
|
|
"new_col = []\n",
|
|
"dict1 = {}\n",
|
|
"\n",
|
|
"new_code = []\n",
|
|
"dict1['code'] = code\n",
|
|
"dict1['name'] = name\n",
|
|
"for n in range(1,sheet.max_row+1):\n",
|
|
" nian = str(sheet.cell(n,1).value)\n",
|
|
" dict2 = {}\n",
|
|
" \n",
|
|
" dict1.setdefault(nian,[])\n",
|
|
" if sheet.cell(n,2).value =='理工':\n",
|
|
" m_lb = 'l'\n",
|
|
" elif sheet.cell(n,2).value =='文史':\n",
|
|
" m_lb = 'w'\n",
|
|
" else:\n",
|
|
" m_lb = 'z'\n",
|
|
" dict2['type'] = m_lb\n",
|
|
" dict2['spe_name'] = sheet.cell(n,4).value\n",
|
|
" dict2['max_score'] = sheet.cell(n,7).value\n",
|
|
" dict2['min_score'] = sheet.cell(n,5).value\n",
|
|
" dict2['avg_score'] = sheet.cell(n,6).value\n",
|
|
" if sheet.cell(n,8).value:\n",
|
|
" dict2['dispense'] = sheet.cell(n,8).value\n",
|
|
" dict1[nian].append(dict2)\n",
|
|
"mycol.insert_one(dict1)\n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 按地区列示高校"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"college\"]\n",
|
|
"m_city = []\n",
|
|
"m_xx = {}\n",
|
|
"myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}\n",
|
|
"colleges = mycol.find(myquery,{ \"_id\": 0, \"name\": 1, \"code\": 1,'city':1 })\n",
|
|
"for x in colleges:\n",
|
|
" m_xx1 = []\n",
|
|
" c = x['city']\n",
|
|
" m_xx.setdefault(c,[])\n",
|
|
" m_xx1.append(x['code'])\n",
|
|
" m_xx1.append(x['name'])\n",
|
|
" m_xx[c].append(m_xx1)\n",
|
|
"#print(m_city)\n",
|
|
"#按照原顺序对高校所在城市排序\n",
|
|
"'''\n",
|
|
"city = list(set(m_city))\n",
|
|
"city.sort(key=m_city.index)\n",
|
|
"for c in city:\n",
|
|
"# m_xx['city'] = c\n",
|
|
" m_xx.setdefault(c,[])\n",
|
|
" for y in colleges:\n",
|
|
" print(y['code'],y['name'])\n",
|
|
" \n",
|
|
"#m_xx \n",
|
|
"''' \n",
|
|
"for k,v in m_xx.items():\n",
|
|
" print(k)\n",
|
|
" for mm in v:\n",
|
|
" print(mm[0],mm[1])"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 显示学校2020年招生信息"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"admission_2020\"]\n",
|
|
"code = 'A422'\n",
|
|
"m_city = []\n",
|
|
"m_xx = {}\n",
|
|
"myquery = {'col_code':code}\n",
|
|
"colleges = mycol.find(myquery,{ \"_id\": 0 , \"col_code\":0,\"col_name\":0,'plan':0,'nian':0}).sort('pos')\n",
|
|
"for x in colleges:\n",
|
|
" print(x)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 2021年拟在山东招生普通高校专业(类)选考科目要求"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from os import mkdir\n",
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"from os.path import isdir,isfile\n",
|
|
"from urllib.request import urlopen\n",
|
|
"from urllib.parse import urlencode,quote\n",
|
|
"from openpyxl import Workbook\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"ssl._create_default_https_context = ssl._create_unverified_context\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"xuankaokemu\"]\n",
|
|
"list2 = []\n",
|
|
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
|
|
" list2.append(x['code'])\n",
|
|
"#m_xx = dict()\n",
|
|
"start_url = 'https://xkkm.sdzk.cn/web/xx.html'\n",
|
|
"with urlopen(start_url) as fp:\n",
|
|
" content = fp.read().decode('utf8')\n",
|
|
" \n",
|
|
"pattern = (r'<tr>.*?<td.+?</td>.*?<td.+?>(.+?)</td>'\n",
|
|
" '.*?<td.+?>(.+?)</td>.*?<td.+?>(.+?)</td>')\n",
|
|
"\n",
|
|
"for item in findall(pattern,content,S):\n",
|
|
" if len(item[0]) > 5:\n",
|
|
" continue\n",
|
|
" \n",
|
|
" shengfen,dm,mc = item\n",
|
|
" print('学校代码:', dm)\n",
|
|
" print('学校名称:', mc)\n",
|
|
" m_xx = {}\n",
|
|
" m_xx['code'] = dm\n",
|
|
" m_xx['name'] = mc\n",
|
|
" m_xx.setdefault('zhuanye',{})\n",
|
|
" if dm in list2:\n",
|
|
" continue\n",
|
|
" url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'\n",
|
|
" data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')\n",
|
|
" with urlopen(url,data) as fp:\n",
|
|
" xuexiao_content = fp.read().decode()\n",
|
|
" soup = BeautifulSoup(xuexiao_content,'lxml')\n",
|
|
" #data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')\n",
|
|
" data = soup.select('#ccc > div > table > tbody > tr ')\n",
|
|
" for data1 in data:\n",
|
|
" list1 =[]\n",
|
|
" m_xx1 = {}\n",
|
|
" for item in data1.stripped_strings: \n",
|
|
" list1.append(item)\n",
|
|
" #del list1[0]\n",
|
|
" #print('层次:',list1[1])\n",
|
|
" #print('专业(类)名称:',list1[2])\n",
|
|
" #print('选考科目范围:',list1[3])\n",
|
|
" #print('类中所含专业:',list1[4:])\n",
|
|
" code = list1[0]\n",
|
|
" m_xx['zhuanye'].setdefault(code,{})\n",
|
|
" \n",
|
|
" m_xx['zhuanye'][code]['name'] = list1[2]\n",
|
|
" m_xx['zhuanye'][code]['level'] = list1[1]\n",
|
|
" m_xx['zhuanye'][code]['fanwei'] = list1[3]\n",
|
|
" m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]\n",
|
|
" mycol.insert_one(m_xx) \n",
|
|
" \n",
|
|
" print('ok')\n",
|
|
" \n",
|
|
" # list1.clear\n",
|
|
" \n",
|
|
"\n",
|
|
" sleep(5)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"xuankaokemu\"]\n",
|
|
"list2 = []\n",
|
|
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
|
|
" list2.append(x['code'])\n",
|
|
"print(list2)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"# 招生简章管理"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 各大学招生简章采集"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'\n",
|
|
"\n",
|
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
|
"for i in sf:\n",
|
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
|
" \n",
|
|
" for item in data:\n",
|
|
" dict1 = {}\n",
|
|
" dict1['name'] = item.text.strip()\n",
|
|
" dict1['url'] = item.get('href')\n",
|
|
" if item.get('style') =='color:gray':\n",
|
|
" dict1['bz'] = 0\n",
|
|
" else:\n",
|
|
" dict1['bz'] = 1\n",
|
|
" mycol.insert_one(dict1) \n",
|
|
"\n",
|
|
"\n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 采集单个学校招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'\n",
|
|
"strhtml = requests.get(url,headers = headers)\n",
|
|
"strhtml.encoding = 'utf8'\n",
|
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
"#data = strhtml.text\n",
|
|
"#data1 = data.split(\"\\r\")\n",
|
|
"#print(soup)\n",
|
|
"data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
|
"for item in data:\n",
|
|
" url1 = item.get('href')\n",
|
|
"url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
|
|
"strhtml = requests.get(url1,headers = headers)\n",
|
|
"strhtml.encoding = 'utf8'\n",
|
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
"data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
|
"nr = ''\n",
|
|
"for item in data:\n",
|
|
" nr += item.text+'\\n'\n",
|
|
"print(nr)\n",
|
|
" \n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 第一次采集招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"for x in mycol.find({\"bz\": 1 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
|
|
" url = 'https://gaokao.chsi.com.cn'+x['url']\n",
|
|
" col_name = x['name']\n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" #data = strhtml.text\n",
|
|
" #data1 = data.split(\"\\r\")\n",
|
|
" #print(soup)\n",
|
|
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
|
" for item in data:\n",
|
|
" url1 = item.get('href')\n",
|
|
" url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
|
|
" strhtml = requests.get(url1,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
|
" nr = ''\n",
|
|
" for item in data:\n",
|
|
" nr += item.text+'\\n' \n",
|
|
" myquery = {'name':col_name}\n",
|
|
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
|
|
" sleep(5)\n",
|
|
" "
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 检查新增的学校及招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"l_name = []\n",
|
|
"for x in mycol.find({},{ \"_id\": 0, \"name\": 1}):\n",
|
|
" l_name.append(x['name'])\n",
|
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
|
"n_url = []\n",
|
|
"for i in sf:\n",
|
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
|
" \n",
|
|
" for item in data:\n",
|
|
" m_url = item.get('href')\n",
|
|
" m_name = item.text.strip()\n",
|
|
" if m_name not in l_name:\n",
|
|
" url = 'https://gaokao.chsi.com.cn'+m_url\n",
|
|
" col_name = m_name \n",
|
|
" dict1 = {}\n",
|
|
" dict1['name'] = m_name\n",
|
|
" dict1['url'] = m_url\n",
|
|
" dict1['bz'] = 0 \n",
|
|
" mycol.insert_one(dict1) \n",
|
|
" print(dict1)\n",
|
|
" sleep(3)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 检查、新增招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"l_name = []\n",
|
|
"for x in mycol.find({\"bz\": 0 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
|
|
" l_name.append(x['name'])\n",
|
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
|
"n_name = []\n",
|
|
"for i in sf:\n",
|
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk' \n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
|
" \n",
|
|
" for item in data:\n",
|
|
" m_url = item.get('href')\n",
|
|
" col_name = item.text.strip()\n",
|
|
" #print(m_url)\n",
|
|
" if (item.get('style') !='color:gray') and (col_name in l_name):\n",
|
|
" url = 'https://gaokao.chsi.com.cn' + m_url\n",
|
|
" \n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
|
" for item in data:\n",
|
|
" url1 = item.get('href')\n",
|
|
" print(url1)\n",
|
|
" url1 = 'https://gaokao.chsi.com.cn' + url1\n",
|
|
" strhtml = requests.get(url1,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
|
" nr = ''\n",
|
|
" for item in data:\n",
|
|
" nr += item.text+'\\n' \n",
|
|
" myquery = {'name':col_name}\n",
|
|
" mycol.update_one(myquery,{'$set':{'bz':1}})\n",
|
|
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
|
|
" print(col_name)\n",
|
|
" sleep(3)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": []
|
|
}
|
|
],
|
|
"metadata": {
|
|
"kernelspec": {
|
|
"display_name": "Python 3",
|
|
"language": "python",
|
|
"name": "python3"
|
|
},
|
|
"language_info": {
|
|
"codemirror_mode": {
|
|
"name": "ipython",
|
|
"version": 3
|
|
},
|
|
"file_extension": ".py",
|
|
"mimetype": "text/x-python",
|
|
"name": "python",
|
|
"nbconvert_exporter": "python",
|
|
"pygments_lexer": "ipython3",
|
|
"version": "3.8.10"
|
|
}
|
|
},
|
|
"nbformat": 4,
|
|
"nbformat_minor": 4
|
|
}
|