691 lines
22 KiB
Plaintext
691 lines
22 KiB
Plaintext
{
|
|
"cells": [
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {
|
|
"tags": [],
|
|
"toc-hr-collapsed": true
|
|
},
|
|
"source": [
|
|
"# 高考志愿管理"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 2020年高考录取信息导入MongoDB"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {
|
|
"tags": []
|
|
},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymysql\n",
|
|
"import pymongo\n",
|
|
"import decimal\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"college\"]\n",
|
|
"mycol1 = mydb[\"admission_2020\"]\n",
|
|
"\n",
|
|
"m_col = {}\n",
|
|
"m_spe = {}\n",
|
|
"m_xx = {}\n",
|
|
"for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n",
|
|
" m_col[x['code']] = x['name']\n",
|
|
"\n",
|
|
"\n",
|
|
"db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n",
|
|
"cursor = db.cursor()\n",
|
|
"sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2020\"'\n",
|
|
"cursor.execute(sql)\n",
|
|
"results = cursor.fetchall()\n",
|
|
"for result in results:\n",
|
|
" m_spe.setdefault(result[1],{}) \n",
|
|
" m_spe[result[1]][result[0]] = result[2]\n",
|
|
"sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'\n",
|
|
"cursor.execute(sql)\n",
|
|
"results = cursor.fetchall()\n",
|
|
"i = 0\n",
|
|
"m_min = 0\n",
|
|
"ii = 0\n",
|
|
"for result in results:\n",
|
|
" m_xx.clear()\n",
|
|
" if result[6] == m_min:\n",
|
|
" ii = ii\n",
|
|
" i = i+1\n",
|
|
" else:\n",
|
|
" i = i+1\n",
|
|
" ii = i\n",
|
|
" m_min = result[6]\n",
|
|
" m_xx['pos'] = ii\n",
|
|
" m_xx['col_code'] = result[1]\n",
|
|
" m_xx['col_name'] = m_col[result[1]]\n",
|
|
" m_xx['spe_code'] = result[2]\n",
|
|
" m_xx['spe_name'] = m_spe[result[1]][result[2]]\n",
|
|
" m_xx['plan'] = result[3]\n",
|
|
" m_xx['dispense'] = result[5]\n",
|
|
" m_xx['num_min'] = result[6]\n",
|
|
" m_xx['num_avg'] = int(result[7])\n",
|
|
" m_xx['rank_min'] = result[8]\n",
|
|
" m_xx['nian'] = '2020' \n",
|
|
" mycol1.insert_one(m_xx)\n",
|
|
"#print(m_spe)\n",
|
|
"\n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 计算志愿分数概率"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import random\n",
|
|
"\n",
|
|
"array2 = []\n",
|
|
"for i in range(300000):\n",
|
|
" array1 = []\n",
|
|
" s = 0\n",
|
|
" for ii in range(60):\n",
|
|
" m1 = random.randint(580,595)\n",
|
|
" array1.append(m1)\n",
|
|
" s = s + m1 \n",
|
|
" m_avg = round(s/60,2)\n",
|
|
" if m_avg == 582.8 and (580 in array1) and (595 in array1):\n",
|
|
" #print(array1)\n",
|
|
" array2.extend(array1)\n",
|
|
" \n",
|
|
"#print(array2)\n",
|
|
"m_set = set(array2)\n",
|
|
"for m in m_set:\n",
|
|
" print(m,array2.count(m))"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 导入山东大学录取明细"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import openpyxl\n",
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"admission_college\"]\n",
|
|
"wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')\n",
|
|
"sheet = wb.active\n",
|
|
"#sheets = wb.sheetnames\n",
|
|
"code = 'A422'\n",
|
|
"name = '山东大学'\n",
|
|
"new_col = []\n",
|
|
"dict1 = {}\n",
|
|
"\n",
|
|
"new_code = []\n",
|
|
"dict1['code'] = code\n",
|
|
"dict1['name'] = name\n",
|
|
"for n in range(1,sheet.max_row+1):\n",
|
|
" nian = str(sheet.cell(n,1).value)\n",
|
|
" dict2 = {}\n",
|
|
" \n",
|
|
" dict1.setdefault(nian,[])\n",
|
|
" if sheet.cell(n,2).value =='理工':\n",
|
|
" m_lb = 'l'\n",
|
|
" elif sheet.cell(n,2).value =='文史':\n",
|
|
" m_lb = 'w'\n",
|
|
" else:\n",
|
|
" m_lb = 'z' \n",
|
|
" dict2['type'] = m_lb\n",
|
|
" dict2['spe_name'] = sheet.cell(n,4).value\n",
|
|
" dict2['max_score'] = sheet.cell(n,5).value\n",
|
|
" dict2['min_score'] = sheet.cell(n,6).value\n",
|
|
" dict2['avg_score'] = sheet.cell(n,7).value\n",
|
|
" dict2['dispense'] = sheet.cell(n,8).value\n",
|
|
" dict1[nian].append(dict2)\n",
|
|
"mycol.insert_one(dict1)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 导入山东师范大学录取明细"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import openpyxl\n",
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"admission_college\"]\n",
|
|
"wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')\n",
|
|
"sheet = wb.active\n",
|
|
"#sheets = wb.sheetnames\n",
|
|
"code = 'A423'\n",
|
|
"name = '中国海洋大学'\n",
|
|
"new_col = []\n",
|
|
"dict1 = {}\n",
|
|
"\n",
|
|
"new_code = []\n",
|
|
"dict1['code'] = code\n",
|
|
"dict1['name'] = name\n",
|
|
"for n in range(1,sheet.max_row+1):\n",
|
|
" nian = str(sheet.cell(n,1).value)\n",
|
|
" dict2 = {}\n",
|
|
" \n",
|
|
" dict1.setdefault(nian,[])\n",
|
|
" if sheet.cell(n,2).value =='理工':\n",
|
|
" m_lb = 'l'\n",
|
|
" elif sheet.cell(n,2).value =='文史':\n",
|
|
" m_lb = 'w'\n",
|
|
" else:\n",
|
|
" m_lb = 'z'\n",
|
|
" dict2['type'] = m_lb\n",
|
|
" dict2['spe_name'] = sheet.cell(n,4).value\n",
|
|
" dict2['max_score'] = sheet.cell(n,7).value\n",
|
|
" dict2['min_score'] = sheet.cell(n,5).value\n",
|
|
" dict2['avg_score'] = sheet.cell(n,6).value\n",
|
|
" if sheet.cell(n,8).value:\n",
|
|
" dict2['dispense'] = sheet.cell(n,8).value\n",
|
|
" dict1[nian].append(dict2)\n",
|
|
"mycol.insert_one(dict1)\n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 按地区列示高校"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"college\"]\n",
|
|
"m_city = []\n",
|
|
"m_xx = {}\n",
|
|
"myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}\n",
|
|
"colleges = mycol.find(myquery,{ \"_id\": 0, \"name\": 1, \"code\": 1,'city':1 })\n",
|
|
"for x in colleges:\n",
|
|
" m_xx1 = []\n",
|
|
" c = x['city']\n",
|
|
" m_xx.setdefault(c,[])\n",
|
|
" m_xx1.append(x['code'])\n",
|
|
" m_xx1.append(x['name'])\n",
|
|
" m_xx[c].append(m_xx1)\n",
|
|
"#print(m_city)\n",
|
|
"#按照原顺序对高校所在城市排序\n",
|
|
"'''\n",
|
|
"city = list(set(m_city))\n",
|
|
"city.sort(key=m_city.index)\n",
|
|
"for c in city:\n",
|
|
"# m_xx['city'] = c\n",
|
|
" m_xx.setdefault(c,[])\n",
|
|
" for y in colleges:\n",
|
|
" print(y['code'],y['name'])\n",
|
|
" \n",
|
|
"#m_xx \n",
|
|
"''' \n",
|
|
"for k,v in m_xx.items():\n",
|
|
" print(k)\n",
|
|
" for mm in v:\n",
|
|
" print(mm[0],mm[1])"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 显示学校2020年招生信息"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"admission_2020\"]\n",
|
|
"code = 'A422'\n",
|
|
"m_city = []\n",
|
|
"m_xx = {}\n",
|
|
"myquery = {'col_code':code}\n",
|
|
"colleges = mycol.find(myquery,{ \"_id\": 0 , \"col_code\":0,\"col_name\":0,'plan':0,'nian':0}).sort('pos')\n",
|
|
"for x in colleges:\n",
|
|
" print(x)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 2021年拟在山东招生普通高校专业(类)选考科目要求"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from os import mkdir\n",
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"from os.path import isdir,isfile\n",
|
|
"from urllib.request import urlopen\n",
|
|
"from urllib.parse import urlencode,quote\n",
|
|
"from openpyxl import Workbook\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"ssl._create_default_https_context = ssl._create_unverified_context\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"xuankaokemu\"]\n",
|
|
"list2 = []\n",
|
|
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
|
|
" list2.append(x['code'])\n",
|
|
"#m_xx = dict()\n",
|
|
"start_url = 'https://xkkm.sdzk.cn/web/xx.html'\n",
|
|
"with urlopen(start_url) as fp:\n",
|
|
" content = fp.read().decode('utf8')\n",
|
|
" \n",
|
|
"pattern = (r'<tr>.*?<td.+?</td>.*?<td.+?>(.+?)</td>'\n",
|
|
" '.*?<td.+?>(.+?)</td>.*?<td.+?>(.+?)</td>')\n",
|
|
"\n",
|
|
"for item in findall(pattern,content,S):\n",
|
|
" if len(item[0]) > 5:\n",
|
|
" continue\n",
|
|
" \n",
|
|
" shengfen,dm,mc = item\n",
|
|
" print('学校代码:', dm)\n",
|
|
" print('学校名称:', mc)\n",
|
|
" m_xx = {}\n",
|
|
" m_xx['code'] = dm\n",
|
|
" m_xx['name'] = mc\n",
|
|
" m_xx.setdefault('zhuanye',{})\n",
|
|
" if dm in list2:\n",
|
|
" continue\n",
|
|
" url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'\n",
|
|
" data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')\n",
|
|
" with urlopen(url,data) as fp:\n",
|
|
" xuexiao_content = fp.read().decode()\n",
|
|
" soup = BeautifulSoup(xuexiao_content,'lxml')\n",
|
|
" #data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')\n",
|
|
" data = soup.select('#ccc > div > table > tbody > tr ')\n",
|
|
" for data1 in data:\n",
|
|
" list1 =[]\n",
|
|
" m_xx1 = {}\n",
|
|
" for item in data1.stripped_strings: \n",
|
|
" list1.append(item)\n",
|
|
" #del list1[0]\n",
|
|
" #print('层次:',list1[1])\n",
|
|
" #print('专业(类)名称:',list1[2])\n",
|
|
" #print('选考科目范围:',list1[3])\n",
|
|
" #print('类中所含专业:',list1[4:])\n",
|
|
" code = list1[0]\n",
|
|
" m_xx['zhuanye'].setdefault(code,{})\n",
|
|
" \n",
|
|
" m_xx['zhuanye'][code]['name'] = list1[2]\n",
|
|
" m_xx['zhuanye'][code]['level'] = list1[1]\n",
|
|
" m_xx['zhuanye'][code]['fanwei'] = list1[3]\n",
|
|
" m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]\n",
|
|
" mycol.insert_one(m_xx) \n",
|
|
" \n",
|
|
" print('ok')\n",
|
|
" \n",
|
|
" # list1.clear\n",
|
|
" \n",
|
|
"\n",
|
|
" sleep(5)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"import pymongo\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"xuankaokemu\"]\n",
|
|
"list2 = []\n",
|
|
"for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n",
|
|
" list2.append(x['code'])\n",
|
|
"print(list2)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"# 招生简章管理"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 各大学招生简章采集"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'\n",
|
|
"\n",
|
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
|
"for i in sf:\n",
|
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
|
" \n",
|
|
" for item in data:\n",
|
|
" dict1 = {}\n",
|
|
" dict1['name'] = item.text.strip()\n",
|
|
" dict1['url'] = item.get('href')\n",
|
|
" if item.get('style') =='color:gray':\n",
|
|
" dict1['bz'] = 0\n",
|
|
" else:\n",
|
|
" dict1['bz'] = 1\n",
|
|
" mycol.insert_one(dict1) \n",
|
|
"\n",
|
|
"\n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 采集单个学校招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'\n",
|
|
"strhtml = requests.get(url,headers = headers)\n",
|
|
"strhtml.encoding = 'utf8'\n",
|
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
"#data = strhtml.text\n",
|
|
"#data1 = data.split(\"\\r\")\n",
|
|
"#print(soup)\n",
|
|
"data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
|
"for item in data:\n",
|
|
" url1 = item.get('href')\n",
|
|
"url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
|
|
"strhtml = requests.get(url1,headers = headers)\n",
|
|
"strhtml.encoding = 'utf8'\n",
|
|
"soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
"data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
|
"nr = ''\n",
|
|
"for item in data:\n",
|
|
" nr += item.text+'\\n'\n",
|
|
"print(nr)\n",
|
|
" \n"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 第一次采集招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"for x in mycol.find({\"bz\": 1 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
|
|
" url = 'https://gaokao.chsi.com.cn'+x['url']\n",
|
|
" col_name = x['name']\n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" #data = strhtml.text\n",
|
|
" #data1 = data.split(\"\\r\")\n",
|
|
" #print(soup)\n",
|
|
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
|
" for item in data:\n",
|
|
" url1 = item.get('href')\n",
|
|
" url1 = 'https://gaokao.chsi.com.cn/' + url1\n",
|
|
" strhtml = requests.get(url1,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
|
" nr = ''\n",
|
|
" for item in data:\n",
|
|
" nr += item.text+'\\n' \n",
|
|
" myquery = {'name':col_name}\n",
|
|
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
|
|
" sleep(5)\n",
|
|
" "
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 检查新增的学校及招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"l_name = []\n",
|
|
"for x in mycol.find({},{ \"_id\": 0, \"name\": 1}):\n",
|
|
" l_name.append(x['name'])\n",
|
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
|
"n_url = []\n",
|
|
"for i in sf:\n",
|
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
|
" \n",
|
|
" for item in data:\n",
|
|
" m_url = item.get('href')\n",
|
|
" m_name = item.text.strip()\n",
|
|
" if m_name not in l_name:\n",
|
|
" url = 'https://gaokao.chsi.com.cn'+m_url\n",
|
|
" col_name = m_name \n",
|
|
" dict1 = {}\n",
|
|
" dict1['name'] = m_name\n",
|
|
" dict1['url'] = m_url\n",
|
|
" dict1['bz'] = 0 \n",
|
|
" mycol.insert_one(dict1) \n",
|
|
" print(dict1)\n",
|
|
" sleep(3)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "markdown",
|
|
"metadata": {},
|
|
"source": [
|
|
"## 检查、新增招生简章"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": 9,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": [
|
|
"from time import sleep\n",
|
|
"from re import findall,sub,S\n",
|
|
"import ssl\n",
|
|
"from bs4 import BeautifulSoup\n",
|
|
"import pymongo\n",
|
|
"import requests\n",
|
|
"\n",
|
|
"myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
|
|
"mydb = myclient[\"gaokao\"]\n",
|
|
"mycol = mydb[\"zhaoshengjianzhang\"]\n",
|
|
"headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n",
|
|
"l_name = []\n",
|
|
"for x in mycol.find({\"bz\": 0 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n",
|
|
" l_name.append(x['name'])\n",
|
|
"sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n",
|
|
"n_name = []\n",
|
|
"for i in sf:\n",
|
|
" url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk' \n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n",
|
|
" \n",
|
|
" for item in data:\n",
|
|
" m_url = item.get('href')\n",
|
|
" col_name = item.text.strip()\n",
|
|
" #print(m_url)\n",
|
|
" if (item.get('style') !='color:gray') and (col_name in l_name):\n",
|
|
" url = 'https://gaokao.chsi.com.cn' + m_url\n",
|
|
" \n",
|
|
" strhtml = requests.get(url,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n",
|
|
" for item in data:\n",
|
|
" url1 = item.get('href')\n",
|
|
" print(url1)\n",
|
|
" url1 = 'https://gaokao.chsi.com.cn' + url1\n",
|
|
" strhtml = requests.get(url1,headers = headers)\n",
|
|
" strhtml.encoding = 'utf8'\n",
|
|
" soup = BeautifulSoup(strhtml.text,'lxml')\n",
|
|
" data = soup.select('body > div.width1000.border.gery > div>p')\n",
|
|
" nr = ''\n",
|
|
" for item in data:\n",
|
|
" nr += item.text+'\\n' \n",
|
|
" myquery = {'name':col_name}\n",
|
|
" mycol.update_one(myquery,{'$set':{'bz':1}})\n",
|
|
" mycol.update_one(myquery,{'$push':{'content':nr}})\n",
|
|
" print(col_name)\n",
|
|
" sleep(3)"
|
|
]
|
|
},
|
|
{
|
|
"cell_type": "code",
|
|
"execution_count": null,
|
|
"metadata": {},
|
|
"outputs": [],
|
|
"source": []
|
|
}
|
|
],
|
|
"metadata": {
|
|
"kernelspec": {
|
|
"display_name": "Python 3",
|
|
"language": "python",
|
|
"name": "python3"
|
|
},
|
|
"language_info": {
|
|
"codemirror_mode": {
|
|
"name": "ipython",
|
|
"version": 3
|
|
},
|
|
"file_extension": ".py",
|
|
"mimetype": "text/x-python",
|
|
"name": "python",
|
|
"nbconvert_exporter": "python",
|
|
"pygments_lexer": "ipython3",
|
|
"version": "3.8.10"
|
|
}
|
|
},
|
|
"nbformat": 4,
|
|
"nbformat_minor": 4
|
|
}
|