{ "cells": [ { "cell_type": "markdown", "metadata": { "tags": [], "toc-hr-collapsed": true }, "source": [ "# 高考志愿管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 2020年高考录取信息导入MongoDB" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "tags": [] }, "outputs": [], "source": [ "import pymysql\n", "import pymongo\n", "import decimal\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "mycol1 = mydb[\"admission_2020\"]\n", "\n", "m_col = {}\n", "m_spe = {}\n", "m_xx = {}\n", "for x in mycol.find({\"code\":{'$exists': 'true'}},{\"_id\": 0, \"code\": 1, \"name\": 1}):\n", " m_col[x['code']] = x['name']\n", "\n", "\n", "db = pymysql.connect(host = \"localhost\",user = \"songyi\",password = \"yylzs\",database = \"gaokao\" )\n", "cursor = db.cursor()\n", "sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian=\"2020\"'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "for result in results:\n", " m_spe.setdefault(result[1],{}) \n", " m_spe[result[1]][result[0]] = result[2]\n", "sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'\n", "cursor.execute(sql)\n", "results = cursor.fetchall()\n", "i = 0\n", "m_min = 0\n", "ii = 0\n", "for result in results:\n", " m_xx.clear()\n", " if result[6] == m_min:\n", " ii = ii\n", " i = i+1\n", " else:\n", " i = i+1\n", " ii = i\n", " m_min = result[6]\n", " m_xx['pos'] = ii\n", " m_xx['col_code'] = result[1]\n", " m_xx['col_name'] = m_col[result[1]]\n", " m_xx['spe_code'] = result[2]\n", " m_xx['spe_name'] = m_spe[result[1]][result[2]]\n", " m_xx['plan'] = result[3]\n", " m_xx['dispense'] = result[5]\n", " m_xx['num_min'] = result[6]\n", " m_xx['num_avg'] = int(result[7])\n", " m_xx['rank_min'] = result[8]\n", " m_xx['nian'] = '2020' \n", " mycol1.insert_one(m_xx)\n", "#print(m_spe)\n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 计算志愿分数概率" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import random\n", "\n", "array2 = []\n", "for i in range(300000):\n", " array1 = []\n", " s = 0\n", " for ii in range(60):\n", " m1 = random.randint(580,595)\n", " array1.append(m1)\n", " s = s + m1 \n", " m_avg = round(s/60,2)\n", " if m_avg == 582.8 and (580 in array1) and (595 in array1):\n", " #print(array1)\n", " array2.extend(array1)\n", " \n", "#print(array2)\n", "m_set = set(array2)\n", "for m in m_set:\n", " print(m,array2.count(m))" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 导入山东大学录取明细" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import openpyxl\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"admission_college\"]\n", "wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')\n", "sheet = wb.active\n", "#sheets = wb.sheetnames\n", "code = 'A422'\n", "name = '山东大学'\n", "new_col = []\n", "dict1 = {}\n", "\n", "new_code = []\n", "dict1['code'] = code\n", "dict1['name'] = name\n", "for n in range(1,sheet.max_row+1):\n", " nian = str(sheet.cell(n,1).value)\n", " dict2 = {}\n", " \n", " dict1.setdefault(nian,[])\n", " if sheet.cell(n,2).value =='理工':\n", " m_lb = 'l'\n", " elif sheet.cell(n,2).value =='文史':\n", " m_lb = 'w'\n", " else:\n", " m_lb = 'z' \n", " dict2['type'] = m_lb\n", " dict2['spe_name'] = sheet.cell(n,4).value\n", " dict2['max_score'] = sheet.cell(n,5).value\n", " dict2['min_score'] = sheet.cell(n,6).value\n", " dict2['avg_score'] = sheet.cell(n,7).value\n", " dict2['dispense'] = sheet.cell(n,8).value\n", " dict1[nian].append(dict2)\n", "mycol.insert_one(dict1)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 导入山东师范大学录取明细" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import openpyxl\n", "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"admission_college\"]\n", "wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')\n", "sheet = wb.active\n", "#sheets = wb.sheetnames\n", "code = 'A423'\n", "name = '中国海洋大学'\n", "new_col = []\n", "dict1 = {}\n", "\n", "new_code = []\n", "dict1['code'] = code\n", "dict1['name'] = name\n", "for n in range(1,sheet.max_row+1):\n", " nian = str(sheet.cell(n,1).value)\n", " dict2 = {}\n", " \n", " dict1.setdefault(nian,[])\n", " if sheet.cell(n,2).value =='理工':\n", " m_lb = 'l'\n", " elif sheet.cell(n,2).value =='文史':\n", " m_lb = 'w'\n", " else:\n", " m_lb = 'z'\n", " dict2['type'] = m_lb\n", " dict2['spe_name'] = sheet.cell(n,4).value\n", " dict2['max_score'] = sheet.cell(n,7).value\n", " dict2['min_score'] = sheet.cell(n,5).value\n", " dict2['avg_score'] = sheet.cell(n,6).value\n", " if sheet.cell(n,8).value:\n", " dict2['dispense'] = sheet.cell(n,8).value\n", " dict1[nian].append(dict2)\n", "mycol.insert_one(dict1)\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 按地区列示高校" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"college\"]\n", "m_city = []\n", "m_xx = {}\n", "myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}\n", "colleges = mycol.find(myquery,{ \"_id\": 0, \"name\": 1, \"code\": 1,'city':1 })\n", "for x in colleges:\n", " m_xx1 = []\n", " c = x['city']\n", " m_xx.setdefault(c,[])\n", " m_xx1.append(x['code'])\n", " m_xx1.append(x['name'])\n", " m_xx[c].append(m_xx1)\n", "#print(m_city)\n", "#按照原顺序对高校所在城市排序\n", "'''\n", "city = list(set(m_city))\n", "city.sort(key=m_city.index)\n", "for c in city:\n", "# m_xx['city'] = c\n", " m_xx.setdefault(c,[])\n", " for y in colleges:\n", " print(y['code'],y['name'])\n", " \n", "#m_xx \n", "''' \n", "for k,v in m_xx.items():\n", " print(k)\n", " for mm in v:\n", " print(mm[0],mm[1])" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 显示学校2020年招生信息" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"admission_2020\"]\n", "code = 'A422'\n", "m_city = []\n", "m_xx = {}\n", "myquery = {'col_code':code}\n", "colleges = mycol.find(myquery,{ \"_id\": 0 , \"col_code\":0,\"col_name\":0,'plan':0,'nian':0}).sort('pos')\n", "for x in colleges:\n", " print(x)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 2021年拟在山东招生普通高校专业(类)选考科目要求" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from os import mkdir\n", "from time import sleep\n", "from re import findall,sub,S\n", "from os.path import isdir,isfile\n", "from urllib.request import urlopen\n", "from urllib.parse import urlencode,quote\n", "from openpyxl import Workbook\n", "import ssl\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "ssl._create_default_https_context = ssl._create_unverified_context\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"xuankaokemu\"]\n", "list2 = []\n", "for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n", " list2.append(x['code'])\n", "#m_xx = dict()\n", "start_url = 'https://xkkm.sdzk.cn/web/xx.html'\n", "with urlopen(start_url) as fp:\n", " content = fp.read().decode('utf8')\n", " \n", "pattern = (r'.*?.*?(.+?)'\n", " '.*?(.+?).*?(.+?)')\n", "\n", "for item in findall(pattern,content,S):\n", " if len(item[0]) > 5:\n", " continue\n", " \n", " shengfen,dm,mc = item\n", " print('学校代码:', dm)\n", " print('学校名称:', mc)\n", " m_xx = {}\n", " m_xx['code'] = dm\n", " m_xx['name'] = mc\n", " m_xx.setdefault('zhuanye',{})\n", " if dm in list2:\n", " continue\n", " url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'\n", " data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')\n", " with urlopen(url,data) as fp:\n", " xuexiao_content = fp.read().decode()\n", " soup = BeautifulSoup(xuexiao_content,'lxml')\n", " #data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')\n", " data = soup.select('#ccc > div > table > tbody > tr ')\n", " for data1 in data:\n", " list1 =[]\n", " m_xx1 = {}\n", " for item in data1.stripped_strings: \n", " list1.append(item)\n", " #del list1[0]\n", " #print('层次:',list1[1])\n", " #print('专业(类)名称:',list1[2])\n", " #print('选考科目范围:',list1[3])\n", " #print('类中所含专业:',list1[4:])\n", " code = list1[0]\n", " m_xx['zhuanye'].setdefault(code,{})\n", " \n", " m_xx['zhuanye'][code]['name'] = list1[2]\n", " m_xx['zhuanye'][code]['level'] = list1[1]\n", " m_xx['zhuanye'][code]['fanwei'] = list1[3]\n", " m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]\n", " mycol.insert_one(m_xx) \n", " \n", " print('ok')\n", " \n", " # list1.clear\n", " \n", "\n", " sleep(5)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "import pymongo\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"xuankaokemu\"]\n", "list2 = []\n", "for x in mycol.find({},{ \"_id\": 0, \"code\": 1}):\n", " list2.append(x['code'])\n", "print(list2)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# 招生简章管理" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 各大学招生简章采集" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from time import sleep\n", "import ssl\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import requests\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"zhaoshengjianzhang\"]\n", "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", "#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'\n", "\n", "sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n", "for i in sf:\n", " url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n", " strhtml = requests.get(url,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n", " \n", " for item in data:\n", " dict1 = {}\n", " dict1['name'] = item.text.strip()\n", " dict1['url'] = item.get('href')\n", " if item.get('style') =='color:gray':\n", " dict1['bz'] = 0\n", " else:\n", " dict1['bz'] = 1\n", " mycol.insert_one(dict1) \n", "\n", "\n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 采集单个学校招生简章" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from time import sleep\n", "from re import findall,sub,S\n", "import ssl\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import requests\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"zhaoshengjianzhang\"]\n", "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", "url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'\n", "strhtml = requests.get(url,headers = headers)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "#data = strhtml.text\n", "#data1 = data.split(\"\\r\")\n", "#print(soup)\n", "data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n", "for item in data:\n", " url1 = item.get('href')\n", "url1 = 'https://gaokao.chsi.com.cn/' + url1\n", "strhtml = requests.get(url1,headers = headers)\n", "strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml.text,'lxml')\n", "data = soup.select('body > div.width1000.border.gery > div>p')\n", "nr = ''\n", "for item in data:\n", " nr += item.text+'\\n'\n", "print(nr)\n", " \n" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 第一次采集招生简章" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from time import sleep\n", "from re import findall,sub,S\n", "import ssl\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import requests\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"zhaoshengjianzhang\"]\n", "for x in mycol.find({\"bz\": 1 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n", " url = 'https://gaokao.chsi.com.cn'+x['url']\n", " col_name = x['name']\n", " strhtml = requests.get(url,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " #data = strhtml.text\n", " #data1 = data.split(\"\\r\")\n", " #print(soup)\n", " data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n", " for item in data:\n", " url1 = item.get('href')\n", " url1 = 'https://gaokao.chsi.com.cn/' + url1\n", " strhtml = requests.get(url1,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data = soup.select('body > div.width1000.border.gery > div>p')\n", " nr = ''\n", " for item in data:\n", " nr += item.text+'\\n' \n", " myquery = {'name':col_name}\n", " mycol.update_one(myquery,{'$push':{'content':nr}})\n", " sleep(5)\n", " " ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 检查新增的学校及招生简章" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from time import sleep\n", "from re import findall,sub,S\n", "import ssl\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import requests\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"zhaoshengjianzhang\"]\n", "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", "l_name = []\n", "for x in mycol.find({},{ \"_id\": 0, \"name\": 1}):\n", " l_name.append(x['name'])\n", "sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n", "n_url = []\n", "for i in sf:\n", " url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'\n", " strhtml = requests.get(url,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n", " \n", " for item in data:\n", " m_url = item.get('href')\n", " m_name = item.text.strip()\n", " if m_name not in l_name:\n", " url = 'https://gaokao.chsi.com.cn'+m_url\n", " col_name = m_name \n", " dict1 = {}\n", " dict1['name'] = m_name\n", " dict1['url'] = m_url\n", " dict1['bz'] = 0 \n", " mycol.insert_one(dict1) \n", " print(dict1)\n", " sleep(3)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## 检查、新增招生简章" ] }, { "cell_type": "code", "execution_count": 9, "metadata": {}, "outputs": [], "source": [ "from time import sleep\n", "from re import findall,sub,S\n", "import ssl\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import requests\n", "\n", "myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "mydb = myclient[\"gaokao\"]\n", "mycol = mydb[\"zhaoshengjianzhang\"]\n", "headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}\n", "l_name = []\n", "for x in mycol.find({\"bz\": 0 },{ \"_id\": 0, \"name\": 1, \"url\": 1 }):\n", " l_name.append(x['name'])\n", "sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]\n", "n_name = []\n", "for i in sf:\n", " url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk' \n", " strhtml = requests.get(url,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')\n", " \n", " for item in data:\n", " m_url = item.get('href')\n", " col_name = item.text.strip()\n", " #print(m_url)\n", " if (item.get('style') !='color:gray') and (col_name in l_name):\n", " url = 'https://gaokao.chsi.com.cn' + m_url\n", " \n", " strhtml = requests.get(url,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')\n", " for item in data:\n", " url1 = item.get('href')\n", " print(url1)\n", " url1 = 'https://gaokao.chsi.com.cn' + url1\n", " strhtml = requests.get(url1,headers = headers)\n", " strhtml.encoding = 'utf8'\n", " soup = BeautifulSoup(strhtml.text,'lxml')\n", " data = soup.select('body > div.width1000.border.gery > div>p')\n", " nr = ''\n", " for item in data:\n", " nr += item.text+'\\n' \n", " myquery = {'name':col_name}\n", " mycol.update_one(myquery,{'$set':{'bz':1}})\n", " mycol.update_one(myquery,{'$push':{'content':nr}})\n", " print(col_name)\n", " sleep(3)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.8.10" } }, "nbformat": 4, "nbformat_minor": 4 }