Files
jupyter/高考志愿管理.ipynb
T
2022-06-16 15:19:28 +08:00

27 KiB

高考志愿管理

2020年高考录取信息导入MongoDB

In [ ]:
import pymysql
import pymongo
import decimal

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college"]
mycol1 = mydb["admission_2020"]

m_col = {}
m_spe = {}
m_xx = {}
for x in mycol.find({"code":{'$exists': 'true'}},{"_id": 0, "code": 1, "name": 1}):
    m_col[x['code']] = x['name']


db = pymysql.connect(host = "localhost",user = "songyi",password = "yylzs",database = "gaokao" )
cursor = db.cursor()
sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian="2020"'
cursor.execute(sql)
results = cursor.fetchall()
for result in results:
    m_spe.setdefault(result[1],{})    
    m_spe[result[1]][result[0]] = result[2]
sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'
cursor.execute(sql)
results = cursor.fetchall()
i = 0
m_min = 0
ii = 0
for result in results:
    m_xx.clear()
    if result[6] == m_min:
        ii = ii
        i = i+1
    else:
        i = i+1
        ii = i
        m_min = result[6]
    m_xx['pos'] = ii
    m_xx['col_code'] = result[1]
    m_xx['col_name'] = m_col[result[1]]
    m_xx['spe_code'] = result[2]
    m_xx['spe_name'] = m_spe[result[1]][result[2]]
    m_xx['plan'] = result[3]
    m_xx['dispense'] = result[5]
    m_xx['num_min'] = result[6]
    m_xx['num_avg'] = int(result[7])
    m_xx['rank_min'] = result[8]
    m_xx['nian'] = '2020'    
    mycol1.insert_one(m_xx)
#print(m_spe)

2021年全国高等学校信息更新

2021年全国高等学校信息导出至json文件

In [21]:
import openpyxl
import json

filename = 'data/2021年全国普通高等学校名单.xlsx'
wb = openpyxl.load_workbook(filename)
sheet = wb.active
data1 =list(sheet.values)
del data1[0:3]
i = 1
dict1 = {}
for items in data1:
    list1 = (items[1],str(items[2]),items[3],items[4],items[5],items[6])
    dict1[i] = list1
    i += 1 
with open('data/2021年全国普通高等学校名单.json','w') as fl2:
    json.dump(dict1,fl2) 

整理2021年全国高等学校信息

In [25]:
import openpyxl
import json
import pymongo


filename = 'data/2021年全国普通高等学校名单.json'
with open(filename,'r') as fl:
    m_xx = json.load(fl)

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college_2021"]

m_col = {}


for x in mycol.find({"id_code":{'$exists': 'true'}},{"_id": 0, "id_code": 1, "name": 1}):
    m_col[x['id_code']] = x['name']
#print(m_col.keys())
for k, v in m_xx.items():
    m_code = v[1]
    #print(m_code)
    if m_code not in m_col.keys():
        dict1 = {}
        dict1['id_code'] = v[1]
        dict1['name'] = v[0]
        dict1['charge'] = v[2]
        dict1['city'] = v[3]
        dict1['grade'] = v[4]
        dict1['note'] = v[5]
        mycol.insert_one(dict1) 
        print(dict1['name'] + '导入成功!')
        #print(v[0],v[1])
        


        
河北工业职业技术大学导入成功!
河北科技工程职业技术大学导入成功!
河北石油职业技术大学导入成功!
邢台应用技术职业学院导入成功!
山西工程科技职业大学导入成功!
吉林通用航空职业技术学院导入成功!
通化医药健康职业学院导入成功!
上海南湖职业技术学院导入成功!
浙江药科职业大学导入成功!
浙江金华科贸职业技术学院导入成功!
宿州航空职业学院导入成功!
和君职业学院导入成功!
滨州科技职业学院导入成功!
洛阳文化旅游职业学院导入成功!
周口文理职业学院导入成功!
信阳艺术职业学院导入成功!
郑州城建职业学院导入成功!
郑州医药健康职业学院导入成功!
湖北孝感美珈职业学院导入成功!
广州幼儿师范高等专科学校导入成功!
广东汕头幼儿师范高等专科学校导入成功!
广东梅州职业技术学院导入成功!
广东潮州卫生健康职业学院导入成功!
广东云浮中医药职业学院导入成功!
广东肇庆航空职业学院导入成功!
广西农业职业技术大学导入成功!
防城港职业技术学院导入成功!
广西信息职业技术学院导入成功!
广西农业工程职业技术学院导入成功!
北海康养职业学院导入成功!
重庆工信职业学院导入成功!
甘孜职业学院导入成功!
自贡职业技术学院导入成功!
贵阳康养职业大学导入成功!
贵州文化旅游职业学院导入成功!
宝鸡中北职业学院导入成功!
兰州石化职业技术大学导入成功!
兰州资源环境职业技术大学导入成功!
兰州航空职业技术学院导入成功!
白银希望职业技术学院导入成功!

计算志愿分数概率

In [ ]:
import random

array2 = []
for i in range(300000):
    array1 = []
    s = 0
    for ii in range(60):
        m1 = random.randint(580,595)
        array1.append(m1)
        s = s + m1    
    m_avg = round(s/60,2)
    if m_avg == 582.8 and (580 in array1) and (595 in array1):
        #print(array1)
        array2.extend(array1)
        
#print(array2)
m_set = set(array2)
for m in m_set:
    print(m,array2.count(m))

导入山东大学录取明细

In [ ]:
import openpyxl
import pymongo

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["admission_college"]
wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')
sheet = wb.active
#sheets = wb.sheetnames
code = 'A422'
name = '山东大学'
new_col = []
dict1 = {}

new_code = []
dict1['code'] = code
dict1['name'] = name
for n in range(1,sheet.max_row+1):
    nian = str(sheet.cell(n,1).value)
    dict2 = {}
    
    dict1.setdefault(nian,[])
    if sheet.cell(n,2).value =='理工':
        m_lb = 'l'
    elif sheet.cell(n,2).value =='文史':
        m_lb = 'w'
    else:
        m_lb = 'z'    
    dict2['type'] = m_lb
    dict2['spe_name'] = sheet.cell(n,4).value
    dict2['max_score'] = sheet.cell(n,5).value
    dict2['min_score'] = sheet.cell(n,6).value
    dict2['avg_score'] = sheet.cell(n,7).value
    dict2['dispense'] = sheet.cell(n,8).value
    dict1[nian].append(dict2)
mycol.insert_one(dict1)

导入山东师范大学录取明细

In [ ]:
import openpyxl
import pymongo

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["admission_college"]
wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')
sheet = wb.active
#sheets = wb.sheetnames
code = 'A423'
name = '中国海洋大学'
new_col = []
dict1 = {}

new_code = []
dict1['code'] = code
dict1['name'] = name
for n in range(1,sheet.max_row+1):
    nian = str(sheet.cell(n,1).value)
    dict2 = {}
    
    dict1.setdefault(nian,[])
    if sheet.cell(n,2).value =='理工':
        m_lb = 'l'
    elif sheet.cell(n,2).value =='文史':
        m_lb = 'w'
    else:
        m_lb = 'z'
    dict2['type'] = m_lb
    dict2['spe_name'] = sheet.cell(n,4).value
    dict2['max_score'] = sheet.cell(n,7).value
    dict2['min_score'] = sheet.cell(n,5).value
    dict2['avg_score'] = sheet.cell(n,6).value
    if sheet.cell(n,8).value:
        dict2['dispense'] = sheet.cell(n,8).value
    dict1[nian].append(dict2)
mycol.insert_one(dict1)

按地区列示高校

In [ ]:
import pymongo

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college"]
m_city = []
m_xx = {}
myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}
colleges = mycol.find(myquery,{ "_id": 0, "name": 1, "code": 1,'city':1 })
for x in colleges:
    m_xx1 = []
    c = x['city']
    m_xx.setdefault(c,[])
    m_xx1.append(x['code'])
    m_xx1.append(x['name'])
    m_xx[c].append(m_xx1)
#print(m_city)
#按照原顺序对高校所在城市排序
'''
city = list(set(m_city))
city.sort(key=m_city.index)
for c in city:
#    m_xx['city'] = c
    m_xx.setdefault(c,[])
    for y in colleges:
        print(y['code'],y['name'])
        
#m_xx    
'''    
for k,v in m_xx.items():
    print(k)
    for mm in v:
        print(mm[0],mm[1])

显示学校2020年招生信息

In [ ]:
import pymongo

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["admission_2020"]
code = 'A422'
m_city = []
m_xx = {}
myquery = {'col_code':code}
colleges = mycol.find(myquery,{ "_id": 0 , "col_code":0,"col_name":0,'plan':0,'nian':0}).sort('pos')
for x in colleges:
    print(x)

2021年拟在山东招生普通高校专业(类)选考科目要求

In [ ]:
from os import mkdir
from time import sleep
from re import findall,sub,S
from os.path import isdir,isfile
from urllib.request import  urlopen
from urllib.parse import  urlencode,quote
from openpyxl import Workbook
import ssl
from bs4 import BeautifulSoup
import pymongo
ssl._create_default_https_context = ssl._create_unverified_context
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["xuankaokemu"]
list2 = []
for x in mycol.find({},{ "_id": 0, "code": 1}):
    list2.append(x['code'])
#m_xx = dict()
start_url = 'https://xkkm.sdzk.cn/web/xx.html'
with urlopen(start_url) as fp:
    content = fp.read().decode('utf8')
    
pattern = (r'<tr>.*?<td.+?</td>.*?<td.+?>(.+?)</td>'
           '.*?<td.+?>(.+?)</td>.*?<td.+?>(.+?)</td>')

for item in findall(pattern,content,S):
    if len(item[0]) > 5:
        continue
    
    shengfen,dm,mc = item
    print('学校代码:', dm)
    print('学校名称:', mc)
    m_xx = {}
    m_xx['code'] = dm
    m_xx['name'] = mc
    m_xx.setdefault('zhuanye',{})
    if dm in list2:
        continue
    url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'
    data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')
    with  urlopen(url,data) as fp:
        xuexiao_content = fp.read().decode()
        soup = BeautifulSoup(xuexiao_content,'lxml')
        #data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')
        data = soup.select('#ccc > div > table > tbody > tr ')
        for data1 in data:
            list1 =[]
            m_xx1 = {}
            for item in data1.stripped_strings:            
                list1.append(item)
            #del list1[0]
            #print('层次:',list1[1])
            #print('专业(类)名称:',list1[2])
            #print('选考科目范围:',list1[3])
            #print('类中所含专业:',list1[4:])
            code = list1[0]
            m_xx['zhuanye'].setdefault(code,{})
            
            m_xx['zhuanye'][code]['name'] = list1[2]
            m_xx['zhuanye'][code]['level'] = list1[1]
            m_xx['zhuanye'][code]['fanwei'] = list1[3]
            m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]
    mycol.insert_one(m_xx) 
            
    print('ok')
            
        #    list1.clear
        

    sleep(5)
In [ ]:
import pymongo

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["xuankaokemu"]
list2 = []
for x in mycol.find({},{ "_id": 0, "code": 1}):
    list2.append(x['code'])
print(list2)

招生简章管理

各大学招生简章采集

In [ ]:
from time import sleep
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'

sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]
for i in sf:
    url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'
    strhtml = requests.get(url,headers = headers)
    strhtml.encoding = 'utf8'
    soup = BeautifulSoup(strhtml.text,'lxml')
    data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')
    
    for item in data:
        dict1 = {}
        dict1['name'] = item.text.strip()
        dict1['url'] = item.get('href')
        if item.get('style') =='color:gray':
            dict1['bz'] = 0
        else:
            dict1['bz'] = 1
        mycol.insert_one(dict1) 


采集单个学校招生简章

In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
#data = strhtml.text
#data1 = data.split("\r")
#print(soup)
data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')
for item in data:
    url1 = item.get('href')
url1 = 'https://gaokao.chsi.com.cn/' + url1
strhtml = requests.get(url1,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000.border.gery > div>p')
nr = ''
for item in data:
    nr += item.text+'\n'
print(nr)
    

第一次采集招生简章

In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
for x in mycol.find({"bz": 1 },{ "_id": 0, "name": 1, "url": 1 }):
    url = 'https://gaokao.chsi.com.cn'+x['url']
    col_name = x['name']
    strhtml = requests.get(url,headers = headers)
    strhtml.encoding = 'utf8'
    soup = BeautifulSoup(strhtml.text,'lxml')
    #data = strhtml.text
    #data1 = data.split("\r")
    #print(soup)
    data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')
    for item in data:
        url1 = item.get('href')
    url1 = 'https://gaokao.chsi.com.cn/' + url1
    strhtml = requests.get(url1,headers = headers)
    strhtml.encoding = 'utf8'
    soup = BeautifulSoup(strhtml.text,'lxml')
    data = soup.select('body > div.width1000.border.gery > div>p')
    nr = ''
    for item in data:
        nr += item.text+'\n'        
    myquery = {'name':col_name}
    mycol.update_one(myquery,{'$push':{'content':nr}})
    sleep(5)
    

检查新增的学校及招生简章

In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
l_name = []
for x in mycol.find({},{ "_id": 0, "name": 1}):
    l_name.append(x['name'])
sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]
n_url = []
for i in sf:
    url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'
    strhtml = requests.get(url,headers = headers)
    strhtml.encoding = 'utf8'
    soup = BeautifulSoup(strhtml.text,'lxml')
    data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')
    
    for item in data:
        m_url = item.get('href')
        m_name = item.text.strip()
        if m_name not in l_name:
            url = 'https://gaokao.chsi.com.cn'+m_url
            col_name = m_name            
            dict1 = {}
            dict1['name'] = m_name
            dict1['url'] = m_url
            dict1['bz'] = 0           
            mycol.insert_one(dict1) 
            print(dict1)
            sleep(3)

检查、新增招生简章

In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests

myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
l_name = []
for x in mycol.find({"bz": 0 },{ "_id": 0, "name": 1, "url": 1 }):
    l_name.append(x['name'])
sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]
n_name = []
for i in sf:
    url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'   
    strhtml = requests.get(url,headers = headers)
    strhtml.encoding = 'utf8'
    soup = BeautifulSoup(strhtml.text,'lxml')
    data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')
    
    for item in data:
        m_url = item.get('href')
        col_name = item.text.strip()
        #print(m_url)
        if (item.get('style') !='color:gray') and (col_name in l_name):
            url = 'https://gaokao.chsi.com.cn' + m_url
            
            strhtml = requests.get(url,headers = headers)
            strhtml.encoding = 'utf8'
            soup = BeautifulSoup(strhtml.text,'lxml')
            data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')
            for item in data:
                url1 = item.get('href')
            print(url1)
            url1 = 'https://gaokao.chsi.com.cn' + url1
            strhtml = requests.get(url1,headers = headers)
            strhtml.encoding = 'utf8'
            soup = BeautifulSoup(strhtml.text,'lxml')
            data = soup.select('body > div.width1000.border.gery > div>p')
            nr = ''
            for item in data:
                nr += item.text+'\n'        
            myquery = {'name':col_name}
            mycol.update_one(myquery,{'$set':{'bz':1}})
            mycol.update_one(myquery,{'$push':{'content':nr}})
            print(col_name)
            sleep(3)
In [ ]: