22 KiB
22 KiB
In [ ]:
import pymysql
import pymongo
import decimal
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college"]
mycol1 = mydb["admission_2020"]
m_col = {}
m_spe = {}
m_xx = {}
for x in mycol.find({"code":{'$exists': 'true'}},{"_id": 0, "code": 1, "name": 1}):
m_col[x['code']] = x['name']
db = pymysql.connect(host = "localhost",user = "songyi",password = "yylzs",database = "gaokao" )
cursor = db.cursor()
sql = 'SELECT a.code,a.college,a.name FROM speciality AS a WHERE a.nian="2020"'
cursor.execute(sql)
results = cursor.fetchall()
for result in results:
m_spe.setdefault(result[1],{})
m_spe[result[1]][result[0]] = result[2]
sql = 'SELECT * FROM admission_2020 AS a ORDER BY a.rank_min'
cursor.execute(sql)
results = cursor.fetchall()
i = 0
m_min = 0
ii = 0
for result in results:
m_xx.clear()
if result[6] == m_min:
ii = ii
i = i+1
else:
i = i+1
ii = i
m_min = result[6]
m_xx['pos'] = ii
m_xx['col_code'] = result[1]
m_xx['col_name'] = m_col[result[1]]
m_xx['spe_code'] = result[2]
m_xx['spe_name'] = m_spe[result[1]][result[2]]
m_xx['plan'] = result[3]
m_xx['dispense'] = result[5]
m_xx['num_min'] = result[6]
m_xx['num_avg'] = int(result[7])
m_xx['rank_min'] = result[8]
m_xx['nian'] = '2020'
mycol1.insert_one(m_xx)
#print(m_spe)
In [ ]:
import random
array2 = []
for i in range(300000):
array1 = []
s = 0
for ii in range(60):
m1 = random.randint(580,595)
array1.append(m1)
s = s + m1
m_avg = round(s/60,2)
if m_avg == 582.8 and (580 in array1) and (595 in array1):
#print(array1)
array2.extend(array1)
#print(array2)
m_set = set(array2)
for m in m_set:
print(m,array2.count(m))In [ ]:
import openpyxl
import pymongo
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["admission_college"]
wb = openpyxl.load_workbook('./data/shandongdaxue.xlsx')
sheet = wb.active
#sheets = wb.sheetnames
code = 'A422'
name = '山东大学'
new_col = []
dict1 = {}
new_code = []
dict1['code'] = code
dict1['name'] = name
for n in range(1,sheet.max_row+1):
nian = str(sheet.cell(n,1).value)
dict2 = {}
dict1.setdefault(nian,[])
if sheet.cell(n,2).value =='理工':
m_lb = 'l'
elif sheet.cell(n,2).value =='文史':
m_lb = 'w'
else:
m_lb = 'z'
dict2['type'] = m_lb
dict2['spe_name'] = sheet.cell(n,4).value
dict2['max_score'] = sheet.cell(n,5).value
dict2['min_score'] = sheet.cell(n,6).value
dict2['avg_score'] = sheet.cell(n,7).value
dict2['dispense'] = sheet.cell(n,8).value
dict1[nian].append(dict2)
mycol.insert_one(dict1)In [ ]:
import openpyxl
import pymongo
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["admission_college"]
wb = openpyxl.load_workbook('./data/中国海洋大学.xlsx')
sheet = wb.active
#sheets = wb.sheetnames
code = 'A423'
name = '中国海洋大学'
new_col = []
dict1 = {}
new_code = []
dict1['code'] = code
dict1['name'] = name
for n in range(1,sheet.max_row+1):
nian = str(sheet.cell(n,1).value)
dict2 = {}
dict1.setdefault(nian,[])
if sheet.cell(n,2).value =='理工':
m_lb = 'l'
elif sheet.cell(n,2).value =='文史':
m_lb = 'w'
else:
m_lb = 'z'
dict2['type'] = m_lb
dict2['spe_name'] = sheet.cell(n,4).value
dict2['max_score'] = sheet.cell(n,7).value
dict2['min_score'] = sheet.cell(n,5).value
dict2['avg_score'] = sheet.cell(n,6).value
if sheet.cell(n,8).value:
dict2['dispense'] = sheet.cell(n,8).value
dict1[nian].append(dict2)
mycol.insert_one(dict1)
In [ ]:
import pymongo
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["college"]
m_city = []
m_xx = {}
myquery = {'code':{'$exists': 'true'},'note':{'$not':{'$regex':'民办'}}}
colleges = mycol.find(myquery,{ "_id": 0, "name": 1, "code": 1,'city':1 })
for x in colleges:
m_xx1 = []
c = x['city']
m_xx.setdefault(c,[])
m_xx1.append(x['code'])
m_xx1.append(x['name'])
m_xx[c].append(m_xx1)
#print(m_city)
#按照原顺序对高校所在城市排序
'''
city = list(set(m_city))
city.sort(key=m_city.index)
for c in city:
# m_xx['city'] = c
m_xx.setdefault(c,[])
for y in colleges:
print(y['code'],y['name'])
#m_xx
'''
for k,v in m_xx.items():
print(k)
for mm in v:
print(mm[0],mm[1])In [ ]:
import pymongo
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["admission_2020"]
code = 'A422'
m_city = []
m_xx = {}
myquery = {'col_code':code}
colleges = mycol.find(myquery,{ "_id": 0 , "col_code":0,"col_name":0,'plan':0,'nian':0}).sort('pos')
for x in colleges:
print(x)In [ ]:
from os import mkdir
from time import sleep
from re import findall,sub,S
from os.path import isdir,isfile
from urllib.request import urlopen
from urllib.parse import urlencode,quote
from openpyxl import Workbook
import ssl
from bs4 import BeautifulSoup
import pymongo
ssl._create_default_https_context = ssl._create_unverified_context
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["xuankaokemu"]
list2 = []
for x in mycol.find({},{ "_id": 0, "code": 1}):
list2.append(x['code'])
#m_xx = dict()
start_url = 'https://xkkm.sdzk.cn/web/xx.html'
with urlopen(start_url) as fp:
content = fp.read().decode('utf8')
pattern = (r'<tr>.*?<td.+?</td>.*?<td.+?>(.+?)</td>'
'.*?<td.+?>(.+?)</td>.*?<td.+?>(.+?)</td>')
for item in findall(pattern,content,S):
if len(item[0]) > 5:
continue
shengfen,dm,mc = item
print('学校代码:', dm)
print('学校名称:', mc)
m_xx = {}
m_xx['code'] = dm
m_xx['name'] = mc
m_xx.setdefault('zhuanye',{})
if dm in list2:
continue
url = r'https://xkkm.sdzk.cn/xkkm/queryXxInfor'
data = urlencode({'dm':dm,'mc':quote(mc),'yzm':'ok'}).encode('ascii')
with urlopen(url,data) as fp:
xuexiao_content = fp.read().decode()
soup = BeautifulSoup(xuexiao_content,'lxml')
#data1 = soup.select('#ccc > div > table > tbody > tr > td:nth-child(5)')
data = soup.select('#ccc > div > table > tbody > tr ')
for data1 in data:
list1 =[]
m_xx1 = {}
for item in data1.stripped_strings:
list1.append(item)
#del list1[0]
#print('层次:',list1[1])
#print('专业(类)名称:',list1[2])
#print('选考科目范围:',list1[3])
#print('类中所含专业:',list1[4:])
code = list1[0]
m_xx['zhuanye'].setdefault(code,{})
m_xx['zhuanye'][code]['name'] = list1[2]
m_xx['zhuanye'][code]['level'] = list1[1]
m_xx['zhuanye'][code]['fanwei'] = list1[3]
m_xx['zhuanye'][code]['suohanzhuanye'] = list1[4:]
mycol.insert_one(m_xx)
print('ok')
# list1.clear
sleep(5)In [ ]:
import pymongo
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["xuankaokemu"]
list2 = []
for x in mycol.find({},{ "_id": 0, "code": 1}):
list2.append(x['code'])
print(list2)In [ ]:
from time import sleep
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
#url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm=11&yxls=&yxlx=&xlcc=bk'
sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]
for i in sf:
url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')
for item in data:
dict1 = {}
dict1['name'] = item.text.strip()
dict1['url'] = item.get('href')
if item.get('style') =='color:gray':
dict1['bz'] = 0
else:
dict1['bz'] = 1
mycol.insert_one(dict1)
In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
url = 'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listZszc--schId-5.dhtml'
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
#data = strhtml.text
#data1 = data.split("\r")
#print(soup)
data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')
for item in data:
url1 = item.get('href')
url1 = 'https://gaokao.chsi.com.cn/' + url1
strhtml = requests.get(url1,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000.border.gery > div>p')
nr = ''
for item in data:
nr += item.text+'\n'
print(nr)
In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
for x in mycol.find({"bz": 1 },{ "_id": 0, "name": 1, "url": 1 }):
url = 'https://gaokao.chsi.com.cn'+x['url']
col_name = x['name']
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
#data = strhtml.text
#data1 = data.split("\r")
#print(soup)
data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')
for item in data:
url1 = item.get('href')
url1 = 'https://gaokao.chsi.com.cn/' + url1
strhtml = requests.get(url1,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000.border.gery > div>p')
nr = ''
for item in data:
nr += item.text+'\n'
myquery = {'name':col_name}
mycol.update_one(myquery,{'$push':{'content':nr}})
sleep(5)
In [ ]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
l_name = []
for x in mycol.find({},{ "_id": 0, "name": 1}):
l_name.append(x['name'])
sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]
n_url = []
for i in sf:
url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')
for item in data:
m_url = item.get('href')
m_name = item.text.strip()
if m_name not in l_name:
url = 'https://gaokao.chsi.com.cn'+m_url
col_name = m_name
dict1 = {}
dict1['name'] = m_name
dict1['url'] = m_url
dict1['bz'] = 0
mycol.insert_one(dict1)
print(dict1)
sleep(3)In [9]:
from time import sleep
from re import findall,sub,S
import ssl
from bs4 import BeautifulSoup
import pymongo
import requests
myclient = pymongo.MongoClient('mongodb://localhost:27017/')
mydb = myclient["gaokao"]
mycol = mydb["zhaoshengjianzhang"]
headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Safari/537.36'}
l_name = []
for x in mycol.find({"bz": 0 },{ "_id": 0, "name": 1, "url": 1 }):
l_name.append(x['name'])
sf = [11,12,13,14,15,21,22,23,31,32,33,34,35,36,37,41,42,43,44,45,46,50,51,52,53,54,61,62,63,64,65]
n_name = []
for i in sf:
url = f'https://gaokao.chsi.com.cn/zsgs/zhangcheng/listVerifedZszc.do?method=index&ssdm={str(i)}&yxls=&yxlx=&xlcc=bk'
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000 > table:nth-child(5) > tbody > tr> td > a')
for item in data:
m_url = item.get('href')
col_name = item.text.strip()
#print(m_url)
if (item.get('style') !='color:gray') and (col_name in l_name):
url = 'https://gaokao.chsi.com.cn' + m_url
strhtml = requests.get(url,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000.gery > div > div.right > table > tr > td > a')
for item in data:
url1 = item.get('href')
print(url1)
url1 = 'https://gaokao.chsi.com.cn' + url1
strhtml = requests.get(url1,headers = headers)
strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml.text,'lxml')
data = soup.select('body > div.width1000.border.gery > div>p')
nr = ''
for item in data:
nr += item.text+'\n'
myquery = {'name':col_name}
mycol.update_one(myquery,{'$set':{'bz':1}})
mycol.update_one(myquery,{'$push':{'content':nr}})
print(col_name)
sleep(3)In [ ]: