Files
jupyter/数据采集.ipynb
T
2024-03-28 19:40:49 +08:00

3.3 KiB

中国戏考数据采集

In [ ]:
import requests
import time
from bs4 import BeautifulSoup
import pymongo
import re

#m_mongo = {}
#requests.packages.urllib3.disable_warnings()
#requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'
#myclient = pymongo.MongoClient('mongodb://localhost:27017/')
#mydb = myclient["gaokao"]
#mycol = mydb["news"]
url = 'data/京剧剧本 - 总目.html'
a_url = 'https://scripts.xikao.com'
#strhtml = requests.get(url)
strhtml = 'data/京剧剧本 - 总目.html'
#strhtml.encoding = 'utf8'
soup = BeautifulSoup(strhtml,'html.parser')
data = soup.select('#master-table')
for item in data:
    print(item)
    print('\br')
In [ ]:
from bs4 import BeautifulSoup

with open("data/京剧剧本 - 总目.html") as fp:
    soup = BeautifulSoup(fp, 'html.parser')

#soup = BeautifulSoup("<html>a web page</html>", 'html.parser')
data = soup.find_all(id = 'master-table')
for item in data:
    list1  = item.find_all(attrs = {'class':'row_c'})
    
    #print(list1)
    for tr in list1:
        print(tr)
        #if tr.b:
        #    print(tr.b.text)
        #    print('end')
In [ ]:
from bs4 import BeautifulSoup

with open("data/京剧剧本 - 总目.html") as fp:
    soup = BeautifulSoup(fp, 'html.parser')

#soup = BeautifulSoup("<html>a web page</html>", 'html.parser')
data = soup.find_all(id = 'master-table')
for item in data:
    list1  = item.find_all(lambda tag: tag.get('class') == ['row_c']  or tag.get('class') == ['row_a'])
    
    #print(list1)
    for tr in list1:
        #print(tr)
        if tr.a:
            print(tr.a.get('href'),tr.b.text)
            print('end')
In [ ]: