Files
jupyter/数据采集.ipynb
T
2024-03-28 19:40:49 +08:00

129 lines
3.3 KiB
Plaintext

{
"cells": [
{
"cell_type": "markdown",
"id": "e01539fe-9ee2-4604-b464-a9dd050bd2d7",
"metadata": {},
"source": [
"## 中国戏考数据采集"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b054c312-d5ee-46b7-a173-f00ac8c2c952",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import requests\n",
"import time\n",
"from bs4 import BeautifulSoup\n",
"import pymongo\n",
"import re\n",
"\n",
"#m_mongo = {}\n",
"#requests.packages.urllib3.disable_warnings()\n",
"#requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n",
"#myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n",
"#mydb = myclient[\"gaokao\"]\n",
"#mycol = mydb[\"news\"]\n",
"url = 'data/京剧剧本 - 总目.html'\n",
"a_url = 'https://scripts.xikao.com'\n",
"#strhtml = requests.get(url)\n",
"strhtml = 'data/京剧剧本 - 总目.html'\n",
"#strhtml.encoding = 'utf8'\n",
"soup = BeautifulSoup(strhtml,'html.parser')\n",
"data = soup.select('#master-table')\n",
"for item in data:\n",
" print(item)\n",
" print('\\br')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "66a2d507-1232-4cb3-abbd-09fa7c4df315",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from bs4 import BeautifulSoup\n",
"\n",
"with open(\"data/京剧剧本 - 总目.html\") as fp:\n",
" soup = BeautifulSoup(fp, 'html.parser')\n",
"\n",
"#soup = BeautifulSoup(\"<html>a web page</html>\", 'html.parser')\n",
"data = soup.find_all(id = 'master-table')\n",
"for item in data:\n",
" list1 = item.find_all(attrs = {'class':'row_c'})\n",
" \n",
" #print(list1)\n",
" for tr in list1:\n",
" print(tr)\n",
" #if tr.b:\n",
" # print(tr.b.text)\n",
" # print('end')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5cc74584-aadc-488b-9426-228838d9c3db",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"from bs4 import BeautifulSoup\n",
"\n",
"with open(\"data/京剧剧本 - 总目.html\") as fp:\n",
" soup = BeautifulSoup(fp, 'html.parser')\n",
"\n",
"#soup = BeautifulSoup(\"<html>a web page</html>\", 'html.parser')\n",
"data = soup.find_all(id = 'master-table')\n",
"for item in data:\n",
" list1 = item.find_all(lambda tag: tag.get('class') == ['row_c'] or tag.get('class') == ['row_a'])\n",
" \n",
" #print(list1)\n",
" for tr in list1:\n",
" #print(tr)\n",
" if tr.a:\n",
" print(tr.a.get('href'),tr.b.text)\n",
" print('end')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "1b3b3b53-d9c1-4e80-8e25-19165d2921e9",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.10.12"
}
},
"nbformat": 4,
"nbformat_minor": 5
}