{ "cells": [ { "cell_type": "markdown", "id": "e01539fe-9ee2-4604-b464-a9dd050bd2d7", "metadata": {}, "source": [ "## 中国戏考数据采集" ] }, { "cell_type": "code", "execution_count": null, "id": "b054c312-d5ee-46b7-a173-f00ac8c2c952", "metadata": { "tags": [] }, "outputs": [], "source": [ "import requests\n", "import time\n", "from bs4 import BeautifulSoup\n", "import pymongo\n", "import re\n", "\n", "#m_mongo = {}\n", "#requests.packages.urllib3.disable_warnings()\n", "#requests.packages.urllib3.util.ssl_.DEFAULT_CIPHERS += 'HIGH:!DH:!aNULL'\n", "#myclient = pymongo.MongoClient('mongodb://localhost:27017/')\n", "#mydb = myclient[\"gaokao\"]\n", "#mycol = mydb[\"news\"]\n", "url = 'data/京剧剧本 - 总目.html'\n", "a_url = 'https://scripts.xikao.com'\n", "#strhtml = requests.get(url)\n", "strhtml = 'data/京剧剧本 - 总目.html'\n", "#strhtml.encoding = 'utf8'\n", "soup = BeautifulSoup(strhtml,'html.parser')\n", "data = soup.select('#master-table')\n", "for item in data:\n", " print(item)\n", " print('\\br')" ] }, { "cell_type": "code", "execution_count": null, "id": "66a2d507-1232-4cb3-abbd-09fa7c4df315", "metadata": { "tags": [] }, "outputs": [], "source": [ "from bs4 import BeautifulSoup\n", "\n", "with open(\"data/京剧剧本 - 总目.html\") as fp:\n", " soup = BeautifulSoup(fp, 'html.parser')\n", "\n", "#soup = BeautifulSoup(\"a web page\", 'html.parser')\n", "data = soup.find_all(id = 'master-table')\n", "for item in data:\n", " list1 = item.find_all(attrs = {'class':'row_c'})\n", " \n", " #print(list1)\n", " for tr in list1:\n", " print(tr)\n", " #if tr.b:\n", " # print(tr.b.text)\n", " # print('end')" ] }, { "cell_type": "code", "execution_count": null, "id": "5cc74584-aadc-488b-9426-228838d9c3db", "metadata": { "tags": [] }, "outputs": [], "source": [ "from bs4 import BeautifulSoup\n", "\n", "with open(\"data/京剧剧本 - 总目.html\") as fp:\n", " soup = BeautifulSoup(fp, 'html.parser')\n", "\n", "#soup = BeautifulSoup(\"a web page\", 'html.parser')\n", "data = soup.find_all(id = 'master-table')\n", "for item in data:\n", " list1 = item.find_all(lambda tag: tag.get('class') == ['row_c'] or tag.get('class') == ['row_a'])\n", " \n", " #print(list1)\n", " for tr in list1:\n", " #print(tr)\n", " if tr.a:\n", " print(tr.a.get('href'),tr.b.text)\n", " print('end')" ] }, { "cell_type": "code", "execution_count": null, "id": "1b3b3b53-d9c1-4e80-8e25-19165d2921e9", "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3 (ipykernel)", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.10.12" } }, "nbformat": 4, "nbformat_minor": 5 }