This commit is contained in:
512song committed 2025-03-14 16:29:36 +08:00
1 parent 4da01b8af8
commit fe91eda8cc
5 files changed
+685 -208

No files matched your search

+126 -7
View File
@@ -1,12 +1,5 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# 数字文件名转换为文本文件名"
]
},
{
"cell_type": "markdown",
"metadata": {},
@@ -412,6 +405,132 @@
" fl1.writelines(list1)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 读取html文件"
]
},
{
"cell_type": "code",
"execution_count": 33,
"metadata": {
"execution": {
"iopub.execute_input": "2025-03-09T02:47:43.965042Z",
"iopub.status.busy": "2025-03-09T02:47:43.964342Z",
"iopub.status.idle": "2025-03-09T02:47:44.398962Z",
"shell.execute_reply": "2025-03-09T02:47:44.398531Z",
"shell.execute_reply.started": "2025-03-09T02:47:43.964980Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"<class 'list'>\n"
]
}
],
"source": [
"from bs4 import BeautifulSoup\n",
"import re\n",
"from pathlib import Path\n",
"\n",
"def get_html_title(html_file_path):\n",
" with open(html_file_path, 'r', encoding='utf-8') as f:\n",
" html_content = f.read()\n",
" \n",
" soup = BeautifulSoup(html_content, 'html.parser')\n",
" title_tag = soup.title\n",
" \n",
" if title_tag and hasattr(title_tag, 'string'):\n",
" return title_tag.string.strip()\n",
" else:\n",
" return None # 标签不存在或内容为空\n",
"\n",
"\n",
"path = Path('./file/guzi')\n",
"markdown_content = []\n",
"files = [file for file in path.iterdir() if file.is_file()]\n",
"sorted_files = sorted(files, key=lambda f: f.name)\n",
"print(type(sorted_files))\n",
"for file in sorted_files:\n",
" if file.suffix=='.html':\n",
" markdown_content.append('## '+get_html_title(file))\n",
" with open(file, 'r', encoding='utf-8') as f:\n",
" html_content = f.read() \n",
" soup = BeautifulSoup(html_content, 'html.parser')\n",
" div = soup.find('div', class_=\"show-content\")\n",
" p_elements = div.find_all('p') \n",
" for div in p_elements:\n",
" markdown_content.append(div.get_text(strip=True))\n",
" #markdown_content.append('\\n')\n",
" if file.suffix=='.md':\n",
" with open(file, 'r', encoding='utf-8') as f:\n",
" txt_content = f.read()\n",
" lines = txt_content.splitlines()\n",
" for line in lines:\n",
" line = line.strip()\n",
" markdown_content.append(line)\n",
" \n",
"markdown_content = '\\n'.join(markdown_content)\n",
"with open('guzi.md', 'w', encoding='utf-8') as file:\n",
" file.write(markdown_content)"
]
},
{
"cell_type": "code",
"execution_count": 29,
"metadata": {
"execution": {
"iopub.execute_input": "2025-03-09T02:36:40.207084Z",
"iopub.status.busy": "2025-03-09T02:36:40.206317Z",
"iopub.status.idle": "2025-03-09T02:36:40.223987Z",
"shell.execute_reply": "2025-03-09T02:36:40.223140Z",
"shell.execute_reply.started": "2025-03-09T02:36:40.207017Z"
}
},
"outputs": [],
"source": [
"from bs4 import BeautifulSoup\n",
"import re\n",
"from pathlib import Path\n",
"\n",
"\n",
"def get_html_title(html_file_path):\n",
" with open(html_file_path, 'r', encoding='utf-8') as f:\n",
" html_content = f.read()\n",
" \n",
" soup = BeautifulSoup(html_content, 'html.parser')\n",
" title_tag = soup.title\n",
" \n",
" if title_tag and hasattr(title_tag, 'string'):\n",
" return title_tag.string.strip()\n",
" else:\n",
" return None # 标签不存在或内容为空\n",
"\n",
"\n",
"file='file/guzi/1-01.html'\n",
"markdown_content = []\n",
"markdown_content.append('## '+get_html_title(file))\n",
"with open(file, 'r', encoding='utf-8') as f:\n",
" html_content = f.read()\n",
" \n",
" soup = BeautifulSoup(html_content, 'html.parser')\n",
" div = soup.find('div', class_=\"show-content\")\n",
" p_elements = div.find_all('p')\n",
" \n",
" for div in p_elements:\n",
" markdown_content.append(div.get_text(strip=True))\n",
" markdown_content.append('\\n')\n",
"markdown_content = '\\n'.join(markdown_content)\n",
"with open('1-01.md', 'w', encoding='utf-8') as file:\n",
" file.write(markdown_content)\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},