20250314
This commit is contained in:
1 parent
4da01b8af8
commit
fe91eda8cc
5 files changed
+685
-208
No files matched your search
+126
-7
@@ -1,12 +1,5 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"# 数字文件名转换为文本文件名"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
@@ -412,6 +405,132 @@
|
||||
" fl1.writelines(list1)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 读取html文件"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 33,
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2025-03-09T02:47:43.965042Z",
|
||||
"iopub.status.busy": "2025-03-09T02:47:43.964342Z",
|
||||
"iopub.status.idle": "2025-03-09T02:47:44.398962Z",
|
||||
"shell.execute_reply": "2025-03-09T02:47:44.398531Z",
|
||||
"shell.execute_reply.started": "2025-03-09T02:47:43.964980Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"<class 'list'>\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from bs4 import BeautifulSoup\n",
|
||||
"import re\n",
|
||||
"from pathlib import Path\n",
|
||||
"\n",
|
||||
"def get_html_title(html_file_path):\n",
|
||||
" with open(html_file_path, 'r', encoding='utf-8') as f:\n",
|
||||
" html_content = f.read()\n",
|
||||
" \n",
|
||||
" soup = BeautifulSoup(html_content, 'html.parser')\n",
|
||||
" title_tag = soup.title\n",
|
||||
" \n",
|
||||
" if title_tag and hasattr(title_tag, 'string'):\n",
|
||||
" return title_tag.string.strip()\n",
|
||||
" else:\n",
|
||||
" return None # 标签不存在或内容为空\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"path = Path('./file/guzi')\n",
|
||||
"markdown_content = []\n",
|
||||
"files = [file for file in path.iterdir() if file.is_file()]\n",
|
||||
"sorted_files = sorted(files, key=lambda f: f.name)\n",
|
||||
"print(type(sorted_files))\n",
|
||||
"for file in sorted_files:\n",
|
||||
" if file.suffix=='.html':\n",
|
||||
" markdown_content.append('## '+get_html_title(file))\n",
|
||||
" with open(file, 'r', encoding='utf-8') as f:\n",
|
||||
" html_content = f.read() \n",
|
||||
" soup = BeautifulSoup(html_content, 'html.parser')\n",
|
||||
" div = soup.find('div', class_=\"show-content\")\n",
|
||||
" p_elements = div.find_all('p') \n",
|
||||
" for div in p_elements:\n",
|
||||
" markdown_content.append(div.get_text(strip=True))\n",
|
||||
" #markdown_content.append('\\n')\n",
|
||||
" if file.suffix=='.md':\n",
|
||||
" with open(file, 'r', encoding='utf-8') as f:\n",
|
||||
" txt_content = f.read()\n",
|
||||
" lines = txt_content.splitlines()\n",
|
||||
" for line in lines:\n",
|
||||
" line = line.strip()\n",
|
||||
" markdown_content.append(line)\n",
|
||||
" \n",
|
||||
"markdown_content = '\\n'.join(markdown_content)\n",
|
||||
"with open('guzi.md', 'w', encoding='utf-8') as file:\n",
|
||||
" file.write(markdown_content)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 29,
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2025-03-09T02:36:40.207084Z",
|
||||
"iopub.status.busy": "2025-03-09T02:36:40.206317Z",
|
||||
"iopub.status.idle": "2025-03-09T02:36:40.223987Z",
|
||||
"shell.execute_reply": "2025-03-09T02:36:40.223140Z",
|
||||
"shell.execute_reply.started": "2025-03-09T02:36:40.207017Z"
|
||||
}
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from bs4 import BeautifulSoup\n",
|
||||
"import re\n",
|
||||
"from pathlib import Path\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_html_title(html_file_path):\n",
|
||||
" with open(html_file_path, 'r', encoding='utf-8') as f:\n",
|
||||
" html_content = f.read()\n",
|
||||
" \n",
|
||||
" soup = BeautifulSoup(html_content, 'html.parser')\n",
|
||||
" title_tag = soup.title\n",
|
||||
" \n",
|
||||
" if title_tag and hasattr(title_tag, 'string'):\n",
|
||||
" return title_tag.string.strip()\n",
|
||||
" else:\n",
|
||||
" return None # 标签不存在或内容为空\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"file='file/guzi/1-01.html'\n",
|
||||
"markdown_content = []\n",
|
||||
"markdown_content.append('## '+get_html_title(file))\n",
|
||||
"with open(file, 'r', encoding='utf-8') as f:\n",
|
||||
" html_content = f.read()\n",
|
||||
" \n",
|
||||
" soup = BeautifulSoup(html_content, 'html.parser')\n",
|
||||
" div = soup.find('div', class_=\"show-content\")\n",
|
||||
" p_elements = div.find_all('p')\n",
|
||||
" \n",
|
||||
" for div in p_elements:\n",
|
||||
" markdown_content.append(div.get_text(strip=True))\n",
|
||||
" markdown_content.append('\\n')\n",
|
||||
"markdown_content = '\\n'.join(markdown_content)\n",
|
||||
"with open('1-01.md', 'w', encoding='utf-8') as file:\n",
|
||||
" file.write(markdown_content)\n",
|
||||
"\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
|
||||
Reference in new issue
Block a user