100 lines
4.1 KiB
Python
100 lines
4.1 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
辅行诀五脏用药法要-药性探真 资料库 ETL 第 11 步(2026-09-20)
|
||
《药性探真》第 2 章 **原件标题层级就地优化**(对应第 1 章之 08_retitle_src_ch1.py)
|
||
|
||
背景:
|
||
第 1 章原件曾按**用户 2026-09-17 明确指令**就地优化标题层级(只改井号、正文逐行校验未动)。
|
||
第 2 章原件标题 25 行原样为「##」(章/节/条目不分级)。README 归档原则为
|
||
「01_来源数据 不可修改」,故本次**默认不做改动**,只在本脚本中备好同办途径。
|
||
用法:
|
||
python3 05_脚本工具/11_retitle_src_ch2.py # 默认 dry-run:只打印将改动的行,不落盘
|
||
python3 05_脚本工具/11_retitle_src_ch2.py --apply # 真正就地优化(先留快照)
|
||
层级体例(同第 1 章):章=二级(##)/节=三级(###)/条目=四级(####)
|
||
安全保障:
|
||
① --apply 前先写逐字节快照 《…》第2章_层级优化前快照_YYYYMMDD.md(已存在则不覆盖)
|
||
② 落盘前断言「去标题行后的正文逐行完全相同」(正文一字未动)
|
||
③ 幂等:已优化后再跑,改动 0 行
|
||
"""
|
||
import os, re, sys, hashlib, datetime
|
||
|
||
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
|
||
SRC = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第2章.md")
|
||
SNAP_FMT = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第2章_层级优化前快照_%s.md")
|
||
|
||
H_RULES = [
|
||
(re.compile(r"^第[一二三四五六七八九十]+章[ ]"), 2, "章"),
|
||
(re.compile(r"^第[一二三四五六七八九十]+节[ ]"), 3, "节"),
|
||
(re.compile(r"^[一二三四五六七八九十]+、"), 4, "条目"),
|
||
]
|
||
|
||
|
||
def strip_headings(t):
|
||
return [l for l in t.split("\n") if not l.startswith("#")]
|
||
|
||
|
||
def plan(text):
|
||
"""返回 (新文本, 改动清单);未识别标题直接报错。"""
|
||
out, changes = [], []
|
||
for i, l in enumerate(text.split("\n"), 1):
|
||
if not l.startswith("#"):
|
||
out.append(l)
|
||
continue
|
||
bare = re.sub(r"^#+\s*", "", l)
|
||
lv = None
|
||
for pat, L, _K in H_RULES:
|
||
if pat.match(bare):
|
||
lv = L
|
||
break
|
||
if lv is None:
|
||
raise SystemExit(f"未识别的标题(行 {i}):{l[:60]}")
|
||
newh = "#" * lv + " " + bare
|
||
if newh != l:
|
||
changes.append((i, l, newh))
|
||
out.append(newh)
|
||
return "\n".join(out), changes
|
||
|
||
|
||
def main():
|
||
apply = "--apply" in sys.argv
|
||
raw = open(SRC, "rb").read()
|
||
text = raw.decode("utf-8")
|
||
md5_before = hashlib.md5(raw).hexdigest()
|
||
new, changes = plan(text)
|
||
|
||
assert strip_headings(text) == strip_headings(new), "正文行被改动,脚本中止"
|
||
|
||
print(f"对象:{SRC}")
|
||
print(f"md5(当前):{md5_before}")
|
||
print(f"标题行合计:{sum(1 for l in text.split(chr(10)) if l.startswith('#'))} 行;需改动 {len(changes)} 行")
|
||
for i, a, b in changes:
|
||
print(f" L{i}: {a} -> {b}")
|
||
if not changes:
|
||
print("✅ 无需改动(已是最优层级或已优化过)——幂等检查通过")
|
||
return
|
||
|
||
if not apply:
|
||
print("\n(dry-run:未落盘。如需就地优化,加 --apply)")
|
||
return
|
||
|
||
snap = SNAP_FMT % datetime.date.today().strftime("%Y%m%d")
|
||
if os.path.exists(snap):
|
||
print(f"快照已存在,不覆盖:{snap}")
|
||
else:
|
||
with open(snap, "wb") as f:
|
||
f.write(raw); f.flush(); os.fsync(f.fileno())
|
||
print(f"快照(改动前逐字节副本):{snap}(md5 {md5_before})")
|
||
with open(SRC, "wb") as f:
|
||
f.write(new.encode("utf-8")); f.flush(); os.fsync(f.fileno())
|
||
md5_after = hashlib.md5(open(SRC, "rb").read()).hexdigest()
|
||
print(f"已就地优化标题层级:{SRC}\n md5 {md5_before} -> {md5_after}")
|
||
# 幂等复验
|
||
again, ch2 = plan(open(SRC, encoding="utf-8").read())
|
||
print(f"幂等复验:再跑需改动 {len(ch2)} 行(应为 0)")
|
||
assert not ch2
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|