Files
health/辅行诀五脏用药法要-药性探真/05_脚本工具/11_retitle_src_ch2.py
T

100 lines
4.1 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辅行诀五脏用药法要-药性探真 资料库 ETL 第 11 步(2026-09-20)
《药性探真》第 2 章 **原件标题层级就地优化**(对应第 1 章之 08_retitle_src_ch1.py)
背景:
第 1 章原件曾按**用户 2026-09-17 明确指令**就地优化标题层级(只改井号、正文逐行校验未动)。
第 2 章原件标题 25 行原样为「##」(章/节/条目不分级)。README 归档原则为
「01_来源数据 不可修改」,故本次**默认不做改动**,只在本脚本中备好同办途径。
用法:
python3 05_脚本工具/11_retitle_src_ch2.py # 默认 dry-run:只打印将改动的行,不落盘
python3 05_脚本工具/11_retitle_src_ch2.py --apply # 真正就地优化(先留快照)
层级体例(同第 1 章):章=二级(##)/节=三级(###)/条目=四级(####)
安全保障:
① --apply 前先写逐字节快照 《…》第2章_层级优化前快照_YYYYMMDD.md(已存在则不覆盖)
② 落盘前断言「去标题行后的正文逐行完全相同」(正文一字未动)
③ 幂等:已优化后再跑,改动 0 行
"""
import os, re, sys, hashlib, datetime
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
SRC = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第2章.md")
SNAP_FMT = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第2章_层级优化前快照_%s.md")
H_RULES = [
(re.compile(r"^第[一二三四五六七八九十]+章[  ]"), 2, "章"),
(re.compile(r"^第[一二三四五六七八九十]+节[  ]"), 3, "节"),
(re.compile(r"^[一二三四五六七八九十]+、"), 4, "条目"),
]
def strip_headings(t):
return [l for l in t.split("\n") if not l.startswith("#")]
def plan(text):
"""返回 (新文本, 改动清单);未识别标题直接报错。"""
out, changes = [], []
for i, l in enumerate(text.split("\n"), 1):
if not l.startswith("#"):
out.append(l)
continue
bare = re.sub(r"^#+\s*", "", l)
lv = None
for pat, L, _K in H_RULES:
if pat.match(bare):
lv = L
break
if lv is None:
raise SystemExit(f"未识别的标题(行 {i}):{l[:60]}")
newh = "#" * lv + " " + bare
if newh != l:
changes.append((i, l, newh))
out.append(newh)
return "\n".join(out), changes
def main():
apply = "--apply" in sys.argv
raw = open(SRC, "rb").read()
text = raw.decode("utf-8")
md5_before = hashlib.md5(raw).hexdigest()
new, changes = plan(text)
assert strip_headings(text) == strip_headings(new), "正文行被改动,脚本中止"
print(f"对象:{SRC}")
print(f"md5(当前):{md5_before}")
print(f"标题行合计:{sum(1 for l in text.split(chr(10)) if l.startswith('#'))} 行;需改动 {len(changes)} 行")
for i, a, b in changes:
print(f" L{i}: {a} -> {b}")
if not changes:
print("✅ 无需改动(已是最优层级或已优化过)——幂等检查通过")
return
if not apply:
print("\n(dry-run:未落盘。如需就地优化,加 --apply)")
return
snap = SNAP_FMT % datetime.date.today().strftime("%Y%m%d")
if os.path.exists(snap):
print(f"快照已存在,不覆盖:{snap}")
else:
with open(snap, "wb") as f:
f.write(raw); f.flush(); os.fsync(f.fileno())
print(f"快照(改动前逐字节副本):{snap}(md5 {md5_before})")
with open(SRC, "wb") as f:
f.write(new.encode("utf-8")); f.flush(); os.fsync(f.fileno())
md5_after = hashlib.md5(open(SRC, "rb").read()).hexdigest()
print(f"已就地优化标题层级:{SRC}\n md5 {md5_before} -> {md5_after}")
# 幂等复验
again, ch2 = plan(open(SRC, encoding="utf-8").read())
print(f"幂等复验:再跑需改动 {len(ch2)} 行(应为 0)")
assert not ch2
if __name__ == "__main__":
main()