初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)

This commit is contained in:
512song committed 2026-09-23 21:59:25 +08:00
commit 80ae3811cf
7712 files changed
+4628547

No files matched your search

@@ -0,0 +1,99 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辅行诀五脏用药法要-药性探真 资料库 ETL 第 11 步(2026-09-20)
《药性探真》第 2 章 **原件标题层级就地优化**(对应第 1 章之 08_retitle_src_ch1.py)
背景:
第 1 章原件曾按**用户 2026-09-17 明确指令**就地优化标题层级(只改井号、正文逐行校验未动)。
第 2 章原件标题 25 行原样为「##」(章/节/条目不分级)。README 归档原则为
「01_来源数据 不可修改」,故本次**默认不做改动**,只在本脚本中备好同办途径。
用法:
python3 05_脚本工具/11_retitle_src_ch2.py # 默认 dry-run:只打印将改动的行,不落盘
python3 05_脚本工具/11_retitle_src_ch2.py --apply # 真正就地优化(先留快照)
层级体例(同第 1 章):章=二级(##)/节=三级(###)/条目=四级(####)
安全保障:
① --apply 前先写逐字节快照 《…》第2章_层级优化前快照_YYYYMMDD.md(已存在则不覆盖)
② 落盘前断言「去标题行后的正文逐行完全相同」(正文一字未动)
③ 幂等:已优化后再跑,改动 0 行
"""
import os, re, sys, hashlib, datetime
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
SRC = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第2章.md")
SNAP_FMT = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第2章_层级优化前快照_%s.md")
H_RULES = [
(re.compile(r"^第[一二三四五六七八九十]+章[  ]"), 2, "章"),
(re.compile(r"^第[一二三四五六七八九十]+节[  ]"), 3, "节"),
(re.compile(r"^[一二三四五六七八九十]+、"), 4, "条目"),
]
def strip_headings(t):
return [l for l in t.split("\n") if not l.startswith("#")]
def plan(text):
"""返回 (新文本, 改动清单);未识别标题直接报错。"""
out, changes = [], []
for i, l in enumerate(text.split("\n"), 1):
if not l.startswith("#"):
out.append(l)
continue
bare = re.sub(r"^#+\s*", "", l)
lv = None
for pat, L, _K in H_RULES:
if pat.match(bare):
lv = L
break
if lv is None:
raise SystemExit(f"未识别的标题(行 {i}):{l[:60]}")
newh = "#" * lv + " " + bare
if newh != l:
changes.append((i, l, newh))
out.append(newh)
return "\n".join(out), changes
def main():
apply = "--apply" in sys.argv
raw = open(SRC, "rb").read()
text = raw.decode("utf-8")
md5_before = hashlib.md5(raw).hexdigest()
new, changes = plan(text)
assert strip_headings(text) == strip_headings(new), "正文行被改动,脚本中止"
print(f"对象:{SRC}")
print(f"md5(当前):{md5_before}")
print(f"标题行合计:{sum(1 for l in text.split(chr(10)) if l.startswith('#'))} 行;需改动 {len(changes)} 行")
for i, a, b in changes:
print(f" L{i}: {a} -> {b}")
if not changes:
print("✅ 无需改动(已是最优层级或已优化过)——幂等检查通过")
return
if not apply:
print("\n(dry-run:未落盘。如需就地优化,加 --apply)")
return
snap = SNAP_FMT % datetime.date.today().strftime("%Y%m%d")
if os.path.exists(snap):
print(f"快照已存在,不覆盖:{snap}")
else:
with open(snap, "wb") as f:
f.write(raw); f.flush(); os.fsync(f.fileno())
print(f"快照(改动前逐字节副本):{snap}(md5 {md5_before})")
with open(SRC, "wb") as f:
f.write(new.encode("utf-8")); f.flush(); os.fsync(f.fileno())
md5_after = hashlib.md5(open(SRC, "rb").read()).hexdigest()
print(f"已就地优化标题层级:{SRC}\n md5 {md5_before} -> {md5_after}")
# 幂等复验
again, ch2 = plan(open(SRC, encoding="utf-8").read())
print(f"幂等复验:再跑需改动 {len(ch2)} 行(应为 0)")
assert not ch2
if __name__ == "__main__":
main()