Files

92 lines
4.0 KiB
Python
Raw Permalink Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辅行诀五脏用药法要-药性探真 资料库 ETL 第 8 步(2026-09-17)
**在 01_来源数据 原件中优化标题层级**(用户 2026-09-17 指令:「直接在 01 原件里改标题层级」)
对象:01_来源数据/《辅行诀五脏用药法要 药性探真》第1章.md(用户精清稿)
改动范围:**只改标题行的井号层级**,正文一字不动(逐行校验非标题行完全一致)
层级体例:章=##(二级)/节=###(三级)/条目=####(四级)/条下附=#####(五级)/附内子目=######(六级)
留痕:
① 变更前原件快照另存 `01_来源数据/《辅行诀五脏用药法要 药性探真》第1章_层级优化前快照_20260917.md`
(内容 = 变更前逐字节副本,md5 与旧值一致,供随时回退);
② 变更前后 md5 记入本脚本 stdout 与 `02_加工数据/药性探真_第1章_检查报告.md`(由 07 脚本刷新)。
幂等:可重复运行;已优化过的文件重跑无副作用(无标题需改时输出 0 处)。
"""
import os, re, hashlib, shutil
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
SRC = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第1章.md")
SNAP = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第1章_层级优化前快照_20260917.md")
H_RULES = [
(re.compile(r"^第[一二三四五六七八九十]+章[  ]"), 2, "章"),
(re.compile(r"^第[一二三四五六七八九十]+节[  ]"), 3, "节"),
(re.compile(r"^[一二三四五六七八九十]+、"), 4, "条目"),
(re.compile(r"^附:"), 5, "条下附"),
(re.compile(r"^\d+\.\s*《"), 6, "附内子目"),
]
def md5b(b):
return hashlib.md5(b).hexdigest()
def main():
raw = open(SRC, "rb").read()
md5_before = md5b(raw)
text = raw.decode("utf-8")
lines = text.split("\n")
# 变更前快照(仅在快照不存在时保存,避免二次运行覆盖真正的原始副本)
if not os.path.exists(SNAP):
shutil.copy2(SRC, SNAP)
print("快照已存:", os.path.basename(SNAP), md5b(open(SNAP, "rb").read()))
out, changes, unrecognized = [], [], []
for i, l in enumerate(lines, 1):
s = l.strip()
if s.startswith("#"):
bare = re.sub(r"^#+\s*", "", s)
lv, kind = None, None
for pat, L, K in H_RULES:
if pat.match(bare):
lv, kind = L, K
break
if lv is None:
unrecognized.append((i, s[:40]))
out.append(l)
continue
new = "#" * lv + " " + bare
if new != s:
changes.append((i, kind, s, new))
out.append(new)
else:
out.append(l)
if unrecognized:
raise SystemExit(f"存在未识别标题,未改动任何内容:{unrecognized}")
new_text = "\n".join(out)
md5_after = md5b(new_text.encode("utf-8"))
# 正文一致性校验:去掉标题井号后逐行比对
def strip_h(t):
return [re.sub(r"^#+\s*", "", x) for x in t.split("\n")]
if strip_h(new_text) != strip_h(text):
raise SystemExit("正文一致性校验失败:非标题内容被改动,已中止")
if len(new_text) != len(text) + sum(len(n) - len(o) for _, _, o, n in changes):
raise SystemExit("长度校验失败,已中止")
with open(SRC, "w", encoding="utf-8") as f:
f.write(new_text); f.flush(); os.fsync(f.fileno())
print(f"原件标题层级:改动 {len(changes)} 行")
for ln, kind, old, new in changes:
print(f" L{ln:<4} {kind:<5} {old[:24]:<26} → {new[:26]}")
print(f"md5 变更前 {md5_before} → 变更后 {md5_after}")
print("正文校验:非标题行逐行一致(一字未动)")
if __name__ == "__main__":
main()