初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)

This commit is contained in:
512song committed 2026-09-23 21:59:25 +08:00
commit 80ae3811cf
7712 files changed
+4628547

No files matched your search

@@ -0,0 +1,91 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辅行诀五脏用药法要-药性探真 资料库 ETL 第 8 步(2026-09-17)
**在 01_来源数据 原件中优化标题层级**(用户 2026-09-17 指令:「直接在 01 原件里改标题层级」)
对象:01_来源数据/《辅行诀五脏用药法要 药性探真》第1章.md(用户精清稿)
改动范围:**只改标题行的井号层级**,正文一字不动(逐行校验非标题行完全一致)
层级体例:章=##(二级)/节=###(三级)/条目=####(四级)/条下附=#####(五级)/附内子目=######(六级)
留痕:
① 变更前原件快照另存 `01_来源数据/《辅行诀五脏用药法要 药性探真》第1章_层级优化前快照_20260917.md`
(内容 = 变更前逐字节副本,md5 与旧值一致,供随时回退);
② 变更前后 md5 记入本脚本 stdout 与 `02_加工数据/药性探真_第1章_检查报告.md`(由 07 脚本刷新)。
幂等:可重复运行;已优化过的文件重跑无副作用(无标题需改时输出 0 处)。
"""
import os, re, hashlib, shutil
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
SRC = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第1章.md")
SNAP = os.path.join(BASE, "01_来源数据/《辅行诀五脏用药法要 药性探真》第1章_层级优化前快照_20260917.md")
H_RULES = [
(re.compile(r"^第[一二三四五六七八九十]+章[  ]"), 2, "章"),
(re.compile(r"^第[一二三四五六七八九十]+节[  ]"), 3, "节"),
(re.compile(r"^[一二三四五六七八九十]+、"), 4, "条目"),
(re.compile(r"^附:"), 5, "条下附"),
(re.compile(r"^\d+\.\s*《"), 6, "附内子目"),
]
def md5b(b):
return hashlib.md5(b).hexdigest()
def main():
raw = open(SRC, "rb").read()
md5_before = md5b(raw)
text = raw.decode("utf-8")
lines = text.split("\n")
# 变更前快照(仅在快照不存在时保存,避免二次运行覆盖真正的原始副本)
if not os.path.exists(SNAP):
shutil.copy2(SRC, SNAP)
print("快照已存:", os.path.basename(SNAP), md5b(open(SNAP, "rb").read()))
out, changes, unrecognized = [], [], []
for i, l in enumerate(lines, 1):
s = l.strip()
if s.startswith("#"):
bare = re.sub(r"^#+\s*", "", s)
lv, kind = None, None
for pat, L, K in H_RULES:
if pat.match(bare):
lv, kind = L, K
break
if lv is None:
unrecognized.append((i, s[:40]))
out.append(l)
continue
new = "#" * lv + " " + bare
if new != s:
changes.append((i, kind, s, new))
out.append(new)
else:
out.append(l)
if unrecognized:
raise SystemExit(f"存在未识别标题,未改动任何内容:{unrecognized}")
new_text = "\n".join(out)
md5_after = md5b(new_text.encode("utf-8"))
# 正文一致性校验:去掉标题井号后逐行比对
def strip_h(t):
return [re.sub(r"^#+\s*", "", x) for x in t.split("\n")]
if strip_h(new_text) != strip_h(text):
raise SystemExit("正文一致性校验失败:非标题内容被改动,已中止")
if len(new_text) != len(text) + sum(len(n) - len(o) for _, _, o, n in changes):
raise SystemExit("长度校验失败,已中止")
with open(SRC, "w", encoding="utf-8") as f:
f.write(new_text); f.flush(); os.fsync(f.fileno())
print(f"原件标题层级:改动 {len(changes)} 行")
for ln, kind, old, new in changes:
print(f" L{ln:<4} {kind:<5} {old[:24]:<26} → {new[:26]}")
print(f"md5 变更前 {md5_before} → 变更后 {md5_after}")
print("正文校验:非标题行逐行一致(一字未动)")
if __name__ == "__main__":
main()