Files
health/辅行诀五脏用药法要-药性探真/05_脚本工具/14_merge_ch123.py
T

276 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辅行诀五脏用药法要-药性探真 资料库 ETL 第 14 步(2026-09-22):
《〈辅行诀五脏用药法要〉药性探真》第 1/2/3 章**清洗版** → 合并为单一清洗版正文,入 02_加工数据。
用户指令(2026-09-22):
「《辅行诀五脏用药法要 药性探真》第1-3章已经清理完成,合并后导入资料库,
级别低于 …/02_加工数据/辅行诀脏腑用药法要.md」
→ 合并(本脚本)+ 定级登记(README/来源说明):本合并本为**现代研究文献**,
**引用优先级低于《辅行诀脏腑用药法要》(参考本),更低于《辅行诀五脏用药法要》(底本)**。
原则(与库内既有规范一致):
① **只合并,不改字**:三章清洗版正文**一字未增删**(改字一律回到 07/10/12 清洗脚本,本脚本不设 FIXES 表);
② **可复核**:合并正文 == 三章清洗版正文按章序拼接(自检断言,逐字节);
③ **来源 md5 钉定**:任一章清洗版与登记值不符即报错(防静默合并到错版本);
④ **幂等**:重跑覆盖,输出与库内一致时不改写文件;
⑤ **行号可回溯**:产物结构 JSON 给出「合卷行号 ⇄ 各章清洗版行号」偏移对照。
产物:
02_加工数据/药性探真_第1-3章_合并清洗版.md (合卷正文)
02_加工数据/药性探真_第1-3章_合并结构.json (章/节/条目/子目 四级 + 合卷行号锚 + 各章偏移)
用法:
cd ~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真
python3 05_脚本工具/14_merge_ch123.py # 生成/刷新(幂等)
python3 05_脚本工具/14_merge_ch123.py --check # 只校验,不写盘
"""
import os
import re
import sys
import json
import hashlib
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
OUT_MD = "02_加工数据/药性探真_第1-3章_合并清洗版.md"
OUT_JSON = "02_加工数据/药性探真_第1-3章_合并结构.json"
# 合卷标题(文书级标识行;**不由来源正文产生**,系合卷所必需,正文本身一字未增删)
TITLE = "# 《辅行诀五脏用药法要 药性探真》第一至三章"
# 三章清洗版(顺序即合卷章序):相对路径 / 登记 md5 / 章标签
SRC = [
("02_加工数据/药性探真_第1章_清洗版.md", "a20b17b7589255889256426b795a86d9", "第一章"),
("02_加工数据/药性探真_第2章_清洗版.md", "ac1a4d325f971279bc583324c95408d8", "第二章"),
("02_加工数据/药性探真_第3章_清洗版.md", "5f1fb57c02de5d370a8aff51013062bb", "第三章"),
]
H_CHAPTER = re.compile(r"^##\s+(第[一二三四五六七八九十]+章)\s*(.*)$")
H_SECTION = re.compile(r"^###\s+(第[一二三四五六七八九十]+节)\s*(.*)$")
H_ENTRY = re.compile(r"^####\s+(.+)$")
H_SUB = re.compile(r"^#####\s+(.+)$")
H_SUB6 = re.compile(r"^######\s+(.+)$")
def md5_of(p):
return hashlib.md5(open(p, "rb").read()).hexdigest()
def read_text(rel):
"""CRLF-safe:文本模式会吞 CRLF,一律按字节读入再解码。"""
with open(os.path.join(BASE, rel), "rb") as f:
return f.read().decode("utf-8")
def load_sources():
"""读入三章清洗版 + 断言(章首行、无一级标题、md5 相符)。"""
srcs, errs = [], []
for rel, want, label in SRC:
p = os.path.join(BASE, rel)
if not os.path.exists(p):
errs.append(f"缺文件:{rel}")
continue
got = md5_of(p)
if got != want:
errs.append(f"md5 不符:{rel}\n 登记 {want}\n 实测 {got}"
f"\n → 清洗版已被重跑刷新,请先核对来源,再更新本脚本 SRC 表")
continue
txt = read_text(rel)
first = txt.split("\n")[0].strip()
m = H_CHAPTER.match(first)
if not m:
errs.append(f"章首行不是二级章标题:{rel} → {first!r}")
continue
if m.group(1) != label:
errs.append(f"章标签不符:{rel} 首行作 {m.group(1)},登记 {label}")
continue
h1 = [ln for ln in txt.split("\n") if ln.startswith("# ")]
if h1:
errs.append(f"来源含一级标题(合卷后与文书标题同级):{rel} → {h1[:3]}")
continue
if not txt.endswith("\n"):
errs.append(f"来源末行无换行:{rel}")
continue
srcs.append({"章": label, "文件": rel, "md5": got,
"行数": len(txt.split("\n")) - 1, # 末行为空行不计
"正文": txt})
return srcs, errs
def build_merged(srcs):
body = "\n\n".join(s["正文"].rstrip("\n") for s in srcs) + "\n"
return TITLE + "\n\n" + body, body
def parse_structure(srcs, body):
"""按合卷正文解析 章/节/条目/子目 四级,并给出「合卷行号 → 各章清洗版行号」回溯。"""
# 各章在合卷正文中的起始行(1-based):标题行 + 空行 + 各章正文
offsets, ln = {}, 1 + 2 # 1 行文书标题 + 1 空行
for s in srcs:
offsets[s["章"]] = ln - 1 # 合卷行号 = 该章清洗版行号 + offset
ln += s["行数"] + 1 # 章正文行数 + 1 空行分隔
lines = body.split("\n")
chapters, cur_ch, cur_sec, cur_ent = [], None, None, None
stats = {"章": 0, "节": 0, "条目": 0, "子目": 0, "附内子目": 0}
for i, raw in enumerate(lines, 1): # i = 合卷行号(正文第 1 行 = 合卷第 3 行)
if not raw.strip():
continue
mc = H_CHAPTER.match(raw)
if mc:
cur_ch = {"章": (mc.group(1) + " " + mc.group(2)).strip(),
"行": i, "节": []}
chapters.append(cur_ch)
stats["章"] += 1
cur_sec = cur_ent = None
continue
ms = H_SECTION.match(raw)
if ms:
if cur_ch is None:
raise SystemExit(f"❌ 第 {i} 行出现节标题但无章标题:{raw[:40]}")
cur_sec = {"节": (ms.group(1) + " " + ms.group(2)).strip(),
"行": i, "条目": []}
cur_ch["节"].append(cur_sec)
stats["节"] += 1
cur_ent = None
continue
me = H_ENTRY.match(raw)
if me:
if cur_sec is None:
raise SystemExit(f"❌ 第 {i} 行出现条目标题但无节标题:{raw[:40]}")
cur_ent = {"条目": me.group(1).strip(), "行": i, "子目": [], "附内子目": []}
cur_sec["条目"].append(cur_ent)
stats["条目"] += 1
continue
msub = H_SUB.match(raw)
if msub:
if cur_ent is None:
raise SystemExit(f"❌ 第 {i} 行出现子目标题但无条目标题:{raw[:40]}")
cur_ent["子目"].append({"子目": msub.group(1).strip(), "行": i})
stats["子目"] += 1
continue
msub6 = H_SUB6.match(raw)
if msub6:
if cur_ent is None:
raise SystemExit(f"❌ 第 {i} 行出现六级标题但无条目标题:{raw[:40]}")
cur_ent["附内子目"].append({"附内子目": msub6.group(1).strip(), "行": i})
stats["附内子目"] += 1
continue
# 逐级补「回溯行号」(合卷行号 − 所属章偏移 = 该章清洗版行号)
def back(ch_label, l):
return l - offsets[ch_label]
for ch in chapters:
for sec in ch["节"]:
sec["源行"] = back(ch["章"][:3], sec["行"])
for ent in sec["条目"]:
ent["源行"] = back(ch["章"][:3], ent["行"])
for sub in ent["子目"] + ent["附内子目"]:
sub["源行"] = back(ch["章"][:3], sub["行"])
return chapters, offsets, stats
def main():
check_only = "--check" in sys.argv
srcs, errs = load_sources()
if errs:
print("❌ 来源校验未通过:")
for e in errs:
print(" -", e)
return 1
merged, body = build_merged(srcs)
out_md = os.path.join(BASE, OUT_MD)
out_json = os.path.join(BASE, OUT_JSON)
old_md = read_text(OUT_MD) if os.path.exists(out_md) else None
chapters, offsets, stats = parse_structure(srcs, body)
total_lines = len(merged.split("\n")) - 1
# ── 自检(断言,不静默)──────────────────────────────────────
checks = []
checks.append(("来源 md5 三章全部相符",
all(md5_of(os.path.join(BASE, s["文件"])) == s["md5"] for s in srcs)))
checks.append(("合并正文 == 三章清洗版按章序逐字节拼接",
merged.split("\n", 1)[1].lstrip("\n")
== "\n\n".join(s["正文"].rstrip("\n") for s in srcs) + "\n"))
checks.append(("行顺序:第1章 < 第2章 < 第3章",
[c["章"][:3] for c in chapters] == ["第一章", "第二章", "第三章"]))
checks.append(("章数 = 3", stats["章"] == 3))
checks.append(("无一级标题(除文书标题行)",
len([l for l in merged.split("\n") if l.startswith("# ")]) == 1))
checks.append(("标题层级合法(## → ### → #### → ##### → ######,无越级)",
all(re.match(r"^#{1,6} ", l) for l in merged.split("\n")
if l.startswith("#"))))
ok = all(v for _, v in checks)
new_md5 = hashlib.md5(merged.encode("utf-8")).hexdigest()
print(f"合卷正文:{OUT_MD}")
print(f" {total_lines} 行 / {len(merged.encode('utf-8'))} 字节 / md5 {new_md5}")
print(f" 章 {stats['章']} · 节 {stats['节']} · 条目 {stats['条目']} · 子目 {stats['子目']}"
f" · 附内子目 {stats['附内子目']}")
print(" 章行号(合卷 ⇄ 各章清洗版回溯):")
for s in srcs:
off = offsets[s["章"]]
print(f" {s['章']} 合卷第 {off + 1} 行起(清洗版行号 + {off}),"
f"该章 {s['行数']} 行,md5 {s['md5'][:8]}…")
print(" 自检:")
for name, v in checks:
print(f" {'✅' if v else '❌'} {name}")
if not ok:
print("\n❌ 自检未通过,未写盘。")
return 1
if check_only:
print("\n--check:只校验,未写盘。")
return 0
changed = (old_md != merged)
if changed:
with open(out_md, "w", encoding="utf-8") as f:
f.write(merged)
print(f"\n{'✅ 已写入' if changed else '≡ 内容与库内一致,未改写'}:{OUT_MD}")
structure = {
"定位": "《〈辅行诀五脏用药法要〉药性探真》第一至三章 合并清洗版(衣之镖撰;学苑出版社 2014)",
"级别": "现代研究文献;引用优先级**低于**《辅行诀脏腑用药法要》(参考本),更低于"
"《辅行诀五脏用药法要》(底本)。**不得以本合并本替代《辅行诀》原书引用**。",
"用户指令": "2026-09-22「第1-3章已经清理完成,合并后导入资料库,级别低于 "
"…/02_加工数据/辅行诀脏腑用药法要.md」",
"产物文件": OUT_MD,
"产物md5": new_md5,
"产物行数": total_lines,
"产物字节": len(merged.encode("utf-8")),
"文书标题行": TITLE,
"标题体例": "文书标题=#(一级)/章=##(二级)/节=###(三级)/条目=####(四级)/子目=#####(五级)"
"/附内子目=######(六级);第 1 章之五级为「附:」、六级为附内子目,第 3 章之五级为子目(1. …),"
"三章体例**原样保留**、不作跨章统一",
"合并规则": "三章清洗版正文按章序逐字节拼接,章间空一行;**一字未增删**(文书标题行为合卷所必需之标识行)",
"来源": [{"章": s["章"], "文件": s["文件"], "md5": s["md5"], "行数": s["行数"],
"合卷起始行": offsets[s["章"]] + 1, "清洗版行号偏移": offsets[s["章"]]}
for s in srcs],
"行号对照": {"规则": "合卷行号 = 该章清洗版行号 + 偏移;各章条目均带「源行」为其清洗版行号",
"各章偏移": {s["章"]: offsets[s["章"]] for s in srcs}},
"章": chapters,
"统计": stats,
"自检": [{"项": n, "结果": "✅ 通过" if v else "❌ 未通过"} for n, v in checks],
"上游脚本": ["07_clean_ch1.py(md5 见 README §七)", "10_clean_ch2.py", "12_clean_ch3.py"],
"重跑": "python3 05_脚本工具/14_merge_ch123.py(幂等;--check 只校验不写盘)",
}
txt = json.dumps(structure, ensure_ascii=False, indent=2) + "\n"
old_json = read_text(OUT_JSON) if os.path.exists(out_json) else None
if old_json != txt:
with open(out_json, "w", encoding="utf-8") as f:
f.write(txt)
print(f"✅ 已写入:{OUT_JSON}")
else:
print(f"≡ 内容与库内一致,未改写:{OUT_JSON}")
return 0
if __name__ == "__main__":
sys.exit(main())