#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 辅行诀五脏用药法要-药性探真 资料库 ETL 第 14 步(2026-09-22): 《〈辅行诀五脏用药法要〉药性探真》第 1/2/3 章**清洗版** → 合并为单一清洗版正文,入 02_加工数据。 用户指令(2026-09-22): 「《辅行诀五脏用药法要 药性探真》第1-3章已经清理完成,合并后导入资料库, 级别低于 …/02_加工数据/辅行诀脏腑用药法要.md」 → 合并(本脚本)+ 定级登记(README/来源说明):本合并本为**现代研究文献**, **引用优先级低于《辅行诀脏腑用药法要》(参考本),更低于《辅行诀五脏用药法要》(底本)**。 原则(与库内既有规范一致): ① **只合并,不改字**:三章清洗版正文**一字未增删**(改字一律回到 07/10/12 清洗脚本,本脚本不设 FIXES 表); ② **可复核**:合并正文 == 三章清洗版正文按章序拼接(自检断言,逐字节); ③ **来源 md5 钉定**:任一章清洗版与登记值不符即报错(防静默合并到错版本); ④ **幂等**:重跑覆盖,输出与库内一致时不改写文件; ⑤ **行号可回溯**:产物结构 JSON 给出「合卷行号 ⇄ 各章清洗版行号」偏移对照。 产物: 02_加工数据/药性探真_第1-3章_合并清洗版.md (合卷正文) 02_加工数据/药性探真_第1-3章_合并结构.json (章/节/条目/子目 四级 + 合卷行号锚 + 各章偏移) 用法: cd ~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真 python3 05_脚本工具/14_merge_ch123.py # 生成/刷新(幂等) python3 05_脚本工具/14_merge_ch123.py --check # 只校验,不写盘 """ import os import re import sys import json import hashlib BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真") OUT_MD = "02_加工数据/药性探真_第1-3章_合并清洗版.md" OUT_JSON = "02_加工数据/药性探真_第1-3章_合并结构.json" # 合卷标题(文书级标识行;**不由来源正文产生**,系合卷所必需,正文本身一字未增删) TITLE = "# 《辅行诀五脏用药法要 药性探真》第一至三章" # 三章清洗版(顺序即合卷章序):相对路径 / 登记 md5 / 章标签 SRC = [ ("02_加工数据/药性探真_第1章_清洗版.md", "a20b17b7589255889256426b795a86d9", "第一章"), ("02_加工数据/药性探真_第2章_清洗版.md", "ac1a4d325f971279bc583324c95408d8", "第二章"), ("02_加工数据/药性探真_第3章_清洗版.md", "5f1fb57c02de5d370a8aff51013062bb", "第三章"), ] H_CHAPTER = re.compile(r"^##\s+(第[一二三四五六七八九十]+章)\s*(.*)$") H_SECTION = re.compile(r"^###\s+(第[一二三四五六七八九十]+节)\s*(.*)$") H_ENTRY = re.compile(r"^####\s+(.+)$") H_SUB = re.compile(r"^#####\s+(.+)$") H_SUB6 = re.compile(r"^######\s+(.+)$") def md5_of(p): return hashlib.md5(open(p, "rb").read()).hexdigest() def read_text(rel): """CRLF-safe:文本模式会吞 CRLF,一律按字节读入再解码。""" with open(os.path.join(BASE, rel), "rb") as f: return f.read().decode("utf-8") def load_sources(): """读入三章清洗版 + 断言(章首行、无一级标题、md5 相符)。""" srcs, errs = [], [] for rel, want, label in SRC: p = os.path.join(BASE, rel) if not os.path.exists(p): errs.append(f"缺文件:{rel}") continue got = md5_of(p) if got != want: errs.append(f"md5 不符:{rel}\n 登记 {want}\n 实测 {got}" f"\n → 清洗版已被重跑刷新,请先核对来源,再更新本脚本 SRC 表") continue txt = read_text(rel) first = txt.split("\n")[0].strip() m = H_CHAPTER.match(first) if not m: errs.append(f"章首行不是二级章标题:{rel} → {first!r}") continue if m.group(1) != label: errs.append(f"章标签不符:{rel} 首行作 {m.group(1)},登记 {label}") continue h1 = [ln for ln in txt.split("\n") if ln.startswith("# ")] if h1: errs.append(f"来源含一级标题(合卷后与文书标题同级):{rel} → {h1[:3]}") continue if not txt.endswith("\n"): errs.append(f"来源末行无换行:{rel}") continue srcs.append({"章": label, "文件": rel, "md5": got, "行数": len(txt.split("\n")) - 1, # 末行为空行不计 "正文": txt}) return srcs, errs def build_merged(srcs): body = "\n\n".join(s["正文"].rstrip("\n") for s in srcs) + "\n" return TITLE + "\n\n" + body, body def parse_structure(srcs, body): """按合卷正文解析 章/节/条目/子目 四级,并给出「合卷行号 → 各章清洗版行号」回溯。""" # 各章在合卷正文中的起始行(1-based):标题行 + 空行 + 各章正文 offsets, ln = {}, 1 + 2 # 1 行文书标题 + 1 空行 for s in srcs: offsets[s["章"]] = ln - 1 # 合卷行号 = 该章清洗版行号 + offset ln += s["行数"] + 1 # 章正文行数 + 1 空行分隔 lines = body.split("\n") chapters, cur_ch, cur_sec, cur_ent = [], None, None, None stats = {"章": 0, "节": 0, "条目": 0, "子目": 0, "附内子目": 0} for i, raw in enumerate(lines, 1): # i = 合卷行号(正文第 1 行 = 合卷第 3 行) if not raw.strip(): continue mc = H_CHAPTER.match(raw) if mc: cur_ch = {"章": (mc.group(1) + " " + mc.group(2)).strip(), "行": i, "节": []} chapters.append(cur_ch) stats["章"] += 1 cur_sec = cur_ent = None continue ms = H_SECTION.match(raw) if ms: if cur_ch is None: raise SystemExit(f"❌ 第 {i} 行出现节标题但无章标题:{raw[:40]}") cur_sec = {"节": (ms.group(1) + " " + ms.group(2)).strip(), "行": i, "条目": []} cur_ch["节"].append(cur_sec) stats["节"] += 1 cur_ent = None continue me = H_ENTRY.match(raw) if me: if cur_sec is None: raise SystemExit(f"❌ 第 {i} 行出现条目标题但无节标题:{raw[:40]}") cur_ent = {"条目": me.group(1).strip(), "行": i, "子目": [], "附内子目": []} cur_sec["条目"].append(cur_ent) stats["条目"] += 1 continue msub = H_SUB.match(raw) if msub: if cur_ent is None: raise SystemExit(f"❌ 第 {i} 行出现子目标题但无条目标题:{raw[:40]}") cur_ent["子目"].append({"子目": msub.group(1).strip(), "行": i}) stats["子目"] += 1 continue msub6 = H_SUB6.match(raw) if msub6: if cur_ent is None: raise SystemExit(f"❌ 第 {i} 行出现六级标题但无条目标题:{raw[:40]}") cur_ent["附内子目"].append({"附内子目": msub6.group(1).strip(), "行": i}) stats["附内子目"] += 1 continue # 逐级补「回溯行号」(合卷行号 − 所属章偏移 = 该章清洗版行号) def back(ch_label, l): return l - offsets[ch_label] for ch in chapters: for sec in ch["节"]: sec["源行"] = back(ch["章"][:3], sec["行"]) for ent in sec["条目"]: ent["源行"] = back(ch["章"][:3], ent["行"]) for sub in ent["子目"] + ent["附内子目"]: sub["源行"] = back(ch["章"][:3], sub["行"]) return chapters, offsets, stats def main(): check_only = "--check" in sys.argv srcs, errs = load_sources() if errs: print("❌ 来源校验未通过:") for e in errs: print(" -", e) return 1 merged, body = build_merged(srcs) out_md = os.path.join(BASE, OUT_MD) out_json = os.path.join(BASE, OUT_JSON) old_md = read_text(OUT_MD) if os.path.exists(out_md) else None chapters, offsets, stats = parse_structure(srcs, body) total_lines = len(merged.split("\n")) - 1 # ── 自检(断言,不静默)────────────────────────────────────── checks = [] checks.append(("来源 md5 三章全部相符", all(md5_of(os.path.join(BASE, s["文件"])) == s["md5"] for s in srcs))) checks.append(("合并正文 == 三章清洗版按章序逐字节拼接", merged.split("\n", 1)[1].lstrip("\n") == "\n\n".join(s["正文"].rstrip("\n") for s in srcs) + "\n")) checks.append(("行顺序:第1章 < 第2章 < 第3章", [c["章"][:3] for c in chapters] == ["第一章", "第二章", "第三章"])) checks.append(("章数 = 3", stats["章"] == 3)) checks.append(("无一级标题(除文书标题行)", len([l for l in merged.split("\n") if l.startswith("# ")]) == 1)) checks.append(("标题层级合法(## → ### → #### → ##### → ######,无越级)", all(re.match(r"^#{1,6} ", l) for l in merged.split("\n") if l.startswith("#")))) ok = all(v for _, v in checks) new_md5 = hashlib.md5(merged.encode("utf-8")).hexdigest() print(f"合卷正文:{OUT_MD}") print(f" {total_lines} 行 / {len(merged.encode('utf-8'))} 字节 / md5 {new_md5}") print(f" 章 {stats['章']} · 节 {stats['节']} · 条目 {stats['条目']} · 子目 {stats['子目']}" f" · 附内子目 {stats['附内子目']}") print(" 章行号(合卷 ⇄ 各章清洗版回溯):") for s in srcs: off = offsets[s["章"]] print(f" {s['章']} 合卷第 {off + 1} 行起(清洗版行号 + {off})," f"该章 {s['行数']} 行,md5 {s['md5'][:8]}…") print(" 自检:") for name, v in checks: print(f" {'✅' if v else '❌'} {name}") if not ok: print("\n❌ 自检未通过,未写盘。") return 1 if check_only: print("\n--check:只校验,未写盘。") return 0 changed = (old_md != merged) if changed: with open(out_md, "w", encoding="utf-8") as f: f.write(merged) print(f"\n{'✅ 已写入' if changed else '≡ 内容与库内一致,未改写'}:{OUT_MD}") structure = { "定位": "《〈辅行诀五脏用药法要〉药性探真》第一至三章 合并清洗版(衣之镖撰;学苑出版社 2014)", "级别": "现代研究文献;引用优先级**低于**《辅行诀脏腑用药法要》(参考本),更低于" "《辅行诀五脏用药法要》(底本)。**不得以本合并本替代《辅行诀》原书引用**。", "用户指令": "2026-09-22「第1-3章已经清理完成,合并后导入资料库,级别低于 " "…/02_加工数据/辅行诀脏腑用药法要.md」", "产物文件": OUT_MD, "产物md5": new_md5, "产物行数": total_lines, "产物字节": len(merged.encode("utf-8")), "文书标题行": TITLE, "标题体例": "文书标题=#(一级)/章=##(二级)/节=###(三级)/条目=####(四级)/子目=#####(五级)" "/附内子目=######(六级);第 1 章之五级为「附:」、六级为附内子目,第 3 章之五级为子目(1. …)," "三章体例**原样保留**、不作跨章统一", "合并规则": "三章清洗版正文按章序逐字节拼接,章间空一行;**一字未增删**(文书标题行为合卷所必需之标识行)", "来源": [{"章": s["章"], "文件": s["文件"], "md5": s["md5"], "行数": s["行数"], "合卷起始行": offsets[s["章"]] + 1, "清洗版行号偏移": offsets[s["章"]]} for s in srcs], "行号对照": {"规则": "合卷行号 = 该章清洗版行号 + 偏移;各章条目均带「源行」为其清洗版行号", "各章偏移": {s["章"]: offsets[s["章"]] for s in srcs}}, "章": chapters, "统计": stats, "自检": [{"项": n, "结果": "✅ 通过" if v else "❌ 未通过"} for n, v in checks], "上游脚本": ["07_clean_ch1.py(md5 见 README §七)", "10_clean_ch2.py", "12_clean_ch3.py"], "重跑": "python3 05_脚本工具/14_merge_ch123.py(幂等;--check 只校验不写盘)", } txt = json.dumps(structure, ensure_ascii=False, indent=2) + "\n" old_json = read_text(OUT_JSON) if os.path.exists(out_json) else None if old_json != txt: with open(out_json, "w", encoding="utf-8") as f: f.write(txt) print(f"✅ 已写入:{OUT_JSON}") else: print(f"≡ 内容与库内一致,未改写:{OUT_JSON}") return 0 if __name__ == "__main__": sys.exit(main())