276 lines
13 KiB
Python
276 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
辅行诀五脏用药法要-药性探真 资料库 ETL 第 14 步(2026-09-22):
|
||
《〈辅行诀五脏用药法要〉药性探真》第 1/2/3 章**清洗版** → 合并为单一清洗版正文,入 02_加工数据。
|
||
|
||
用户指令(2026-09-22):
|
||
「《辅行诀五脏用药法要 药性探真》第1-3章已经清理完成,合并后导入资料库,
|
||
级别低于 …/02_加工数据/辅行诀脏腑用药法要.md」
|
||
→ 合并(本脚本)+ 定级登记(README/来源说明):本合并本为**现代研究文献**,
|
||
**引用优先级低于《辅行诀脏腑用药法要》(参考本),更低于《辅行诀五脏用药法要》(底本)**。
|
||
|
||
原则(与库内既有规范一致):
|
||
① **只合并,不改字**:三章清洗版正文**一字未增删**(改字一律回到 07/10/12 清洗脚本,本脚本不设 FIXES 表);
|
||
② **可复核**:合并正文 == 三章清洗版正文按章序拼接(自检断言,逐字节);
|
||
③ **来源 md5 钉定**:任一章清洗版与登记值不符即报错(防静默合并到错版本);
|
||
④ **幂等**:重跑覆盖,输出与库内一致时不改写文件;
|
||
⑤ **行号可回溯**:产物结构 JSON 给出「合卷行号 ⇄ 各章清洗版行号」偏移对照。
|
||
|
||
产物:
|
||
02_加工数据/药性探真_第1-3章_合并清洗版.md (合卷正文)
|
||
02_加工数据/药性探真_第1-3章_合并结构.json (章/节/条目/子目 四级 + 合卷行号锚 + 各章偏移)
|
||
|
||
用法:
|
||
cd ~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真
|
||
python3 05_脚本工具/14_merge_ch123.py # 生成/刷新(幂等)
|
||
python3 05_脚本工具/14_merge_ch123.py --check # 只校验,不写盘
|
||
"""
|
||
import os
|
||
import re
|
||
import sys
|
||
import json
|
||
import hashlib
|
||
|
||
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
|
||
|
||
OUT_MD = "02_加工数据/药性探真_第1-3章_合并清洗版.md"
|
||
OUT_JSON = "02_加工数据/药性探真_第1-3章_合并结构.json"
|
||
|
||
# 合卷标题(文书级标识行;**不由来源正文产生**,系合卷所必需,正文本身一字未增删)
|
||
TITLE = "# 《辅行诀五脏用药法要 药性探真》第一至三章"
|
||
|
||
# 三章清洗版(顺序即合卷章序):相对路径 / 登记 md5 / 章标签
|
||
SRC = [
|
||
("02_加工数据/药性探真_第1章_清洗版.md", "a20b17b7589255889256426b795a86d9", "第一章"),
|
||
("02_加工数据/药性探真_第2章_清洗版.md", "ac1a4d325f971279bc583324c95408d8", "第二章"),
|
||
("02_加工数据/药性探真_第3章_清洗版.md", "5f1fb57c02de5d370a8aff51013062bb", "第三章"),
|
||
]
|
||
|
||
H_CHAPTER = re.compile(r"^##\s+(第[一二三四五六七八九十]+章)\s*(.*)$")
|
||
H_SECTION = re.compile(r"^###\s+(第[一二三四五六七八九十]+节)\s*(.*)$")
|
||
H_ENTRY = re.compile(r"^####\s+(.+)$")
|
||
H_SUB = re.compile(r"^#####\s+(.+)$")
|
||
H_SUB6 = re.compile(r"^######\s+(.+)$")
|
||
|
||
|
||
def md5_of(p):
|
||
return hashlib.md5(open(p, "rb").read()).hexdigest()
|
||
|
||
|
||
def read_text(rel):
|
||
"""CRLF-safe:文本模式会吞 CRLF,一律按字节读入再解码。"""
|
||
with open(os.path.join(BASE, rel), "rb") as f:
|
||
return f.read().decode("utf-8")
|
||
|
||
|
||
def load_sources():
|
||
"""读入三章清洗版 + 断言(章首行、无一级标题、md5 相符)。"""
|
||
srcs, errs = [], []
|
||
for rel, want, label in SRC:
|
||
p = os.path.join(BASE, rel)
|
||
if not os.path.exists(p):
|
||
errs.append(f"缺文件:{rel}")
|
||
continue
|
||
got = md5_of(p)
|
||
if got != want:
|
||
errs.append(f"md5 不符:{rel}\n 登记 {want}\n 实测 {got}"
|
||
f"\n → 清洗版已被重跑刷新,请先核对来源,再更新本脚本 SRC 表")
|
||
continue
|
||
txt = read_text(rel)
|
||
first = txt.split("\n")[0].strip()
|
||
m = H_CHAPTER.match(first)
|
||
if not m:
|
||
errs.append(f"章首行不是二级章标题:{rel} → {first!r}")
|
||
continue
|
||
if m.group(1) != label:
|
||
errs.append(f"章标签不符:{rel} 首行作 {m.group(1)},登记 {label}")
|
||
continue
|
||
h1 = [ln for ln in txt.split("\n") if ln.startswith("# ")]
|
||
if h1:
|
||
errs.append(f"来源含一级标题(合卷后与文书标题同级):{rel} → {h1[:3]}")
|
||
continue
|
||
if not txt.endswith("\n"):
|
||
errs.append(f"来源末行无换行:{rel}")
|
||
continue
|
||
srcs.append({"章": label, "文件": rel, "md5": got,
|
||
"行数": len(txt.split("\n")) - 1, # 末行为空行不计
|
||
"正文": txt})
|
||
return srcs, errs
|
||
|
||
|
||
def build_merged(srcs):
|
||
body = "\n\n".join(s["正文"].rstrip("\n") for s in srcs) + "\n"
|
||
return TITLE + "\n\n" + body, body
|
||
|
||
|
||
def parse_structure(srcs, body):
|
||
"""按合卷正文解析 章/节/条目/子目 四级,并给出「合卷行号 → 各章清洗版行号」回溯。"""
|
||
# 各章在合卷正文中的起始行(1-based):标题行 + 空行 + 各章正文
|
||
offsets, ln = {}, 1 + 2 # 1 行文书标题 + 1 空行
|
||
for s in srcs:
|
||
offsets[s["章"]] = ln - 1 # 合卷行号 = 该章清洗版行号 + offset
|
||
ln += s["行数"] + 1 # 章正文行数 + 1 空行分隔
|
||
|
||
lines = body.split("\n")
|
||
chapters, cur_ch, cur_sec, cur_ent = [], None, None, None
|
||
stats = {"章": 0, "节": 0, "条目": 0, "子目": 0, "附内子目": 0}
|
||
for i, raw in enumerate(lines, 1): # i = 合卷行号(正文第 1 行 = 合卷第 3 行)
|
||
if not raw.strip():
|
||
continue
|
||
mc = H_CHAPTER.match(raw)
|
||
if mc:
|
||
cur_ch = {"章": (mc.group(1) + " " + mc.group(2)).strip(),
|
||
"行": i, "节": []}
|
||
chapters.append(cur_ch)
|
||
stats["章"] += 1
|
||
cur_sec = cur_ent = None
|
||
continue
|
||
ms = H_SECTION.match(raw)
|
||
if ms:
|
||
if cur_ch is None:
|
||
raise SystemExit(f"❌ 第 {i} 行出现节标题但无章标题:{raw[:40]}")
|
||
cur_sec = {"节": (ms.group(1) + " " + ms.group(2)).strip(),
|
||
"行": i, "条目": []}
|
||
cur_ch["节"].append(cur_sec)
|
||
stats["节"] += 1
|
||
cur_ent = None
|
||
continue
|
||
me = H_ENTRY.match(raw)
|
||
if me:
|
||
if cur_sec is None:
|
||
raise SystemExit(f"❌ 第 {i} 行出现条目标题但无节标题:{raw[:40]}")
|
||
cur_ent = {"条目": me.group(1).strip(), "行": i, "子目": [], "附内子目": []}
|
||
cur_sec["条目"].append(cur_ent)
|
||
stats["条目"] += 1
|
||
continue
|
||
msub = H_SUB.match(raw)
|
||
if msub:
|
||
if cur_ent is None:
|
||
raise SystemExit(f"❌ 第 {i} 行出现子目标题但无条目标题:{raw[:40]}")
|
||
cur_ent["子目"].append({"子目": msub.group(1).strip(), "行": i})
|
||
stats["子目"] += 1
|
||
continue
|
||
msub6 = H_SUB6.match(raw)
|
||
if msub6:
|
||
if cur_ent is None:
|
||
raise SystemExit(f"❌ 第 {i} 行出现六级标题但无条目标题:{raw[:40]}")
|
||
cur_ent["附内子目"].append({"附内子目": msub6.group(1).strip(), "行": i})
|
||
stats["附内子目"] += 1
|
||
continue
|
||
|
||
# 逐级补「回溯行号」(合卷行号 − 所属章偏移 = 该章清洗版行号)
|
||
def back(ch_label, l):
|
||
return l - offsets[ch_label]
|
||
|
||
for ch in chapters:
|
||
for sec in ch["节"]:
|
||
sec["源行"] = back(ch["章"][:3], sec["行"])
|
||
for ent in sec["条目"]:
|
||
ent["源行"] = back(ch["章"][:3], ent["行"])
|
||
for sub in ent["子目"] + ent["附内子目"]:
|
||
sub["源行"] = back(ch["章"][:3], sub["行"])
|
||
return chapters, offsets, stats
|
||
|
||
|
||
def main():
|
||
check_only = "--check" in sys.argv
|
||
srcs, errs = load_sources()
|
||
if errs:
|
||
print("❌ 来源校验未通过:")
|
||
for e in errs:
|
||
print(" -", e)
|
||
return 1
|
||
|
||
merged, body = build_merged(srcs)
|
||
out_md = os.path.join(BASE, OUT_MD)
|
||
out_json = os.path.join(BASE, OUT_JSON)
|
||
old_md = read_text(OUT_MD) if os.path.exists(out_md) else None
|
||
|
||
chapters, offsets, stats = parse_structure(srcs, body)
|
||
total_lines = len(merged.split("\n")) - 1
|
||
|
||
# ── 自检(断言,不静默)──────────────────────────────────────
|
||
checks = []
|
||
checks.append(("来源 md5 三章全部相符",
|
||
all(md5_of(os.path.join(BASE, s["文件"])) == s["md5"] for s in srcs)))
|
||
checks.append(("合并正文 == 三章清洗版按章序逐字节拼接",
|
||
merged.split("\n", 1)[1].lstrip("\n")
|
||
== "\n\n".join(s["正文"].rstrip("\n") for s in srcs) + "\n"))
|
||
checks.append(("行顺序:第1章 < 第2章 < 第3章",
|
||
[c["章"][:3] for c in chapters] == ["第一章", "第二章", "第三章"]))
|
||
checks.append(("章数 = 3", stats["章"] == 3))
|
||
checks.append(("无一级标题(除文书标题行)",
|
||
len([l for l in merged.split("\n") if l.startswith("# ")]) == 1))
|
||
checks.append(("标题层级合法(## → ### → #### → ##### → ######,无越级)",
|
||
all(re.match(r"^#{1,6} ", l) for l in merged.split("\n")
|
||
if l.startswith("#"))))
|
||
ok = all(v for _, v in checks)
|
||
|
||
new_md5 = hashlib.md5(merged.encode("utf-8")).hexdigest()
|
||
print(f"合卷正文:{OUT_MD}")
|
||
print(f" {total_lines} 行 / {len(merged.encode('utf-8'))} 字节 / md5 {new_md5}")
|
||
print(f" 章 {stats['章']} · 节 {stats['节']} · 条目 {stats['条目']} · 子目 {stats['子目']}"
|
||
f" · 附内子目 {stats['附内子目']}")
|
||
print(" 章行号(合卷 ⇄ 各章清洗版回溯):")
|
||
for s in srcs:
|
||
off = offsets[s["章"]]
|
||
print(f" {s['章']} 合卷第 {off + 1} 行起(清洗版行号 + {off}),"
|
||
f"该章 {s['行数']} 行,md5 {s['md5'][:8]}…")
|
||
print(" 自检:")
|
||
for name, v in checks:
|
||
print(f" {'✅' if v else '❌'} {name}")
|
||
if not ok:
|
||
print("\n❌ 自检未通过,未写盘。")
|
||
return 1
|
||
|
||
if check_only:
|
||
print("\n--check:只校验,未写盘。")
|
||
return 0
|
||
|
||
changed = (old_md != merged)
|
||
if changed:
|
||
with open(out_md, "w", encoding="utf-8") as f:
|
||
f.write(merged)
|
||
print(f"\n{'✅ 已写入' if changed else '≡ 内容与库内一致,未改写'}:{OUT_MD}")
|
||
|
||
structure = {
|
||
"定位": "《〈辅行诀五脏用药法要〉药性探真》第一至三章 合并清洗版(衣之镖撰;学苑出版社 2014)",
|
||
"级别": "现代研究文献;引用优先级**低于**《辅行诀脏腑用药法要》(参考本),更低于"
|
||
"《辅行诀五脏用药法要》(底本)。**不得以本合并本替代《辅行诀》原书引用**。",
|
||
"用户指令": "2026-09-22「第1-3章已经清理完成,合并后导入资料库,级别低于 "
|
||
"…/02_加工数据/辅行诀脏腑用药法要.md」",
|
||
"产物文件": OUT_MD,
|
||
"产物md5": new_md5,
|
||
"产物行数": total_lines,
|
||
"产物字节": len(merged.encode("utf-8")),
|
||
"文书标题行": TITLE,
|
||
"标题体例": "文书标题=#(一级)/章=##(二级)/节=###(三级)/条目=####(四级)/子目=#####(五级)"
|
||
"/附内子目=######(六级);第 1 章之五级为「附:」、六级为附内子目,第 3 章之五级为子目(1. …),"
|
||
"三章体例**原样保留**、不作跨章统一",
|
||
"合并规则": "三章清洗版正文按章序逐字节拼接,章间空一行;**一字未增删**(文书标题行为合卷所必需之标识行)",
|
||
"来源": [{"章": s["章"], "文件": s["文件"], "md5": s["md5"], "行数": s["行数"],
|
||
"合卷起始行": offsets[s["章"]] + 1, "清洗版行号偏移": offsets[s["章"]]}
|
||
for s in srcs],
|
||
"行号对照": {"规则": "合卷行号 = 该章清洗版行号 + 偏移;各章条目均带「源行」为其清洗版行号",
|
||
"各章偏移": {s["章"]: offsets[s["章"]] for s in srcs}},
|
||
"章": chapters,
|
||
"统计": stats,
|
||
"自检": [{"项": n, "结果": "✅ 通过" if v else "❌ 未通过"} for n, v in checks],
|
||
"上游脚本": ["07_clean_ch1.py(md5 见 README §七)", "10_clean_ch2.py", "12_clean_ch3.py"],
|
||
"重跑": "python3 05_脚本工具/14_merge_ch123.py(幂等;--check 只校验不写盘)",
|
||
}
|
||
txt = json.dumps(structure, ensure_ascii=False, indent=2) + "\n"
|
||
old_json = read_text(OUT_JSON) if os.path.exists(out_json) else None
|
||
if old_json != txt:
|
||
with open(out_json, "w", encoding="utf-8") as f:
|
||
f.write(txt)
|
||
print(f"✅ 已写入:{OUT_JSON}")
|
||
else:
|
||
print(f"≡ 内容与库内一致,未改写:{OUT_JSON}")
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|