Files
health/辅行诀五脏用药法要-药性探真/05_脚本工具/04_ref_index.py
T

169 lines
8.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
04_ref_index.py —— 从《辅行诀五脏用药法要》基础文本(印刷本清洗版)抽取
「核校参考索引」,供《药性探真》后续章节清洗时卡药名 / 方名 / 五行互含名位。
产出:
02_加工数据/参考索引_辅行诀药名方名.json 机器可读索引
02_加工数据/参考索引_辅行诀药名方名.md 人工速查表
底本:02_加工数据/辅行诀五脏用药法要_基础文本_清洗版.md(baseline,md5 见 README)
性质:只读底本、不修改任何原文;幂等可重跑。
用法:
python3 05_脚本工具/04_ref_index.py # 生成索引
python3 05_脚本工具/04_ref_index.py 药名 # 查某个词是否在参考表内
"""
import json
import re
import sys
import hashlib
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
SRC = ROOT / "02_加工数据" / "辅行诀五脏用药法要_基础文本_清洗版.md"
OUT_JSON = ROOT / "02_加工数据" / "参考索引_辅行诀药名方名.json"
OUT_MD = ROOT / "02_加工数据" / "参考索引_辅行诀药名方名.md"
TASTE_TO_XING = {"辛": "木", "咸": "火", "甘": "土", "酸": "金", "苦": "水"}
# 底本中同物异写的形态(同一底本内两形并见者,仅登记不改动底本)
ALIAS_NOTES = [
{"规范形": "栝楼", "异写": ["栝蒌"], "依据": "底本第429行「栝楼为土中土」,方剂篇作「栝蒌」;《药性探真》第一章作「栝蒌」"},
{"规范形": "山茱萸", "异写": ["萸肉"], "依据": "药精表作「萸肉」,方剂篇加减法作「山茱萸」"},
{"规范形": "干姜", "异写": ["甘草炙→炙甘草"], "依据": "药精表「甘草炙」「甘草」分列二味(土中金/土中火),方剂篇作「甘草,炙」"},
]
def parse(lines):
yaojing, hunhan, fangming = [], [], []
# ── 1. 二十五味药精 ───────────────────────────────
for i, line in enumerate(lines, 1):
m = re.match(r"^味([辛咸甘酸苦])皆属([木火土金水])[,,]", line)
if not m:
continue
taste, benxing = m.group(1), m.group(2)
assert TASTE_TO_XING[taste] == benxing, f"第{i}行 味-行不符:{taste}≠{benxing}"
for seg in re.split(r"[;;。]", line[m.end():]):
seg = seg.strip()
if not seg:
continue
mm = re.match(r"^(\S+?)\s+(\S+?)(?:为之主|为([木火土金水]))$", seg)
if not mm:
raise SystemExit(f"第{i}行 药精条无法解析:{seg}")
drug, mineral, hanxing = mm.group(1), mm.group(2), mm.group(3)
is_main = hanxing is None
hanxing = benxing if is_main else hanxing
yaojing.append({
"药名": drug, "味": taste, "本行": benxing,
"互含名位": f"{benxing}中{hanxing}",
"配对金石药": mineral, "药精位次": "主" if is_main else f"含{hanxing}",
"行号": i, "底本依据": line.strip(),
})
# ── 2. 十三种药(五行互含,备心病方之用) ──────────
start = None
for i, line in enumerate(lines, 1):
if "又有药十三种" in line:
start = i
break
if start is None:
raise SystemExit("未找到「又有药十三种」条")
body = lines[start] # 该条正文紧随其后
for i in range(start, min(start + 4, len(lines))):
if re.search(r"为[木火土金水]中[木火土金水]", lines[i]):
body = lines[i]
body_line = i + 1
break
for seg in re.split(r"[;;。,,]", body):
seg = seg.strip()
if not seg:
continue
if seg.startswith("又为"):
if not hunhan:
raise SystemExit(f"第{body_line}行「又为」无前项:{seg}")
hunhan[-1]["互含名位"].append(seg[2:])
continue
mm = re.match(r"^(\S+?)为([木火土金水]中[木火土金水])$", seg)
if not mm:
raise SystemExit(f"第{body_line}行 十三种药条无法解析:{seg}")
hunhan.append({"药名": mm.group(1), "互含名位": [mm.group(2)], "行号": body_line,
"底本依据": body.strip()})
# ── 3. 方名 ───────────────────────────────────────
seen = set()
for i, line in enumerate(lines, 1):
m = re.match(r"^([^::\s]{2,16}?(?:汤|散|丸|煎|膏|酒|方))[::]", line)
if not m:
continue
name = m.group(1)
if name in seen:
continue
seen.add(name)
fangming.append({"方名": name, "行号": i, "主治": line.split(":", 1)[-1][:40]})
return yaojing, hunhan, fangming, body_line
def main():
text = SRC.read_text(encoding="utf-8")
lines = text.split("\n")
yaojing, hunhan, fangming, hunhan_line = parse(lines)
drugs = [d["药名"] for d in yaojing] + [d["药名"] for d in hunhan]
index = {
"来源": SRC.name,
"来源md5": hashlib.md5(text.encode("utf-8")).hexdigest(),
"生成脚本": "05_脚本工具/04_ref_index.py",
"说明": "《药性探真》后续章节清洗的核校参考表:药名、五行互含名位、配对金石药、方名。",
"药精二十五味": yaojing,
"五行互含十三种药": hunhan,
"方名清单": fangming,
"药名一览": sorted(set(drugs)),
"同物异写登记": ALIAS_NOTES,
"计数": {"药精": len(yaojing), "十三种药": len(hunhan),
"药名合计": len(set(drugs)), "方名": len(fangming)},
}
OUT_JSON.write_text(json.dumps(index, ensure_ascii=False, indent=2), encoding="utf-8")
# ── 人工速查 md ───────────────────────────────────
L = ["# 参考索引|《辅行诀五脏用药法要》药名·方名·五行互含名位", "",
f"> 生成:`05_脚本工具/04_ref_index.py`(幂等)|底本:`{SRC.name}`(md5 `{index['来源md5']}`)",
"> 用途:《药性探真》后续章节清洗时的**核校参考**(药名规范形、五行互含名位、配对金石药、方名)。",
"> 原则:只作旁证,不改动《药性探真》原文;发现不一致→登记人工核对清单。", "",
"## 一、二十五味药精(《汤液经法》五行互含)", "",
"| 味 | 本行 | 互含名位 | 药名 | 配对金石药 | 底本行号 |", "|---|---|---|---|---|---|"]
for d in yaojing:
L.append(f"| {d['味']} | {d['本行']} | {d['互含名位']} | {d['药名']} | {d['配对金石药']} | {d['行号']} |")
L += ["", "## 二、五行互含十三种药(备心病方之用)", "", "| 药名 | 互含名位 | 底本行号 |", "|---|---|---|"]
for d in hunhan:
L.append(f"| {d['药名']} | {'、'.join(d['互含名位'])} | {d['行号']} |")
L += ["", "## 三、方名清单", "", "| 方名 | 底本行号 | 主治(截断) |", "|---|---|---|"]
for f in fangming:
L.append(f"| {f['方名']} | {f['行号']} | {f['主治']} |")
L += ["", "## 四、同物异写登记(只登记,不改底本)", ""]
for a in ALIAS_NOTES:
L.append(f"- **{a['规范形']}** /异写:{'、'.join(a['异写'])} —— {a['依据']}")
L += ["", "## 五、药名一览(去重)", "", "、".join(index["药名一览"]), ""]
OUT_MD.write_text("\n".join(L), encoding="utf-8")
print(f"药精 {len(yaojing)} 味 | 十三种药 {len(hunhan)} 味 | 药名去重 {len(set(drugs))} | 方名 {len(fangming)}")
print(f"索引:{OUT_JSON.relative_to(ROOT)}")
print(f"速查:{OUT_MD.relative_to(ROOT)}")
def lookup(word):
"""查词是否在参考表内(供后续章节清洗时快速卡药名)。"""
idx = json.loads(OUT_JSON.read_text(encoding="utf-8"))
hits = [d for d in idx["药精二十五味"] + idx["五行互含十三种药"] if word in d["药名"]]
fangs = [f for f in idx["方名清单"] if word in f["方名"]]
print(f"「{word}」 药名命中 {len(hits)} / 方名命中 {len(fangs)}")
for h in hits:
print(" 药:", h["药名"], h.get("互含名位"), h.get("配对金石药", ""))
for f in fangs:
print(" 方:", f["方名"], "→", f["主治"])
if __name__ == "__main__":
if len(sys.argv) > 1:
lookup(sys.argv[1])
else:
main()