Files
health/中医入门资料库/05_脚本工具/build_theory_basis.py
T

181 lines
9.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
构建《中医入门》应用理论依据库 → 02_加工数据/理论依据库.json
用途: 为体质调理/药膳推荐/处方分析等应用提供可调用的理论依据(原文+行号溯源)
每条依据含: 原文引文 + 行号(指向 02_加工数据/中医入门.md)
"""
import json, os, re, datetime
BOOK = "/home/songyi/Documents/ai_agent_scraper_study/data/中医入门资料库/02_加工数据/中医入门.md"
OUT = "/home/songyi/Documents/ai_agent_scraper_study/data/中医入门资料库/02_加工数据/理论依据库.json"
TODAY = datetime.date.today().isoformat()
lines = open(BOOK, encoding="utf-8").read().splitlines()
text = "\n".join(lines)
def ln(idx): # 字符偏移 → 行号(1-based)
return text[:idx].count("\n") + 1
def find_seg(start, end):
i, j = text.find(start), text.find(end)
return (i, j) if 0 <= i < j else (i, len(text))
lib = {"metadata": {
"生成日期": TODAY,
"来源": "中医入门资料库/02_加工数据/中医入门.md(秦伯未《中医入门》体例,用户定级: 重要理论基础)",
"用途": "应用调用层——体质调理/药膳推荐/处方分析/配伍安全核查时,调用本库返回《中医入门》理论依据(原文+行号)",
"查询接口": "../05_脚本工具/theory_api.py"}}
# ============ 八法 ============
i0, j0 = find_seg("三、八法", "四、常用治法")
bafa = {}
for m in re.finditer(r'^\d+[\.、]\s*([^\s::,。]{1,2}法)[::]\s*(.+)$', text[i0:j0], re.M):
name = m.group(1)
seg_after = text[i0+j0 if False else i0+m.end(): j0]
first = seg_after.split("。")[0]
bafa[name] = {"定义原文": (m.group(2).split("。")[0] + "。").strip() if m.group(2) else "",
"行号": ln(i0 + m.start())}
lib["八法"] = bafa
# ============ 常用治法 72 ============
i0, j0 = find_seg("四、常用治法", "## 第三章")
zhifa, missing = {}, []
for m in re.finditer(r'^(\d+)[\.、,,]?\s*([^\s::,。]{2,12}法)[::]\s*(.*)$', text[i0:j0], re.M):
num, name, rest = int(m.group(1)), m.group(2), m.group(3)
use = re.search(r'用于(.*?)(?:药如|。|$)', rest)
yao = re.search(r'药如([^.。;;]*)', rest)
zhifa[str(num)] = {"名称": name,
"适应症": use.group(1).strip(",,。 ") if use else None,
"药如": ([x.strip() for x in re.split(r'[、,,]', yao.group(1)) if x.strip()] if yao else []),
"行号": ln(i0 + m.start())}
nums = sorted(int(k) for k in zhifa)
gaps = [n for n in range(nums[0], nums[-1]+1) if n not in nums]
lib["治法72"] = {"条目": zhifa, "总数": len(zhifa), "编号范围": [nums[0], nums[-1]],
"缺号": gaps, "缺号说明": "原书编号缺号或OCR漏行,如实记录" if gaps else None}
# ============ 七方 ============
# 切片止于七方段结尾(其后是张景岳八阵与汪昂22类剂型分类,"22. 救急方"属剂型分类,非七方)
i0, j0 = find_seg("二、七方", "三、剂型")
cut = text.find("七方是方剂组成的法则之一", i0)
if 0 < cut < j0: j0 = cut
qifang, cur, cur_pos = {}, None, None
for m in re.finditer(r'^\d+[\.、]\s*([^\s::,。]{1,3})方[::]\s*(.+)$|^(二、七方)', text[i0:j0], re.M):
if m.group(3):
cur = None; continue
cur = m.group(1) + "方"
para = m.group(2) # (.+)$ 已捕获整段(每段一行),无需再找空行
# 清洗举例药名: 剥离连接词前缀("如下法中的大承气汤"→"大承气汤")
examples = []
for tok in re.findall(r'([\u4e00-\u9fff]{2,9}(?:汤|丸|散|饮|膏))', para):
tok = re.split(r'[如用和及等是的中的将]', tok)[-1]
if len(tok) >= 3 and tok not in examples: examples.append(tok)
qifang[cur] = {"定义原文": para[:150].strip(),
"书内举例": examples,
"行号": ln(i0 + m.start())}
lib["七方"] = qifang
# ============ 效能分类 15 ============
i0, j0 = find_seg("二、效能", "#### 1. 扶正类")
xiaoneng = {}
for m in re.finditer(r'^(\d+)[\.、]\s*([^\s::,。]{2,8}药)[::]\s*(.+)$', text[i0:j0], re.M):
seg = m.group(3)
reps = []
for km in re.finditer(r'([\u4e00-\u9fff]{2,8})如([、\u4e00-\u9fff()]{2,60})', seg):
for h in re.split(r'[、]', km.group(2)):
h = re.sub(r'([^)]*)', '', h).strip("等")
if h: reps.append(h)
xiaoneng[m.group(2)] = {"定义原文": seg.split("。")[0] + "。", "代表药": reps[:24],
"行号": ln(i0 + m.start())}
lib["效能分类"] = xiaoneng
# ============ 扶正/祛邪功效药组 ============
groups = []
def parse_groups(start, end, kind):
i0, j0 = find_seg(start, end)
cur_sys = ""
for m in re.finditer(r'^\((\d+)\)\s*属于(.+?):|^[^\n#\d(].*?——.*$', text[i0:j0], re.M):
line = m.group(0).strip()
if m.group(1):
cur_sys = m.group(2); continue
eff, _, rhs = line.partition("——")
label = f"{kind}·{cur_sys}·{eff.strip()}" if cur_sys else f"{kind}·{eff.strip()}"
herbs = []
for h in re.split(r'[、,,]', rhs.strip().strip("。")):
h = re.sub(r'([^)]*)', '', h).strip()
h = re.sub(r'^(亦可用鲜|亦可用)', '', h).strip()
if h: herbs.append(h)
groups.append({"组": label, "药": herbs, "行号": ln(i0 + m.start())})
parse_groups("#### 1. 扶正类", "#### 2. 祛邪类", "扶正类")
parse_groups("#### 2. 祛邪类", "三、归经", "祛邪类")
lib["功效药组"] = {"条目": groups, "总数": len(groups)}
# ============ 示例方 40 ============
i0, j0 = find_seg("一、基本方剂", "## 第四章")
shili = {}
for m in re.finditer(r'^(\d+)[\.、]\s*([^::\n]+)[::]\s*(.+)$', text[i0:j0], re.M):
name = m.group(2).strip()
rest = m.group(3)
comp = [c.strip() for c in rest.split("。")[0].split("、") if c.strip()]
pos = re.search(r'为([^,。]{2,12})主方', rest)
zhi = re.search(r'用于([^。]*)', rest)
shili[name] = {"序号": int(m.group(1)), "组成_书": comp,
"主方定位": pos.group(1) if pos else None,
"主治": zhi.group(1).strip() if zhi else None,
"行号": ln(i0 + m.start())}
lib["示例方40"] = shili
# ============ 二十八脉主症 ============
i0, j0 = find_seg("二十八脉的主症", "## 第三章")
mai = {}
for m in re.finditer(r'^([^\s,。]{1,2})脉主(.+)$', text[i0:j0], re.M):
mai[m.group(1) + "脉"] = {"主症原文": m.group(2).strip().rstrip("。"), "行号": ln(i0 + m.start())}
lib["脉象主症"] = {"条目": mai, "总数": len(mai),
"说明": "标题为二十八脉,实列27条,长脉仅见总括韵文(原书编排,如实记录)"}
# ============ 用药安全 ============
ph_i = text.find("一、配合"); ph_j = text.find("二、用量")
ph = text[ph_i:ph_j]
i18 = ph.find("十八反歌:")
seg18_prose = ph[ph.find("歌中所提十八种药"):ph.find("歌中所提十九种药")]
seg19_prose = ph[ph.find("歌中所提十九种药"):ph.find("此外,妊娠禁忌")]
# 畏对提取: 从"如硫黄畏朴硝"起取,避开首句"即表示相畏比较显著"的噪音
iex = seg19_prose.find("如硫黄畏朴硝")
pairs = [(a.replace("如", "", 1) if a.startswith("如") else a, b)
for a, b in re.findall(r'([\u4e00-\u9fff]{1,4})畏([\u4e00-\u9fff]{1,5})', seg19_prose[iex:])
if len(a) >= 2 and len(b) >= 2]
for extra in [("川乌", "犀角"), ("草乌", "犀角")]: # "川乌、草乌畏犀角"正则只捕草乌,补川乌
if extra not in pairs: pairs.append(extra)
i19 = ph.find("十九畏歌:")
seg_preg = ph[ph.find("此外,妊娠禁忌"):ph.find("经验告诉我们")]
safety = {"十八反": {
"相反组": [
{"组": ["乌头"], "组_通行扩展": ["川乌", "草乌", "附子"],
"反": ["半夏", "瓜蒌", "贝母", "白蔹", "白及"], "反_通行扩展": [],
"说明": "书列'乌头';附子为乌头子根,通行本明列附子反半夏/瓜蒌等,安全核查按通行扩展执行"},
{"组": ["甘草"], "组_通行扩展": [], "反": ["海藻", "大戟", "甘遂", "芫花"], "反_通行扩展": [],
"说明": None},
{"组": ["藜芦"], "组_通行扩展": [],
"反": ["人参", "沙参", "细辛", "芍药"], "反_通行扩展": ["丹参", "玄参", "苦参"],
"说明": "书列'诸参'散文明列人参、沙参;通行本扩展丹参/玄参/苦参"}],
"歌诀原文": re.search(r'十八反歌:[^\n]+', ph).group(0),
"散文原文": seg18_prose.strip().split("\n")[0],
"行号": ln(ph_i + i18),
"通行扩展说明": "歌诀'诸参辛芍反藜芦',通行本藜芦组扩展含丹参/玄参/苦参;乌头组扩展含附子。扩展项在核查结果中标注[通行扩展]"},
"十九畏": {
"畏对": pairs,
"散文原文": seg19_prose.strip().split("\n")[0],
"行号": ln(ph_i + i19)},
"妊娠禁忌": {"药": [], "原文摘录": seg_preg.strip()[:400], "行号": ln(ph_i + ph.find("此外,妊娠禁忌"))}}
for m in re.finditer(r'(植物|动物|矿物)药如([^;。]+)', seg_preg):
for x in re.split(r'[、]', m.group(2)):
x = x.strip()
if 2 <= len(x) <= 6: safety["妊娠禁忌"]["药"].append(x)
lib["用药安全"] = safety
# ============ 输出 ============
json.dump(lib, open(OUT, "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("理论依据库:", OUT, f"({os.path.getsize(OUT)} bytes)")
print(f"八法 {len(bafa)} | 治法 {len(zhifa)}(缺号{gaps}) | 七方 {len(qifang)} | 效能 {len(xiaoneng)} | "
f"药组 {len(groups)} | 示例方 {len(shili)} | 脉 {len(mai)} | 十八反组3 | 十九畏 {len(safety['十九畏']['畏对'])}对 | 妊娠禁忌 {len(safety['妊娠禁忌']['药'])}味")