Files
health/四季调养药膳/05_脚本工具/cross_link_multilibrary.py
T

288 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
四季调养药膳 × 多库全量关联 (参照 中华药膳全书-大医网 关联方法)
策略(按cross-database-linking.md):
①名称匹配(去剂型后缀, 精确3.0/包含1.5) ②成分重叠(≥2共现, 2.0/对) ③功效关键词(1.0/个)
④症状/疾病桥接(功效关键词→疾病名称/常见症状)
目标库:
A. 中华药膳全书 637方(配方库301富关联+调理药膳336) — 名称+成分重叠
B. 大医网药膳食疗1000 — 名称+功效关键词
C. 大医网中药材1000 — 配方药材成分(canonical, 含性味归经/功效溯源)
D. 大医网疾病998 — 症状桥接(功效适用症→疾病名/常见症状)
输出: 03_关联融合/四季调养-多库关联.json (全字段)
03_关联融合/四季调养-多库关联_强关联TOP50.json
"""
import json, os, re, collections
BASE = os.path.expanduser('~/Documents/ai_agent_scraper_study/data')
SJSY_DIR = os.path.join(BASE, '四季调养药膳')
sjsy = json.load(open(SJSY_DIR + '/02_加工数据/四季调养药膳.json', encoding='utf-8'))['recipes']
# ---------- 通用工具 ----------
DOSAGE_FORM = r'[汤粥羹茶酒饮膏糊汁露浆煎饼糕面饭包卷丸散丹锭条片剂液糖]'
def norm_name(s):
return re.sub(DOSAGE_FORM, '', re.sub(r'[\s()()。,,、的]', '', s or ''))
def extract_efficacy_keywords(text):
pats = [
r'[滋阴补阳益气养血健脾疏肝理肺补肾和胃安神定志祛风散寒清热解暑利湿化痰活血化瘀消肿止痛通经活络]',
r'补[气血阴阳肝肾脾肺心精骨髓]', r'滋[阴肾]|益[气精血肝脾肾肺心]',
r'养[血阴心神肝胃肺肾]|健[脾胃]', r'祛[风寒湿暑热痰瘀邪]|散[寒风热瘀结]',
r'清[热暑湿解毒火肝肺胃心]|解[表毒暑热郁]', r'止[咳痛泻血带痒吐]|化[痰瘀湿积石]',
r'消[肿食积炎毒]|利[水尿湿咽]', r'通[经络便乳脉窍]|润[肺肠燥肤喉]',
r'疏[风肝散]|活[血络]|降[气逆火压糖]', r'调[经中补和]|平[喘肝逆]|宣[肺痹通]',
r'助[消化眠阳]|安[神胎]|定[喘惊]',
]
kw = set()
for p in pats:
kw.update(re.findall(p, text or ''))
return kw
# ---------- 四季库: 配方成分抽取(复用cross_link的黑名单/前缀后缀规则) ----------
herb_dir = os.path.join(BASE, '大医网', '01_来源数据', '中药材')
herb_names, herb_alias = [], collections.defaultdict(set)
for fn in sorted(os.listdir(herb_dir)):
if not fn.endswith('.json'):
continue
try:
d = json.load(open(os.path.join(herb_dir, fn), encoding='utf-8'))
except Exception:
continue
name = (d.get('名称') or '').strip()
if not name:
continue
herb_names.append((name, fn.split('_')[0]))
for al in re.split(r'[,,、;;\s]+', d.get('别名', '') or ''):
if 1 < len(al) <= 6:
herb_alias[al].add(name)
NON_HERB = {'料酒','绍酒','黄酒','米酒','白酒','素油','菜油','麻油','香油','味精','食盐','精盐',
'白糖','冰糖','红糖','蜂蜜','酱油','醋','淀粉','水淀粉','玉米粉','面粉','米粉','糯米粉',
'药包','网油','清汤','高汤','鸡汤','上汤','开水','温水','凉水','淘米水'}
FOOD_COLLISION_ALIAS = {'猪肚','猪肚子','羊肉','鱼肚','鱼白','青鱼','蜂乳'}
EXTRA_ALIAS = {'杞子':'枸杞子','红枣':'大枣','圆肉':'龙眼肉','苡仁':'薏苡仁','薏仁':'薏苡仁',
'苡米':'薏苡仁','薏苡米':'薏苡仁','麦门冬':'麦冬','天门冬':'天冬','云苓':'茯苓',
'白苓':'茯苓','银花':'金银花','双花':'金银花','虫草':'冬虫夏草','广木香':'木香',
'首乌':'何首乌','制首乌':'何首乌','燕菜':'燕窝'}
# 注意: 不设 '白果'→'银杏叶' — 白果(种子)与银杏叶(叶)是不同药材, 大医网无白果条目时宁缺勿错
for a, n in EXTRA_ALIAS.items():
herb_alias[a].add(n)
all_terms = sorted(set([n for n, _ in herb_names] + list(herb_alias.keys()))
- NON_HERB - FOOD_COLLISION_ALIAS, key=len, reverse=True)
term2names = collections.defaultdict(set)
for n, _ in herb_names:
term2names[n].add(n)
for a, names in herb_alias.items():
for n in names:
term2names[a].add(n)
LEFT_OK = set('炙制炒煨煅焦法霜飞生熟粉鲜嫩干白')
RIGHT_OK = set('片末粉霜茸丝仁心皮边')
def extract_herbs(material_text):
found = collections.defaultdict(set)
occupied = []
def taken(a, b):
return any(not (b <= s or a >= e) for s, e in occupied)
ONE_CHAR_OK = {'葱', '梨', '藕'} # 大医网仅有的3个安全1字药名(边界检查防雪梨/料酒类误配)
for term in all_terms:
if len(term) < 2 and term not in ONE_CHAR_OK:
continue # 1字词仅放行白名单
for m in re.finditer(re.escape(term), material_text or ''):
a, b = m.start(), m.end()
if taken(a, b):
continue
if a > 0 and '\u4e00' <= material_text[a-1] <= '\u9fff':
if material_text[a-1] not in LEFT_OK or (material_text[a-1] + term) in all_terms:
continue
if b < len(material_text) and '\u4e00' <= material_text[b] <= '\u9fff':
if material_text[b] not in RIGHT_OK or (term + material_text[b]) in all_terms:
continue
occupied.append((a, b))
for canon in term2names[term]:
found[canon].add(m.group(0))
return found
# 每条四季药膳预计算: 药材成分 + 功效关键词
for r in sjsy:
herbs = extract_herbs(r['配方'])
r['药材成分'] = [{'canonical_name': k, 'raw_name': sorted(v)[0]} for k, v in sorted(herbs.items())]
r['功效关键词'] = sorted(extract_efficacy_keywords(r['功效'] + ' ' + r['解析']))
r['normalized_name'] = norm_name(r['名称'])
# ---------- 目标库A: 中华药膳全书 637 ----------
BJ = os.path.join(BASE, '中华药膳全书学做药膳不生病')
zhong = []
rich = json.load(open(BJ + '/02_加工数据/药膳配方库_recipes_富关联.json', encoding='utf-8'))
for rec in rich:
ings = [x['canonical_name'] for x in rec.get('药材成分', [])] + \
[x['canonical_name'] for x in rec.get('食材成分', [])]
zhong.append({'库': '配方库', 'id': rec['id'], '名称': rec['name'],
'成分': set(ings), '功效': extract_efficacy_keywords(rec.get('功能效用', '')),
'normalized': norm_name(rec['name'])})
for rec in json.load(open(BJ + '/01_来源数据/调理药膳.json', encoding='utf-8')):
ings = set(extract_herbs(rec.get('材料准备', '')).keys())
# 补充食材词(简单抽取: 2-6字中文段)
for p in re.split(r'[、,,。;;()()克毫升适量各少许\d]+', rec.get('材料准备', '')):
p = p.strip()
if 2 <= len(p) <= 6:
ings.add(p)
zhong.append({'库': '调理药膳', 'id': rec['id'], '名称': rec['名称'],
'成分': ings, '功效': extract_efficacy_keywords(rec.get('功能效用', '')),
'normalized': norm_name(rec['名称'])})
print(f"药膳全书目标: {len(zhong)} 方")
# ---------- 目标库B: 大医网药膳食疗 ----------
diet_dir = os.path.join(BASE, '大医网', '01_来源数据', '药膳食疗')
diets = []
for fn in sorted(os.listdir(diet_dir)):
if fn.endswith('.json'):
try:
d = json.load(open(os.path.join(diet_dir, fn), encoding='utf-8'))
except Exception:
continue
nm = (d.get('名称') or '').strip()
if nm:
diets.append({'id': fn.split('_')[0], '名称': nm,
'功效': extract_efficacy_keywords((d.get('功效') or '') + (d.get('简介') or '')[:200]),
'normalized': norm_name(nm)})
print(f"大医网药膳食疗目标: {len(diets)} 条")
# ---------- 目标库C: 大医网中药材(名称→file_id索引) ----------
herb_id = {n: fid for n, fid in herb_names}
# ---------- 目标库D: 大医网疾病 ----------
dis_dir = os.path.join(BASE, '大医网', '01_来源数据', '疾病')
diseases = []
for fn in sorted(os.listdir(dis_dir)):
if fn.endswith('.json'):
try:
d = json.load(open(os.path.join(dis_dir, fn), encoding='utf-8'))
except Exception:
continue
nm = (d.get('名称') or '').strip()
if not nm:
continue
# 常见症状字段可能为dict/str
cs = d.get('常见症状', '')
if isinstance(cs, dict):
cs = ' '.join(str(v) for v in cs.values())
sym = d.get('症状', '')
if isinstance(sym, dict):
sym = ' '.join(str(v) for v in sym.values())
diseases.append({'id': fn.split('_')[0], '名称': nm, '常见症状': cs or '', '症状': sym or ''})
print(f"大医网疾病目标: {len(diseases)} 条")
# ---------- 主循环: 四策略 ----------
cross = []
stats = collections.Counter()
for r in sjsy:
rec = {'id': r['id'], '名称': r['名称'], '季节': r['季节'], '性味': r['性味']}
my_ing = {x['canonical_name'] for x in r['药材成分']}
my_kw = set(r['功效关键词'])
# ①名称 + ②成分 + ③功效 → 药膳全书
m_zhong, ing_zhong, eff_zhong = [], [], []
my_norm = r['normalized_name']
for z in zhong:
if my_norm and z['normalized'] == my_norm:
m_zhong.append([z['库'], z['id'], z['名称'], 'exact'])
elif len(z['normalized']) >= 2 and len(my_norm) >= 2 and \
(z['normalized'] in my_norm or my_norm in z['normalized']):
m_zhong.append([z['库'], z['id'], z['名称'], 'partial'])
common = my_ing & z['成分']
if len(common) >= 2:
ratio = round(len(common) / max(len(my_ing), len(z['成分'])), 3)
ing_zhong.append([z['库'], z['id'], z['名称'], ratio, sorted(common)])
eff = my_kw & z['功效']
if len(eff) >= 2:
eff_zhong.append([z['库'], z['id'], z['名称'], len(eff), sorted(eff)])
ing_zhong.sort(key=lambda x: -x[3]); eff_zhong.sort(key=lambda x: -x[3])
rec['药膳全书_名称'] = m_zhong[:5]
rec['药膳全书_成分重叠'] = ing_zhong[:5]
rec['药膳全书_功效重叠'] = eff_zhong[:5]
# B: 药膳食疗 (名称+功效)
m_diet, eff_diet = [], []
for z in diets:
if my_norm and z['normalized'] == my_norm:
m_diet.append([z['id'], z['名称'], 'exact'])
elif len(z['normalized']) >= 2 and len(my_norm) >= 2 and \
(z['normalized'] in my_norm or my_norm in z['normalized']):
m_diet.append([z['id'], z['名称'], 'partial'])
eff = my_kw & z['功效']
if len(eff) >= 2:
eff_diet.append([z['id'], z['名称'], len(eff), sorted(eff)])
eff_diet.sort(key=lambda x: -x[2])
rec['大医网药膳食疗_名称'] = m_diet[:5]
rec['大医网药膳食疗_功效重叠'] = eff_diet[:5]
# C: 中药材 (成分→性味归经溯源)
rec['大医网中药材'] = [{'canonical_name': x['canonical_name'], 'raw_name': x['raw_name'],
'file_id': herb_id.get(x['canonical_name'], '')} for x in r['药材成分']]
# D: 疾病桥接 (功效文本中的适用症 → 疾病名/常见症状)
eff_text = r['功效']
d_hits = []
for z in diseases:
score = 0
why = []
if z['名称'] and z['名称'] in eff_text:
score += 3; why.append('名称直配')
else:
# 症状词桥接: 疾病名2-4字出现在功效文本; 或功效关键词在疾病常见症状中
for kw in my_kw:
if len(kw) >= 2 and kw in (z['常见症状'] or ''):
score += 1; why.append(kw)
if score >= 2:
d_hits.append([z['id'], z['名称'], score, sorted(set(why))])
d_hits.sort(key=lambda x: -x[2])
rec['大医网疾病_桥接'] = d_hits[:8]
# 综合评分
strong = bool(m_zhong and (ing_zhong or eff_zhong))
score = 0
if m_zhong:
score += 3 if m_zhong[0][3] == 'exact' else 1.5
if ing_zhong:
score += ing_zhong[0][3] * 2
if eff_zhong:
score += min(eff_zhong[0][3], 5)
if m_diet:
score += 3 if m_diet[0][2] == 'exact' else 1.5
if d_hits:
score += d_hits[0][2] * 0.5
rec['综合关联分'] = round(score, 1)
rec['强关联'] = strong
stats['强关联' if strong else '普通'] += 1
if any([m_zhong, ing_zhong, eff_zhong, m_diet, eff_diet, rec['大医网中药材'], d_hits]):
stats['有任意关联'] += 1
cross.append(rec)
# ---------- 输出 ----------
out_dir = os.path.join(SJSY_DIR, '03_关联融合')
top = sorted(cross, key=lambda x: -x['综合关联分'])[:50]
with open(os.path.join(out_dir, '四季调养-多库关联.json'), 'w', encoding='utf-8') as f:
json.dump({'metadata': {'本库': '四季调养药膳(156条)',
'目标库': {'A': '中华药膳全书637方(配方库301富关联+调理药膳336)',
'B': '大医网药膳食疗1000', 'C': '大医网中药材1000',
'D': '大医网疾病998'},
'策略': ['①名称匹配', '②成分重叠≥2', '③功效关键词≥2', '④症状疾病桥接'],
'统计': dict(stats),
'药材成分去重数': len({x['canonical_name'] for r in sjsy for x in r['药材成分']}),
'生成日期': '2026-09-08'},
'cross_references': cross}, f, ensure_ascii=False, indent=1)
with open(os.path.join(out_dir, '四季调养-多库关联_强关联TOP50.json'), 'w', encoding='utf-8') as f:
json.dump({'metadata': {'说明': '按综合关联分排序的TOP50', '字段': '同主文件'},
'top50': top}, f, ensure_ascii=False, indent=1)
print(f"\n强关联: {stats['强关联']}, 有任意关联: {stats['有任意关联']}/156")
print("\n=== TOP5 综合关联示例 ===")
for r in top[:5]:
print(f"{r['综合关联分']:5.1f} {r['id']} {r['名称']}({r['季节']}/{r['性味']})")
print(f" 药膳全书名称: {r['药膳全书_名称'][:2]}")
print(f" 成分重叠TOP2: {[x[2] for x in r['药膳全书_成分重叠'][:2]]}")
print(f" 疾病桥接TOP3: {[x[1] for x in r['大医网疾病_桥接'][:3]]}")