225 lines
11 KiB
Python
225 lines
11 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
四季调养药膳 — 跨库交叉关联脚本
|
||
方向:
|
||
A. 四季调养药膳 × 中华药膳全书(调理药膳336/配方库301): 名称精确+包含匹配
|
||
B. 四季调养药膳 × 大医网中药材1000: 配方中的药材名匹配(最长前缀+别名表)
|
||
C. 四季调养药膳 × 大医网方剂1000: 名称精确+包含匹配
|
||
输出: 03_关联融合/四季调养-药膳全书_交叉关联.json
|
||
03_关联融合/四季调养-大医网_交叉关联.json
|
||
"""
|
||
import json, os, re, collections
|
||
|
||
BASE = os.path.expanduser('~/Documents/ai_agent_scraper_study/data')
|
||
SJSY = os.path.join(BASE, '四季调养药膳', '02_加工数据', '四季调养药膳.json')
|
||
FUSION = os.path.join(BASE, '四季调养药膳', '03_关联融合')
|
||
|
||
sjsy = json.load(open(SJSY, encoding='utf-8'))['recipes']
|
||
|
||
# ============ B 前置: 大医网药材名与别名表 ============
|
||
herb_dir = os.path.join(BASE, '大医网', '01_来源数据', '中药材')
|
||
herb_names = [] # (名称, file_id)
|
||
herb_alias = collections.defaultdict(set) # 别名 -> {正名}
|
||
for fn in sorted(os.listdir(herb_dir)):
|
||
if not fn.endswith('.json'):
|
||
continue
|
||
try:
|
||
d = json.load(open(os.path.join(herb_dir, fn), encoding='utf-8'))
|
||
except Exception:
|
||
continue
|
||
name = d.get('名称', '').strip()
|
||
if not name:
|
||
continue
|
||
fid = fn.split('_')[0]
|
||
herb_names.append((name, fid))
|
||
for al in re.split(r'[,,、;;\s]+', d.get('别名', '') or ''):
|
||
al = al.strip()
|
||
if 1 < len(al) <= 6:
|
||
herb_alias[al].add(name)
|
||
|
||
# 别名补充(药膳常用):
|
||
EXTRA_ALIAS = {'杞子': '枸杞子', '红枣': '大枣', '圆肉': '龙眼肉',
|
||
'苡仁': '薏苡仁', '薏仁': '薏苡仁',
|
||
'苡米': '薏苡仁', '薏苡米': '薏苡仁', '麦门冬': '麦冬', '天门冬': '天冬',
|
||
'云苓': '茯苓', '白苓': '茯苓', '银花': '金银花', '双花': '金银花',
|
||
'虫草': '冬虫夏草', '广木香': '木香', '首乌': '何首乌', '制首乌': '何首乌',
|
||
'白果': '银杏叶', '燕菜': '燕窝'}
|
||
for a, n in EXTRA_ALIAS.items():
|
||
herb_alias[a].add(n)
|
||
|
||
# 非药材词(调味/厨料, 不做药材匹配; 大医网无对应条目或属食材常识)
|
||
NON_HERB = {'料酒', '绍酒', '黄酒', '米酒', '白酒', '素油', '菜油', '麻油', '香油',
|
||
'味精', '食盐', '精盐', '白糖', '冰糖', '红糖', '蜂蜜', '酱油', '醋',
|
||
'淀粉', '水淀粉', '玉米粉', '面粉', '米粉', '糯米粉', '药包', '网油',
|
||
'清汤', '高汤', '鸡汤', '上汤', '开水', '温水', '凉水', '淘米水'}
|
||
|
||
# 食材撞名别名黑名单: 大医网民间别名与食材同名, 禁止用作匹配
|
||
# (牛耳枫子别名"猪肚"、黄芪别名"羊肉"、车前草别名"猪肚子"、鱼胶别名"鱼肚/鱼白"、鲃鱼别名"青鱼")
|
||
FOOD_COLLISION_ALIAS = {'猪肚', '猪肚子', '羊肉', '鱼肚', '鱼白', '青鱼', '蜂乳'}
|
||
|
||
# 处理前缀(匹配时剥去): 炙/制/炒/煨/煅/焦/霜/法/飞/熟/生
|
||
PROC_PREFIX = re.compile(r'^(炙|制|炒|煨|煅|焦|霜|法|飞|熟|生)(?=[\u4e00-\u9fff]{2,})')
|
||
|
||
# 最长前缀排序列表(含别名), 排除非药材词
|
||
all_terms = sorted(set([n for n, _ in herb_names] + list(herb_alias.keys()))
|
||
- NON_HERB - FOOD_COLLISION_ALIAS,
|
||
key=len, reverse=True)
|
||
term2names = collections.defaultdict(set)
|
||
for n, _ in herb_names:
|
||
term2names[n].add(n)
|
||
for a, names in herb_alias.items():
|
||
for n in names:
|
||
term2names[a].add(n)
|
||
|
||
# 炮制前缀(允许出现在药材名左侧)与形态后缀(允许出现在右侧)
|
||
LEFT_OK = set('炙制炒煨煅焦法霜飞生熟粉鲜嫩干白')
|
||
RIGHT_OK = set('片末粉霜茸丝')
|
||
|
||
def extract_herbs(material_text):
|
||
"""从配方文本抽取药材名(最长前缀+span防重叠+炮制前缀/形态后缀宽容)"""
|
||
found = collections.defaultdict(set) # 大医网正名 -> 原文写法
|
||
occupied = [] # 已命中的字符区间, 防止短词嵌在长词/调味词内部
|
||
def taken(a, b):
|
||
return any(not (b <= s or a >= e) for s, e in occupied)
|
||
ONE_CHAR_OK = {'葱', '梨', '藕'} # 大医网仅有的3个安全1字药名(边界检查防雪梨/料酒类误配)
|
||
for term in all_terms:
|
||
if len(term) < 2 and term not in ONE_CHAR_OK:
|
||
continue # 1字词仅放行白名单
|
||
for m in re.finditer(re.escape(term), material_text):
|
||
a, b = m.start(), m.end()
|
||
if taken(a, b):
|
||
continue
|
||
# 左界: 前一字符须非汉字, 或为炮制前缀(且前缀+词不是更长的药材名)
|
||
if a > 0 and '\u4e00' <= material_text[a-1] <= '\u9fff':
|
||
if material_text[a-1] not in LEFT_OK or (material_text[a-1] + term) in all_terms:
|
||
continue
|
||
# 右界: 后一字符须非汉字, 或为形态后缀(且词+后缀不是更长的药材名)
|
||
if b < len(material_text) and '\u4e00' <= material_text[b] <= '\u9fff':
|
||
if material_text[b] not in RIGHT_OK or (term + material_text[b]) in all_terms:
|
||
continue
|
||
occupied.append((a, b))
|
||
for canon in term2names[term]:
|
||
found[canon].add(m.group(0))
|
||
return found
|
||
|
||
# ============ A: × 中华药膳全书 ============
|
||
book_files = {
|
||
'调理药膳': os.path.join(BASE, '中华药膳全书学做药膳不生病', '01_来源数据', '调理药膳.json'),
|
||
'药膳配方库': os.path.join(BASE, '中华药膳全书学做药膳不生病', '01_来源数据', '药膳配方库_recipes.json'),
|
||
}
|
||
book_records = []
|
||
for src, fp in book_files.items():
|
||
if os.path.exists(fp):
|
||
for rec in json.load(open(fp, encoding='utf-8')):
|
||
book_records.append({'来源': src, **{k: rec.get(k, '') for k in ('id', '名称', 'name', '症状', 'category')}})
|
||
|
||
norm = lambda s: re.sub(r'[\s()()。,,、的]', '', s or '')
|
||
book_by_norm = collections.defaultdict(list)
|
||
for rec in book_records:
|
||
nm = rec.get('名称') or rec.get('name')
|
||
if nm:
|
||
book_by_norm[norm(nm)].append(rec)
|
||
|
||
link_book = []
|
||
for r in sjsy:
|
||
hits = []
|
||
key = norm(r['名称'])
|
||
if key in book_by_norm:
|
||
for rec in book_by_norm[key]:
|
||
hits.append({'匹配类型': '名称精确', '对方库': rec['来源'],
|
||
'对方id': rec.get('id') or rec.get('category', ''), '对方名称': rec.get('名称') or rec.get('name')})
|
||
else:
|
||
for rec in book_records:
|
||
other = rec.get('名称') or rec.get('name') or ''
|
||
if not other:
|
||
continue
|
||
no, nk = norm(other), key
|
||
if len(no) >= 3 and (no in nk or nk in no):
|
||
hits.append({'匹配类型': '名称包含', '对方库': rec['来源'],
|
||
'对方id': rec.get('id') or rec.get('category', ''), '对方名称': other})
|
||
if hits:
|
||
link_book.append({'本库id': r['id'], '本库名称': r['名称'], '季节': r['季节'],
|
||
'关联数': len(hits), '关联': hits})
|
||
|
||
# ============ D: × 大医网药膳食疗1000 ============
|
||
diet_dir = os.path.join(BASE, '大医网', '01_来源数据', '药膳食疗')
|
||
diet_idx = []
|
||
for fn in sorted(os.listdir(diet_dir)):
|
||
if fn.endswith('.json'):
|
||
try:
|
||
d = json.load(open(os.path.join(diet_dir, fn), encoding='utf-8'))
|
||
nm = (d.get('名称') or '').strip()
|
||
if nm:
|
||
diet_idx.append({'名称': nm, 'file_id': fn.split('_')[0]})
|
||
except Exception:
|
||
pass
|
||
|
||
# ============ C: × 大医网方剂 ============
|
||
formula_dir = os.path.join(BASE, '大医网', '01_来源数据', '方剂')
|
||
formula_idx = {}
|
||
for fn in sorted(os.listdir(formula_dir)):
|
||
if fn.endswith('.json'):
|
||
try:
|
||
d = json.load(open(os.path.join(formula_dir, fn), encoding='utf-8'))
|
||
nm = (d.get('名称') or d.get('方名') or '').strip()
|
||
if nm:
|
||
formula_idx[nm] = fn.split('_')[0]
|
||
except Exception:
|
||
pass
|
||
|
||
link_dayi = []
|
||
herb_hit_count = collections.Counter()
|
||
for r in sjsy:
|
||
rec = {'本库id': r['id'], '本库名称': r['名称'], '季节': r['季节'], '性味': r['性味']}
|
||
key = norm(r['名称'])
|
||
# C: 方剂名称匹配
|
||
f_hits = []
|
||
for nm, fid in formula_idx.items():
|
||
nk = norm(nm)
|
||
if nk == key or (len(nk) >= 3 and (nk in key or key in nk)):
|
||
f_hits.append({'方剂': nm, 'file_id': fid})
|
||
rec['关联方剂'] = f_hits
|
||
# D: 药膳食疗名称匹配(本库名称↔药膳食疗名称, 单向包含)
|
||
d_hits = []
|
||
for it in diet_idx:
|
||
nk = norm(it['名称'])
|
||
if nk == key or (len(nk) >= 3 and (nk in key or key in nk)):
|
||
d_hits.append({'药膳食疗': it['名称'], 'file_id': it['file_id']})
|
||
rec['关联药膳食疗'] = d_hits
|
||
# B: 药材匹配(配方文本)
|
||
herbs = extract_herbs(r['配方'])
|
||
rec['关联药材'] = sorted(herbs.keys())
|
||
rec['药材原文写法'] = {k: sorted(v) for k, v in sorted(herbs.items())}
|
||
for h in herbs:
|
||
herb_hit_count[h] += 1
|
||
link_dayi.append(rec)
|
||
|
||
os.makedirs(FUSION, exist_ok=True)
|
||
with open(os.path.join(FUSION, '四季调养-药膳全书_交叉关联.json'), 'w', encoding='utf-8') as f:
|
||
json.dump({'metadata': {'本库条数': len(sjsy), '对方库': '中华药膳全书(调理药膳336+配方库301)',
|
||
'关联条数': len(link_book), '生成日期': '2026-09-08'},
|
||
'cross_references': link_book}, f, ensure_ascii=False, indent=1)
|
||
|
||
with open(os.path.join(FUSION, '四季调养-大医网_交叉关联.json'), 'w', encoding='utf-8') as f:
|
||
json.dump({'metadata': {'本库条数': len(sjsy),
|
||
'对方库': '大医网(方剂1000+药膳食疗1000+中药材1000)',
|
||
'药材命中TOP30': herb_hit_count.most_common(30),
|
||
'有药材关联条数': sum(1 for x in link_dayi if x['关联药材']),
|
||
'有方剂关联条数': sum(1 for x in link_dayi if x['关联方剂']),
|
||
'有药膳食疗关联条数': sum(1 for x in link_dayi if x['关联药膳食疗']),
|
||
'生成日期': '2026-09-08'},
|
||
'cross_references': link_dayi}, f, ensure_ascii=False, indent=1)
|
||
|
||
print(f"药膳全书关联: {len(link_book)}/{len(sjsy)} 条")
|
||
print(f"大医网: 有药材关联 {sum(1 for x in link_dayi if x['关联药材'])} 条, 有方剂关联 {sum(1 for x in link_dayi if x['关联方剂'])} 条, 有药膳食疗关联 {sum(1 for x in link_dayi if x['关联药膳食疗'])} 条")
|
||
print("药材命中TOP15:", herb_hit_count.most_common(15))
|
||
print("\n抽样(苡仁炖猪蹄):")
|
||
for x in link_dayi:
|
||
if x['本库名称'] == '苡仁炖猪蹄':
|
||
print(json.dumps(x, ensure_ascii=False, indent=1))
|
||
print("\n抽样(当归生姜羊肉汤):")
|
||
for x in link_dayi:
|
||
if x['本库名称'] == '当归生姜羊肉汤':
|
||
print(json.dumps({k: x[k] for k in ('本库id','关联方剂','关联药膳食疗','关联药材')}, ensure_ascii=False, indent=1))
|