#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 四季调养药膳 — 跨库交叉关联脚本 方向: A. 四季调养药膳 × 中华药膳全书(调理药膳336/配方库301): 名称精确+包含匹配 B. 四季调养药膳 × 大医网中药材1000: 配方中的药材名匹配(最长前缀+别名表) C. 四季调养药膳 × 大医网方剂1000: 名称精确+包含匹配 输出: 03_关联融合/四季调养-药膳全书_交叉关联.json 03_关联融合/四季调养-大医网_交叉关联.json """ import json, os, re, collections BASE = os.path.expanduser('~/Documents/ai_agent_scraper_study/data') SJSY = os.path.join(BASE, '四季调养药膳', '02_加工数据', '四季调养药膳.json') FUSION = os.path.join(BASE, '四季调养药膳', '03_关联融合') sjsy = json.load(open(SJSY, encoding='utf-8'))['recipes'] # ============ B 前置: 大医网药材名与别名表 ============ herb_dir = os.path.join(BASE, '大医网', '01_来源数据', '中药材') herb_names = [] # (名称, file_id) herb_alias = collections.defaultdict(set) # 别名 -> {正名} for fn in sorted(os.listdir(herb_dir)): if not fn.endswith('.json'): continue try: d = json.load(open(os.path.join(herb_dir, fn), encoding='utf-8')) except Exception: continue name = d.get('名称', '').strip() if not name: continue fid = fn.split('_')[0] herb_names.append((name, fid)) for al in re.split(r'[,,、;;\s]+', d.get('别名', '') or ''): al = al.strip() if 1 < len(al) <= 6: herb_alias[al].add(name) # 别名补充(药膳常用): EXTRA_ALIAS = {'杞子': '枸杞子', '红枣': '大枣', '圆肉': '龙眼肉', '苡仁': '薏苡仁', '薏仁': '薏苡仁', '苡米': '薏苡仁', '薏苡米': '薏苡仁', '麦门冬': '麦冬', '天门冬': '天冬', '云苓': '茯苓', '白苓': '茯苓', '银花': '金银花', '双花': '金银花', '虫草': '冬虫夏草', '广木香': '木香', '首乌': '何首乌', '制首乌': '何首乌', '白果': '银杏叶', '燕菜': '燕窝'} for a, n in EXTRA_ALIAS.items(): herb_alias[a].add(n) # 非药材词(调味/厨料, 不做药材匹配; 大医网无对应条目或属食材常识) NON_HERB = {'料酒', '绍酒', '黄酒', '米酒', '白酒', '素油', '菜油', '麻油', '香油', '味精', '食盐', '精盐', '白糖', '冰糖', '红糖', '蜂蜜', '酱油', '醋', '淀粉', '水淀粉', '玉米粉', '面粉', '米粉', '糯米粉', '药包', '网油', '清汤', '高汤', '鸡汤', '上汤', '开水', '温水', '凉水', '淘米水'} # 食材撞名别名黑名单: 大医网民间别名与食材同名, 禁止用作匹配 # (牛耳枫子别名"猪肚"、黄芪别名"羊肉"、车前草别名"猪肚子"、鱼胶别名"鱼肚/鱼白"、鲃鱼别名"青鱼") FOOD_COLLISION_ALIAS = {'猪肚', '猪肚子', '羊肉', '鱼肚', '鱼白', '青鱼', '蜂乳'} # 处理前缀(匹配时剥去): 炙/制/炒/煨/煅/焦/霜/法/飞/熟/生 PROC_PREFIX = re.compile(r'^(炙|制|炒|煨|煅|焦|霜|法|飞|熟|生)(?=[\u4e00-\u9fff]{2,})') # 最长前缀排序列表(含别名), 排除非药材词 all_terms = sorted(set([n for n, _ in herb_names] + list(herb_alias.keys())) - NON_HERB - FOOD_COLLISION_ALIAS, key=len, reverse=True) term2names = collections.defaultdict(set) for n, _ in herb_names: term2names[n].add(n) for a, names in herb_alias.items(): for n in names: term2names[a].add(n) # 炮制前缀(允许出现在药材名左侧)与形态后缀(允许出现在右侧) LEFT_OK = set('炙制炒煨煅焦法霜飞生熟粉鲜嫩干白') RIGHT_OK = set('片末粉霜茸丝') def extract_herbs(material_text): """从配方文本抽取药材名(最长前缀+span防重叠+炮制前缀/形态后缀宽容)""" found = collections.defaultdict(set) # 大医网正名 -> 原文写法 occupied = [] # 已命中的字符区间, 防止短词嵌在长词/调味词内部 def taken(a, b): return any(not (b <= s or a >= e) for s, e in occupied) ONE_CHAR_OK = {'葱', '梨', '藕'} # 大医网仅有的3个安全1字药名(边界检查防雪梨/料酒类误配) for term in all_terms: if len(term) < 2 and term not in ONE_CHAR_OK: continue # 1字词仅放行白名单 for m in re.finditer(re.escape(term), material_text): a, b = m.start(), m.end() if taken(a, b): continue # 左界: 前一字符须非汉字, 或为炮制前缀(且前缀+词不是更长的药材名) if a > 0 and '\u4e00' <= material_text[a-1] <= '\u9fff': if material_text[a-1] not in LEFT_OK or (material_text[a-1] + term) in all_terms: continue # 右界: 后一字符须非汉字, 或为形态后缀(且词+后缀不是更长的药材名) if b < len(material_text) and '\u4e00' <= material_text[b] <= '\u9fff': if material_text[b] not in RIGHT_OK or (term + material_text[b]) in all_terms: continue occupied.append((a, b)) for canon in term2names[term]: found[canon].add(m.group(0)) return found # ============ A: × 中华药膳全书 ============ book_files = { '调理药膳': os.path.join(BASE, '中华药膳全书学做药膳不生病', '01_来源数据', '调理药膳.json'), '药膳配方库': os.path.join(BASE, '中华药膳全书学做药膳不生病', '01_来源数据', '药膳配方库_recipes.json'), } book_records = [] for src, fp in book_files.items(): if os.path.exists(fp): for rec in json.load(open(fp, encoding='utf-8')): book_records.append({'来源': src, **{k: rec.get(k, '') for k in ('id', '名称', 'name', '症状', 'category')}}) norm = lambda s: re.sub(r'[\s()()。,,、的]', '', s or '') book_by_norm = collections.defaultdict(list) for rec in book_records: nm = rec.get('名称') or rec.get('name') if nm: book_by_norm[norm(nm)].append(rec) link_book = [] for r in sjsy: hits = [] key = norm(r['名称']) if key in book_by_norm: for rec in book_by_norm[key]: hits.append({'匹配类型': '名称精确', '对方库': rec['来源'], '对方id': rec.get('id') or rec.get('category', ''), '对方名称': rec.get('名称') or rec.get('name')}) else: for rec in book_records: other = rec.get('名称') or rec.get('name') or '' if not other: continue no, nk = norm(other), key if len(no) >= 3 and (no in nk or nk in no): hits.append({'匹配类型': '名称包含', '对方库': rec['来源'], '对方id': rec.get('id') or rec.get('category', ''), '对方名称': other}) if hits: link_book.append({'本库id': r['id'], '本库名称': r['名称'], '季节': r['季节'], '关联数': len(hits), '关联': hits}) # ============ D: × 大医网药膳食疗1000 ============ diet_dir = os.path.join(BASE, '大医网', '01_来源数据', '药膳食疗') diet_idx = [] for fn in sorted(os.listdir(diet_dir)): if fn.endswith('.json'): try: d = json.load(open(os.path.join(diet_dir, fn), encoding='utf-8')) nm = (d.get('名称') or '').strip() if nm: diet_idx.append({'名称': nm, 'file_id': fn.split('_')[0]}) except Exception: pass # ============ C: × 大医网方剂 ============ formula_dir = os.path.join(BASE, '大医网', '01_来源数据', '方剂') formula_idx = {} for fn in sorted(os.listdir(formula_dir)): if fn.endswith('.json'): try: d = json.load(open(os.path.join(formula_dir, fn), encoding='utf-8')) nm = (d.get('名称') or d.get('方名') or '').strip() if nm: formula_idx[nm] = fn.split('_')[0] except Exception: pass link_dayi = [] herb_hit_count = collections.Counter() for r in sjsy: rec = {'本库id': r['id'], '本库名称': r['名称'], '季节': r['季节'], '性味': r['性味']} key = norm(r['名称']) # C: 方剂名称匹配 f_hits = [] for nm, fid in formula_idx.items(): nk = norm(nm) if nk == key or (len(nk) >= 3 and (nk in key or key in nk)): f_hits.append({'方剂': nm, 'file_id': fid}) rec['关联方剂'] = f_hits # D: 药膳食疗名称匹配(本库名称↔药膳食疗名称, 单向包含) d_hits = [] for it in diet_idx: nk = norm(it['名称']) if nk == key or (len(nk) >= 3 and (nk in key or key in nk)): d_hits.append({'药膳食疗': it['名称'], 'file_id': it['file_id']}) rec['关联药膳食疗'] = d_hits # B: 药材匹配(配方文本) herbs = extract_herbs(r['配方']) rec['关联药材'] = sorted(herbs.keys()) rec['药材原文写法'] = {k: sorted(v) for k, v in sorted(herbs.items())} for h in herbs: herb_hit_count[h] += 1 link_dayi.append(rec) os.makedirs(FUSION, exist_ok=True) with open(os.path.join(FUSION, '四季调养-药膳全书_交叉关联.json'), 'w', encoding='utf-8') as f: json.dump({'metadata': {'本库条数': len(sjsy), '对方库': '中华药膳全书(调理药膳336+配方库301)', '关联条数': len(link_book), '生成日期': '2026-09-08'}, 'cross_references': link_book}, f, ensure_ascii=False, indent=1) with open(os.path.join(FUSION, '四季调养-大医网_交叉关联.json'), 'w', encoding='utf-8') as f: json.dump({'metadata': {'本库条数': len(sjsy), '对方库': '大医网(方剂1000+药膳食疗1000+中药材1000)', '药材命中TOP30': herb_hit_count.most_common(30), '有药材关联条数': sum(1 for x in link_dayi if x['关联药材']), '有方剂关联条数': sum(1 for x in link_dayi if x['关联方剂']), '有药膳食疗关联条数': sum(1 for x in link_dayi if x['关联药膳食疗']), '生成日期': '2026-09-08'}, 'cross_references': link_dayi}, f, ensure_ascii=False, indent=1) print(f"药膳全书关联: {len(link_book)}/{len(sjsy)} 条") print(f"大医网: 有药材关联 {sum(1 for x in link_dayi if x['关联药材'])} 条, 有方剂关联 {sum(1 for x in link_dayi if x['关联方剂'])} 条, 有药膳食疗关联 {sum(1 for x in link_dayi if x['关联药膳食疗'])} 条") print("药材命中TOP15:", herb_hit_count.most_common(15)) print("\n抽样(苡仁炖猪蹄):") for x in link_dayi: if x['本库名称'] == '苡仁炖猪蹄': print(json.dumps(x, ensure_ascii=False, indent=1)) print("\n抽样(当归生姜羊肉汤):") for x in link_dayi: if x['本库名称'] == '当归生姜羊肉汤': print(json.dumps({k: x[k] for k in ('本库id','关联方剂','关联药膳食疗','关联药材')}, ensure_ascii=False, indent=1))