Files

225 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
四季调养药膳 — 跨库交叉关联脚本
方向:
A. 四季调养药膳 × 中华药膳全书(调理药膳336/配方库301): 名称精确+包含匹配
B. 四季调养药膳 × 大医网中药材1000: 配方中的药材名匹配(最长前缀+别名表)
C. 四季调养药膳 × 大医网方剂1000: 名称精确+包含匹配
输出: 03_关联融合/四季调养-药膳全书_交叉关联.json
03_关联融合/四季调养-大医网_交叉关联.json
"""
import json, os, re, collections
BASE = os.path.expanduser('~/Documents/ai_agent_scraper_study/data')
SJSY = os.path.join(BASE, '四季调养药膳', '02_加工数据', '四季调养药膳.json')
FUSION = os.path.join(BASE, '四季调养药膳', '03_关联融合')
sjsy = json.load(open(SJSY, encoding='utf-8'))['recipes']
# ============ B 前置: 大医网药材名与别名表 ============
herb_dir = os.path.join(BASE, '大医网', '01_来源数据', '中药材')
herb_names = [] # (名称, file_id)
herb_alias = collections.defaultdict(set) # 别名 -> {正名}
for fn in sorted(os.listdir(herb_dir)):
if not fn.endswith('.json'):
continue
try:
d = json.load(open(os.path.join(herb_dir, fn), encoding='utf-8'))
except Exception:
continue
name = d.get('名称', '').strip()
if not name:
continue
fid = fn.split('_')[0]
herb_names.append((name, fid))
for al in re.split(r'[,,、;;\s]+', d.get('别名', '') or ''):
al = al.strip()
if 1 < len(al) <= 6:
herb_alias[al].add(name)
# 别名补充(药膳常用):
EXTRA_ALIAS = {'杞子': '枸杞子', '红枣': '大枣', '圆肉': '龙眼肉',
'苡仁': '薏苡仁', '薏仁': '薏苡仁',
'苡米': '薏苡仁', '薏苡米': '薏苡仁', '麦门冬': '麦冬', '天门冬': '天冬',
'云苓': '茯苓', '白苓': '茯苓', '银花': '金银花', '双花': '金银花',
'虫草': '冬虫夏草', '广木香': '木香', '首乌': '何首乌', '制首乌': '何首乌',
'白果': '银杏叶', '燕菜': '燕窝'}
for a, n in EXTRA_ALIAS.items():
herb_alias[a].add(n)
# 非药材词(调味/厨料, 不做药材匹配; 大医网无对应条目或属食材常识)
NON_HERB = {'料酒', '绍酒', '黄酒', '米酒', '白酒', '素油', '菜油', '麻油', '香油',
'味精', '食盐', '精盐', '白糖', '冰糖', '红糖', '蜂蜜', '酱油', '醋',
'淀粉', '水淀粉', '玉米粉', '面粉', '米粉', '糯米粉', '药包', '网油',
'清汤', '高汤', '鸡汤', '上汤', '开水', '温水', '凉水', '淘米水'}
# 食材撞名别名黑名单: 大医网民间别名与食材同名, 禁止用作匹配
# (牛耳枫子别名"猪肚"、黄芪别名"羊肉"、车前草别名"猪肚子"、鱼胶别名"鱼肚/鱼白"、鲃鱼别名"青鱼")
FOOD_COLLISION_ALIAS = {'猪肚', '猪肚子', '羊肉', '鱼肚', '鱼白', '青鱼', '蜂乳'}
# 处理前缀(匹配时剥去): 炙/制/炒/煨/煅/焦/霜/法/飞/熟/生
PROC_PREFIX = re.compile(r'^(炙|制|炒|煨|煅|焦|霜|法|飞|熟|生)(?=[\u4e00-\u9fff]{2,})')
# 最长前缀排序列表(含别名), 排除非药材词
all_terms = sorted(set([n for n, _ in herb_names] + list(herb_alias.keys()))
- NON_HERB - FOOD_COLLISION_ALIAS,
key=len, reverse=True)
term2names = collections.defaultdict(set)
for n, _ in herb_names:
term2names[n].add(n)
for a, names in herb_alias.items():
for n in names:
term2names[a].add(n)
# 炮制前缀(允许出现在药材名左侧)与形态后缀(允许出现在右侧)
LEFT_OK = set('炙制炒煨煅焦法霜飞生熟粉鲜嫩干白')
RIGHT_OK = set('片末粉霜茸丝')
def extract_herbs(material_text):
"""从配方文本抽取药材名(最长前缀+span防重叠+炮制前缀/形态后缀宽容)"""
found = collections.defaultdict(set) # 大医网正名 -> 原文写法
occupied = [] # 已命中的字符区间, 防止短词嵌在长词/调味词内部
def taken(a, b):
return any(not (b <= s or a >= e) for s, e in occupied)
ONE_CHAR_OK = {'葱', '梨', '藕'} # 大医网仅有的3个安全1字药名(边界检查防雪梨/料酒类误配)
for term in all_terms:
if len(term) < 2 and term not in ONE_CHAR_OK:
continue # 1字词仅放行白名单
for m in re.finditer(re.escape(term), material_text):
a, b = m.start(), m.end()
if taken(a, b):
continue
# 左界: 前一字符须非汉字, 或为炮制前缀(且前缀+词不是更长的药材名)
if a > 0 and '\u4e00' <= material_text[a-1] <= '\u9fff':
if material_text[a-1] not in LEFT_OK or (material_text[a-1] + term) in all_terms:
continue
# 右界: 后一字符须非汉字, 或为形态后缀(且词+后缀不是更长的药材名)
if b < len(material_text) and '\u4e00' <= material_text[b] <= '\u9fff':
if material_text[b] not in RIGHT_OK or (term + material_text[b]) in all_terms:
continue
occupied.append((a, b))
for canon in term2names[term]:
found[canon].add(m.group(0))
return found
# ============ A: × 中华药膳全书 ============
book_files = {
'调理药膳': os.path.join(BASE, '中华药膳全书学做药膳不生病', '01_来源数据', '调理药膳.json'),
'药膳配方库': os.path.join(BASE, '中华药膳全书学做药膳不生病', '01_来源数据', '药膳配方库_recipes.json'),
}
book_records = []
for src, fp in book_files.items():
if os.path.exists(fp):
for rec in json.load(open(fp, encoding='utf-8')):
book_records.append({'来源': src, **{k: rec.get(k, '') for k in ('id', '名称', 'name', '症状', 'category')}})
norm = lambda s: re.sub(r'[\s()()。,,、的]', '', s or '')
book_by_norm = collections.defaultdict(list)
for rec in book_records:
nm = rec.get('名称') or rec.get('name')
if nm:
book_by_norm[norm(nm)].append(rec)
link_book = []
for r in sjsy:
hits = []
key = norm(r['名称'])
if key in book_by_norm:
for rec in book_by_norm[key]:
hits.append({'匹配类型': '名称精确', '对方库': rec['来源'],
'对方id': rec.get('id') or rec.get('category', ''), '对方名称': rec.get('名称') or rec.get('name')})
else:
for rec in book_records:
other = rec.get('名称') or rec.get('name') or ''
if not other:
continue
no, nk = norm(other), key
if len(no) >= 3 and (no in nk or nk in no):
hits.append({'匹配类型': '名称包含', '对方库': rec['来源'],
'对方id': rec.get('id') or rec.get('category', ''), '对方名称': other})
if hits:
link_book.append({'本库id': r['id'], '本库名称': r['名称'], '季节': r['季节'],
'关联数': len(hits), '关联': hits})
# ============ D: × 大医网药膳食疗1000 ============
diet_dir = os.path.join(BASE, '大医网', '01_来源数据', '药膳食疗')
diet_idx = []
for fn in sorted(os.listdir(diet_dir)):
if fn.endswith('.json'):
try:
d = json.load(open(os.path.join(diet_dir, fn), encoding='utf-8'))
nm = (d.get('名称') or '').strip()
if nm:
diet_idx.append({'名称': nm, 'file_id': fn.split('_')[0]})
except Exception:
pass
# ============ C: × 大医网方剂 ============
formula_dir = os.path.join(BASE, '大医网', '01_来源数据', '方剂')
formula_idx = {}
for fn in sorted(os.listdir(formula_dir)):
if fn.endswith('.json'):
try:
d = json.load(open(os.path.join(formula_dir, fn), encoding='utf-8'))
nm = (d.get('名称') or d.get('方名') or '').strip()
if nm:
formula_idx[nm] = fn.split('_')[0]
except Exception:
pass
link_dayi = []
herb_hit_count = collections.Counter()
for r in sjsy:
rec = {'本库id': r['id'], '本库名称': r['名称'], '季节': r['季节'], '性味': r['性味']}
key = norm(r['名称'])
# C: 方剂名称匹配
f_hits = []
for nm, fid in formula_idx.items():
nk = norm(nm)
if nk == key or (len(nk) >= 3 and (nk in key or key in nk)):
f_hits.append({'方剂': nm, 'file_id': fid})
rec['关联方剂'] = f_hits
# D: 药膳食疗名称匹配(本库名称↔药膳食疗名称, 单向包含)
d_hits = []
for it in diet_idx:
nk = norm(it['名称'])
if nk == key or (len(nk) >= 3 and (nk in key or key in nk)):
d_hits.append({'药膳食疗': it['名称'], 'file_id': it['file_id']})
rec['关联药膳食疗'] = d_hits
# B: 药材匹配(配方文本)
herbs = extract_herbs(r['配方'])
rec['关联药材'] = sorted(herbs.keys())
rec['药材原文写法'] = {k: sorted(v) for k, v in sorted(herbs.items())}
for h in herbs:
herb_hit_count[h] += 1
link_dayi.append(rec)
os.makedirs(FUSION, exist_ok=True)
with open(os.path.join(FUSION, '四季调养-药膳全书_交叉关联.json'), 'w', encoding='utf-8') as f:
json.dump({'metadata': {'本库条数': len(sjsy), '对方库': '中华药膳全书(调理药膳336+配方库301)',
'关联条数': len(link_book), '生成日期': '2026-09-08'},
'cross_references': link_book}, f, ensure_ascii=False, indent=1)
with open(os.path.join(FUSION, '四季调养-大医网_交叉关联.json'), 'w', encoding='utf-8') as f:
json.dump({'metadata': {'本库条数': len(sjsy),
'对方库': '大医网(方剂1000+药膳食疗1000+中药材1000)',
'药材命中TOP30': herb_hit_count.most_common(30),
'有药材关联条数': sum(1 for x in link_dayi if x['关联药材']),
'有方剂关联条数': sum(1 for x in link_dayi if x['关联方剂']),
'有药膳食疗关联条数': sum(1 for x in link_dayi if x['关联药膳食疗']),
'生成日期': '2026-09-08'},
'cross_references': link_dayi}, f, ensure_ascii=False, indent=1)
print(f"药膳全书关联: {len(link_book)}/{len(sjsy)} 条")
print(f"大医网: 有药材关联 {sum(1 for x in link_dayi if x['关联药材'])} 条, 有方剂关联 {sum(1 for x in link_dayi if x['关联方剂'])} 条, 有药膳食疗关联 {sum(1 for x in link_dayi if x['关联药膳食疗'])} 条")
print("药材命中TOP15:", herb_hit_count.most_common(15))
print("\n抽样(苡仁炖猪蹄):")
for x in link_dayi:
if x['本库名称'] == '苡仁炖猪蹄':
print(json.dumps(x, ensure_ascii=False, indent=1))
print("\n抽样(当归生姜羊肉汤):")
for x in link_dayi:
if x['本库名称'] == '当归生姜羊肉汤':
print(json.dumps({k: x[k] for k in ('本库id','关联方剂','关联药膳食疗','关联药材')}, ensure_ascii=False, indent=1))