#!/usr/bin/env python3 """ 体质-药膳推荐体系构建脚本 多因子评分模型 + 功效分类统计 + 来源分析 + 共享分析 """ import json import os import glob import re from collections import defaultdict, Counter # ===== 1. 加载数据 ===== BASE_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网" # 加载9种体质详细数据 with open(os.path.join(BASE_DIR, "02_加工数据/中医体质/九种体质详细数据.json"), "r", encoding="utf-8") as f: constitutions_data = json.load(f) # 构建体质字典 constitutions = {} for c in constitutions_data: name = c["名称"] constitutions[name] = c print(f"加载体质数据: {len(constitutions)} 种") for name in constitutions: c = constitutions[name] print(f" {name}: 推荐药膳={c.get('推荐药膳', [])}") # 加载所有药膳文件 diet_dir = os.path.join(BASE_DIR, "01_来源数据/药膳食疗") diet_files = glob.glob(os.path.join(diet_dir, "*.json")) print(f"\n药膳食疗文件数: {len(diet_files)}") dietary_list = [] for fpath in sorted(diet_files): try: with open(fpath, "r", encoding="utf-8") as f: item = json.load(f) dietary_list.append(item) except Exception as e: print(f" 加载失败: {os.path.basename(fpath)}: {e}") print(f"成功加载药膳食疗: {len(dietary_list)} 条") # 加载已有交叉关联 cross_assoc_path = os.path.join(BASE_DIR, "02_加工数据/中医体质/体质-大医网交叉关联.json") if os.path.exists(cross_assoc_path): with open(cross_assoc_path, "r", encoding="utf-8") as f: cross_data = json.load(f) print(f"加载已有交叉关联, keys: {list(cross_data.keys())}") else: cross_data = {} print("未找到已有交叉关联文件") # ===== 2. 定义体质相关配置 ===== # 经典文献列表(来源字段含以下文献名视为经典文献) CLASSIC_LITERATURE = [ "伤寒论", "金匮要略", "黄帝内经", "本草纲目", "食疗本草", "饮膳正要", "食医心鉴", "太平圣惠方", "圣济总录", "中国药膳大辞典", "中国药膳学", "中医药膳学", "药膳宝典", "药膳学", "千金要方", "千金翼方", "外台秘要", "肘后备急方", "普济方", "中医食疗学", "季节养生药膳指南" ] # 九种体质的配置(关联关键字、推荐食材、推荐药膳、调理关键词) CONSTITUTION_CONFIG = { "气虚质": { "keywords": ["气虚", "气短", "乏力", "自汗", "懒言", "神疲", "补气", "益气", "培元", "固本"], "ingredients": ["山药", "大枣", "红枣", "黄芪", "党参", "人参", "白术", "茯苓", "糯米", "粳米", "小米", "土豆", "香菇", "鸡肉", "牛肉", "鳝鱼", "泥鳅", "蜂蜜", "黄豆", "扁豆", "豇豆", "南瓜", "胡萝卜", "桂圆", "龙眼"], "efficacy_keywords": ["补气", "益气", "健脾", "培元", "固本", "补中益气"], "type_name": "气虚" }, "阳虚质": { "keywords": ["阳虚", "畏寒", "怕冷", "四肢不温", "寒凝", "温阳", "散寒", "壮阳", "肾阳", "脾阳"], "ingredients": ["羊肉", "牛肉", "韭菜", "生姜", "肉桂", "核桃", "栗子", "荔枝", "龙眼", "茴香", "丁香", "花椒", "小茴香", "干姜", "当归", "附子", "狗肉"], "efficacy_keywords": ["温阳", "散寒", "壮阳", "温中", "暖胃", "温补脾肾", "补肾壮阳"], "type_name": "阳虚" }, "阴虚质": { "keywords": ["阴虚", "口干", "咽干", "手足心热", "潮热", "盗汗", "滋阴", "降火", "生津", "润燥"], "ingredients": ["百合", "银耳", "莲子", "枸杞", "黑芝麻", "鸭肉", "甲鱼", "龟肉", "海参", "牡蛎", "蜂蜜", "梨", "甘蔗", "枇杷", "桑葚", "黑木耳", "沙参", "玉竹", "麦冬", "生地", "生地黄"], "efficacy_keywords": ["滋阴", "降火", "生津", "润燥", "养阴", "清热", "滋补肝肾"], "type_name": "阴虚" }, "痰湿质": { "keywords": ["痰湿", "肥胖", "痰多", "胸闷", "苔腻", "健脾", "祛湿", "化痰", "降浊", "利水"], "ingredients": ["薏苡仁", "冬瓜", "白萝卜", "赤小豆", "茯苓", "荷叶", "山楂", "陈皮", "燕麦", "荞麦", "海带", "紫菜", "生姜", "莱菔子"], "efficacy_keywords": ["祛湿", "化痰", "健脾", "利水", "降浊", "消食", "导滞", "除湿"], "type_name": "痰湿" }, "湿热质": { "keywords": ["湿热", "口苦", "苔黄腻", "痤疮", "湿疹", "清热", "利湿", "解毒", "化浊", "泻火"], "ingredients": ["绿豆", "赤小豆", "薏苡仁", "冬瓜", "苦瓜", "黄瓜", "芹菜", "蒲公英", "马齿苋", "菊花", "金银花", "莲藕", "茭白", "栀子"], "efficacy_keywords": ["清热", "利湿", "解毒", "化浊", "泻火", "凉血", "祛湿"], "type_name": "湿热" }, "血瘀质": { "keywords": ["血瘀", "瘀血", "面色晦暗", "瘀斑", "刺痛", "活血", "化瘀", "行气", "通络", "消癥"], "ingredients": ["山楂", "黑豆", "黑木耳", "醋", "玫瑰花", "红糖", "桃仁", "油菜", "茄子", "藕", "香菇", "海带", "川芎", "红花", "当归"], "efficacy_keywords": ["活血", "化瘀", "行气", "通络", "散瘀", "消癥", "止痛"], "type_name": "血瘀" }, "气郁质": { "keywords": ["气郁", "抑郁", "焦虑", "胸闷", "胁痛", "疏肝", "解郁", "理气", "调中", "安神"], "ingredients": ["柑橘", "佛手", "玫瑰花", "小麦", "大麦", "荞麦", "香橼", "橙子", "柚", "洋葱", "大蒜", "萝卜", "茴香", "柴胡", "薄荷", "合欢花", "陈皮", "柠檬"], "efficacy_keywords": ["疏肝", "解郁", "理气", "安神", "调中", "行气", "开郁"], "type_name": "气郁" }, "特禀质": { "keywords": ["过敏", "哮喘", "荨麻疹", "鼻炎", "风团", "益气", "固表", "祛风", "抗敏", "特禀"], "ingredients": ["黄芪", "白术", "防风", "红枣", "蜂蜜", "山药", "人参", "灵芝", "薏苡仁", "莲子", "糯米", "花生", "党参", "紫苏", "生姜"], "efficacy_keywords": ["益气", "固表", "祛风", "抗敏", "扶正", "补肺", "健脾"], "type_name": "特禀" }, "平和质": { "keywords": ["平和", "平衡", "调和", "保健", "养生", "预防"], "ingredients": ["五谷杂粮", "蔬菜", "水果", "鱼肉", "蛋奶", "豆制品", "坚果", "绿茶", "山药", "红枣", "枸杞", "茯苓"], "efficacy_keywords": ["平和", "调和", "保健", "养生", "平衡", "滋补"], "type_name": "平和" } } print("\n体质配置加载完成") # ===== 3. 多因子评分模型 ===== def is_classic_literature(source): """判断是否来自经典文献""" if not source: return False for lit in CLASSIC_LITERATURE: if lit in source: return True return False def text_contains_keywords(text, keywords): """检查文本是否包含任意关键字""" if not text: return False text_lower = text.lower() for kw in keywords: if kw in text: return True return False def count_keyword_matches(text, keywords): """统计文本中匹配的关键字数量""" if not text: return 0 count = 0 for kw in keywords: if kw in text: count += 1 return count def score_dietary_for_constitution(diet_item, constitution_name): """ 多因子评分模型 返回:(总分, 各因子详细得分) """ cfg = CONSTITUTION_CONFIG[constitution_name] keywords = cfg["keywords"] ingredients = cfg["ingredients"] # 获取各字段 efficacy = diet_item.get("功效", "") or "" intro = diet_item.get("简介", "") or "" name = diet_item.get("名称", "") or "" recipe = diet_item.get("配方", "") or "" # HTML可能包含食材 suitable_pop = diet_item.get("适宜人群", "") or "" source = diet_item.get("来源", "") or "" related = diet_item.get("相关配伍", "") or "" # 移除HTML标签 recipe_clean = re.sub(r'<[^>]+>', '', recipe) suitable_clean = re.sub(r'<[^>]+>', '', suitable_pop) related_clean = re.sub(r'<[^>]+>', '', related) intro_clean = re.sub(r'<[^>]+>', '', intro) scores = {} # A) 功效字段含体质关键字 +4(高权重) score_efficacy = 0 for kw in keywords: if kw in efficacy: score_efficacy += 4 # 同时检查功效关键词 for kw in cfg["efficacy_keywords"]: if kw in efficacy: score_efficacy += 4 scores["功效匹配"] = score_efficacy # B) 简介/名称字段含体质关键字 +3 score_intro_name = 0 combined_intro_name = intro_clean + " " + name for kw in keywords: if kw in combined_intro_name: score_intro_name += 3 for kw in cfg["efficacy_keywords"]: if kw in combined_intro_name: score_intro_name += 3 scores["简介/名称匹配"] = score_intro_name # C) 配方字段含体质推荐食材 +2 score_recipe = 0 for ing in ingredients: if ing in recipe_clean: score_recipe += 2 # 也检查相关配伍字段 for ing in ingredients: if ing in related_clean: score_recipe += 1 # 配伍中匹配权重略低 scores["配方食材匹配"] = score_recipe # D) 适宜人群字段含体质关键字 +3 score_suitable = 0 for kw in keywords: if kw in suitable_clean: score_suitable += 3 for kw in cfg["efficacy_keywords"]: if kw in suitable_clean: score_suitable += 3 scores["适宜人群匹配"] = score_suitable # E) 来源字段含经典文献 +1 score_source = 1 if is_classic_literature(source) else 0 scores["经典文献加分"] = score_source total = sum(scores.values()) return total, scores # ===== 4. 对每种体质评分 ===== print("\n开始多因子评分...") constitution_order = ["气虚质", "阳虚质", "阴虚质", "痰湿质", "湿热质", "血瘀质", "气郁质", "特禀质", "平和质"] # 存储结果 dietary_recommendations = {} dietary_efficacy_stats = {} all_shared_dietary = {} # name -> list of constitution types THRESHOLD = 4.0 for cname in constitution_order: scored_items = [] for item in dietary_list: total, scores = score_dietary_for_constitution(item, cname) if total >= THRESHOLD: scored_items.append({ "名称": item.get("名称", ""), "来源": item.get("来源", ""), "功效": item.get("功效", ""), "简介": item.get("简介", ""), "得分": total, "各因子得分": scores, "配方": item.get("配方", ""), "适宜人群": item.get("适宜人群", ""), "相关配伍": item.get("相关配伍", "") }) # 按得分降序排列,取TOP30 scored_items.sort(key=lambda x: (-x["得分"], x["名称"])) top_items = scored_items[:30] print(f"{cname}: 得分≥{THRESHOLD}的药膳数={len(scored_items)}, TOP3={[t['名称'] for t in top_items[:3]]}") # 记录共享信息 for item in top_items: name = item["名称"] if name not in all_shared_dietary: all_shared_dietary[name] = {"体质": [], "功效": item["功效"], "来源": item["来源"]} all_shared_dietary[name]["体质"].append(cname) # 构建推荐列表(含共享标记) recommendations = [] for item in top_items: name = item["名称"] total = item["得分"] source = item["来源"] efficacy = item["功效"] recommendations.append({ "名称": name, "来源": source, "功效": efficacy, "得分": total, "是否共享": False, # 后面更新 "共享体质": [] }) dietary_recommendations[cname] = recommendations # ===== 5. 更新共享信息 ===== # 找出跨体质药膳 shared_count = 0 cross_constitution_items = [] for name, info in all_shared_dietary.items(): if len(info["体质"]) >= 2: shared_count += 1 cross_constitution_items.append({ "名称": name, "适用体质": info["体质"], "功效": info["功效"], "来源": info["来源"] }) # 更新推荐列表中的共享标记 for cname in constitution_order: for rec in dietary_recommendations[cname]: name = rec["名称"] if name in all_shared_dietary and len(all_shared_dietary[name]["体质"]) >= 2: rec["是否共享"] = True rec["共享体质"] = [t for t in all_shared_dietary[name]["体质"] if t != cname] print(f"\n跨体质药膳数: {shared_count}") for item in cross_constitution_items[:5]: print(f" {item['名称']}: {item['适用体质']}") # ===== 6. 功效分类统计 ===== print("\n开始功效分类统计...") EFFICACY_KEYWORDS_MAP = { "补气": ["补气", "益气", "补中益气"], "健脾": ["健脾", "补脾", "益脾", "温脾"], "养血": ["养血", "补血", "活血养血"], "滋阴": ["滋阴", "养阴", "育阴"], "温阳": ["温阳", "壮阳", "补阳"], "散寒": ["散寒", "温中", "暖胃", "祛寒"], "清热": ["清热", "解毒", "泻火", "凉血"], "祛湿": ["祛湿", "利湿", "除湿", "化湿", "渗湿"], "化痰": ["化痰", "祛痰", "消痰"], "疏肝": ["疏肝", "解郁", "理气", "行气"], "安神": ["安神", "宁心", "养心"], "补肾": ["补肾", "益肾", "滋肾", "温肾"], "润肺": ["润肺", "补肺", "养肺"], "消食": ["消食", "导滞", "开胃", "健胃"], "生津": ["生津", "止渴", "润燥"], "活血": ["活血", "化瘀", "散瘀", "通络"], "固表": ["固表", "固涩", "止汗", "收敛"], "祛风": ["祛风", "疏风", "散风"], "利水": ["利水", "消肿", "利尿"], "补虚": ["补虚", "补益", "滋补", "扶正"] } for cname in constitution_order: top_items = dietary_recommendations[cname][:30] efficacy_stats = defaultdict(int) for item in top_items: efficacy_text = item["功效"] if not efficacy_text: continue # 统计功效关键词 for category, keywords in EFFICACY_KEYWORDS_MAP.items(): for kw in keywords: if kw in efficacy_text: efficacy_stats[category] += 1 break # 每个药膳每个分类只计1次 dietary_efficacy_stats[cname] = dict(sorted(efficacy_stats.items(), key=lambda x: -x[1])) top_cats = list(dietary_efficacy_stats[cname].keys())[:5] print(f"{cname}: {dict(top_cats)}...") # ===== 7. 来源分析 ===== print("\n开始来源分析...") source_stats = defaultdict(int) for cname in constitution_order: for rec in dietary_recommendations[cname]: source = rec["来源"] if source: # 简化来源名 simple_source = source.strip() source_stats[simple_source] += 1 top_sources = sorted(source_stats.items(), key=lambda x: -x[1])[:20] print("TOP来源:") for s, c in top_sources: print(f" {s}: {c}次") # ===== 8. 构建完整JSON输出 ===== print("\n构建JSON输出...") # 统计摘要 total_recommendations = sum(len(v) for v in dietary_recommendations.values()) total_unique = len(all_shared_dietary) summary_stats = { "总药膳数": len(dietary_list), "体质类型数": len(constitution_order), "各体质推荐药膳数": {c: len(dietary_recommendations[c]) for c in constitution_order}, "总推荐人次": total_recommendations, "去重推荐药膳数": total_unique, "跨体质共享药膳数": shared_count, "评分阈值": THRESHOLD, "评分模型因子": { "功效字段含体质关键字": "+4", "简介/名称字段含体质关键字": "+3", "配方字段含体质推荐食材": "+2", "适宜人群字段含体质关键字": "+3", "来源字段含经典文献": "+1" } } output_json = { "体质-药膳推荐": dietary_recommendations, "体质-药膳功效统计": dietary_efficacy_stats, "体质-药膳来源统计": dict(top_sources), "体质-药膳共享分析": { "数量": shared_count, "跨体质药膳": cross_constitution_items }, "统计摘要": summary_stats } # 保存JSON output_dir = os.path.join(BASE_DIR, "03_关联融合/交叉关联分析") os.makedirs(output_dir, exist_ok=True) output_path = os.path.join(output_dir, "体质-药膳推荐体系.json") with open(output_path, "w", encoding="utf-8") as f: json.dump(output_json, f, ensure_ascii=False, indent=2) print(f"\nJSON已保存: {output_path}") print(f"文件大小: {os.path.getsize(output_path) / 1024:.1f} KB") # ===== 9. 验证JSON完整性 ===== with open(output_path, "r", encoding="utf-8") as f: verify = json.load(f) print(f"\n验证JSON:") print(f" 体质-药膳推荐: {len(verify['体质-药膳推荐'])} 种体质") for c in constitution_order: print(f" {c}: {len(verify['体质-药膳推荐'][c])} 条推荐") print(f" 体质-药膳功效统计: {len(verify['体质-药膳功效统计'])} 种") print(f" 体质-药膳共享分析: {verify['体质-药膳共享分析']['数量']} 条共享") print(f" 统计摘要: 总推荐={verify['统计摘要']['总推荐人次']}人次") print("\n===== 药膳推荐体系构建完成 =====")