初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)

This commit is contained in:
512song committed 2026-09-23 21:59:25 +08:00
commit 80ae3811cf
7712 files changed
+4628547

No files matched your search

@@ -0,0 +1,451 @@
#!/usr/bin/env python3
"""
体质-药膳推荐体系构建脚本
多因子评分模型 + 功效分类统计 + 来源分析 + 共享分析
"""
import json
import os
import glob
import re
from collections import defaultdict, Counter
# ===== 1. 加载数据 =====
BASE_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
# 加载9种体质详细数据
with open(os.path.join(BASE_DIR, "02_加工数据/中医体质/九种体质详细数据.json"), "r", encoding="utf-8") as f:
constitutions_data = json.load(f)
# 构建体质字典
constitutions = {}
for c in constitutions_data:
name = c["名称"]
constitutions[name] = c
print(f"加载体质数据: {len(constitutions)} 种")
for name in constitutions:
c = constitutions[name]
print(f" {name}: 推荐药膳={c.get('推荐药膳', [])}")
# 加载所有药膳文件
diet_dir = os.path.join(BASE_DIR, "01_来源数据/药膳食疗")
diet_files = glob.glob(os.path.join(diet_dir, "*.json"))
print(f"\n药膳食疗文件数: {len(diet_files)}")
dietary_list = []
for fpath in sorted(diet_files):
try:
with open(fpath, "r", encoding="utf-8") as f:
item = json.load(f)
dietary_list.append(item)
except Exception as e:
print(f" 加载失败: {os.path.basename(fpath)}: {e}")
print(f"成功加载药膳食疗: {len(dietary_list)} 条")
# 加载已有交叉关联
cross_assoc_path = os.path.join(BASE_DIR, "02_加工数据/中医体质/体质-大医网交叉关联.json")
if os.path.exists(cross_assoc_path):
with open(cross_assoc_path, "r", encoding="utf-8") as f:
cross_data = json.load(f)
print(f"加载已有交叉关联, keys: {list(cross_data.keys())}")
else:
cross_data = {}
print("未找到已有交叉关联文件")
# ===== 2. 定义体质相关配置 =====
# 经典文献列表(来源字段含以下文献名视为经典文献)
CLASSIC_LITERATURE = [
"伤寒论", "金匮要略", "黄帝内经", "本草纲目", "食疗本草",
"饮膳正要", "食医心鉴", "太平圣惠方", "圣济总录",
"中国药膳大辞典", "中国药膳学", "中医药膳学", "药膳宝典",
"药膳学", "千金要方", "千金翼方", "外台秘要",
"肘后备急方", "普济方", "中医食疗学",
"季节养生药膳指南"
]
# 九种体质的配置(关联关键字、推荐食材、推荐药膳、调理关键词)
CONSTITUTION_CONFIG = {
"气虚质": {
"keywords": ["气虚", "气短", "乏力", "自汗", "懒言", "神疲", "补气", "益气", "培元", "固本"],
"ingredients": ["山药", "大枣", "红枣", "黄芪", "党参", "人参", "白术", "茯苓", "糯米",
"粳米", "小米", "土豆", "香菇", "鸡肉", "牛肉", "鳝鱼", "泥鳅", "蜂蜜",
"黄豆", "扁豆", "豇豆", "南瓜", "胡萝卜", "桂圆", "龙眼"],
"efficacy_keywords": ["补气", "益气", "健脾", "培元", "固本", "补中益气"],
"type_name": "气虚"
},
"阳虚质": {
"keywords": ["阳虚", "畏寒", "怕冷", "四肢不温", "寒凝", "温阳", "散寒", "壮阳", "肾阳", "脾阳"],
"ingredients": ["羊肉", "牛肉", "韭菜", "生姜", "肉桂", "核桃", "栗子", "荔枝", "龙眼",
"茴香", "丁香", "花椒", "小茴香", "干姜", "当归", "附子", "狗肉"],
"efficacy_keywords": ["温阳", "散寒", "壮阳", "温中", "暖胃", "温补脾肾", "补肾壮阳"],
"type_name": "阳虚"
},
"阴虚质": {
"keywords": ["阴虚", "口干", "咽干", "手足心热", "潮热", "盗汗", "滋阴", "降火", "生津", "润燥"],
"ingredients": ["百合", "银耳", "莲子", "枸杞", "黑芝麻", "鸭肉", "甲鱼", "龟肉", "海参",
"牡蛎", "蜂蜜", "梨", "甘蔗", "枇杷", "桑葚", "黑木耳", "沙参", "玉竹",
"麦冬", "生地", "生地黄"],
"efficacy_keywords": ["滋阴", "降火", "生津", "润燥", "养阴", "清热", "滋补肝肾"],
"type_name": "阴虚"
},
"痰湿质": {
"keywords": ["痰湿", "肥胖", "痰多", "胸闷", "苔腻", "健脾", "祛湿", "化痰", "降浊", "利水"],
"ingredients": ["薏苡仁", "冬瓜", "白萝卜", "赤小豆", "茯苓", "荷叶", "山楂", "陈皮",
"燕麦", "荞麦", "海带", "紫菜", "生姜", "莱菔子"],
"efficacy_keywords": ["祛湿", "化痰", "健脾", "利水", "降浊", "消食", "导滞", "除湿"],
"type_name": "痰湿"
},
"湿热质": {
"keywords": ["湿热", "口苦", "苔黄腻", "痤疮", "湿疹", "清热", "利湿", "解毒", "化浊", "泻火"],
"ingredients": ["绿豆", "赤小豆", "薏苡仁", "冬瓜", "苦瓜", "黄瓜", "芹菜", "蒲公英",
"马齿苋", "菊花", "金银花", "莲藕", "茭白", "栀子"],
"efficacy_keywords": ["清热", "利湿", "解毒", "化浊", "泻火", "凉血", "祛湿"],
"type_name": "湿热"
},
"血瘀质": {
"keywords": ["血瘀", "瘀血", "面色晦暗", "瘀斑", "刺痛", "活血", "化瘀", "行气", "通络", "消癥"],
"ingredients": ["山楂", "黑豆", "黑木耳", "醋", "玫瑰花", "红糖", "桃仁", "油菜", "茄子",
"藕", "香菇", "海带", "川芎", "红花", "当归"],
"efficacy_keywords": ["活血", "化瘀", "行气", "通络", "散瘀", "消癥", "止痛"],
"type_name": "血瘀"
},
"气郁质": {
"keywords": ["气郁", "抑郁", "焦虑", "胸闷", "胁痛", "疏肝", "解郁", "理气", "调中", "安神"],
"ingredients": ["柑橘", "佛手", "玫瑰花", "小麦", "大麦", "荞麦", "香橼", "橙子", "柚",
"洋葱", "大蒜", "萝卜", "茴香", "柴胡", "薄荷", "合欢花", "陈皮", "柠檬"],
"efficacy_keywords": ["疏肝", "解郁", "理气", "安神", "调中", "行气", "开郁"],
"type_name": "气郁"
},
"特禀质": {
"keywords": ["过敏", "哮喘", "荨麻疹", "鼻炎", "风团", "益气", "固表", "祛风", "抗敏", "特禀"],
"ingredients": ["黄芪", "白术", "防风", "红枣", "蜂蜜", "山药", "人参", "灵芝", "薏苡仁",
"莲子", "糯米", "花生", "党参", "紫苏", "生姜"],
"efficacy_keywords": ["益气", "固表", "祛风", "抗敏", "扶正", "补肺", "健脾"],
"type_name": "特禀"
},
"平和质": {
"keywords": ["平和", "平衡", "调和", "保健", "养生", "预防"],
"ingredients": ["五谷杂粮", "蔬菜", "水果", "鱼肉", "蛋奶", "豆制品", "坚果", "绿茶",
"山药", "红枣", "枸杞", "茯苓"],
"efficacy_keywords": ["平和", "调和", "保健", "养生", "平衡", "滋补"],
"type_name": "平和"
}
}
print("\n体质配置加载完成")
# ===== 3. 多因子评分模型 =====
def is_classic_literature(source):
"""判断是否来自经典文献"""
if not source:
return False
for lit in CLASSIC_LITERATURE:
if lit in source:
return True
return False
def text_contains_keywords(text, keywords):
"""检查文本是否包含任意关键字"""
if not text:
return False
text_lower = text.lower()
for kw in keywords:
if kw in text:
return True
return False
def count_keyword_matches(text, keywords):
"""统计文本中匹配的关键字数量"""
if not text:
return 0
count = 0
for kw in keywords:
if kw in text:
count += 1
return count
def score_dietary_for_constitution(diet_item, constitution_name):
"""
多因子评分模型
返回:(总分, 各因子详细得分)
"""
cfg = CONSTITUTION_CONFIG[constitution_name]
keywords = cfg["keywords"]
ingredients = cfg["ingredients"]
# 获取各字段
efficacy = diet_item.get("功效", "") or ""
intro = diet_item.get("简介", "") or ""
name = diet_item.get("名称", "") or ""
recipe = diet_item.get("配方", "") or "" # HTML可能包含食材
suitable_pop = diet_item.get("适宜人群", "") or ""
source = diet_item.get("来源", "") or ""
related = diet_item.get("相关配伍", "") or ""
# 移除HTML标签
recipe_clean = re.sub(r'<[^>]+>', '', recipe)
suitable_clean = re.sub(r'<[^>]+>', '', suitable_pop)
related_clean = re.sub(r'<[^>]+>', '', related)
intro_clean = re.sub(r'<[^>]+>', '', intro)
scores = {}
# A) 功效字段含体质关键字 +4(高权重)
score_efficacy = 0
for kw in keywords:
if kw in efficacy:
score_efficacy += 4
# 同时检查功效关键词
for kw in cfg["efficacy_keywords"]:
if kw in efficacy:
score_efficacy += 4
scores["功效匹配"] = score_efficacy
# B) 简介/名称字段含体质关键字 +3
score_intro_name = 0
combined_intro_name = intro_clean + " " + name
for kw in keywords:
if kw in combined_intro_name:
score_intro_name += 3
for kw in cfg["efficacy_keywords"]:
if kw in combined_intro_name:
score_intro_name += 3
scores["简介/名称匹配"] = score_intro_name
# C) 配方字段含体质推荐食材 +2
score_recipe = 0
for ing in ingredients:
if ing in recipe_clean:
score_recipe += 2
# 也检查相关配伍字段
for ing in ingredients:
if ing in related_clean:
score_recipe += 1 # 配伍中匹配权重略低
scores["配方食材匹配"] = score_recipe
# D) 适宜人群字段含体质关键字 +3
score_suitable = 0
for kw in keywords:
if kw in suitable_clean:
score_suitable += 3
for kw in cfg["efficacy_keywords"]:
if kw in suitable_clean:
score_suitable += 3
scores["适宜人群匹配"] = score_suitable
# E) 来源字段含经典文献 +1
score_source = 1 if is_classic_literature(source) else 0
scores["经典文献加分"] = score_source
total = sum(scores.values())
return total, scores
# ===== 4. 对每种体质评分 =====
print("\n开始多因子评分...")
constitution_order = ["气虚质", "阳虚质", "阴虚质", "痰湿质", "湿热质", "血瘀质", "气郁质", "特禀质", "平和质"]
# 存储结果
dietary_recommendations = {}
dietary_efficacy_stats = {}
all_shared_dietary = {} # name -> list of constitution types
THRESHOLD = 4.0
for cname in constitution_order:
scored_items = []
for item in dietary_list:
total, scores = score_dietary_for_constitution(item, cname)
if total >= THRESHOLD:
scored_items.append({
"名称": item.get("名称", ""),
"来源": item.get("来源", ""),
"功效": item.get("功效", ""),
"简介": item.get("简介", ""),
"得分": total,
"各因子得分": scores,
"配方": item.get("配方", ""),
"适宜人群": item.get("适宜人群", ""),
"相关配伍": item.get("相关配伍", "")
})
# 按得分降序排列,取TOP30
scored_items.sort(key=lambda x: (-x["得分"], x["名称"]))
top_items = scored_items[:30]
print(f"{cname}: 得分≥{THRESHOLD}的药膳数={len(scored_items)}, TOP3={[t['名称'] for t in top_items[:3]]}")
# 记录共享信息
for item in top_items:
name = item["名称"]
if name not in all_shared_dietary:
all_shared_dietary[name] = {"体质": [], "功效": item["功效"], "来源": item["来源"]}
all_shared_dietary[name]["体质"].append(cname)
# 构建推荐列表(含共享标记)
recommendations = []
for item in top_items:
name = item["名称"]
total = item["得分"]
source = item["来源"]
efficacy = item["功效"]
recommendations.append({
"名称": name,
"来源": source,
"功效": efficacy,
"得分": total,
"是否共享": False, # 后面更新
"共享体质": []
})
dietary_recommendations[cname] = recommendations
# ===== 5. 更新共享信息 =====
# 找出跨体质药膳
shared_count = 0
cross_constitution_items = []
for name, info in all_shared_dietary.items():
if len(info["体质"]) >= 2:
shared_count += 1
cross_constitution_items.append({
"名称": name,
"适用体质": info["体质"],
"功效": info["功效"],
"来源": info["来源"]
})
# 更新推荐列表中的共享标记
for cname in constitution_order:
for rec in dietary_recommendations[cname]:
name = rec["名称"]
if name in all_shared_dietary and len(all_shared_dietary[name]["体质"]) >= 2:
rec["是否共享"] = True
rec["共享体质"] = [t for t in all_shared_dietary[name]["体质"] if t != cname]
print(f"\n跨体质药膳数: {shared_count}")
for item in cross_constitution_items[:5]:
print(f" {item['名称']}: {item['适用体质']}")
# ===== 6. 功效分类统计 =====
print("\n开始功效分类统计...")
EFFICACY_KEYWORDS_MAP = {
"补气": ["补气", "益气", "补中益气"],
"健脾": ["健脾", "补脾", "益脾", "温脾"],
"养血": ["养血", "补血", "活血养血"],
"滋阴": ["滋阴", "养阴", "育阴"],
"温阳": ["温阳", "壮阳", "补阳"],
"散寒": ["散寒", "温中", "暖胃", "祛寒"],
"清热": ["清热", "解毒", "泻火", "凉血"],
"祛湿": ["祛湿", "利湿", "除湿", "化湿", "渗湿"],
"化痰": ["化痰", "祛痰", "消痰"],
"疏肝": ["疏肝", "解郁", "理气", "行气"],
"安神": ["安神", "宁心", "养心"],
"补肾": ["补肾", "益肾", "滋肾", "温肾"],
"润肺": ["润肺", "补肺", "养肺"],
"消食": ["消食", "导滞", "开胃", "健胃"],
"生津": ["生津", "止渴", "润燥"],
"活血": ["活血", "化瘀", "散瘀", "通络"],
"固表": ["固表", "固涩", "止汗", "收敛"],
"祛风": ["祛风", "疏风", "散风"],
"利水": ["利水", "消肿", "利尿"],
"补虚": ["补虚", "补益", "滋补", "扶正"]
}
for cname in constitution_order:
top_items = dietary_recommendations[cname][:30]
efficacy_stats = defaultdict(int)
for item in top_items:
efficacy_text = item["功效"]
if not efficacy_text:
continue
# 统计功效关键词
for category, keywords in EFFICACY_KEYWORDS_MAP.items():
for kw in keywords:
if kw in efficacy_text:
efficacy_stats[category] += 1
break # 每个药膳每个分类只计1次
dietary_efficacy_stats[cname] = dict(sorted(efficacy_stats.items(), key=lambda x: -x[1]))
top_cats = list(dietary_efficacy_stats[cname].keys())[:5]
print(f"{cname}: {dict(top_cats)}...")
# ===== 7. 来源分析 =====
print("\n开始来源分析...")
source_stats = defaultdict(int)
for cname in constitution_order:
for rec in dietary_recommendations[cname]:
source = rec["来源"]
if source:
# 简化来源名
simple_source = source.strip()
source_stats[simple_source] += 1
top_sources = sorted(source_stats.items(), key=lambda x: -x[1])[:20]
print("TOP来源:")
for s, c in top_sources:
print(f" {s}: {c}次")
# ===== 8. 构建完整JSON输出 =====
print("\n构建JSON输出...")
# 统计摘要
total_recommendations = sum(len(v) for v in dietary_recommendations.values())
total_unique = len(all_shared_dietary)
summary_stats = {
"总药膳数": len(dietary_list),
"体质类型数": len(constitution_order),
"各体质推荐药膳数": {c: len(dietary_recommendations[c]) for c in constitution_order},
"总推荐人次": total_recommendations,
"去重推荐药膳数": total_unique,
"跨体质共享药膳数": shared_count,
"评分阈值": THRESHOLD,
"评分模型因子": {
"功效字段含体质关键字": "+4",
"简介/名称字段含体质关键字": "+3",
"配方字段含体质推荐食材": "+2",
"适宜人群字段含体质关键字": "+3",
"来源字段含经典文献": "+1"
}
}
output_json = {
"体质-药膳推荐": dietary_recommendations,
"体质-药膳功效统计": dietary_efficacy_stats,
"体质-药膳来源统计": dict(top_sources),
"体质-药膳共享分析": {
"数量": shared_count,
"跨体质药膳": cross_constitution_items
},
"统计摘要": summary_stats
}
# 保存JSON
output_dir = os.path.join(BASE_DIR, "03_关联融合/交叉关联分析")
os.makedirs(output_dir, exist_ok=True)
output_path = os.path.join(output_dir, "体质-药膳推荐体系.json")
with open(output_path, "w", encoding="utf-8") as f:
json.dump(output_json, f, ensure_ascii=False, indent=2)
print(f"\nJSON已保存: {output_path}")
print(f"文件大小: {os.path.getsize(output_path) / 1024:.1f} KB")
# ===== 9. 验证JSON完整性 =====
with open(output_path, "r", encoding="utf-8") as f:
verify = json.load(f)
print(f"\n验证JSON:")
print(f" 体质-药膳推荐: {len(verify['体质-药膳推荐'])} 种体质")
for c in constitution_order:
print(f" {c}: {len(verify['体质-药膳推荐'][c])} 条推荐")
print(f" 体质-药膳功效统计: {len(verify['体质-药膳功效统计'])} 种")
print(f" 体质-药膳共享分析: {verify['体质-药膳共享分析']['数量']} 条共享")
print(f" 统计摘要: 总推荐={verify['统计摘要']['总推荐人次']}人次")
print("\n===== 药膳推荐体系构建完成 =====")