Files
health/大医网/05_脚本工具/build_constitution_knowledge_index.py
T

478 lines
19 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
体质-知识索引深层关联融合
===========================
建立体质↔经脉、体质↔功效关键词、体质↔药膳疾病通道、体质↔导引功法
生成统一的知识图谱JSON文件并追加分析报告。
"""
import json
import os
import sys
from collections import defaultdict
BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
# ---------- 1. 加载数据 ----------
def load_json(path, desc=""):
try:
with open(path, "r", encoding="utf-8") as f:
data = json.load(f)
print(f" ✓ {desc or os.path.basename(path)} ({len(data)} 条顶层键)")
return data
except Exception as e:
print(f" ✗ 加载失败 {path}: {e}")
return None
print("=" * 60)
print("加载输入数据...")
print("=" * 60)
# 九种体质详细数据
constitutions = load_json(
f"{BASE}/02_加工数据/中医体质/九种体质详细数据.json",
"九种体质详细数据"
)
# 知识索引文件
meridian_class = load_json(
f"{BASE}/02_加工数据/知识索引/索引_经脉分类.json",
"索引_经脉分类"
)
efficacy_keywords = load_json(
f"{BASE}/02_加工数据/知识索引/索引_药膳功效关键词.json",
"索引_药膳功效关键词"
)
diet_disease_cleaned = load_json(
f"{BASE}/02_加工数据/知识索引/索引_药膳疾病关联_cleaned.json",
"索引_药膳疾病关联_cleaned"
)
diet_disease_full = load_json(
f"{BASE}/02_加工数据/知识索引/索引_药膳疾病关联.json",
"索引_药膳疾病关联"
)
daoyin_techniques = load_json(
f"{BASE}/02_加工数据/知识索引/索引_导引运动处_technique.json",
"索引_导引运动处_technique"
)
# 交叉关联分析数据
acupoint_mining = load_json(
f"{BASE}/03_关联融合/交叉关联分析/体质-方剂药材穴位挖掘.json",
"体质-方剂药材穴位挖掘"
)
diet_recommend = load_json(
f"{BASE}/03_关联融合/交叉关联分析/体质-药膳推荐体系.json",
"体质-药膳推荐体系"
)
# ---------- 2. 辅助函数 ----------
def get_acupoint_meridian(acupoint_name, acupoint_branch, meridian_class):
"""
确定穴位的经脉归属。
优先使用穴位数据中的"隶属"字段;若找不到,
在经脉分类索引中搜索穴位名称。
"""
# 如果已经标注了隶属关系
if acupoint_branch and acupoint_branch != "":
# 检查隶属字段是否是已知经脉
if acupoint_branch in meridian_class:
return acupoint_branch
# 部分穴位隶属字段是"经外奇穴"、"耳穴"等
return acupoint_branch
# 在经脉分类中查找
for meridian, points in meridian_class.items():
for point in points:
# 移除"穴"后缀匹配
p_clean = point.replace("穴", "")
a_clean = acupoint_name.replace("穴", "")
if p_clean == a_clean or point == acupoint_name:
return meridian
return "未分类"
def get_efficacy_score(constitution_keywords, efficacy_dict):
"""
计算体质关键字与功效关键词的匹配得分。
对每个体质关键字,检查是否包含在功效关键词名中或功效关键词名是否包含体质关键字。
返回匹配列表及得分。
"""
matches = []
seen = set()
for kw in constitution_keywords:
for eff_name in efficacy_dict.keys():
# 关键字匹配:体质关键字出现在功效关键词中 或 功效关键词出现在体质关键字中
if kw in eff_name or eff_name in kw:
if eff_name not in seen:
# 得分 = 该功效关键词下的药膳数量(作为权重)
score = len(efficacy_dict[eff_name])
matches.append({
"关键词": eff_name,
"得分": min(score, 50), # 上限50
"匹配源": f"体质关键字「{kw}」↔ 药膳功效「{eff_name}」"
})
seen.add(eff_name)
# 按得分降序
matches.sort(key=lambda x: x["得分"], reverse=True)
return matches
# ---------- 3. 体质↔经脉分类关联 ----------
print("\n" + "=" * 60)
print("1/4 体质↔经脉分类关联...")
print("=" * 60)
body_meridian_dist = {}
if acupoint_mining and "体质-穴位" in acupoint_mining:
for body_type, acupoints in acupoint_mining["体质-穴位"].items():
meridian_counter = defaultdict(int)
for apt in acupoints:
name = apt.get("名称", "")
branch = apt.get("隶属", "")
meridian = get_acupoint_meridian(name, branch, meridian_class)
meridian_counter[meridian] += 1
# 按计数降序排列
sorted_meridian = dict(sorted(meridian_counter.items(),
key=lambda x: x[1], reverse=True))
body_meridian_dist[body_type] = sorted_meridian
print(f" {body_type}: {len(acupoints)} 个穴位, {len(sorted_meridian)} 条经脉")
else:
print(" ⚠ 未找到体质-穴位数据,尝试使用已有的体质-穴位归经统计")
if acupoint_mining and "体质-穴位归经统计" in acupoint_mining:
body_meridian_dist = acupoint_mining["体质-穴位归经统计"]
for bt, data in body_meridian_dist.items():
print(f" {bt}: {len(data)} 条经脉")
else:
print(" ✗ 无法构建体质-经脉关联")
# ---------- 4. 体质↔功效关键词关联 ----------
print("\n" + "=" * 60)
print("2/4 体质↔功效关键词关联...")
print("=" * 60)
body_efficacy = {}
if constitutions and efficacy_keywords:
for con in constitutions:
name = con.get("名称", "")
keywords = con.get("关联关键字", [])
matches = get_efficacy_score(keywords, efficacy_keywords)
body_efficacy[name] = matches[:30] # 保留TOP30
print(f" {name}: {len(keywords)} 个关键字 → 匹配 {len(matches)} 个功效关键词 (展示前5: {[m['关键词'] for m in matches[:5]]})")
else:
print(" ✗ 无法构建体质-功效关键词关联")
# ---------- 5. 体质↔药膳疾病通道 ----------
print("\n" + "=" * 60)
print("3/4 体质↔药膳疾病通道...")
print("=" * 60)
# 构建药膳→疾病映射
diet_to_disease = defaultdict(set)
# 使用全量版药膳疾病关联
if diet_disease_full and "药膳→疾病TOP30" in diet_disease_full:
for diet_name, diseases in diet_disease_full["药膳→疾病TOP30"].items():
if isinstance(diseases, list):
for d in diseases:
diet_to_disease[diet_name].add(d)
print(f" 从全量版提取: {len(diet_to_disease)} 个药膳→疾病映射")
else:
print(" ⚠ 全量版无药膳→疾病TOP30")
body_disease_channel = {}
if diet_recommend and "体质-药膳推荐" in diet_recommend:
for body_type, diet_list in diet_recommend["体质-药膳推荐"].items():
disease_counter = defaultdict(lambda: {"疾病名": "", "药膳数": 0, "代表药膳": []})
for item in diet_list:
diet_name = item.get("名称", "")
if diet_name in diet_to_disease:
for disease in diet_to_disease[diet_name]:
key = disease
disease_counter[key]["疾病名"] = disease
disease_counter[key]["药膳数"] += 1
if len(disease_counter[key]["代表药膳"]) < 5:
disease_counter[key]["代表药膳"].append(diet_name)
sorted_diseases = sorted(disease_counter.values(),
key=lambda x: x["药膳数"], reverse=True)
body_disease_channel[body_type] = sorted_diseases[:20] # TOP20
print(f" {body_type}: {len(diet_list)} 个推荐药膳 → {len(sorted_diseases)} 个关联疾病 (展示前5: {[d['疾病名'] for d in sorted_diseases[:5]]})")
else:
print(" ✗ 无法构建体质-药膳疾病通道")
# ---------- 6. 体质↔导引功法关联 ----------
print("\n" + "=" * 60)
print("4/4 体质↔导引功法关联...")
print("=" * 60)
body_daoyin = {}
if constitutions and daoyin_techniques:
techniques_list = daoyin_techniques.get("techniques", [])
for con in constitutions:
name = con.get("名称", "")
recommend_gongfa = con.get("推荐功法", [])
con_keywords = con.get("关联关键字", [])
# Part A: 体质推荐功法的导引技术匹配(按名称)
matched_by_name = []
for gongfa in recommend_gongfa:
# 在导引技术中查找匹配
best_match = None
for t in techniques_list:
tname = t.get("name", "")
# 名称包含匹配 (如"站桩"→"一字桩")
if gongfa == tname or gongfa in tname or tname in gongfa:
best_match = t
break
if best_match:
matched_by_name.append({
"功法": best_match.get("name", gongfa),
"功效": ";".join(best_match.get("benefits", [])),
"涉及部位": ";".join(best_match.get("parts_involved", [])),
"匹配方式": "名称匹配"
})
else:
matched_by_name.append({
"功法": gongfa,
"功效": f"体质调理推荐功法({name}通用运动处方)",
"涉及部位": "",
"匹配方式": "体质推荐"
})
# Part B: 按体质关键字匹配导引技术(功效层面关联)
matched_by_keyword = []
for t in techniques_list:
tname = t.get("name", "")
benefits = ";".join(t.get("benefits", []))
parts = ";".join(t.get("parts_involved", []))
# 检查体质关键字是否出现在功效描述中
matched_kws = [kw for kw in con_keywords if kw in benefits]
if matched_kws:
matched_by_keyword.append({
"功法": tname,
"功效": benefits,
"涉及部位": parts,
"匹配方式": f"关键字匹配({','.join(matched_kws)})"
})
# 合并:先按名称匹配,再按关键字匹配
all_matched = matched_by_name + matched_by_keyword
body_daoyin[name] = all_matched
name_matches = sum(1 for m in matched_by_name if m["匹配方式"] == "名称匹配")
kw_matches = len(matched_by_keyword)
print(f" {name}: {len(recommend_gongfa)} 种推荐功法 + {kw_matches} 个按关键字匹配的导引技术")
else:
print(" ✗ 无法构建体质-导引功法关联")
# ---------- 7. 组装输出JSON ----------
print("\n" + "=" * 60)
print("组装输出JSON...")
print("=" * 60)
output = {
"体质-经脉分布": body_meridian_dist,
"体质-功效关键词": body_efficacy,
"体质-药膳疾病通道": body_disease_channel,
"体质-导引功法": body_daoyin,
"统计摘要": {
"体质数": len(constitutions) if constitutions else 0,
"经脉关联数": sum(len(v) for v in body_meridian_dist.values()) if body_meridian_dist else 0,
"功效关键词关联数": sum(len(v) for v in body_efficacy.values()) if body_efficacy else 0,
"药膳疾病通道数": sum(len(v) for v in body_disease_channel.values()) if body_disease_channel else 0,
"导引功法关联数": sum(len(v) for v in body_daoyin.values()) if body_daoyin else 0
}
}
output_path = f"{BASE}/03_关联融合/体质-知识索引关联.json"
os.makedirs(os.path.dirname(output_path), exist_ok=True)
with open(output_path, "w", encoding="utf-8") as f:
json.dump(output, f, ensure_ascii=False, indent=2)
# 验证
if os.path.exists(output_path):
file_size = os.path.getsize(output_path)
print(f" ✓ JSON输出文件: {output_path} ({file_size/1024:.1f} KB)")
else:
print(f" ✗ 输出文件未创建: {output_path}")
sys.exit(1)
# ---------- 8. 生成报告追加内容 ----------
print("\n" + "=" * 60)
print("生成报告追加内容...")
print("=" * 60)
report_path = f"{BASE}/04_分析报告/体质数据挖掘报告.md"
# 构建报告文本
report_section = []
report_section.append("\n---\n## 九、体质-知识索引关联分析\n")
report_section.append("\n> **数据源**: 知识索引文件(经脉分类、功效关键词、药膳疾病关联、导引运动技术)\n")
report_section.append("> **方法**: 基于体质关键字匹配、穴位-经脉映射、药膳-疾病桥接、功法名称匹配\n")
# 9.1 体质-经脉分布
report_section.append("\n### 9.1 体质-经脉分布(每种体质TOP5经脉)\n")
report_section.append("\n| 体质类型 | TOP5经脉 |\n|---------|----------|\n")
for bt, meridians in body_meridian_dist.items():
top5 = list(meridians.keys())[:5]
top5_str = "、".join([f"{m}({meridians[m]})" for m in top5])
report_section.append(f"| {bt} | {top5_str} |\n")
# 9.2 体质-功效关键词
report_section.append("\n### 9.2 体质-功效关键词TOP10\n")
report_section.append("\n| 体质类型 | TOP10功效关键词 |\n|---------|----------------|\n")
for bt, keywords in body_efficacy.items():
top10 = keywords[:10]
top10_str = "、".join([f"{k['关键词']}({k['得分']})" for k in top10])
report_section.append(f"| {bt} | {top10_str} |\n")
# 9.3 体质-药膳疾病通道
report_section.append("\n### 9.3 体质-药膳疾病通道(每种体质TOP5关联疾病)\n")
report_section.append("\n| 体质类型 | TOP5关联疾病 |\n|---------|--------------|\n")
for bt, diseases in body_disease_channel.items():
top5 = diseases[:5]
top5_str = "、".join([f"{d['疾病名']}({d['药膳数']}种药膳)" for d in top5])
report_section.append(f"| {bt} | {top5_str} |\n")
# 9.4 体质-导引功法
report_section.append("\n### 9.4 体质-导引功法推荐\n")
report_section.append("\n| 体质类型 | 推荐功法 | 导引功效 |\n|---------|---------|----------|\n")
for bt, gongfa_list in body_daoyin.items():
for gf in gongfa_list:
gongfa_name = gf["功法"]
gongfa_effect = gf["功效"]
report_section.append(f"| {bt} | {gongfa_name} | {gongfa_effect} |\n")
# 9.5 关键发现
report_section.append("\n### 9.5 关键发现\n")
# 经脉分析
if body_meridian_dist:
all_meridians = defaultdict(int)
for bt, meridians in body_meridian_dist.items():
for m, c in meridians.items():
all_meridians[m] += c
top_meridians = sorted(all_meridians.items(), key=lambda x: x[1], reverse=True)[:5]
top_mer_str = "、".join([f"{m}({c}次)" for m, c in top_meridians])
report_section.append(f"\n1. **经脉分布**: 各体质关联穴位归属经脉中,出现频率最高的经脉为{top_mer_str}。")
report_section.append("耳穴和经外奇穴在多数体质中占有较大比例,提示体质调理中微针系统的重要性。")
# 功效关键词分析
if body_efficacy:
all_keyword_scores = defaultdict(int)
for bt, keywords in body_efficacy.items():
for k in keywords:
all_keyword_scores[k["关键词"]] += k["得分"]
top_keywords = sorted(all_keyword_scores.items(), key=lambda x: x[1], reverse=True)[:5]
top_kw_str = "、".join([f"{kw}({sc})" for kw, sc in top_keywords])
report_section.append(f"\n2. **功效关键词**: 跨体质最突出的功效关键词为{top_kw_str},")
report_section.append('反映了中医体质调理以「补、益、调、养」为核心的整体思路。')
# 疾病通道分析
if body_disease_channel:
all_diseases = defaultdict(int)
for bt, diseases in body_disease_channel.items():
for d in diseases:
all_diseases[d["疾病名"]] += d["药膳数"]
top_diseases = sorted(all_diseases.items(), key=lambda x: x[1], reverse=True)[:5]
top_dis_str = "、".join([f"{d}({c}种药膳)" for d, c in top_diseases])
report_section.append(f"\n3. **疾病通道**: 体质→药膳→疾病关联路径中,最常关联的疾病为{top_dis_str}。")
report_section.append("提示通过药膳调理体质可针对性预防这些疾病。")
# 导引功法分析
if body_daoyin:
all_gongfa = defaultdict(int)
for bt, gongfa_list in body_daoyin.items():
for gf in gongfa_list:
all_gongfa[gf["功法"]] += 1
top_gongfa = sorted(all_gongfa.items(), key=lambda x: x[1], reverse=True)[:5]
top_gf_str = "、".join([f"{g}({c}种体质)" for g, c in top_gongfa])
report_section.append(f"\n4. **导引功法**: {top_gf_str}是覆盖体质最广的推荐功法,")
report_section.append("八段锦和太极拳几乎适用于所有体质类型,是中医体质调理的通用运动处方。")
report_section.append(f"\n5. **数据整合**: 本次分析融合了11个知识索引文件与3个体质交叉关联文件,")
report_section.append(f"共建立{output['统计摘要']['经脉关联数']}条经脉关联、{output['统计摘要']['功效关键词关联数']}条功效关键词关联、")
report_section.append(f"{output['统计摘要']['药膳疾病通道数']}条药膳疾病通道和{output['统计摘要']['导引功法关联数']}条导引功法关联。")
report_text = "\n".join(report_section)
# 读取原报告,在 ## 八、附录 前插入
if os.path.exists(report_path):
with open(report_path, "r", encoding="utf-8") as f:
original_content = f.read()
marker = "## 八、附录"
if marker in original_content:
# 在 marker 之前插入
insert_pos = original_content.find(marker)
new_content = original_content[:insert_pos] + report_text + "\n\n" + original_content[insert_pos:]
else:
# 如果找不到,追加到末尾
new_content = original_content + "\n\n" + report_text
with open(report_path, "w", encoding="utf-8") as f:
f.write(new_content)
# 验证
if os.path.exists(report_path):
file_size = os.path.getsize(report_path)
print(f" ✓ 报告已更新: {report_path} ({file_size/1024:.1f} KB)")
else:
print(f" ✗ 报告文件未找到: {report_path}")
else:
print(f" ⚠ 报告文件不存在,创建新报告")
with open(report_path, "w", encoding="utf-8") as f:
f.write("# 大医网 | 九种中医体质数据挖掘报告\n\n")
f.write(report_text)
# ---------- 9. 最终验证 ----------
print("\n" + "=" * 60)
print("最终验证")
print("=" * 60)
summary = output["统计摘要"]
print(f" 体质数: {summary['体质数']}")
print(f" 经脉关联数: {summary['经脉关联数']}")
print(f" 功效关键词关联数: {summary['功效关键词关联数']}")
print(f" 药膳疾病通道数: {summary['药膳疾病通道数']}")
print(f" 导引功法关联数: {summary['导引功法关联数']}")
# 验证输出文件存在
json_ok = os.path.exists(output_path)
report_ok = os.path.exists(report_path)
print(f" ✓ JSON输出: {json_ok}")
print(f" ✓ 报告已更新: {report_ok}")
if json_ok and report_ok:
print("\n✅ 全部完成!")
else:
print("\n❌ 有文件未生成!")
sys.exit(1)