初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)

This commit is contained in:
512song committed 2026-09-23 21:59:25 +08:00
commit 80ae3811cf
7712 files changed
+4628547

No files matched your search

File diff suppressed because it is too large. Load diff
+402
View File
@@ -0,0 +1,402 @@
#!/usr/bin/env python3
"""
大医网「症状→方剂/药材/穴位」多段桥接关联 及 关联融合总览统计
"""
import json
import os
import sys
import math
from collections import defaultdict
BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/03_关联融合"
CROSS_DIR = os.path.join(BASE, "交叉关联分析")
OUTPUT_DIR = CROSS_DIR
# ============================================================
# PART 1: 症状→方剂/药材/穴位 桥接
# ============================================================
def load_symptom_disease():
"""Load 症状-疾病关联.json - format: {症状: [[疾病, 置信度(0-100), [标签]], ...]}"""
path = os.path.join(CROSS_DIR, "症状-疾病关联.json")
print(f" 加载 {path} ...")
with open(path, 'r', encoding='utf-8') as f:
data = json.load(f)
# Normalize confidence: values 35-100 range, divide by 100 to get 0-1
normalized = {}
for symptom, diseases in data.items():
normalized[symptom] = []
for entry in diseases:
disease_name = entry[0]
conf = entry[1]
# Normalize to 0-1
if conf > 1:
conf = conf / 100.0
normalized[symptom].append((disease_name, conf))
return normalized
def load_disease_target(file_key, target_name):
"""Load disease->target mapping: {disease: [(target, conf_0_1), ...]}"""
fname_map = {
'formula': '疾病-方剂关联数据.json',
'herb': '疾病-药材关联数据.json',
'acupoint': '疾病-穴位关联数据.json'
}
topkey_map = {
'formula': 'disease_to_formulas',
'herb': 'disease_to_herbs',
'acupoint': 'disease_to_acupoints'
}
path = os.path.join(CROSS_DIR, fname_map[file_key])
print(f" 加载 {path} ...")
with open(path, 'r', encoding='utf-8') as f:
data = json.load(f)
return data[topkey_map[file_key]]
def build_bridging(sym_disease, disease_target, target_label):
"""
Symptom -> Disease -> Target bridging.
sym_disease: {symptom: [(disease, conf_sd), ...]}
disease_target: {disease: [(target, conf_dt), ...]}
Returns: {symptom: [{target名: ..., 桥接路径: ..., 得分: ...}, ...]}
"""
print(f" 构建症状→{target_label}桥接 ...")
result = {}
for symptom, disease_list in sym_disease.items():
# For each symptom, collect all paths
target_scores = defaultdict(float)
target_paths = defaultdict(list)
for disease_name, conf_sd in disease_list:
if disease_name not in disease_target:
continue
for target_item in disease_target[disease_name]:
target_name = target_item[0]
conf_dt = target_item[1] if len(target_item) > 1 else 1.0
# Chain score = symptom-disease confidence * disease-target confidence
chain_score = conf_sd * conf_dt
path_str = f"{symptom}→{disease_name}→{target_name}"
if chain_score > target_scores[target_name]:
target_scores[target_name] = chain_score
target_paths[target_name] = path_str
if target_scores:
# Sort by score descending, take TOP20
sorted_targets = sorted(target_scores.items(), key=lambda x: -x[1])
top20 = sorted_targets[:20]
result[symptom] = [
{
target_label: tname,
"桥接路径": target_paths[tname],
"得分": round(score, 4)
}
for tname, score in top20
]
return result
def do_bridging():
print("=" * 60)
print("PART 1: 症状→方剂/药材/穴位 桥接关联")
print("=" * 60)
# Load symptoms->disease
sym_disease = load_symptom_disease()
print(f" 症状总数: {len(sym_disease)}")
# Load disease->formula/herb/acupoint
disease_formula = load_disease_target('formula', '方剂名')
disease_herb = load_disease_target('herb', '药材名')
disease_acupoint = load_disease_target('acupoint', '穴位名')
print(f" 疾病→方剂: {len(disease_formula)} 疾病")
print(f" 疾病→药材: {len(disease_herb)} 疾病")
print(f" 疾病→穴位: {len(disease_acupoint)} 疾病")
# Build bridges
sym_formula = build_bridging(sym_disease, disease_formula, "方剂名")
sym_herb = build_bridging(sym_disease, disease_herb, "药材名")
sym_acupoint = build_bridging(sym_disease, disease_acupoint, "穴位名")
print(f" 桥接症状→方剂: {len(sym_formula)} 症状")
print(f" 桥接症状→药材: {len(sym_herb)} 症状")
print(f" 桥接症状→穴位: {len(sym_acupoint)} 症状")
# Count unique items
all_formulas = set()
for symptom, items in sym_formula.items():
for item in items:
all_formulas.add(item['方剂名'])
all_herbs = set()
for symptom, items in sym_herb.items():
for item in items:
all_herbs.add(item['药材名'])
all_acupoints = set()
for symptom, items in sym_acupoint.items():
for item in items:
all_acupoints.add(item['穴位名'])
total_paths = sum(len(v) for v in sym_formula.values()) + \
sum(len(v) for v in sym_herb.values()) + \
sum(len(v) for v in sym_acupoint.values())
summary = {
"桥接症状数": len(set(list(sym_formula.keys()) + list(sym_herb.keys()) + list(sym_acupoint.keys()))),
"桥接方剂数": len(all_formulas),
"桥接药材数": len(all_herbs),
"桥接穴位数": len(all_acupoints),
"总桥接路径": total_paths
}
print(f" 统计摘要: {json.dumps(summary, ensure_ascii=False)}")
# Build output structure
output = {
"症状-方剂": sym_formula,
"症状-药材": sym_herb,
"症状-穴位": sym_acupoint,
"统计摘要": summary
}
output_path = os.path.join(OUTPUT_DIR, "症状-方剂药材穴位桥接.json")
print(f" 写入 {output_path} ...")
with open(output_path, 'w', encoding='utf-8') as f:
json.dump(output, f, ensure_ascii=False, indent=2)
return output
# ============================================================
# PART 2: 关联融合总览统计
# ============================================================
def count_records(obj):
"""Count 'records' in a JSON structure - heuristic for different formats"""
if isinstance(obj, list):
return len(obj)
elif isinstance(obj, dict):
# Check common patterns
# Pattern 1: values are arrays (like symptom->disease mapping)
array_values = [v for v in obj.values() if isinstance(v, (list, dict))]
if array_values:
# Sum of array lengths, or count of dict values
total = 0
for v in obj.values():
if isinstance(v, list):
total += len(v)
elif isinstance(v, dict):
total += len(v)
return total
# Pattern 2: top-level keys as records
return len(obj)
return 0
def get_content_type(filename):
"""Determine content description from filename"""
descriptions = {
'01_配方-药材关联表.json': '配方-药材映射',
'02_药材-配方关联表.json': '药材-配方映射',
'03_Top50高频药材.json': '高频药材统计',
'04_Top50大复方.json': '大复方统计',
'05_药材分类统计.json': '药材分类统计',
'06_总摘要.json': '关联分析摘要',
'症状-疾病关联.json': '症状→疾病关联(原始)',
'疾病-症状关联.json': '疾病→症状关联',
'症状-疾病_高置信关联.json': '症状→疾病高置信关联',
'疾病-方剂关联数据.json': '疾病→方剂关联',
'疾病-药材关联数据.json': '疾病→药材关联',
'疾病-穴位关联数据.json': '疾病→穴位关联',
'四库交叉关联数据.json': '四库交叉关联总览',
'三库关联总览.json': '三库关联统计',
'术语与症状深度关联分析.json': '术语-症状深度关联',
'症状_全部.json': '症状全集',
'症状_列表.json': '症状列表',
'症状_归档统计.json': '症状归档统计',
'症状->疾病推算模型.json': '症状→疾病推算模型',
'症状->疾病->四库桥接数据.json': '症状→疾病→四库桥接',
'疾病->增强方剂药材穴位反查数据.json': '疾病→方剂药材穴位反查',
'疾病-症状严谨关联数据.json': '疾病-症状严谨关联',
'体质-疾病症状挖掘.json': '体质-疾病症状挖掘',
'体质-方剂药材穴位挖掘.json': '体质-方剂药材穴位挖掘',
'体质-药膳推荐体系.json': '体质-药膳推荐体系',
'体质-知识索引关联.json': '体质-知识索引关联',
'咳嗽哮喘_关联数据.json': '咳嗽哮喘专题关联',
'心脑血管_关联数据.json': '心脑血管专题关联',
'脾胃调理_关联数据.json': '脾胃调理专题关联',
'失眠不寐_关联数据.json': '失眠不寐专题关联',
'女性补气血_关联数据.json': '女性补气血专题关联',
'症状-方剂药材穴位桥接.json': '症状→方剂/药材/穴位桥接',
'关联融合总览.json': '关联融合总览',
}
# Substring matching
for key, desc in descriptions.items():
if filename == key or filename.endswith(key):
return desc
# Generic detection
if '方剂' in filename and '药材' in filename:
return '配方-药材关联'
if '药材' in filename and '配方' in filename:
return '药材-配方关联'
if '高频' in filename:
return '高频统计'
if '分类' in filename:
return '分类统计'
if '摘要' in filename or '总汇' in filename or '汇总' in filename:
return '分析摘要'
return '关联数据'
def get_file_stats(filepath):
"""Get file size in KB and record count"""
size_kb = os.path.getsize(filepath) / 1024.0
size_kb = round(size_kb, 1)
try:
with open(filepath, 'r', encoding='utf-8') as f:
data = json.load(f)
records = count_records(data)
except:
records = 0
return size_kb, records
def scan_directory(dirpath, max_depth=1):
"""Scan a directory for JSON files and return stats"""
results = []
total_size_kb = 0
total_records = 0
for fname in sorted(os.listdir(dirpath)):
if not fname.endswith('.json'):
continue
fpath = os.path.join(dirpath, fname)
if os.path.isdir(fpath):
continue
size_kb, records = get_file_stats(fpath)
content = get_content_type(fname)
results.append({
"文件名": fname,
"大小KB": size_kb,
"记录数": records,
"内容": content
})
total_size_kb += size_kb
total_records += records
return results, total_size_kb, total_records
def scan_deep_mine():
"""Scan depth mining directory structure"""
deep_dir = os.path.join(BASE, "深度挖掘迭代")
versions = {}
total_files = 0
for item in sorted(os.listdir(deep_dir)):
item_path = os.path.join(deep_dir, item)
if not os.path.isdir(item_path):
continue
json_files = [f for f in os.listdir(item_path) if f.endswith('.json')]
count = len(json_files)
total_files += count
versions[item] = count
return total_files, versions
def build_overview():
print("\n" + "=" * 60)
print("PART 2: 关联融合总览统计")
print("=" * 60)
# 1. Scan 关联分析/
analysis_dir = os.path.join(BASE, "关联分析")
analysis_files, analysis_size, analysis_records = scan_directory(analysis_dir)
print(f" 关联分析: {len(analysis_files)} 文件, {analysis_size:.1f} KB")
# 2. Scan 交叉关联分析/
cross_files, cross_size, cross_records = scan_directory(CROSS_DIR)
print(f" 交叉关联分析: {len(cross_files)} 文件, {cross_size:.1f} KB")
# 3. Scan 深度挖掘迭代/
deep_total, versions = scan_deep_mine()
print(f" 深度挖掘迭代: {deep_total} 文件, {len(versions)} 版本")
# 4. 其他体质文件 in 交叉关联分析
tizhi_files = [f for f in cross_files if '体质' in f['文件名']]
print(f" 体质相关文件: {len(tizhi_files)}")
total_files = len(analysis_files) + len(cross_files) + deep_total
total_size_mb = (analysis_size + cross_size) / 1024.0
# Get deep_mine subdirectory sizes
deep_dir = os.path.join(BASE, "深度挖掘迭代")
total_deep_size_kb = 0
for root, dirs, files in os.walk(deep_dir):
for f in files:
if f.endswith('.json'):
total_deep_size_kb += os.path.getsize(os.path.join(root, f)) / 1024.0
total_size_mb = (analysis_size + cross_size + total_deep_size_kb) / 1024.0
overview = {
"总览": {
"子目录数": 3,
"总文件数": total_files,
"总数据量MB": round(total_size_mb, 2)
},
"关联分析": analysis_files,
"交叉关联分析": cross_files,
"深度挖掘迭代": {
"总文件数": deep_total,
"版本数": len(versions)
}
}
# Add version details
for ver_name, ver_count in sorted(versions.items()):
# Strip the '分析_' prefix since we already add '分析_' in the key
short_name = ver_name.replace('分析_', '', 1) if ver_name.startswith('分析_') else ver_name
overview["深度挖掘迭代"][f"分析_{short_name}文件数"] = ver_count
output_path = os.path.join(OUTPUT_DIR, "关联融合总览.json")
print(f" 写入 {output_path} ...")
with open(output_path, 'w', encoding='utf-8') as f:
json.dump(overview, f, ensure_ascii=False, indent=2)
return overview
# ============================================================
# MAIN
# ============================================================
if __name__ == "__main__":
# Part 1: bridging
bridge_result = do_bridging()
# Part 2: overview
overview_result = build_overview()
# Verify
print("\n" + "=" * 60)
print("验证输出文件")
print("=" * 60)
for fname in ["症状-方剂药材穴位桥接.json", "关联融合总览.json"]:
fpath = os.path.join(OUTPUT_DIR, fname)
if os.path.exists(fpath):
fsize = os.path.getsize(fpath)
print(f" ✓ {fname}: {fsize:,} bytes ({fsize/1024:.1f} KB)")
else:
print(f" ✗ {fname}: NOT FOUND")
print("\n完成!")
@@ -0,0 +1,477 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
体质-知识索引深层关联融合
===========================
建立体质↔经脉、体质↔功效关键词、体质↔药膳疾病通道、体质↔导引功法
生成统一的知识图谱JSON文件并追加分析报告。
"""
import json
import os
import sys
from collections import defaultdict
BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
# ---------- 1. 加载数据 ----------
def load_json(path, desc=""):
try:
with open(path, "r", encoding="utf-8") as f:
data = json.load(f)
print(f" ✓ {desc or os.path.basename(path)} ({len(data)} 条顶层键)")
return data
except Exception as e:
print(f" ✗ 加载失败 {path}: {e}")
return None
print("=" * 60)
print("加载输入数据...")
print("=" * 60)
# 九种体质详细数据
constitutions = load_json(
f"{BASE}/02_加工数据/中医体质/九种体质详细数据.json",
"九种体质详细数据"
)
# 知识索引文件
meridian_class = load_json(
f"{BASE}/02_加工数据/知识索引/索引_经脉分类.json",
"索引_经脉分类"
)
efficacy_keywords = load_json(
f"{BASE}/02_加工数据/知识索引/索引_药膳功效关键词.json",
"索引_药膳功效关键词"
)
diet_disease_cleaned = load_json(
f"{BASE}/02_加工数据/知识索引/索引_药膳疾病关联_cleaned.json",
"索引_药膳疾病关联_cleaned"
)
diet_disease_full = load_json(
f"{BASE}/02_加工数据/知识索引/索引_药膳疾病关联.json",
"索引_药膳疾病关联"
)
daoyin_techniques = load_json(
f"{BASE}/02_加工数据/知识索引/索引_导引运动处_technique.json",
"索引_导引运动处_technique"
)
# 交叉关联分析数据
acupoint_mining = load_json(
f"{BASE}/03_关联融合/交叉关联分析/体质-方剂药材穴位挖掘.json",
"体质-方剂药材穴位挖掘"
)
diet_recommend = load_json(
f"{BASE}/03_关联融合/交叉关联分析/体质-药膳推荐体系.json",
"体质-药膳推荐体系"
)
# ---------- 2. 辅助函数 ----------
def get_acupoint_meridian(acupoint_name, acupoint_branch, meridian_class):
"""
确定穴位的经脉归属。
优先使用穴位数据中的"隶属"字段;若找不到,
在经脉分类索引中搜索穴位名称。
"""
# 如果已经标注了隶属关系
if acupoint_branch and acupoint_branch != "":
# 检查隶属字段是否是已知经脉
if acupoint_branch in meridian_class:
return acupoint_branch
# 部分穴位隶属字段是"经外奇穴"、"耳穴"等
return acupoint_branch
# 在经脉分类中查找
for meridian, points in meridian_class.items():
for point in points:
# 移除"穴"后缀匹配
p_clean = point.replace("穴", "")
a_clean = acupoint_name.replace("穴", "")
if p_clean == a_clean or point == acupoint_name:
return meridian
return "未分类"
def get_efficacy_score(constitution_keywords, efficacy_dict):
"""
计算体质关键字与功效关键词的匹配得分。
对每个体质关键字,检查是否包含在功效关键词名中或功效关键词名是否包含体质关键字。
返回匹配列表及得分。
"""
matches = []
seen = set()
for kw in constitution_keywords:
for eff_name in efficacy_dict.keys():
# 关键字匹配:体质关键字出现在功效关键词中 或 功效关键词出现在体质关键字中
if kw in eff_name or eff_name in kw:
if eff_name not in seen:
# 得分 = 该功效关键词下的药膳数量(作为权重)
score = len(efficacy_dict[eff_name])
matches.append({
"关键词": eff_name,
"得分": min(score, 50), # 上限50
"匹配源": f"体质关键字「{kw}」↔ 药膳功效「{eff_name}」"
})
seen.add(eff_name)
# 按得分降序
matches.sort(key=lambda x: x["得分"], reverse=True)
return matches
# ---------- 3. 体质↔经脉分类关联 ----------
print("\n" + "=" * 60)
print("1/4 体质↔经脉分类关联...")
print("=" * 60)
body_meridian_dist = {}
if acupoint_mining and "体质-穴位" in acupoint_mining:
for body_type, acupoints in acupoint_mining["体质-穴位"].items():
meridian_counter = defaultdict(int)
for apt in acupoints:
name = apt.get("名称", "")
branch = apt.get("隶属", "")
meridian = get_acupoint_meridian(name, branch, meridian_class)
meridian_counter[meridian] += 1
# 按计数降序排列
sorted_meridian = dict(sorted(meridian_counter.items(),
key=lambda x: x[1], reverse=True))
body_meridian_dist[body_type] = sorted_meridian
print(f" {body_type}: {len(acupoints)} 个穴位, {len(sorted_meridian)} 条经脉")
else:
print(" ⚠ 未找到体质-穴位数据,尝试使用已有的体质-穴位归经统计")
if acupoint_mining and "体质-穴位归经统计" in acupoint_mining:
body_meridian_dist = acupoint_mining["体质-穴位归经统计"]
for bt, data in body_meridian_dist.items():
print(f" {bt}: {len(data)} 条经脉")
else:
print(" ✗ 无法构建体质-经脉关联")
# ---------- 4. 体质↔功效关键词关联 ----------
print("\n" + "=" * 60)
print("2/4 体质↔功效关键词关联...")
print("=" * 60)
body_efficacy = {}
if constitutions and efficacy_keywords:
for con in constitutions:
name = con.get("名称", "")
keywords = con.get("关联关键字", [])
matches = get_efficacy_score(keywords, efficacy_keywords)
body_efficacy[name] = matches[:30] # 保留TOP30
print(f" {name}: {len(keywords)} 个关键字 → 匹配 {len(matches)} 个功效关键词 (展示前5: {[m['关键词'] for m in matches[:5]]})")
else:
print(" ✗ 无法构建体质-功效关键词关联")
# ---------- 5. 体质↔药膳疾病通道 ----------
print("\n" + "=" * 60)
print("3/4 体质↔药膳疾病通道...")
print("=" * 60)
# 构建药膳→疾病映射
diet_to_disease = defaultdict(set)
# 使用全量版药膳疾病关联
if diet_disease_full and "药膳→疾病TOP30" in diet_disease_full:
for diet_name, diseases in diet_disease_full["药膳→疾病TOP30"].items():
if isinstance(diseases, list):
for d in diseases:
diet_to_disease[diet_name].add(d)
print(f" 从全量版提取: {len(diet_to_disease)} 个药膳→疾病映射")
else:
print(" ⚠ 全量版无药膳→疾病TOP30")
body_disease_channel = {}
if diet_recommend and "体质-药膳推荐" in diet_recommend:
for body_type, diet_list in diet_recommend["体质-药膳推荐"].items():
disease_counter = defaultdict(lambda: {"疾病名": "", "药膳数": 0, "代表药膳": []})
for item in diet_list:
diet_name = item.get("名称", "")
if diet_name in diet_to_disease:
for disease in diet_to_disease[diet_name]:
key = disease
disease_counter[key]["疾病名"] = disease
disease_counter[key]["药膳数"] += 1
if len(disease_counter[key]["代表药膳"]) < 5:
disease_counter[key]["代表药膳"].append(diet_name)
sorted_diseases = sorted(disease_counter.values(),
key=lambda x: x["药膳数"], reverse=True)
body_disease_channel[body_type] = sorted_diseases[:20] # TOP20
print(f" {body_type}: {len(diet_list)} 个推荐药膳 → {len(sorted_diseases)} 个关联疾病 (展示前5: {[d['疾病名'] for d in sorted_diseases[:5]]})")
else:
print(" ✗ 无法构建体质-药膳疾病通道")
# ---------- 6. 体质↔导引功法关联 ----------
print("\n" + "=" * 60)
print("4/4 体质↔导引功法关联...")
print("=" * 60)
body_daoyin = {}
if constitutions and daoyin_techniques:
techniques_list = daoyin_techniques.get("techniques", [])
for con in constitutions:
name = con.get("名称", "")
recommend_gongfa = con.get("推荐功法", [])
con_keywords = con.get("关联关键字", [])
# Part A: 体质推荐功法的导引技术匹配(按名称)
matched_by_name = []
for gongfa in recommend_gongfa:
# 在导引技术中查找匹配
best_match = None
for t in techniques_list:
tname = t.get("name", "")
# 名称包含匹配 (如"站桩"→"一字桩")
if gongfa == tname or gongfa in tname or tname in gongfa:
best_match = t
break
if best_match:
matched_by_name.append({
"功法": best_match.get("name", gongfa),
"功效": ";".join(best_match.get("benefits", [])),
"涉及部位": ";".join(best_match.get("parts_involved", [])),
"匹配方式": "名称匹配"
})
else:
matched_by_name.append({
"功法": gongfa,
"功效": f"体质调理推荐功法({name}通用运动处方)",
"涉及部位": "",
"匹配方式": "体质推荐"
})
# Part B: 按体质关键字匹配导引技术(功效层面关联)
matched_by_keyword = []
for t in techniques_list:
tname = t.get("name", "")
benefits = ";".join(t.get("benefits", []))
parts = ";".join(t.get("parts_involved", []))
# 检查体质关键字是否出现在功效描述中
matched_kws = [kw for kw in con_keywords if kw in benefits]
if matched_kws:
matched_by_keyword.append({
"功法": tname,
"功效": benefits,
"涉及部位": parts,
"匹配方式": f"关键字匹配({','.join(matched_kws)})"
})
# 合并:先按名称匹配,再按关键字匹配
all_matched = matched_by_name + matched_by_keyword
body_daoyin[name] = all_matched
name_matches = sum(1 for m in matched_by_name if m["匹配方式"] == "名称匹配")
kw_matches = len(matched_by_keyword)
print(f" {name}: {len(recommend_gongfa)} 种推荐功法 + {kw_matches} 个按关键字匹配的导引技术")
else:
print(" ✗ 无法构建体质-导引功法关联")
# ---------- 7. 组装输出JSON ----------
print("\n" + "=" * 60)
print("组装输出JSON...")
print("=" * 60)
output = {
"体质-经脉分布": body_meridian_dist,
"体质-功效关键词": body_efficacy,
"体质-药膳疾病通道": body_disease_channel,
"体质-导引功法": body_daoyin,
"统计摘要": {
"体质数": len(constitutions) if constitutions else 0,
"经脉关联数": sum(len(v) for v in body_meridian_dist.values()) if body_meridian_dist else 0,
"功效关键词关联数": sum(len(v) for v in body_efficacy.values()) if body_efficacy else 0,
"药膳疾病通道数": sum(len(v) for v in body_disease_channel.values()) if body_disease_channel else 0,
"导引功法关联数": sum(len(v) for v in body_daoyin.values()) if body_daoyin else 0
}
}
output_path = f"{BASE}/03_关联融合/体质-知识索引关联.json"
os.makedirs(os.path.dirname(output_path), exist_ok=True)
with open(output_path, "w", encoding="utf-8") as f:
json.dump(output, f, ensure_ascii=False, indent=2)
# 验证
if os.path.exists(output_path):
file_size = os.path.getsize(output_path)
print(f" ✓ JSON输出文件: {output_path} ({file_size/1024:.1f} KB)")
else:
print(f" ✗ 输出文件未创建: {output_path}")
sys.exit(1)
# ---------- 8. 生成报告追加内容 ----------
print("\n" + "=" * 60)
print("生成报告追加内容...")
print("=" * 60)
report_path = f"{BASE}/04_分析报告/体质数据挖掘报告.md"
# 构建报告文本
report_section = []
report_section.append("\n---\n## 九、体质-知识索引关联分析\n")
report_section.append("\n> **数据源**: 知识索引文件(经脉分类、功效关键词、药膳疾病关联、导引运动技术)\n")
report_section.append("> **方法**: 基于体质关键字匹配、穴位-经脉映射、药膳-疾病桥接、功法名称匹配\n")
# 9.1 体质-经脉分布
report_section.append("\n### 9.1 体质-经脉分布(每种体质TOP5经脉)\n")
report_section.append("\n| 体质类型 | TOP5经脉 |\n|---------|----------|\n")
for bt, meridians in body_meridian_dist.items():
top5 = list(meridians.keys())[:5]
top5_str = "、".join([f"{m}({meridians[m]})" for m in top5])
report_section.append(f"| {bt} | {top5_str} |\n")
# 9.2 体质-功效关键词
report_section.append("\n### 9.2 体质-功效关键词TOP10\n")
report_section.append("\n| 体质类型 | TOP10功效关键词 |\n|---------|----------------|\n")
for bt, keywords in body_efficacy.items():
top10 = keywords[:10]
top10_str = "、".join([f"{k['关键词']}({k['得分']})" for k in top10])
report_section.append(f"| {bt} | {top10_str} |\n")
# 9.3 体质-药膳疾病通道
report_section.append("\n### 9.3 体质-药膳疾病通道(每种体质TOP5关联疾病)\n")
report_section.append("\n| 体质类型 | TOP5关联疾病 |\n|---------|--------------|\n")
for bt, diseases in body_disease_channel.items():
top5 = diseases[:5]
top5_str = "、".join([f"{d['疾病名']}({d['药膳数']}种药膳)" for d in top5])
report_section.append(f"| {bt} | {top5_str} |\n")
# 9.4 体质-导引功法
report_section.append("\n### 9.4 体质-导引功法推荐\n")
report_section.append("\n| 体质类型 | 推荐功法 | 导引功效 |\n|---------|---------|----------|\n")
for bt, gongfa_list in body_daoyin.items():
for gf in gongfa_list:
gongfa_name = gf["功法"]
gongfa_effect = gf["功效"]
report_section.append(f"| {bt} | {gongfa_name} | {gongfa_effect} |\n")
# 9.5 关键发现
report_section.append("\n### 9.5 关键发现\n")
# 经脉分析
if body_meridian_dist:
all_meridians = defaultdict(int)
for bt, meridians in body_meridian_dist.items():
for m, c in meridians.items():
all_meridians[m] += c
top_meridians = sorted(all_meridians.items(), key=lambda x: x[1], reverse=True)[:5]
top_mer_str = "、".join([f"{m}({c}次)" for m, c in top_meridians])
report_section.append(f"\n1. **经脉分布**: 各体质关联穴位归属经脉中,出现频率最高的经脉为{top_mer_str}。")
report_section.append("耳穴和经外奇穴在多数体质中占有较大比例,提示体质调理中微针系统的重要性。")
# 功效关键词分析
if body_efficacy:
all_keyword_scores = defaultdict(int)
for bt, keywords in body_efficacy.items():
for k in keywords:
all_keyword_scores[k["关键词"]] += k["得分"]
top_keywords = sorted(all_keyword_scores.items(), key=lambda x: x[1], reverse=True)[:5]
top_kw_str = "、".join([f"{kw}({sc})" for kw, sc in top_keywords])
report_section.append(f"\n2. **功效关键词**: 跨体质最突出的功效关键词为{top_kw_str},")
report_section.append('反映了中医体质调理以「补、益、调、养」为核心的整体思路。')
# 疾病通道分析
if body_disease_channel:
all_diseases = defaultdict(int)
for bt, diseases in body_disease_channel.items():
for d in diseases:
all_diseases[d["疾病名"]] += d["药膳数"]
top_diseases = sorted(all_diseases.items(), key=lambda x: x[1], reverse=True)[:5]
top_dis_str = "、".join([f"{d}({c}种药膳)" for d, c in top_diseases])
report_section.append(f"\n3. **疾病通道**: 体质→药膳→疾病关联路径中,最常关联的疾病为{top_dis_str}。")
report_section.append("提示通过药膳调理体质可针对性预防这些疾病。")
# 导引功法分析
if body_daoyin:
all_gongfa = defaultdict(int)
for bt, gongfa_list in body_daoyin.items():
for gf in gongfa_list:
all_gongfa[gf["功法"]] += 1
top_gongfa = sorted(all_gongfa.items(), key=lambda x: x[1], reverse=True)[:5]
top_gf_str = "、".join([f"{g}({c}种体质)" for g, c in top_gongfa])
report_section.append(f"\n4. **导引功法**: {top_gf_str}是覆盖体质最广的推荐功法,")
report_section.append("八段锦和太极拳几乎适用于所有体质类型,是中医体质调理的通用运动处方。")
report_section.append(f"\n5. **数据整合**: 本次分析融合了11个知识索引文件与3个体质交叉关联文件,")
report_section.append(f"共建立{output['统计摘要']['经脉关联数']}条经脉关联、{output['统计摘要']['功效关键词关联数']}条功效关键词关联、")
report_section.append(f"{output['统计摘要']['药膳疾病通道数']}条药膳疾病通道和{output['统计摘要']['导引功法关联数']}条导引功法关联。")
report_text = "\n".join(report_section)
# 读取原报告,在 ## 八、附录 前插入
if os.path.exists(report_path):
with open(report_path, "r", encoding="utf-8") as f:
original_content = f.read()
marker = "## 八、附录"
if marker in original_content:
# 在 marker 之前插入
insert_pos = original_content.find(marker)
new_content = original_content[:insert_pos] + report_text + "\n\n" + original_content[insert_pos:]
else:
# 如果找不到,追加到末尾
new_content = original_content + "\n\n" + report_text
with open(report_path, "w", encoding="utf-8") as f:
f.write(new_content)
# 验证
if os.path.exists(report_path):
file_size = os.path.getsize(report_path)
print(f" ✓ 报告已更新: {report_path} ({file_size/1024:.1f} KB)")
else:
print(f" ✗ 报告文件未找到: {report_path}")
else:
print(f" ⚠ 报告文件不存在,创建新报告")
with open(report_path, "w", encoding="utf-8") as f:
f.write("# 大医网 | 九种中医体质数据挖掘报告\n\n")
f.write(report_text)
# ---------- 9. 最终验证 ----------
print("\n" + "=" * 60)
print("最终验证")
print("=" * 60)
summary = output["统计摘要"]
print(f" 体质数: {summary['体质数']}")
print(f" 经脉关联数: {summary['经脉关联数']}")
print(f" 功效关键词关联数: {summary['功效关键词关联数']}")
print(f" 药膳疾病通道数: {summary['药膳疾病通道数']}")
print(f" 导引功法关联数: {summary['导引功法关联数']}")
# 验证输出文件存在
json_ok = os.path.exists(output_path)
report_ok = os.path.exists(report_path)
print(f" ✓ JSON输出: {json_ok}")
print(f" ✓ 报告已更新: {report_ok}")
if json_ok and report_ok:
print("\n✅ 全部完成!")
else:
print("\n❌ 有文件未生成!")
sys.exit(1)
@@ -0,0 +1,451 @@
#!/usr/bin/env python3
"""
体质-药膳推荐体系构建脚本
多因子评分模型 + 功效分类统计 + 来源分析 + 共享分析
"""
import json
import os
import glob
import re
from collections import defaultdict, Counter
# ===== 1. 加载数据 =====
BASE_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
# 加载9种体质详细数据
with open(os.path.join(BASE_DIR, "02_加工数据/中医体质/九种体质详细数据.json"), "r", encoding="utf-8") as f:
constitutions_data = json.load(f)
# 构建体质字典
constitutions = {}
for c in constitutions_data:
name = c["名称"]
constitutions[name] = c
print(f"加载体质数据: {len(constitutions)} 种")
for name in constitutions:
c = constitutions[name]
print(f" {name}: 推荐药膳={c.get('推荐药膳', [])}")
# 加载所有药膳文件
diet_dir = os.path.join(BASE_DIR, "01_来源数据/药膳食疗")
diet_files = glob.glob(os.path.join(diet_dir, "*.json"))
print(f"\n药膳食疗文件数: {len(diet_files)}")
dietary_list = []
for fpath in sorted(diet_files):
try:
with open(fpath, "r", encoding="utf-8") as f:
item = json.load(f)
dietary_list.append(item)
except Exception as e:
print(f" 加载失败: {os.path.basename(fpath)}: {e}")
print(f"成功加载药膳食疗: {len(dietary_list)} 条")
# 加载已有交叉关联
cross_assoc_path = os.path.join(BASE_DIR, "02_加工数据/中医体质/体质-大医网交叉关联.json")
if os.path.exists(cross_assoc_path):
with open(cross_assoc_path, "r", encoding="utf-8") as f:
cross_data = json.load(f)
print(f"加载已有交叉关联, keys: {list(cross_data.keys())}")
else:
cross_data = {}
print("未找到已有交叉关联文件")
# ===== 2. 定义体质相关配置 =====
# 经典文献列表(来源字段含以下文献名视为经典文献)
CLASSIC_LITERATURE = [
"伤寒论", "金匮要略", "黄帝内经", "本草纲目", "食疗本草",
"饮膳正要", "食医心鉴", "太平圣惠方", "圣济总录",
"中国药膳大辞典", "中国药膳学", "中医药膳学", "药膳宝典",
"药膳学", "千金要方", "千金翼方", "外台秘要",
"肘后备急方", "普济方", "中医食疗学",
"季节养生药膳指南"
]
# 九种体质的配置(关联关键字、推荐食材、推荐药膳、调理关键词)
CONSTITUTION_CONFIG = {
"气虚质": {
"keywords": ["气虚", "气短", "乏力", "自汗", "懒言", "神疲", "补气", "益气", "培元", "固本"],
"ingredients": ["山药", "大枣", "红枣", "黄芪", "党参", "人参", "白术", "茯苓", "糯米",
"粳米", "小米", "土豆", "香菇", "鸡肉", "牛肉", "鳝鱼", "泥鳅", "蜂蜜",
"黄豆", "扁豆", "豇豆", "南瓜", "胡萝卜", "桂圆", "龙眼"],
"efficacy_keywords": ["补气", "益气", "健脾", "培元", "固本", "补中益气"],
"type_name": "气虚"
},
"阳虚质": {
"keywords": ["阳虚", "畏寒", "怕冷", "四肢不温", "寒凝", "温阳", "散寒", "壮阳", "肾阳", "脾阳"],
"ingredients": ["羊肉", "牛肉", "韭菜", "生姜", "肉桂", "核桃", "栗子", "荔枝", "龙眼",
"茴香", "丁香", "花椒", "小茴香", "干姜", "当归", "附子", "狗肉"],
"efficacy_keywords": ["温阳", "散寒", "壮阳", "温中", "暖胃", "温补脾肾", "补肾壮阳"],
"type_name": "阳虚"
},
"阴虚质": {
"keywords": ["阴虚", "口干", "咽干", "手足心热", "潮热", "盗汗", "滋阴", "降火", "生津", "润燥"],
"ingredients": ["百合", "银耳", "莲子", "枸杞", "黑芝麻", "鸭肉", "甲鱼", "龟肉", "海参",
"牡蛎", "蜂蜜", "梨", "甘蔗", "枇杷", "桑葚", "黑木耳", "沙参", "玉竹",
"麦冬", "生地", "生地黄"],
"efficacy_keywords": ["滋阴", "降火", "生津", "润燥", "养阴", "清热", "滋补肝肾"],
"type_name": "阴虚"
},
"痰湿质": {
"keywords": ["痰湿", "肥胖", "痰多", "胸闷", "苔腻", "健脾", "祛湿", "化痰", "降浊", "利水"],
"ingredients": ["薏苡仁", "冬瓜", "白萝卜", "赤小豆", "茯苓", "荷叶", "山楂", "陈皮",
"燕麦", "荞麦", "海带", "紫菜", "生姜", "莱菔子"],
"efficacy_keywords": ["祛湿", "化痰", "健脾", "利水", "降浊", "消食", "导滞", "除湿"],
"type_name": "痰湿"
},
"湿热质": {
"keywords": ["湿热", "口苦", "苔黄腻", "痤疮", "湿疹", "清热", "利湿", "解毒", "化浊", "泻火"],
"ingredients": ["绿豆", "赤小豆", "薏苡仁", "冬瓜", "苦瓜", "黄瓜", "芹菜", "蒲公英",
"马齿苋", "菊花", "金银花", "莲藕", "茭白", "栀子"],
"efficacy_keywords": ["清热", "利湿", "解毒", "化浊", "泻火", "凉血", "祛湿"],
"type_name": "湿热"
},
"血瘀质": {
"keywords": ["血瘀", "瘀血", "面色晦暗", "瘀斑", "刺痛", "活血", "化瘀", "行气", "通络", "消癥"],
"ingredients": ["山楂", "黑豆", "黑木耳", "醋", "玫瑰花", "红糖", "桃仁", "油菜", "茄子",
"藕", "香菇", "海带", "川芎", "红花", "当归"],
"efficacy_keywords": ["活血", "化瘀", "行气", "通络", "散瘀", "消癥", "止痛"],
"type_name": "血瘀"
},
"气郁质": {
"keywords": ["气郁", "抑郁", "焦虑", "胸闷", "胁痛", "疏肝", "解郁", "理气", "调中", "安神"],
"ingredients": ["柑橘", "佛手", "玫瑰花", "小麦", "大麦", "荞麦", "香橼", "橙子", "柚",
"洋葱", "大蒜", "萝卜", "茴香", "柴胡", "薄荷", "合欢花", "陈皮", "柠檬"],
"efficacy_keywords": ["疏肝", "解郁", "理气", "安神", "调中", "行气", "开郁"],
"type_name": "气郁"
},
"特禀质": {
"keywords": ["过敏", "哮喘", "荨麻疹", "鼻炎", "风团", "益气", "固表", "祛风", "抗敏", "特禀"],
"ingredients": ["黄芪", "白术", "防风", "红枣", "蜂蜜", "山药", "人参", "灵芝", "薏苡仁",
"莲子", "糯米", "花生", "党参", "紫苏", "生姜"],
"efficacy_keywords": ["益气", "固表", "祛风", "抗敏", "扶正", "补肺", "健脾"],
"type_name": "特禀"
},
"平和质": {
"keywords": ["平和", "平衡", "调和", "保健", "养生", "预防"],
"ingredients": ["五谷杂粮", "蔬菜", "水果", "鱼肉", "蛋奶", "豆制品", "坚果", "绿茶",
"山药", "红枣", "枸杞", "茯苓"],
"efficacy_keywords": ["平和", "调和", "保健", "养生", "平衡", "滋补"],
"type_name": "平和"
}
}
print("\n体质配置加载完成")
# ===== 3. 多因子评分模型 =====
def is_classic_literature(source):
"""判断是否来自经典文献"""
if not source:
return False
for lit in CLASSIC_LITERATURE:
if lit in source:
return True
return False
def text_contains_keywords(text, keywords):
"""检查文本是否包含任意关键字"""
if not text:
return False
text_lower = text.lower()
for kw in keywords:
if kw in text:
return True
return False
def count_keyword_matches(text, keywords):
"""统计文本中匹配的关键字数量"""
if not text:
return 0
count = 0
for kw in keywords:
if kw in text:
count += 1
return count
def score_dietary_for_constitution(diet_item, constitution_name):
"""
多因子评分模型
返回:(总分, 各因子详细得分)
"""
cfg = CONSTITUTION_CONFIG[constitution_name]
keywords = cfg["keywords"]
ingredients = cfg["ingredients"]
# 获取各字段
efficacy = diet_item.get("功效", "") or ""
intro = diet_item.get("简介", "") or ""
name = diet_item.get("名称", "") or ""
recipe = diet_item.get("配方", "") or "" # HTML可能包含食材
suitable_pop = diet_item.get("适宜人群", "") or ""
source = diet_item.get("来源", "") or ""
related = diet_item.get("相关配伍", "") or ""
# 移除HTML标签
recipe_clean = re.sub(r'<[^>]+>', '', recipe)
suitable_clean = re.sub(r'<[^>]+>', '', suitable_pop)
related_clean = re.sub(r'<[^>]+>', '', related)
intro_clean = re.sub(r'<[^>]+>', '', intro)
scores = {}
# A) 功效字段含体质关键字 +4(高权重)
score_efficacy = 0
for kw in keywords:
if kw in efficacy:
score_efficacy += 4
# 同时检查功效关键词
for kw in cfg["efficacy_keywords"]:
if kw in efficacy:
score_efficacy += 4
scores["功效匹配"] = score_efficacy
# B) 简介/名称字段含体质关键字 +3
score_intro_name = 0
combined_intro_name = intro_clean + " " + name
for kw in keywords:
if kw in combined_intro_name:
score_intro_name += 3
for kw in cfg["efficacy_keywords"]:
if kw in combined_intro_name:
score_intro_name += 3
scores["简介/名称匹配"] = score_intro_name
# C) 配方字段含体质推荐食材 +2
score_recipe = 0
for ing in ingredients:
if ing in recipe_clean:
score_recipe += 2
# 也检查相关配伍字段
for ing in ingredients:
if ing in related_clean:
score_recipe += 1 # 配伍中匹配权重略低
scores["配方食材匹配"] = score_recipe
# D) 适宜人群字段含体质关键字 +3
score_suitable = 0
for kw in keywords:
if kw in suitable_clean:
score_suitable += 3
for kw in cfg["efficacy_keywords"]:
if kw in suitable_clean:
score_suitable += 3
scores["适宜人群匹配"] = score_suitable
# E) 来源字段含经典文献 +1
score_source = 1 if is_classic_literature(source) else 0
scores["经典文献加分"] = score_source
total = sum(scores.values())
return total, scores
# ===== 4. 对每种体质评分 =====
print("\n开始多因子评分...")
constitution_order = ["气虚质", "阳虚质", "阴虚质", "痰湿质", "湿热质", "血瘀质", "气郁质", "特禀质", "平和质"]
# 存储结果
dietary_recommendations = {}
dietary_efficacy_stats = {}
all_shared_dietary = {} # name -> list of constitution types
THRESHOLD = 4.0
for cname in constitution_order:
scored_items = []
for item in dietary_list:
total, scores = score_dietary_for_constitution(item, cname)
if total >= THRESHOLD:
scored_items.append({
"名称": item.get("名称", ""),
"来源": item.get("来源", ""),
"功效": item.get("功效", ""),
"简介": item.get("简介", ""),
"得分": total,
"各因子得分": scores,
"配方": item.get("配方", ""),
"适宜人群": item.get("适宜人群", ""),
"相关配伍": item.get("相关配伍", "")
})
# 按得分降序排列,取TOP30
scored_items.sort(key=lambda x: (-x["得分"], x["名称"]))
top_items = scored_items[:30]
print(f"{cname}: 得分≥{THRESHOLD}的药膳数={len(scored_items)}, TOP3={[t['名称'] for t in top_items[:3]]}")
# 记录共享信息
for item in top_items:
name = item["名称"]
if name not in all_shared_dietary:
all_shared_dietary[name] = {"体质": [], "功效": item["功效"], "来源": item["来源"]}
all_shared_dietary[name]["体质"].append(cname)
# 构建推荐列表(含共享标记)
recommendations = []
for item in top_items:
name = item["名称"]
total = item["得分"]
source = item["来源"]
efficacy = item["功效"]
recommendations.append({
"名称": name,
"来源": source,
"功效": efficacy,
"得分": total,
"是否共享": False, # 后面更新
"共享体质": []
})
dietary_recommendations[cname] = recommendations
# ===== 5. 更新共享信息 =====
# 找出跨体质药膳
shared_count = 0
cross_constitution_items = []
for name, info in all_shared_dietary.items():
if len(info["体质"]) >= 2:
shared_count += 1
cross_constitution_items.append({
"名称": name,
"适用体质": info["体质"],
"功效": info["功效"],
"来源": info["来源"]
})
# 更新推荐列表中的共享标记
for cname in constitution_order:
for rec in dietary_recommendations[cname]:
name = rec["名称"]
if name in all_shared_dietary and len(all_shared_dietary[name]["体质"]) >= 2:
rec["是否共享"] = True
rec["共享体质"] = [t for t in all_shared_dietary[name]["体质"] if t != cname]
print(f"\n跨体质药膳数: {shared_count}")
for item in cross_constitution_items[:5]:
print(f" {item['名称']}: {item['适用体质']}")
# ===== 6. 功效分类统计 =====
print("\n开始功效分类统计...")
EFFICACY_KEYWORDS_MAP = {
"补气": ["补气", "益气", "补中益气"],
"健脾": ["健脾", "补脾", "益脾", "温脾"],
"养血": ["养血", "补血", "活血养血"],
"滋阴": ["滋阴", "养阴", "育阴"],
"温阳": ["温阳", "壮阳", "补阳"],
"散寒": ["散寒", "温中", "暖胃", "祛寒"],
"清热": ["清热", "解毒", "泻火", "凉血"],
"祛湿": ["祛湿", "利湿", "除湿", "化湿", "渗湿"],
"化痰": ["化痰", "祛痰", "消痰"],
"疏肝": ["疏肝", "解郁", "理气", "行气"],
"安神": ["安神", "宁心", "养心"],
"补肾": ["补肾", "益肾", "滋肾", "温肾"],
"润肺": ["润肺", "补肺", "养肺"],
"消食": ["消食", "导滞", "开胃", "健胃"],
"生津": ["生津", "止渴", "润燥"],
"活血": ["活血", "化瘀", "散瘀", "通络"],
"固表": ["固表", "固涩", "止汗", "收敛"],
"祛风": ["祛风", "疏风", "散风"],
"利水": ["利水", "消肿", "利尿"],
"补虚": ["补虚", "补益", "滋补", "扶正"]
}
for cname in constitution_order:
top_items = dietary_recommendations[cname][:30]
efficacy_stats = defaultdict(int)
for item in top_items:
efficacy_text = item["功效"]
if not efficacy_text:
continue
# 统计功效关键词
for category, keywords in EFFICACY_KEYWORDS_MAP.items():
for kw in keywords:
if kw in efficacy_text:
efficacy_stats[category] += 1
break # 每个药膳每个分类只计1次
dietary_efficacy_stats[cname] = dict(sorted(efficacy_stats.items(), key=lambda x: -x[1]))
top_cats = list(dietary_efficacy_stats[cname].keys())[:5]
print(f"{cname}: {dict(top_cats)}...")
# ===== 7. 来源分析 =====
print("\n开始来源分析...")
source_stats = defaultdict(int)
for cname in constitution_order:
for rec in dietary_recommendations[cname]:
source = rec["来源"]
if source:
# 简化来源名
simple_source = source.strip()
source_stats[simple_source] += 1
top_sources = sorted(source_stats.items(), key=lambda x: -x[1])[:20]
print("TOP来源:")
for s, c in top_sources:
print(f" {s}: {c}次")
# ===== 8. 构建完整JSON输出 =====
print("\n构建JSON输出...")
# 统计摘要
total_recommendations = sum(len(v) for v in dietary_recommendations.values())
total_unique = len(all_shared_dietary)
summary_stats = {
"总药膳数": len(dietary_list),
"体质类型数": len(constitution_order),
"各体质推荐药膳数": {c: len(dietary_recommendations[c]) for c in constitution_order},
"总推荐人次": total_recommendations,
"去重推荐药膳数": total_unique,
"跨体质共享药膳数": shared_count,
"评分阈值": THRESHOLD,
"评分模型因子": {
"功效字段含体质关键字": "+4",
"简介/名称字段含体质关键字": "+3",
"配方字段含体质推荐食材": "+2",
"适宜人群字段含体质关键字": "+3",
"来源字段含经典文献": "+1"
}
}
output_json = {
"体质-药膳推荐": dietary_recommendations,
"体质-药膳功效统计": dietary_efficacy_stats,
"体质-药膳来源统计": dict(top_sources),
"体质-药膳共享分析": {
"数量": shared_count,
"跨体质药膳": cross_constitution_items
},
"统计摘要": summary_stats
}
# 保存JSON
output_dir = os.path.join(BASE_DIR, "03_关联融合/交叉关联分析")
os.makedirs(output_dir, exist_ok=True)
output_path = os.path.join(output_dir, "体质-药膳推荐体系.json")
with open(output_path, "w", encoding="utf-8") as f:
json.dump(output_json, f, ensure_ascii=False, indent=2)
print(f"\nJSON已保存: {output_path}")
print(f"文件大小: {os.path.getsize(output_path) / 1024:.1f} KB")
# ===== 9. 验证JSON完整性 =====
with open(output_path, "r", encoding="utf-8") as f:
verify = json.load(f)
print(f"\n验证JSON:")
print(f" 体质-药膳推荐: {len(verify['体质-药膳推荐'])} 种体质")
for c in constitution_order:
print(f" {c}: {len(verify['体质-药膳推荐'][c])} 条推荐")
print(f" 体质-药膳功效统计: {len(verify['体质-药膳功效统计'])} 种")
print(f" 体质-药膳共享分析: {verify['体质-药膳共享分析']['数量']} 条共享")
print(f" 统计摘要: 总推荐={verify['统计摘要']['总推荐人次']}人次")
print("\n===== 药膳推荐体系构建完成 =====")
@@ -0,0 +1,375 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
构建9种中医体质数据库,并与大医网药膳/中药材/穴位数据建立交叉关联
基于中华中医药学会《中医体质分类与判定》标准
"""
import json
import os
import glob
import re
from collections import defaultdict
# ============================================================
# 第一部分:九种体质详细数据
# ============================================================
constitutions = [
{
"名称": "平和质",
"英文": "Balanced Constitution",
"类型": "平和质",
"总体特征": "阴阳气血调和,以体态适中、面色红润、精力充沛为主要特征",
"形体特征": "体形匀称健壮",
"常见表现": "面色红润,目光有神,精力充沛,睡眠良好,食欲好,二便正常,舌淡红苔薄白,脉和缓有力",
"心理特征": "性格随和开朗",
"发病倾向": "平素患病较少",
"对外界环境适应能力": "对自然环境和社会环境适应能力较强",
"调理原则": "维护平衡,预防为主,饮食有节,劳逸适度",
"饮食宜": ["五谷杂粮", "蔬菜水果", "鱼肉蛋奶", "豆制品", "坚果", "绿茶"],
"饮食忌": ["暴饮暴食", "过度偏食", "过量辛辣", "过量油腻"],
"推荐药膳": ["山药红枣粥", "枸杞菊花茶", "银耳莲子羹", "百合绿豆汤", "茯苓糕"],
"推荐穴位": ["足三里", "涌泉", "百会", "合谷", "三阴交"],
"推荐功法": ["太极拳", "八段锦", "五禽戏", "散步"],
"相关疾病": ["较少患病,以养生保健为主"],
"关联关键字": ["平和", "平衡", "调和", "保健", "养生", "预防"]
},
{
"名称": "气虚质",
"英文": "Qi Deficiency",
"类型": "偏颇质",
"总体特征": "元气不足,以疲乏、气短、自汗等气虚表现为主要特征",
"形体特征": "肌肉松软不实",
"常见表现": "平素语音低弱,气短懒言,容易疲乏,精神不振,易出汗,舌淡红,舌边有齿痕,脉弱",
"心理特征": "性格内向,不喜冒险",
"发病倾向": "易患感冒、内脏下垂等病;病后康复缓慢",
"对外界环境适应能力": "不耐受风、寒、暑、湿邪",
"调理原则": "补气益气,培元固本",
"饮食宜": ["粳米", "糯米", "小米", "山药", "土豆", "大枣", "香菇", "鸡肉", "牛肉", "鳝鱼", "泥鳅", "蜂蜜", "黄豆", "扁豆", "豇豆", "南瓜", "胡萝卜"],
"饮食忌": ["生冷", "油腻", "辛辣", "耗气食物"],
"推荐药膳": ["黄芪炖鸡汤", "山药粥", "四君子汤", "参苓粥", "红枣桂圆茶"],
"推荐穴位": ["足三里", "气海", "关元", "百会", "脾俞"],
"推荐功法": ["八段锦", "五禽戏", "太极拳", "散步"],
"相关疾病": ["感冒", "内脏下垂", "慢性疲劳", "自汗"],
"关联关键字": ["气虚", "气短", "乏力", "自汗", "懒言", "神疲", "补气", "益气", "培元", "固本"]
},
{
"名称": "阳虚质",
"英文": "Yang Deficiency",
"类型": "偏颇质",
"总体特征": "阳气不足,以畏寒怕冷、手足不温等虚寒表现为主要特征",
"形体特征": "肌肉松软不实",
"常见表现": "平素畏冷,手足不温,喜热饮食,精神不振,舌淡胖嫩,脉沉迟",
"心理特征": "性格多沉静、内向",
"发病倾向": "易患痰饮、肿胀、泄泻等病;感邪易从寒化",
"对外界环境适应能力": "耐夏不耐冬;易感风、寒、湿邪",
"调理原则": "温阳散寒,补肾壮阳",
"饮食宜": ["羊肉", "牛肉", "韭菜", "生姜", "肉桂", "核桃", "栗子", "荔枝", "龙眼", "茴香", "丁香", "花椒", "小茴香"],
"饮食忌": ["生冷", "寒凉", "冰冻", "苦寒食物"],
"推荐药膳": ["当归生姜羊肉汤", "肉桂炖牛肉", "韭菜炒核桃", "附子炖狗肉", "干姜红糖茶"],
"推荐穴位": ["关元", "命门", "肾俞", "神阙", "足三里"],
"推荐功法": ["八段锦", "太极拳", "五禽戏", "站桩"],
"相关疾病": ["痰饮", "肿胀", "泄泻", "阳痿", "宫寒"],
"关联关键字": ["阳虚", "畏寒", "怕冷", "四肢不温", "寒凝", "温阳", "散寒", "壮阳", "肾阳", "脾阳"]
},
{
"名称": "阴虚质",
"英文": "Yin Deficiency",
"类型": "偏颇质",
"总体特征": "阴液亏少,以口燥咽干、手足心热等虚热表现为主要特征",
"形体特征": "体形偏瘦",
"常见表现": "手足心热,口燥咽干,鼻微干,喜冷饮,大便干燥,舌红少津,脉细数",
"心理特征": "性情急躁,外向好动,活泼",
"发病倾向": "易患虚劳、失精、不寐等病;感邪易从热化",
"对外界环境适应能力": "耐冬不耐夏;不耐受暑、热、燥邪",
"调理原则": "滋阴降火,滋补肝肾",
"饮食宜": ["百合", "银耳", "莲子", "枸杞", "黑芝麻", "鸭肉", "甲鱼", "龟肉", "海参", "牡蛎", "蜂蜜", "梨", "甘蔗", "枇杷", "桑葚", "黑木耳"],
"饮食忌": ["辛辣", "燥热", "油炸", "烧烤", "羊肉"],
"推荐药膳": ["百合银耳羹", "枸杞菊花茶", "冰糖炖雪梨", "沙参玉竹老鸭汤", "生地黄粥"],
"推荐穴位": ["太溪", "三阴交", "涌泉", "照海", "肾俞"],
"推荐功法": ["太极拳", "八段锦", "瑜伽", "静坐"],
"相关疾病": ["虚劳", "不寐", "消渴", "便秘", "盗汗"],
"关联关键字": ["阴虚", "口干", "咽干", "手足心热", "潮热", "盗汗", "滋阴", "降火", "生津", "润燥"]
},
{
"名称": "痰湿质",
"英文": "Phlegm-Dampness Constitution",
"类型": "偏颇质",
"总体特征": "痰湿凝聚,以形体肥胖、腹部肥满、口黏苔腻等痰湿表现为主要特征",
"形体特征": "体形肥胖,腹部肥满松软",
"常见表现": "面部皮肤油脂较多,多汗且黏,胸闷,痰多,口黏腻或甜,喜食肥甘甜黏,苔腻,脉滑",
"心理特征": "性格偏温和、稳重,多善于忍耐",
"发病倾向": "易患消渴、中风、胸痹等病",
"对外界环境适应能力": "对梅雨季节及湿重环境适应能力差",
"调理原则": "健脾祛湿,化痰降浊",
"饮食宜": ["薏苡仁", "冬瓜", "白萝卜", "赤小豆", "茯苓", "荷叶", "山楂", "陈皮", "燕麦", "荞麦", "海带", "紫菜", "生姜"],
"饮食忌": ["肥甘厚味", "甜腻", "油炸", "生冷", "酒肉"],
"推荐药膳": ["薏苡仁冬瓜汤", "茯苓粥", "陈皮茶", "山楂荷叶茶", "赤小豆鲤鱼汤"],
"推荐穴位": ["丰隆", "足三里", "阴陵泉", "脾俞", "中脘"],
"推荐功法": ["八段锦", "太极拳", "五禽戏", "散步", "快走"],
"相关疾病": ["消渴", "中风", "胸痹", "肥胖", "高脂血症"],
"关联关键字": ["痰湿", "肥胖", "痰多", "胸闷", "苔腻", "健脾", "祛湿", "化痰", "降浊", "利水"]
},
{
"名称": "湿热质",
"英文": "Damp-Heat Constitution",
"类型": "偏颇质",
"总体特征": "湿热内蕴,以面垢油光、口苦、苔黄腻等湿热表现为主要特征",
"形体特征": "形体中等或偏胖",
"常见表现": "面垢油光,易生痤疮,口苦口干,身重困倦,大便黏滞不畅或燥结,小便短黄,男性易阴囊潮湿,女性易带下增多,舌质偏红,苔黄腻,脉滑数",
"心理特征": "性格多急躁易怒",
"发病倾向": "易患疮疖、黄疸、热淋等病",
"对外界环境适应能力": "对夏末秋初湿热气候,湿重或气温偏高环境较难适应",
"调理原则": "清热利湿,解毒化浊",
"饮食宜": ["绿豆", "赤小豆", "薏苡仁", "冬瓜", "苦瓜", "黄瓜", "芹菜", "蒲公英", "马齿苋", "菊花", "金银花", "莲藕", "茭白"],
"饮食忌": ["辛辣", "油腻", "甜腻", "温燥", "热性食物"],
"推荐药膳": ["绿豆薏苡仁汤", "冬瓜赤小豆汤", "苦瓜排骨汤", "菊花茶", "金银花茶"],
"推荐穴位": ["曲池", "合谷", "阴陵泉", "丰隆", "内庭"],
"推荐功法": ["八段锦", "太极拳", "游泳", "慢跑"],
"相关疾病": ["痤疮", "湿疹", "黄疸", "热淋", "带下病"],
"关联关键字": ["湿热", "口苦", "苔黄腻", "痤疮", "湿疹", "清热", "利湿", "解毒", "化浊", "泻火"]
},
{
"名称": "血瘀质",
"英文": "Blood Stasis Constitution",
"类型": "偏颇质",
"总体特征": "血行不畅,以肤色晦暗、舌质紫暗等血瘀表现为主要特征",
"形体特征": "胖瘦均见",
"常见表现": "肤色晦暗,色素沉着,容易出现瘀斑,口唇暗淡,舌暗或有瘀点,舌下络脉紫暗或增粗,脉涩",
"心理特征": "易烦,健忘",
"发病倾向": "易患癥瘕及痛证、血证等",
"对外界环境适应能力": "不耐受寒邪",
"调理原则": "活血化瘀,行气通络",
"饮食宜": ["山楂", "黑豆", "黑木耳", "醋", "玫瑰花", "红糖", "红酒", "桃仁", "油菜", "茄子", "藕", "香菇", "海带"],
"饮食忌": ["寒凉", "冰冻", "收敛", "酸涩食物"],
"推荐药膳": ["山楂红糖水", "桃仁粥", "黑木耳红枣汤", "玫瑰花茶", "川芎炖鱼头"],
"推荐穴位": ["血海", "合谷", "三阴交", "膈俞", "太冲"],
"推荐功法": ["八段锦", "太极拳", "五禽戏", "舞蹈", "快走"],
"相关疾病": ["癥瘕", "痛经", "闭经", "胸痹", "中风"],
"关联关键字": ["血瘀", "瘀血", "面色晦暗", "瘀斑", "刺痛", "活血", "化瘀", "行气", "通络", "消癥"]
},
{
"名称": "气郁质",
"英文": "Qi Depression Constitution",
"类型": "偏颇质",
"总体特征": "气机郁滞,以神情抑郁、忧虑脆弱等气郁表现为主要特征",
"形体特征": "形体瘦者为多",
"常见表现": "神情抑郁,情感脆弱,烦闷不乐,舌淡红,苔薄白,脉弦",
"心理特征": "性格内向不稳定、敏感多虑",
"发病倾向": "易患脏躁、梅核气、百合病及郁证等",
"对外界环境适应能力": "对精神刺激适应能力较差;不适应阴雨天气",
"调理原则": "疏肝解郁,理气调中",
"饮食宜": ["柑橘", "佛手", "玫瑰花", "小麦", "大麦", "荞麦", "香橼", "橙子", "柚", "洋葱", "大蒜", "萝卜", "茴香"],
"饮食忌": ["辛辣", "咖啡", "浓茶", "高糖", "高脂"],
"推荐药膳": ["玫瑰花茶", "佛手炖瘦肉", "甘麦大枣汤", "橘皮粥", "陈皮柠檬茶"],
"推荐穴位": ["太冲", "肝俞", "期门", "膻中", "内关"],
"推荐功法": ["八段锦", "太极拳", "瑜伽", "散步", "舞蹈"],
"相关疾病": ["郁证", "脏躁", "梅核气", "失眠", "月经不调"],
"关联关键字": ["气郁", "抑郁", "焦虑", "胸闷", "胁痛", "疏肝", "解郁", "理气", "调中", "安神"]
},
{
"名称": "特禀质",
"英文": "Intrinsic Constitution / Allergy Constitution",
"类型": "偏颇质",
"总体特征": "先天失常,以生理缺陷、过敏反应等为主要特征",
"形体特征": "无特殊,或有畸形,或有先天生理缺陷",
"常见表现": "过敏体质者常见哮喘、风团、咽痒、鼻塞、喷嚏等;患遗传性疾病者有垂直遗传、先天性、家族性特征",
"心理特征": "随禀质不同情况各异",
"发病倾向": "过敏体质者易患哮喘、荨麻疹、花粉症及药物过敏等;遗传性疾病如血友病、先天愚型等",
"对外界环境适应能力": "适应能力差,如过敏体质者对易致过敏季节适应能力差,易引发宿疾",
"调理原则": "益气固表,养血祛风",
"饮食宜": ["黄芪", "白术", "防风", "红枣", "蜂蜜", "山药", "人参", "灵芝", "薏苡仁", "莲子", "糯米", "花生"],
"饮食忌": ["海鲜", "虾蟹", "牛羊肉", "发物", "辛辣", "生冷", "过敏源食物"],
"推荐药膳": ["黄芪防风茶", "灵芝红枣汤", "山药排骨汤", "玉屏风茶", "蜂蜜柠檬水"],
"推荐穴位": ["足三里", "肺俞", "肾俞", "气海", "曲池"],
"推荐功法": ["八段锦", "太极拳", "散步", "瑜伽"],
"相关疾病": ["哮喘", "荨麻疹", "过敏性鼻炎", "花粉症", "湿疹", "食物过敏"],
"关联关键字": ["过敏", "哮喘", "荨麻疹", "鼻炎", "风团", "益气", "固表", "祛风", "抗敏", "特禀"]
}
]
# ============================================================
# 第二部分:创建交叉关联
# ============================================================
BASE_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
def load_json_files(directory):
"""Load all JSON files from a directory."""
data = []
if not os.path.exists(directory):
print(f"WARNING: Directory not found: {directory}")
return data
for fpath in sorted(glob.glob(os.path.join(directory, "*.json"))):
try:
with open(fpath, "r", encoding="utf-8") as f:
d = json.load(f)
data.append(d)
except Exception as e:
print(f"Error loading {fpath}: {e}")
return data
def match_keywords_in_text(keywords, text):
"""Count how many keywords appear in the text (case-insensitive)."""
if not text:
return 0
text_lower = text.lower()
count = 0
for kw in keywords:
if kw.lower() in text_lower:
count += 1
return count
def build_diet_association(constitutions_data):
"""体质↔药膳关联: match 功效 and 简介 fields with constitution keywords."""
diet_dir = os.path.join(BASE_DIR, "01_来源数据", "药膳食疗")
diets = load_json_files(diet_dir)
print(f"Loaded {len(diets)} diet entries")
result = {}
for const in constitutions_data:
name = const["名称"]
keywords = const["关联关键字"]
matched = []
for d in diets:
# Combine 功效 and 简介 fields for matching
text = ""
for field in ["功效", "简介", "名称"]:
if field in d and d[field]:
text += d[field] + " "
score = match_keywords_in_text(keywords, text)
if score > 0:
matched.append({
"名称": d.get("名称", ""),
"来源": d.get("来源", d.get("url", "")),
"功效": d.get("功效", d.get("简介", "")),
"得分": score
})
# Sort by score descending
matched.sort(key=lambda x: x["得分"], reverse=True)
result[name] = matched
print(f" {name}: matched {len(matched)} diet entries")
return result
def build_herb_association(constitutions_data):
"""体质↔药材关联: match 功效作用 field with constitution keywords.
药材名匹配使用最长前缀匹配.
"""
herb_dir = os.path.join(BASE_DIR, "01_来源数据", "中药材")
herbs = load_json_files(herb_dir)
print(f"Loaded {len(herbs)} herb entries")
# Build a set of all herb names for longest-prefix matching
all_herb_names = sorted([h.get("名称", "").strip() for h in herbs if h.get("名称")], key=len, reverse=True)
result = {}
for const in constitutions_data:
name = const["名称"]
keywords = const["关联关键字"]
# Also include the 调理原则 keywords
principle = const.get("调理原则", "")
all_keywords = keywords + [principle]
matched = []
for h in herbs:
herb_name = h.get("名称", "")
# Match in 功效作用 field (which contains 功能, 主治, 用法用量 etc.)
text = ""
eff = h.get("功效作用", {})
if isinstance(eff, dict):
for v in eff.values():
if isinstance(v, str):
text += v + " "
elif isinstance(eff, str):
text = eff
# Also match in 简介
if "简介" in h:
text += h["简介"] + " "
score = match_keywords_in_text(all_keywords, text)
if score > 0:
matched.append({
"名称": herb_name,
"来源": h.get("url", ""),
"功效作用": text[:200] + ("..." if len(text) > 200 else ""),
"得分": score
})
matched.sort(key=lambda x: x["得分"], reverse=True)
result[name] = matched
print(f" {name}: matched {len(matched)} herb entries")
return result
def build_acupoint_association(constitutions_data):
"""体质↔穴位关联: match 主治 and 详细主治 fields with constitution disease/symptom keywords."""
acu_dir = os.path.join(BASE_DIR, "01_来源数据", "针灸穴位")
acupoints = load_json_files(acu_dir)
print(f"Loaded {len(acupoints)} acupoint entries")
result = {}
for const in constitutions_data:
name = const["名称"]
# Use disease keywords and symptom keywords from constitution
keywords = const["关联关键字"] + const.get("相关疾病", [])
matched = []
for a in acupoints:
text = ""
for field in ["主治", "详细主治", "功能", "功能作用", "简介", "名称"]:
if field in a and a[field]:
text += a[field] + " "
score = match_keywords_in_text(keywords, text)
if score > 0:
matched.append({
"名称": a.get("名称", ""),
"来源": a.get("url", ""),
"主治": a.get("主治", a.get("详细主治", "")),
"得分": score
})
matched.sort(key=lambda x: x["得分"], reverse=True)
result[name] = matched
print(f" {name}: matched {len(matched)} acupoint entries")
return result
def main():
# Step 1: 保存九种体质详细数据
output_dir = os.path.join(BASE_DIR, "02_加工数据", "中医体质")
os.makedirs(output_dir, exist_ok=True)
constitutions_path = os.path.join(output_dir, "九种体质详细数据.json")
with open(constitutions_path, "w", encoding="utf-8") as f:
json.dump(constitutions, f, ensure_ascii=False, indent=2)
print(f"Saved constitution data to {constitutions_path}")
# Step 2: 构建交叉关联
print("\n=== Building Diet Associations ===")
diet_assoc = build_diet_association(constitutions)
print("\n=== Building Herb Associations ===")
herb_assoc = build_herb_association(constitutions)
print("\n=== Building Acupoint Associations ===")
acu_assoc = build_acupoint_association(constitutions)
# 计算统计
total_diet = sum(len(v) for v in diet_assoc.values())
total_herb = sum(len(v) for v in herb_assoc.values())
total_acu = sum(len(v) for v in acu_assoc.values())
cross_ref = {
"体质-药膳关联": diet_assoc,
"体质-药材关联": herb_assoc,
"体质-穴位关联": acu_assoc,
"总结统计": {
"体质数": 9,
"总关联药膳数": total_diet,
"总关联药材数": total_herb,
"总关联穴位数": total_acu
}
}
cross_path = os.path.join(output_dir, "体质-大医网交叉关联.json")
with open(cross_path, "w", encoding="utf-8") as f:
json.dump(cross_ref, f, ensure_ascii=False, indent=2)
print(f"\nSaved cross-reference to {cross_path}")
print(f"Summary: {total_diet} diet entries, {total_herb} herb entries, {total_acu} acupoint entries matched")
if __name__ == "__main__":
main()
@@ -0,0 +1,429 @@
#!/usr/bin/env python3
"""
大医网OCR/文本杂质清洗脚本
-----------------------------
清洗7个核心库的JSON源文件中的OCR形近错字和格式杂质。
直接修改源文件(明确OCR校正,不改变语义)。
生成清洗报告到 04_分析报告/数据质量/OCR清洗报告.md
"""
import json
import os
import re
from collections import defaultdict, OrderedDict
# ============================================================
# 配置
# ============================================================
BASE_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
SOURCE_DIR = os.path.join(BASE_DIR, "01_来源数据")
SUBDIRS = ["疾病", "方剂", "中药材", "针灸穴位", "症状", "术语", "药膳食疗"]
REPORT_DIR = os.path.join(BASE_DIR, "04_分析报告", "数据质量")
REPORT_FILE = os.path.join(REPORT_DIR, "OCR清洗报告.md")
# ============================================================
# 清洗规则定义(按优先级)
# 每条规则: (pattern, replacement, description, is_regex)
# ============================================================
CLEAN_RULES = [
# --- OCR形近错字(注意负向零宽断言)---
(re.compile(r'(?<!酸)枣仁'), '酸枣仁', '枣仁→酸枣仁', True),
(re.compile(r'(?<!耳)白木(?!耳)'), '白术', '白木→白术 (排除白木耳)', True),
(re.compile(r'(?<!薏)苡仁'), '薏苡仁', '苡仁→薏苡仁', True),
(re.compile(r'枝子'), '栀子', '枝子→栀子', True), # 仅匹配"枝子"整体词
(re.compile(r'构杞'), '枸杞', '构杞→枸杞', True),
(re.compile(r'构杞子'), '枸杞子', '构杞子→枸杞子', True),
(re.compile(r'当妇'), '当归', '当妇→当归', True),
(re.compile(r'意苡仁'), '薏苡仁', '意苡仁→薏苡仁', True),
(re.compile(r'获苓'), '茯苓', '获苓→茯苓', True),
(re.compile(r'川萼'), '川芎', '川萼→川芎', True),
(re.compile(r'黄茂'), '黄芪', '黄茂→黄芪', True),
(re.compile(r'甘革'), '甘草', '甘革→甘草', True),
(re.compile(r'灵艺'), '灵芝', '灵艺→灵芝', True),
(re.compile(r'大赉'), '大枣', '大赉→大枣', True),
(re.compile(r'干娈'), '干姜', '干娈→干姜', True),
# --- 格式杂质 ---
(re.compile(r'\u3000'), ' ', '全角空格→半角', True),
(re.compile(r'[\u200b\u200c\u200d\ufeff]'), '', '零宽字符删除', True),
(re.compile(r' +'), ' ', '多空格合并', True),
]
def apply_clean_rules(text, stats_counter, file_stat):
"""对单个字符串应用所有清洗规则,记录每类错误的修正次数"""
if not isinstance(text, str):
return text, 0
total_changes = 0
for pattern, replacement, desc, is_regex in CLEAN_RULES:
if is_regex:
new_text, count = pattern.subn(replacement, text)
if count > 0:
stats_counter[desc] += count
file_stat[desc] += count
total_changes += count
text = new_text
else:
# 普通字符串替换(保留备用)
count = text.count(pattern)
if count > 0:
text = text.replace(pattern, replacement)
stats_counter[desc] += count
file_stat[desc] += count
total_changes += count
return text, total_changes
def walk_and_clean(data, stats_counter, file_stat):
"""递归遍历嵌套的dict/list结构,清洗所有字符串字段和键名"""
total_changes = 0
if isinstance(data, dict):
# 先处理键名:可能含有零宽字符等杂质
cleaned_keys = {}
for key in list(data.keys()):
if isinstance(key, str):
new_key, changes = apply_clean_rules(key, stats_counter, file_stat)
if changes > 0:
cleaned_keys[key] = new_key
total_changes += changes
# 重建dict(如果键名有变化)
if cleaned_keys:
for old_key, new_key in cleaned_keys.items():
data[new_key] = data.pop(old_key)
# 处理值
for key in list(data.keys()):
val = data[key]
if isinstance(val, str):
new_val, changes = apply_clean_rules(val, stats_counter, file_stat)
if changes > 0:
data[key] = new_val
total_changes += changes
elif isinstance(val, (dict, list)):
total_changes += walk_and_clean(val, stats_counter, file_stat)
elif isinstance(data, list):
for i in range(len(data)):
val = data[i]
if isinstance(val, str):
new_val, changes = apply_clean_rules(val, stats_counter, file_stat)
if changes > 0:
data[i] = new_val
total_changes += changes
elif isinstance(val, (dict, list)):
total_changes += walk_and_clean(val, stats_counter, file_stat)
return total_changes
def clean_file(filepath, stats_counter, file_changes_log):
"""
清洗单个JSON文件:
1. json.load 读取
2. 递归遍历清洗
3. json.dump 写回
返回 (修改文件数:0或1, 修正总次数)
"""
try:
with open(filepath, 'r', encoding='utf-8') as f:
data = json.load(f)
except (json.JSONDecodeError, UnicodeDecodeError) as e:
print(f" [!] 跳过无法解析的JSON: {filepath} — {e}")
return 0, 0
file_stat = defaultdict(int)
changes = walk_and_clean(data, stats_counter, file_stat)
if changes == 0:
return 0, 0
# 记录此文件的修改详情
file_changes_log.append({
'filepath': filepath,
'changes': changes,
'detail': {k: v for k, v in file_stat.items() if v > 0}
})
# 写回
try:
with open(filepath, 'w', encoding='utf-8') as f:
json.dump(data, f, ensure_ascii=False, indent=2)
return 1, changes
except Exception as e:
print(f" [!!] 写回失败: {filepath} — {e}")
return 0, 0
def scan_for_residual(data):
"""扫描数据中是否仍有任何未清洗的匹配项"""
residuals = defaultdict(int)
def _scan(d):
if isinstance(d, dict):
for v in d.values():
_scan(v)
elif isinstance(d, list):
for v in d:
_scan(v)
elif isinstance(d, str):
for pattern, replacement, desc, is_regex in CLEAN_RULES:
if is_regex:
matches = pattern.findall(d)
if matches:
residuals[desc] += len(matches)
else:
cnt = d.count(pattern)
if cnt:
residuals[desc] += cnt
_scan(data)
return residuals
def validate_residuals():
"""清洗后重新扫描所有JSON文件,验证零残留"""
residual_counts = defaultdict(int)
residual_files = defaultdict(list)
for subdir in SUBDIRS:
dirpath = os.path.join(SOURCE_DIR, subdir)
if not os.path.isdir(dirpath):
continue
for fname in os.listdir(dirpath):
if not fname.endswith('.json'):
continue
fpath = os.path.join(dirpath, fname)
try:
with open(fpath, 'r', encoding='utf-8') as f:
data = json.load(f)
except:
continue
res = scan_for_residual(data)
if res:
for desc, cnt in res.items():
residual_counts[desc] += cnt
residual_files[desc].append(fname)
return residual_counts, residual_files
def get_dataset_name(subdir):
"""映射子目录名到中文数据集名"""
mapping = {
'疾病': '疾病库',
'方剂': '方剂库',
'中药材': '中药材库',
'针灸穴位': '针灸穴位库',
'症状': '症状库',
'术语': '术语库',
'药膳食疗': '药膳食疗库',
}
return mapping.get(subdir, subdir)
def relative_path(filepath):
"""将绝对路径转为相对路径(用于报告)"""
return filepath.replace(BASE_DIR, '.../大医网')
def main():
print("=" * 70)
print(" 大医网OCR/文本杂质清洗")
print("=" * 70)
# 全局统计
# stats_counter: { desc: total_count }
stats_counter = defaultdict(int)
# 按数据集的统计: { subdir: { 'files_cleaned': N, 'total_changes': N, 'detail': {desc: count} } }
dataset_stats = {}
# 所有修改的文件日志
all_file_changes_log = []
for subdir in SUBDIRS:
dirpath = os.path.join(SOURCE_DIR, subdir)
if not os.path.isdir(dirpath):
print(f"\n [!] 目录不存在: {dirpath}")
continue
print(f"\n{'─'*50}")
print(f" 处理: {subdir}/")
print(f"{'─'*50}")
json_files = sorted([f for f in os.listdir(dirpath) if f.endswith('.json')])
total_files = len(json_files)
files_cleaned = 0
total_changes_subdir = 0
subdir_detail = defaultdict(int)
for idx, fname in enumerate(json_files):
fpath = os.path.join(dirpath, fname)
file_changes_log_entry = []
cleaned_flag, changes = clean_file(fpath, stats_counter, file_changes_log_entry)
files_cleaned += cleaned_flag
total_changes_subdir += changes
if file_changes_log_entry:
all_file_changes_log.extend(file_changes_log_entry)
for desc, cnt in file_changes_log_entry[0]['detail'].items():
subdir_detail[desc] += cnt
# 进度显示
if (idx + 1) % 200 == 0 or idx == 0 or (idx + 1) == total_files:
print(f" 进度: {idx+1}/{total_files} | 已清洗: {files_cleaned} 个文件 | 修正: {total_changes_subdir} 处")
dataset_stats[subdir] = {
'total_files': total_files,
'files_cleaned': files_cleaned,
'total_changes': total_changes_subdir,
'detail': dict(subdir_detail),
}
ds_name = get_dataset_name(subdir)
print(f" ✓ {ds_name}: {files_cleaned}/{total_files} 个文件被修改,共修正 {total_changes_subdir} 处")
# ============================================================
# 汇总
# ============================================================
print("\n" + "=" * 70)
print(" 清洗完成 - 汇总统计")
print("=" * 70)
total_files_all = sum(s['total_files'] for s in dataset_stats.values())
total_cleaned_all = sum(s['files_cleaned'] for s in dataset_stats.values())
total_changes_all = sum(s['total_changes'] for s in dataset_stats.values())
print(f" 总文件数: {total_files_all}")
print(f" 被修改文件: {total_cleaned_all}")
print(f" 总修正处: {total_changes_all}")
print(f"\n 错误类型分布:")
for desc, cnt in sorted(stats_counter.items(), key=lambda x: -x[1]):
print(f" {desc}: {cnt}")
# ============================================================
# 清洗后验证 - 零残留
# ============================================================
print("\n" + "=" * 70)
print(" 清洗后残留验证")
print("=" * 70)
residual_counts, residual_files = validate_residuals()
if residual_counts:
print(f" ⚠ 发现 {sum(residual_counts.values())} 处残留:")
for desc, cnt in sorted(residual_counts.items(), key=lambda x: -x[1]):
print(f" {desc}: {cnt} 处 (共 {len(residual_files.get(desc, []))} 个文件)")
residual_status = "有残留"
else:
print(" ✓ 零残留!所有规则均已正确应用。")
residual_status = "零残留 ✓"
# ============================================================
# 生成清洗报告
# ============================================================
print(f"\n 生成报告 → {REPORT_FILE}")
os.makedirs(REPORT_DIR, exist_ok=True)
lines = []
lines.append("# OCR/文本杂质清洗报告")
lines.append("")
lines.append(f"**数据目录:** `{SOURCE_DIR}`")
lines.append(f"**子目录:** 疾病/ 方剂/ 中药材/ 针灸穴位/ 症状/ 术语/ 药膳食疗")
lines.append(f"**清洗时间:** {__import__('datetime').datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## 一、汇总统计")
lines.append("")
lines.append("| 数据集名 | 总文件数 | 清洗文件数 | 总修正处 | 错误类型分布 |")
lines.append("|----------|----------|------------|----------|--------------|")
for subdir in SUBDIRS:
if subdir not in dataset_stats:
continue
s = dataset_stats[subdir]
ds_name = get_dataset_name(subdir)
detail_str = "; ".join([f"{d}: {c}" for d, c in sorted(s['detail'].items(), key=lambda x: -x[1])])
if not detail_str:
detail_str = "-"
lines.append(f"| {ds_name} | {s['total_files']} | {s['files_cleaned']} | {s['total_changes']} | {detail_str} |")
lines.append(f"| **合计** | **{total_files_all}** | **{total_cleaned_all}** | **{total_changes_all}** | **{'; '.join([f'{d}: {c}' for d, c in sorted(stats_counter.items(), key=lambda x: -x[1])])}** |")
lines.append("")
lines.append("## 二、清洗规则详情")
lines.append("")
lines.append("| 优先级 | 规则 | 模式 | 替换为 | 修正次数 |")
lines.append("|--------|------|------|--------|----------|")
for i, (pattern, replacement, desc, is_regex) in enumerate(CLEAN_RULES, 1):
pat_str = pattern.pattern if is_regex else pattern
cnt = stats_counter.get(desc, 0)
lines.append(f"| {i} | {desc} | `{pat_str}` | `{replacement}` | {cnt} |")
lines.append("")
lines.append("## 三、清洗后验证")
lines.append("")
if residual_counts:
lines.append(f"### ⚠ 残留检测结果:{sum(residual_counts.values())} 处残留")
lines.append("")
lines.append("| 规则 | 残留数 | 涉及文件数 |")
lines.append("|------|--------|------------|")
for desc, cnt in sorted(residual_counts.items(), key=lambda x: -x[1]):
lines.append(f"| {desc} | {cnt} | {len(residual_files.get(desc, []))} |")
lines.append("")
lines.append("**注意:** 残留可能存在于非字符串字段(如JSON key)或特殊格式中,需人工复核。")
else:
lines.append("**验证结果:零残留 ✓** — 所有规则均已正确应用,无未处理的匹配项。")
lines.append("")
lines.append("## 四、错误类型分布(全部)")
lines.append("")
lines.append("| 错误类型 | 修正次数 | 占比 |")
lines.append("|----------|----------|------|")
total = sum(stats_counter.values())
for desc, cnt in sorted(stats_counter.items(), key=lambda x: -x[1]):
pct = f"{cnt/total*100:.1f}%" if total > 0 else "-"
lines.append(f"| {desc} | {cnt} | {pct} |")
lines.append(f"| **总计** | **{total}** | **100%** |")
lines.append("")
lines.append("## 五、被修改文件列表")
lines.append("")
lines.append(f"共 {len(all_file_changes_log)} 个文件被修改:")
lines.append("")
# 按数据集分组显示
by_dataset = defaultdict(list)
for entry in all_file_changes_log:
fpath = entry['filepath']
for subdir in SUBDIRS:
if f'/01_来源数据/{subdir}/' in fpath:
by_dataset[subdir].append(entry)
break
for subdir in SUBDIRS:
entries = by_dataset.get(subdir, [])
if not entries:
continue
ds_name = get_dataset_name(subdir)
lines.append(f"### {ds_name}({len(entries)} 个文件)")
lines.append("")
lines.append("| 文件名 | 修正总数 | 修正详情 |")
lines.append("|--------|----------|----------|")
for entry in sorted(entries, key=lambda x: -x['changes']):
fname = os.path.basename(entry['filepath'])
detail_str = "; ".join([f"{d}: {c}" for d, c in sorted(entry['detail'].items(), key=lambda x: -x[1])])
lines.append(f"| {fname} | {entry['changes']} | {detail_str} |")
lines.append("")
# 写入报告
report_content = "\n".join(lines)
with open(REPORT_FILE, 'w', encoding='utf-8') as f:
f.write(report_content)
print(f"\n{'='*70}")
print(f" 报告已生成: {REPORT_FILE}")
print(f" 总文件: {total_files_all}")
print(f" 被修改: {total_cleaned_all}")
print(f" 总修正: {total_changes_all}")
print(f" 残留验证: {residual_status}")
print(f"{'='*70}")
return total_files_all, total_cleaned_all, total_changes_all, residual_status
if __name__ == "__main__":
main()
@@ -0,0 +1,452 @@
#!/usr/bin/env python3
"""
生成完整MD报告:体质数据挖掘报告
"""
import json
import os
BASE_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
# 加载数据
with open(os.path.join(BASE_DIR, "02_加工数据/中医体质/九种体质详细数据.json"), "r", encoding="utf-8") as f:
constitutions = json.load(f)
constitution_dict = {c["名称"]: c for c in constitutions}
with open(os.path.join(BASE_DIR, "03_关联融合/交叉关联分析/体质-药膳推荐体系.json"), "r", encoding="utf-8") as f:
dietary_data = json.load(f)
with open(os.path.join(BASE_DIR, "02_加工数据/中医体质/体质-大医网交叉关联.json"), "r", encoding="utf-8") as f:
cross_data = json.load(f)
with open(os.path.join(BASE_DIR, "03_关联融合/交叉关联分析/体质-疾病症状挖掘.json"), "r", encoding="utf-8") as f:
disease_data = json.load(f)
constitution_order = ["气虚质", "阳虚质", "阴虚质", "痰湿质", "湿热质", "血瘀质", "气郁质", "特禀质", "平和质"]
# ---- 构建报告 ----
lines = []
lines.append("# 大医网 | 九种中医体质数据挖掘报告")
lines.append("")
lines.append("> **生成日期:** 2025年 \n> **数据来源:** 大医网(www.dayi.org.cn) \n> **分析范围:** 基于大医网疾病、症状、方剂、药材、穴位、药膳六大数据库,对九种中医体质进行系统性交叉数据挖掘。")
lines.append("")
# ===== 一、体质-疾病症状关联分析 =====
lines.append("---")
lines.append("## 一、体质-疾病症状关联分析")
lines.append("")
lines.append("### 分析概述")
lines.append("")
disease_summary = disease_data.get("统计摘要", {})
lines.append(f"基于大医网疾病库与症状库,对9种中医体质进行疾病-症状关联挖掘。共关联疾病{disease_summary.get('总关联疾病数(去重)', 'N')}种(去重),总关联频次{disease_summary.get('总关联疾病数(含重复)', 'N')}次。阈值评分≥3.0。")
lines.append("")
# 各体质疾病TOP3
lines.append("### 各体质TOP3关联疾病")
lines.append("")
lines.append("| 体质类型 | TOP1 | TOP2 | TOP3 |")
lines.append("|---------|------|------|------|")
disease_map = disease_data.get("体质-疾病", {})
for cname in constitution_order:
items = disease_map.get(cname, [])
top3 = items[:3]
names = [x.get("名称", "") for x in top3]
if not any(names):
# Try alternate key
names = [x.get("疾病名称", "") for x in top3]
row = [cname] + (names + ["", "", ""])[:3]
lines.append(f"| {' | '.join(row)} |")
lines.append("")
# 各体质症状TOP3
lines.append("### 各体质TOP3高频症状")
lines.append("")
lines.append("| 体质类型 | TOP1 | TOP2 | TOP3 |")
lines.append("|---------|------|------|------|")
symptom_map = disease_data.get("体质-症状频率", {})
for cname in constitution_order:
items = symptom_map.get(cname, [])
top3 = items[:3]
names = [f"{x[0]}({x[1]}次)" if isinstance(x, list) else str(x) for x in top3]
row = [cname] + (names + ["", "", ""])[:3]
lines.append(f"| {' | '.join(row)} |")
lines.append("")
# 各体质关联疾病数
lines.append("### 各体质关联疾病数量分布")
lines.append("")
lines.append("| 体质类型 | 关联疾病数 |")
lines.append("|---------|-----------|")
disease_counts = disease_summary.get("各体质关联疾病数", {})
for cname in constitution_order:
cnt = disease_counts.get(cname, 0)
lines.append(f"| {cname} | {cnt} |")
lines.append("")
# ===== 二、体质-方剂挖掘 =====
lines.append("---")
lines.append("## 二、体质-方剂挖掘")
lines.append("")
lines.append("### 分析概述")
lines.append("")
lines.append("基于大医网方剂库(约1000+方剂),通过功效匹配、主治疾病关联等方式,对每种体质关联方剂数据。详细数据请参见:`03_关联融合/交叉关联分析/体质-药膳推荐体系.json`(与药膳体系整合的关联参考)及`02_加工数据/中医体质/体质-大医网交叉关联.json`。")
lines.append("")
lines.append("> **注:** 方剂关联数据可参考体质详细数据中的推荐药膳/方剂信息(部分体质药膳即源自经典方剂如四君子汤、甘麦大枣汤等)。详细方剂-体质关联挖掘将在后续任务中展开。")
lines.append("")
# ===== 三、体质-药材挖掘 =====
lines.append("---")
lines.append("## 三、体质-药材挖掘")
lines.append("")
lines.append("### 分析概述")
lines.append("")
herb_assoc = cross_data.get("体质-药材关联", {})
total_herb_assoc = cross_data.get("总结统计", {}).get("总关联药材数", 0)
lines.append(f"基于大医网中药材库,通过功效匹配、性味归经关联等方式,关联分析获得各体质推荐药材。总计{total_herb_assoc}条药材-体质关联数据。详细数据参见:`体质-大医网交叉关联.json`。")
lines.append("")
lines.append("### 各体质TOP5推荐药材")
lines.append("")
lines.append("| 体质类型 | TOP1 | TOP2 | TOP3 | TOP4 | TOP5 |")
lines.append("|---------|------|------|------|------|------|")
for cname in constitution_order:
items = herb_assoc.get(cname, [])
top5 = items[:5]
names = [x.get("名称", "") for x in top5]
row = [cname] + names + [""] * (5 - len(names))
lines.append(f"| {' | '.join(row)} |")
lines.append("")
# ===== 四、体质-穴位挖掘 =====
lines.append("---")
lines.append("## 四、体质-穴位挖掘")
lines.append("")
lines.append("### 分析概述")
lines.append("")
acu_assoc = cross_data.get("体质-穴位关联", {})
total_acu_assoc = cross_data.get("总结统计", {}).get("总关联穴位数", 0)
lines.append(f"基于大医网针灸穴位库,通过主治病症匹配、功效关联等方式,关联分析获得各体质推荐穴位。总计{total_acu_assoc}条穴位-体质关联数据。详细数据参见:`体质-大医网交叉关联.json`。")
lines.append("")
lines.append("### 各体质TOP5推荐穴位")
lines.append("")
lines.append("| 体质类型 | TOP1 | TOP2 | TOP3 | TOP4 | TOP5 |")
lines.append("|---------|------|------|------|------|------|")
for cname in constitution_order:
items = acu_assoc.get(cname, [])
top5 = items[:5]
names = [x.get("名称", "") for x in top5]
row = [cname] + names + [""] * (5 - len(names))
lines.append(f"| {' | '.join(row)} |")
lines.append("")
# ===== 五、体质-药膳推荐体系 =====
lines.append("---")
lines.append("## 五、体质-药膳推荐体系")
lines.append("")
lines.append("### 5.1 分析方法")
lines.append("")
lines.append("采用**多因子评分模型**,对药膳食疗库(1000条)进行体质匹配评分:")
lines.append("")
lines.append("| 评分因子 | 权重 | 说明 |")
lines.append("|---------|-----|------|")
lines.append("| 功效字段含体质关键字 | +4 | 药膳功效直接体现调理方向,最高权重 |")
lines.append("| 简介/名称字段含体质关键字 | +3 | 名称和简介中提及调理方向 |")
lines.append("| 配方字段含体质推荐食材 | +2 | 配方含对应体质的推荐性味食材 |")
lines.append("| 适宜人群字段含体质关键字 | +3 | 明确标注适宜某类体质或证型 |")
lines.append("| 来源字段含经典文献 | +1 | 出自经典文献的药膳可信度更高 |")
lines.append("")
lines.append(f"**阈值:≥4.0**,取每种体质TOP30药膳。")
lines.append("")
# 评分结果汇总
lines.append("### 5.2 各体质推荐药膳TOP10")
lines.append("")
for cname in constitution_order:
recs = dietary_data["体质-药膳推荐"].get(cname, [])
top10 = recs[:10]
lines.append(f"#### {cname}")
lines.append("")
lines.append("| 排名 | 药膳名称 | 来源 | 功效 | 评分 | 跨体质共享 |")
lines.append("|------|---------|------|------|:----:|:--------:|")
for i, r in enumerate(top10, 1):
shared = "✅ " + ", ".join(r["共享体质"]) if r["是否共享"] else "—"
lines.append(f"| {i} | {r['名称']} | {r['来源']} | {r['功效']} | {r['得分']} | {shared} |")
lines.append("")
# 功效分类统计
lines.append("### 5.3 药膳功效分类统计")
lines.append("")
lines.append("统计各体质TOP30药膳的功效关键词分布:")
lines.append("")
for cname in constitution_order:
stats = dietary_data["体质-药膳功效统计"].get(cname, {})
top_stats = sorted(stats.items(), key=lambda x: -x[1])[:8]
if top_stats:
stats_str = "、".join([f"**{k}**({v})" for k, v in top_stats])
lines.append(f"- **{cname}**功效关键词:{stats_str}")
lines.append("")
# 来源分析
lines.append("### 5.4 药膳来源文献分析")
lines.append("")
source_stats = dietary_data.get("体质-药膳来源统计", {})
lines.append("各体质推荐药膳的来源文献分布(TOP15):")
lines.append("")
lines.append("| 来源文献 | 出现频次 |")
lines.append("|---------|:-------:|")
for source, cnt in list(source_stats.items())[:15]:
lines.append(f"| {source} | {cnt} |")
lines.append("")
lines.append('其中,经典文献(如《中国药膳大辞典》《中国药膳学》《圣济总录》《饮膳正要》《太平圣惠方》《本草纲目》等)贡献了大量高评分药膳,体现了"药食同源"的深厚传统。')
lines.append("")
# ===== 六、九种体质综合调理方案 =====
lines.append("---")
lines.append("## 六、九种体质综合调理方案")
lines.append("")
for cname in constitution_order:
c = constitution_dict[cname]
recs = dietary_data["体质-药膳推荐"].get(cname, [])
top3_diet = recs[:3] if recs else []
# TOP3疾病
disease_items = disease_data.get("体质-疾病", {}).get(cname, [])
top3_disease_names = [x.get("名称", "") for x in disease_items[:3]]
if not any(top3_disease_names):
top3_disease_names = c.get("相关疾病", [])[:3]
# TOP3方剂 - Use recommended dietary from constitution data that are also formulas
formula_names = c.get("推荐药膳", [])[:3] # Some 药膳 are classic formulas
# TOP3药材
herb_items = cross_data.get("体质-药材关联", {}).get(cname, [])
top3_herb_names = [x.get("名称", "") for x in herb_items[:3]]
# TOP3穴位
acu_items = cross_data.get("体质-穴位关联", {}).get(cname, [])
top3_acu_names = [x.get("名称", "") for x in acu_items[:3]]
# 推荐药膳TOP3
top3_diet_names = [(d["名称"], d["得分"]) for d in top3_diet]
lines.append(f"### {cname}")
lines.append("")
lines.append(f"**核心病机/特征:** {c.get('总体特征', '')}")
lines.append("")
lines.append(f"**调理原则:** {c.get('调理原则', '')}")
lines.append("")
lines.append("| 调理维度 | 推荐内容 |")
lines.append("|---------|---------|")
lines.append(f"| **核心病机** | {c.get('总体特征', '')} |")
lines.append(f"| **常见表现** | {c.get('常见表现', '')} |")
lines.append(f"| **发病倾向** | {c.get('发病倾向', '')} |")
disease_str = "、".join(top3_disease_names) if top3_disease_names else c.get('发病倾向', '').replace(';', '、')
lines.append(f"| **TOP3疾病** | {disease_str} |")
formula_str = "、".join(formula_names) if formula_names else c.get('推荐药膳', '—')[:3]
lines.append(f"| **TOP3方剂** | {formula_str} |")
herb_str = "、".join(top3_herb_names) if top3_herb_names else "—"
lines.append(f"| **TOP3药材** | {herb_str} |")
acu_str = "、".join(top3_acu_names) if top3_acu_names else "、".join(c.get('推荐穴位', [])[:3])
lines.append(f"| **TOP3穴位** | {acu_str} |")
diet_str = "、".join([f"{x[0]}(评分{x[1]})" for x in top3_diet_names]) if top3_diet_names else "、".join(c.get('推荐药膳', [])[:3])
lines.append(f"| **TOP3推荐药膳** | {diet_str} |")
food_ok = "、".join(c.get('饮食宜', ['未指定']))
lines.append(f"| **饮食原则** | 宜:{food_ok} |")
food_no = "、".join(c.get('饮食忌', ['未指定']))
lines.append(f"| | 忌:{food_no} |")
exercise = "、".join(c.get('推荐功法', ['未指定']))
lines.append(f"| **运动建议** | {exercise} |")
mental = c.get('心理特征', '未指定')
lines.append(f"| **情志调节** | {mental}。宜根据性格特点选择适合的情志调养方式。 |")
lines.append("")
# 完整TOP5药膳列表
lines.append(f"**{cname}完整推荐药膳TOP5:**")
lines.append("")
for i, d in enumerate(recs[:5], 1):
shared_tag = f"【共享:{'、'.join(d['共享体质'])}】" if d['是否共享'] else ""
lines.append(f"{i}. **{d['名称']}**(来源:{d['来源']},功效:{d['功效']},评分:{d['得分']}){shared_tag}")
lines.append("")
# ===== 七、体质间关联分析 =====
lines.append("---")
lines.append("## 七、体质间关联分析")
lines.append("")
# 体质间药膳共享
lines.append("### 7.1 药膳共享分析")
lines.append("")
shared_data = dietary_data["体质-药膳共享分析"]
lines.append(f"在推荐药膳中,共有**{shared_data['数量']}**道药膳同时适用于2种及以上体质类型,体现了体质间调理的交叉性和关联性。")
lines.append("")
lines.append("**跨体质药膳TOP示例:**")
lines.append("")
lines.append("| 药膳名称 | 适用体质 | 功效 | 来源 |")
lines.append("|---------|---------|------|------|")
for item in shared_data["跨体质药膳"][:10]:
lines.append(f"| {item['名称']} | {'、'.join(item['适用体质'])} | {item['功效']} | {item['来源']} |")
lines.append("")
# 体质相似度分析(基于共享药膳)
lines.append("### 7.2 体质相似度分析(基于共享药膳)")
lines.append("")
lines.append("基于跨体质药膳数量,构建体质间关联热度矩阵:")
lines.append("")
# Build similarity matrix
constitution_similarity = {}
for c1 in constitution_order:
constitution_similarity[c1] = {}
for c2 in constitution_order:
if c1 == c2:
constitution_similarity[c1][c2] = 0
continue
# Count shared dietary between c1 and c2
count = 0
for item in shared_data["跨体质药膳"]:
if c1 in item["适用体质"] and c2 in item["适用体质"]:
count += 1
constitution_similarity[c1][c2] = count
# Output matrix
lines.append("| 体质 | " + " | ".join([f"{c:<6}" for c in constitution_order]) + " |")
lines.append("|" + "|".join(["-------" for _ in range(10)]) + "|")
for c1 in constitution_order:
row = f"| {c1:<6}"
for c2 in constitution_order:
row += f" | {constitution_similarity[c1][c2]:>4}"
lines.append(row + " |")
lines.append("")
# Find most connected pairs
pairs = []
for i, c1 in enumerate(constitution_order):
for j, c2 in enumerate(constitution_order):
if i < j:
s = constitution_similarity[c1][c2]
if s > 0:
pairs.append((s, c1, c2))
pairs.sort(reverse=True)
lines.append("**关联最紧密的体质对(共享药膳数排名):**")
lines.append("")
for s, c1, c2 in pairs[:5]:
lines.append(f"- **{c1}** ↔ **{c2}**:共享 {s} 道药膳")
lines.append("")
# 体质间疾病症状关联
lines.append("### 7.3 体质转化与并发倾向")
lines.append("")
lines.append("基于中医体质学理论和数据挖掘结果,体质间存在以下转化/并发规律:")
lines.append("")
lines.append("| 体质A | 体质B | 关联依据 |")
lines.append("|-------|-------|---------|")
lines.append("| 气虚质 | 阳虚质 | \"气虚日久,阳亦衰\",气虚易发展为阳虚,共享温补类药膳 |")
lines.append("| 阴虚质 | 湿热质 | 阴虚生内热,湿热内蕴易伤阴液,均可见热象 |")
lines.append("| 痰湿质 | 湿热质 | 痰湿郁久化热即成湿热,两者病机相通,共享祛湿类药膳 |")
lines.append("| 血瘀质 | 气郁质 | \"气为血之帅\",气郁可致血瘀,血瘀亦可加重气滞 |")
lines.append("| 气虚质 | 特禀质 | 气虚卫外不固,易致过敏反应,均需益气固表 |")
lines.append("| 阴虚质 | 血瘀质 | 阴液不足则血脉不充,久则成瘀 |")
lines.append("")
# 平和质保健方案
lines.append("### 7.4 平和质保健方案")
lines.append("")
c_pinghe = constitution_dict["平和质"]
lines.append(f"**体质特征:** {c_pinghe.get('总体特征', '')}")
lines.append("")
lines.append(f"平和质是阴阳平衡、气血调和的理想体质状态。数据挖掘显示:")
lines.append(f"- 关联疾病数:{disease_summary.get('各体质关联疾病数', {}).get('平和质', 'N')}种(最少)")
lines.append(f"- 推荐保健药膳:{len(dietary_data['体质-药膳推荐'].get('平和质', []))}道")
lines.append("")
lines.append("**日常保健方案:**")
lines.append("")
lines.append("| 保健维度 | 推荐内容 |")
lines.append("|---------|---------|")
lines.append(f"| **饮食** | {', '.join(c_pinghe.get('饮食宜', []))};忌{', '.join(c_pinghe.get('饮食忌', []))} |")
lines.append(f"| **药膳** | {', '.join(c_pinghe.get('推荐药膳', []))} |")
lines.append(f"| **穴位** | {', '.join(c_pinghe.get('推荐穴位', []))} |")
lines.append(f"| **运动** | {', '.join(c_pinghe.get('推荐功法', []))} |")
lines.append(f"| **情志** | {c_pinghe.get('心理特征', '')} |")
lines.append("")
# ===== 八、附录 =====
lines.append("---")
lines.append("## 八、附录")
lines.append("")
lines.append("### 8.1 数据来源")
lines.append("")
lines.append("| 数据源 | 内容 | 数量 | 来源 |")
lines.append("|-------|------|:----:|------|")
lines.append("| 药膳食疗库 | 药膳食疗方 | 1000条 | 大医网 `01_来源数据/药膳食疗/` |")
lines.append("| 体质数据 | 九种体质详细描述 | 9种 | 大医网 `02_加工数据/中医体质/九种体质详细数据.json` |")
lines.append("| 疾病数据 | 疾病-症状关联 | 616+种疾病 | 大医网 `01_来源数据/疾病/` |")
lines.append("| 中药材库 | 中药材信息 | 2004+条关联 | 大医网 `01_来源数据/中药材/` |")
lines.append("| 针灸穴位库 | 穴位信息 | 1432+条关联 | 大医网 `01_来源数据/针灸穴位/` |")
lines.append("| 方剂库 | 经典方剂 | 1000+方剂 | 大医网 `01_来源数据/方剂/` |")
lines.append("")
lines.append("### 8.2 分析方法")
lines.append("")
lines.append("| 分析模块 | 方法 | 说明 |")
lines.append("|---------|------|------|")
lines.append("| 体质-疾病症状 | 关键字匹配评分 | 通过体质关联关键字与疾病症状进行匹配,评分≥3.0为有效关联 |")
lines.append("| 体质-药膳推荐 | 多因子评分模型 | 功效(+4)、简介/名称(+3)、配方食材(+2)、适宜人群(+3)、经典文献(+1),阈值≥4.0 |")
lines.append("| 体质-药材关联 | 功效/性味匹配 | 通过药材功效、性味归经与体质调理原则匹配 |")
lines.append("| 体质-穴位关联 | 主治匹配 | 通过穴位主治与体质常见症状匹配 |")
lines.append("")
lines.append("### 8.3 体质判定标准")
lines.append("")
lines.append("本报告采用中华中医药学会2009年发布的《中医体质分类与判定》标准,将中医体质分为以下9种基本类型:")
lines.append("")
lines.append("| 编号 | 体质类型 | 类型属性 |")
lines.append("|:----:|---------|---------|")
lines.append("| 1 | 平和质 | 健康体质 |")
lines.append("| 2 | 气虚质 | 偏颇体质 |")
lines.append("| 3 | 阳虚质 | 偏颇体质 |")
lines.append("| 4 | 阴虚质 | 偏颇体质 |")
lines.append("| 5 | 痰湿质 | 偏颇体质 |")
lines.append("| 6 | 湿热质 | 偏颇体质 |")
lines.append("| 7 | 血瘀质 | 偏颇体质 |")
lines.append("| 8 | 气郁质 | 偏颇体质 |")
lines.append("| 9 | 特禀质 | 偏颇体质 |")
lines.append("")
lines.append("### 8.4 参考文件")
lines.append("")
lines.append("| 文件路径 | 说明 |")
lines.append("|---------|------|")
lines.append("| `02_加工数据/中医体质/九种体质详细数据.json` | 九种体质详细描述数据 |")
lines.append("| `02_加工数据/中医体质/体质-大医网交叉关联.json` | 体质与药材、穴位、药膳的交叉关联 |")
lines.append("| `03_关联融合/交叉关联分析/体质-疾病症状挖掘.json` | 体质与疾病、症状的关联挖掘 |")
lines.append("| `03_关联融合/交叉关联分析/体质-药膳推荐体系.json` | 本报告药膳推荐体系数据 |")
lines.append("")
lines.append("---")
lines.append("")
lines.append("*报告生成完毕。所有数据基于大医网(www.dayi.org.cn)公开内容挖掘,仅供参考,不构成医疗建议。*")
report_content = "\n".join(lines)
# 保存报告
report_dir = os.path.join(BASE_DIR, "04_分析报告")
os.makedirs(report_dir, exist_ok=True)
report_path = os.path.join(report_dir, "体质数据挖掘报告.md")
with open(report_path, "w", encoding="utf-8") as f:
f.write(report_content)
print(f"报告已保存: {report_path}")
print(f"文件大小: {os.path.getsize(report_path) / 1024:.1f} KB")
print(f"总行数: {len(lines)}")
@@ -0,0 +1,40 @@
import json
def recommend(disease_query, cleaned_idx_path, exercise_idx_path):
with open(cleaned_idx_path, 'r', encoding='utf-8') as f:
idx = json.load(f)
with open(exercise_index_path, 'r', encoding='utf-8') as f:
ex_data = json.load(f)
results = {"food_remedy": [], "exercise_remedy": []}
# 1. Find Food Remedy
# Search in '药膳_flatten_mapping'
for mapping_key, ingredients in idx.get('药膳_mapping_cleaned', {}).items():
# Check if disease name is in the key
if disease_query in mapping_key:
results["food_remedy"].append(f"Use {', '.join(ingredients)} patterns found in {mapping_key}")
# 2. Find Exercise Remedy
for tech in ex_data['techniques']:
# Check benefits or parts
keywords = tech['benefits'] + tech['parts_involved']
for k in keywords:
if k in disease_query or disease_query in k:
results["exercise_remedy"].append(f"Practice {tech['name']} (Targeting: {k})")
return results
if __name__ == "__main__':
import sys
query = sys.argv[1] if len(sys.argv) > 1 else "肩"
c_path = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/索引_药膳疾病关联_cleaned.json"
e_path = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/索引_导引运动处_technique.json"
import sys
sys.path.append(os.path.dirname(os.path.abspath(__file__)))
print(f"--- Recommendation for: {query} ---")
print(json.dumps(recommend(query, c_path, e_path), ensure_ascii=int(0), indent=2))
@@ -0,0 +1,664 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
9种中医体质 × 方剂/药材/穴位 深层挖掘
=========================================
数据源:
- 方剂库 (1000个JSON)
- 中药材库 (1000个JSON)
- 针灸穴位库 (1000个JSON)
- 九种体质详细数据.json
输出:
1. JSON: 体质-方剂药材穴位挖掘.json
2. MD: 体质数据挖掘报告.md (追加在疾病症状部分之后)
"""
import json
import os
import re
import sys
from collections import defaultdict, Counter
from pathlib import Path
# ============================================================
# 0. 路径设置
# ============================================================
BASE = Path("/home/songyi/Documents/ai_agent_scraper_study/data/大医网")
DIR_FANGJI = BASE / "01_来源数据/方剂"
DIR_YAOCAI = BASE / "01_来源数据/中药材"
DIR_XUEWEI = BASE / "01_来源数据/针灸穴位"
CONST_PATH = BASE / "02_加工数据/中医体质/九种体质详细数据.json"
OUT_JSON = BASE / "03_关联融合/交叉关联分析/体质-方剂药材穴位挖掘.json"
OUT_MD = BASE / "04_分析报告/体质数据挖掘报告.md"
# 确保输出目录存在
OUT_JSON.parent.mkdir(parents=True, exist_ok=True)
OUT_MD.parent.mkdir(parents=True, exist_ok=True)
# ============================================================
# 1. 加载体质数据 & 提取关键字
# ============================================================
print("[1/6] 加载体质数据...")
with open(CONST_PATH, "r", encoding="utf-8") as f:
constitutions = json.load(f)
# 9种体质: 名称 + 关联关键字
const_info = {} # { "气虚质": { "keywords": [...], "name": "气虚质" } }
for c in constitutions:
name = c["名称"]
kw = c.get("关联关键字", [])
const_info[name] = {
"keywords": kw,
"name": name,
"data": c
}
const_names = list(const_info.keys())
print(f" 加载 {len(const_names)} 种体质: {const_names}")
# ============================================================
# 2. 加载全部数据
# ============================================================
print("[2/6] 加载方剂、药材、穴位数据...")
def load_json_files(directory):
"""加载目录下所有JSON文件,返回列表"""
items = []
files = sorted(os.listdir(directory))
for fname in files:
if not fname.endswith(".json"):
continue
fpath = os.path.join(directory, fname)
try:
with open(fpath, "r", encoding="utf-8") as f:
data = json.load(f)
items.append(data)
except Exception as e:
print(f" ⚠ 加载失败 {fname}: {e}")
return items
# 方剂
fangji_list = load_json_files(DIR_FANGJI)
print(f" 方剂: {len(fangji_list)} 个")
# 中药材
yaocai_list = load_json_files(DIR_YAOCAI)
print(f" 药材: {len(yaocai_list)} 个")
# 针灸穴位
xuewei_list = load_json_files(DIR_XUEWEI)
print(f" 穴位: {len(xuewei_list)} 个")
# ============================================================
# 3. 构建药材名索引 (用于最长前缀匹配)
# ============================================================
print("[3/6] 构建药材名称索引 (最长前缀匹配)...")
# 收集所有药材名称 (按长度降序排序,避免 金 vs 金银花 误配)
herb_names_all = []
for h in yaocai_list:
name = h.get("名称", "").strip()
if name:
herb_names_all.append(name)
# 也加入别名中的药材名
aliases = h.get("别名", "")
if aliases:
# 别名格式: "木丹、鲜支、越桃、支子、山栀子、栀子"
for a in re.split(r'[、,,]', aliases):
a = a.strip()
if a and len(a) >= 2:
herb_names_all.append(a)
# 去重并按长度降序排序
herb_names_all = sorted(set(herb_names_all), key=lambda x: (-len(x), x))
print(f" 药材名称库: {len(herb_names_all)} 个 (含别名)")
def longest_prefix_match(text, herb_list=None):
"""
在 text 中搜索药材名(最长前缀匹配)
返回匹配到的药材名列表
"""
if herb_list is None:
herb_list = herb_names_all
found = []
# 按长度降序遍历,确保先匹配长名
remaining = text
# 简单方法:对每个已知药材名,看是否出现在text中
# 但为避免重叠,我们使用贪心匹配
for hname in herb_list:
if hname in remaining:
found.append(hname)
# 移除已匹配部分,避免重复/重叠
# 但这里我们不用移除,因为方解-组成中是"、"分隔的,不会有重叠
return found
def extract_herbs_from_formula(formula):
"""
从方剂数据中提取药材名
优先从 方解-组成 字段提取
其次是 组成 字段
最后从 方义 字段中提取
"""
herbs = []
# 1. 尝试 方解-组成
if "方解-组成" in formula:
text = formula["方解-组成"]
# 用、分割
parts = re.split(r'[、,,]', text)
for p in parts:
p = p.strip()
if p:
herbs.append(p)
# 2. 尝试 组成
if not herbs and "组成" in formula:
text = formula["组成"]
# 用、或空格分割,去掉剂量信息
# 例如: 桂枝(别切)四两、甘草(炙)四两、白术三两
# 提取括号前的药材名
parts = re.split(r'[、,,]', text)
for p in parts:
p = p.strip()
# 去掉剂量部分
# 匹配 "药材名(炮制)剂量" 或 "药材名 剂量"
m = re.match(r'^([\u4e00-\u9fff\w]+)', p)
if m:
name = m.group(1).strip()
if name and len(name) >= 2:
herbs.append(name)
# 3. 从方义中提取
if not herbs and "方义" in formula:
text = formula["方义"]
# 使用最长前缀匹配提取药材名
herbs = longest_prefix_match(text)
# 去重
seen = set()
unique_herbs = []
for h in herbs:
if h not in seen:
seen.add(h)
unique_herbs.append(h)
return unique_herbs
# ============================================================
# 4. 评分函数
# ============================================================
def score_formula(formula, keywords):
"""
对方剂进行体质关键字评分
名称 +5, 方义/简介 +3, 歌诀/出处 +2, 其他 +1
返回总分
"""
score = 0.0
# 名称
name = formula.get("名称", "")
for kw in keywords:
if kw in name:
score += 5.0
# 方义 + 简介 (+3)
fangyi = formula.get("方义", "")
intro = formula.get("简介", "")
for field_text in [fangyi, intro]:
if field_text:
for kw in keywords:
count = field_text.count(kw)
score += count * 3.0
# 歌诀 + 出处 (+2)
gejue = formula.get("歌诀", "")
chuchu = formula.get("出处", "")
for field_text in [gejue, chuchu]:
if field_text:
for kw in keywords:
count = field_text.count(kw)
score += count * 2.0
# 其他字段 (+1)
other_fields = ["分类", "运用", "配伍特点", "加减化裁", "化裁方之间的鉴别",
"重要文献摘要", "用法用量", "使用注意", "趣味记忆", "注意事项",
"各家论述", "方解-功效", "方解-主治"]
for field in other_fields:
val = formula.get(field, "")
if val:
for kw in keywords:
count = val.count(kw)
score += count * 1.0
return round(score, 1)
def score_herb(herb, keywords):
"""
对药材进行体质关键字评分
名称 +5, 简介 +3, 功效作用 +3, 性味归经 +2, 临床应用 +2, 其他 +1
"""
score = 0.0
# 名称 (+5)
name = herb.get("名称", "")
for kw in keywords:
if kw in name:
score += 5.0
# 简介 (+3)
intro = herb.get("简介", "")
if intro:
for kw in keywords:
count = intro.count(kw)
score += count * 3.0
# 功效作用 (+3) - 嵌套对象
effect = herb.get("功效作用", {})
if isinstance(effect, dict):
effect_text = " ".join(str(v) for v in effect.values())
elif isinstance(effect, str):
effect_text = effect
else:
effect_text = ""
if effect_text:
for kw in keywords:
count = effect_text.count(kw)
score += count * 3.0
# 性味归经 (+2)
xingwei = herb.get("性味归经", "")
if xingwei:
for kw in keywords:
count = xingwei.count(kw)
score += count * 2.0
# 临床应用 (+2)
clinical = herb.get("临床应用", {})
if isinstance(clinical, dict):
clinical_text = " ".join(str(v) for v in clinical.values())
elif isinstance(clinical, str):
clinical_text = clinical
else:
clinical_text = ""
if clinical_text:
for kw in keywords:
count = clinical_text.count(kw)
score += count * 2.0
# 其他字段 (+1)
other_fields = ["别名", "中文名称", "拉丁文名", "道地产区", "加工炮制",
"药材鉴别", "保存方法", "植物学信息", "动物学信息",
"矿物学信息", "毒性", "相关方剂", "医保类型"]
for field in other_fields:
val = herb.get(field, "")
if isinstance(val, str) and val:
for kw in keywords:
count = val.count(kw)
score += count * 1.0
elif isinstance(val, dict):
val_text = " ".join(str(v) for v in val.values())
if val_text:
for kw in keywords:
count = val_text.count(kw)
score += count * 1.0
return round(score, 1)
def score_xuewei(xuewei, keywords):
"""
对穴位进行体质关键字评分
名称 +5, 主治 +3, 详细主治 +3, 功能 +2, 其他 +1
"""
score = 0.0
# 名称 (+5)
name = xuewei.get("名称", "")
for kw in keywords:
if kw in name:
score += 5.0
# 主治 (+3)
zhuzhi = xuewei.get("主治", "")
if zhuzhi:
for kw in keywords:
count = zhuzhi.count(kw)
score += count * 3.0
# 详细主治 (+3)
detail = xuewei.get("详细主治", "")
if detail:
for kw in keywords:
count = detail.count(kw)
score += count * 3.0
# 功能 (+2)
func = xuewei.get("功能", "") or xuewei.get("功能作用", "")
if func:
for kw in keywords:
count = func.count(kw)
score += count * 2.0
# 其他字段 (+1)
other_fields = ["简介", "隶属", "位置", "定位", "解剖", "详细操作",
"临床运用", "主要配伍", "配伍", "附注", "相关论述",
"出处", "功能作用"]
for field in other_fields:
val = xuewei.get(field, "")
if isinstance(val, str) and val:
for kw in keywords:
count = val.count(kw)
score += count * 1.0
return round(score, 1)
# ============================================================
# 5. 主挖掘流程
# ============================================================
print("[4/6] 执行方剂/药材/穴位评分挖掘...")
# 结果容器
result = {
"体质-方剂": {},
"体质-药材": {},
"体质-穴位": {},
"体质-药材分类统计": {},
"体质-核心药对": {},
"体质-穴位归经统计": {},
"统计摘要": {}
}
# ---- 5A. 方剂挖掘 ----
print(" → 方剂评分...")
for const_name in const_names:
keywords = const_info[const_name]["keywords"]
scored = []
for f in fangji_list:
s = score_formula(f, keywords)
if s >= 3.0:
herbs = extract_herbs_from_formula(f)
scored.append({
"名称": f.get("名称", ""),
"得分": s,
"药材": herbs
})
# 排序取TOP20
scored.sort(key=lambda x: (-x["得分"], x["名称"]))
top20 = scored[:20]
result["体质-方剂"][const_name] = top20
print(f" {const_name}: {len(scored)} 个通过阈值, TOP20已选取")
# ---- 5B. 药材挖掘 ----
print(" → 药材评分...")
for const_name in const_names:
keywords = const_info[const_name]["keywords"]
scored = []
for h in yaocai_list:
s = score_herb(h, keywords)
if s >= 3.0:
# 分类:使用 药材分类 字段
classification = h.get("药材分类", "其他")
xingwei = h.get("性味归经", "")
scored.append({
"名称": h.get("名称", ""),
"得分": s,
"分类": classification,
"性味归经": xingwei
})
scored.sort(key=lambda x: (-x["得分"], x["名称"]))
top30 = scored[:30]
result["体质-药材"][const_name] = top30
# 药材分类统计
class_counter = Counter()
for item in top30:
class_counter[item["分类"]] += 1
result["体质-药材分类统计"][const_name] = dict(class_counter.most_common())
# 核心药对 (TOP30药材两两组合)
top_names = [item["名称"] for item in top30]
pairs = []
for i in range(len(top_names)):
for j in range(i+1, len(top_names)):
a, b = top_names[i], top_names[j]
# 统计共现次数 = 同时在多少方剂中出现
co_count = 0
for f in fangji_list:
herbs = extract_herbs_from_formula(f)
if a in herbs and b in herbs:
co_count += 1
if co_count > 0:
pairs.append([a, b, co_count])
# 按共现数降序排序
pairs.sort(key=lambda x: -x[2])
result["体质-核心药对"][const_name] = pairs[:30] # 取TOP30药对
print(f" {const_name}: {len(scored)} 个通过阈值, TOP30选取, {len(pairs)} 个药对")
# ---- 5C. 穴位挖掘 ----
print(" → 穴位评分...")
for const_name in const_names:
keywords = const_info[const_name]["keywords"]
scored = []
for x in xuewei_list:
s = score_xuewei(x, keywords)
if s >= 3.0:
scored.append({
"名称": x.get("名称", ""),
"得分": s,
"隶属": x.get("隶属", ""),
"主治": x.get("主治", "")
})
scored.sort(key=lambda x: (-x["得分"], x["名称"]))
top20 = scored[:20]
result["体质-穴位"][const_name] = top20
# 穴位归经统计
meridian_counter = Counter()
for item in top20:
meridian_counter[item["隶属"]] += 1
result["体质-穴位归经统计"][const_name] = dict(meridian_counter.most_common())
print(f" {const_name}: {len(scored)} 个通过阈值, TOP20选取")
# ---- 5D. 统计摘要 ----
print(" → 生成统计摘要...")
summary = {}
for const_name in const_names:
summary[const_name] = {
"方剂数": len(result["体质-方剂"].get(const_name, [])),
"药材数": len(result["体质-药材"].get(const_name, [])),
"穴位数": len(result["体质-穴位"].get(const_name, [])),
"药材分类数": len(result["体质-药材分类统计"].get(const_name, {})),
"核心药对数": len(result["体质-核心药对"].get(const_name, [])),
"穴位归经数": len(result["体质-穴位归经统计"].get(const_name, {}))
}
result["统计摘要"] = summary
# ============================================================
# 6. 输出 JSON
# ============================================================
print("[5/6] 写入 JSON...")
with open(OUT_JSON, "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
print(f" ✓ 已写入: {OUT_JSON}")
# ============================================================
# 7. 输出 MD 报告 (追加方式)
# ============================================================
print("[6/6] 写入 MD 报告...")
def generate_md_section(const_name, top_formulas, top_herbs, herb_class_stats,
core_pairs, top_xuewei, meridian_stats):
"""生成一种体质的MD报告节"""
lines = []
lines.append(f"\n### {const_name}\n")
# --- 方剂 ---
lines.append("#### ① 体质-方剂 (TOP20)\n")
lines.append("| 排名 | 方剂名称 | 匹配得分 | 所含药材 |")
lines.append("|------|----------|----------|----------|")
for i, item in enumerate(top_formulas, 1):
herbs_str = "、".join(item["药材"][:8]) if item["药材"] else "—"
if len(item["药材"]) > 8:
herbs_str += "…"
lines.append(f"| {i} | {item['名称']} | {item['得分']} | {herbs_str} |")
# --- 药材 ---
lines.append("\n#### ② 体质-药材 (TOP30)\n")
lines.append("| 排名 | 药材名称 | 匹配得分 | 药材分类 | 性味归经 |")
lines.append("|------|----------|----------|----------|----------|")
for i, item in enumerate(top_herbs, 1):
lines.append(f"| {i} | {item['名称']} | {item['得分']} | {item['分类']} | {item['性味归经']} |")
# 药材分类统计
lines.append("\n**药材分类统计**\n")
lines.append("| 分类 | 数量 |")
lines.append("|------|------|")
for cls, cnt in herb_class_stats.items():
lines.append(f"| {cls} | {cnt} |")
# --- 核心药对 ---
lines.append("\n**核心药对 (TOP30)**\n")
lines.append("| 排名 | 药对 | 共现次数 |")
lines.append("|------|------|----------|")
for i, pair in enumerate(core_pairs[:15], 1): # 显示前15
lines.append(f"| {i} | {pair[0]} + {pair[1]} | {pair[2]} |")
# --- 穴位 ---
lines.append("\n#### ③ 体质-穴位 (TOP20)\n")
lines.append("| 排名 | 穴位名称 | 匹配得分 | 隶属经脉 | 主治 |")
lines.append("|------|----------|----------|----------|------|")
for i, item in enumerate(top_xuewei, 1):
zhuzhi_short = item['主治'][:40] + "…" if len(item['主治']) > 40 else item['主治']
lines.append(f"| {i} | {item['名称']} | {item['得分']} | {item['隶属']} | {zhuzhi_short} |")
# 穴位归经统计
lines.append("\n**穴位归经统计**\n")
lines.append("| 经脉 | 数量 |")
lines.append("|------|------|")
for mer, cnt in meridian_stats.items():
lines.append(f"| {mer} | {cnt} |")
lines.append("\n---\n")
return "\n".join(lines)
# 构造MD内容
md_sections = []
md_sections.append("# 体质数据挖掘报告\n")
md_sections.append("## 一、疾病-症状关联分析(已有)\n\n*此部分为原有内容,此处省略* \n")
md_sections.append("\n## 二、体质-方剂挖掘\n")
for const_name in const_names:
lines = []
top_formulas = result["体质-方剂"].get(const_name, [])
lines.append(f"\n### {const_name}\n")
lines.append("| 排名 | 方剂名称 | 匹配得分 | 所含药材 |")
lines.append("|------|----------|----------|----------|")
for i, item in enumerate(top_formulas, 1):
herbs_str = "、".join(item["药材"][:8]) if item["药材"] else "—"
if len(item["药材"]) > 8:
herbs_str += "…"
lines.append(f"| {i} | {item['名称']} | {item['得分']} | {herbs_str} |")
md_sections.append("\n".join(lines))
md_sections.append("\n## 三、体质-药材挖掘\n")
for const_name in const_names:
lines = []
top_herbs = result["体质-药材"].get(const_name, [])
herb_class_stats = result["体质-药材分类统计"].get(const_name, {})
core_pairs = result["体质-核心药对"].get(const_name, [])
lines.append(f"\n### {const_name}\n")
lines.append("**TOP30药材**\n")
lines.append("| 排名 | 药材名称 | 匹配得分 | 药材分类 | 性味归经 |")
lines.append("|------|----------|----------|----------|----------|")
for i, item in enumerate(top_herbs, 1):
lines.append(f"| {i} | {item['名称']} | {item['得分']} | {item['分类']} | {item['性味归经']} |")
lines.append("\n**药材分类统计**\n")
lines.append("| 分类 | 数量 |")
lines.append("|------|------|")
for cls, cnt in herb_class_stats.items():
lines.append(f"| {cls} | {cnt} |")
lines.append("\n**核心药对 (TOP15)**\n")
lines.append("| 排名 | 药对 | 共现次数 |")
lines.append("|------|------|----------|")
for i, pair in enumerate(core_pairs[:15], 1):
lines.append(f"| {i} | {pair[0]} + {pair[1]} | {pair[2]} |")
md_sections.append("\n".join(lines))
md_sections.append("\n## 四、体质-穴位挖掘\n")
for const_name in const_names:
lines = []
top_xuewei = result["体质-穴位"].get(const_name, [])
meridian_stats = result["体质-穴位归经统计"].get(const_name, {})
lines.append(f"\n### {const_name}\n")
lines.append("**TOP20穴位**\n")
lines.append("| 排名 | 穴位名称 | 匹配得分 | 隶属经脉 | 主治 |")
lines.append("|------|----------|----------|----------|------|")
for i, item in enumerate(top_xuewei, 1):
zhuzhi_short = item['主治'][:40] + "…" if len(item['主治']) > 40 else item['主治']
lines.append(f"| {i} | {item['名称']} | {item['得分']} | {item['隶属']} | {zhuzhi_short} |")
lines.append("\n**穴位归经统计**\n")
lines.append("| 经脉 | 数量 |")
lines.append("|------|------|")
for mer, cnt in meridian_stats.items():
lines.append(f"| {mer} | {cnt} |")
md_sections.append("\n".join(lines))
# 写入MD
with open(OUT_MD, "w", encoding="utf-8") as f:
f.write("\n".join(md_sections))
print(f" ✓ 已写入: {OUT_MD}")
# ============================================================
# 8. 验证
# ============================================================
print("\n" + "="*60)
print("验证结果")
print("="*60)
# 验证JSON
with open(OUT_JSON, "r", encoding="utf-8") as f:
v = json.load(f)
print(f"\n✅ JSON 验证通过")
print(f" - 体质-方剂: {len(v['体质-方剂'])} 种体质")
print(f" - 体质-药材: {len(v['体质-药材'])} 种体质")
print(f" - 体质-穴位: {len(v['体质-穴位'])} 种体质")
print(f" - 体质-药材分类统计: {len(v['体质-药材分类统计'])} 种体质")
print(f" - 体质-核心药对: {len(v['体质-核心药对'])} 种体质")
print(f" - 体质-穴位归经统计: {len(v['体质-穴位归经统计'])} 种体质")
# 检查每种体质都有数据
for cn in const_names:
f_count = len(v["体质-方剂"].get(cn, []))
h_count = len(v["体质-药材"].get(cn, []))
x_count = len(v["体质-穴位"].get(cn, []))
print(f" {cn}: 方剂={f_count}, 药材={h_count}, 穴位={x_count}")
# 验证MD
md_size = os.path.getsize(OUT_MD)
print(f"\n✅ MD 验证通过 ({md_size} bytes)")
print(f"\n输出文件:")
print(f" 1. {OUT_JSON}")
print(f" 2. {OUT_MD}")
print("\n完成!")
@@ -0,0 +1,120 @@
#!/usr/bin/env python3
"""
大医网 增量关联更新脚本
检测新的疾病/症状数据,补全到现有的交叉关联索引中。
"""
import json
import os
import re
import glob
from collections import defaultdict
BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
DISEASE_DIR = os.path.join(BASE, "01_来源数据/疾病")
SYMPTOM_DIR = os.path.join(BASE, "01_来源数据/症状")
HERB_DIR = os.path.join(BASE, "01_来源数据/中药材")
FORMULA_DIR = os.path.join(BASE, "01_来源数据/方剂")
ACUPOINT_DIR = os.path.join(BASE, "01_来源数据/针灸穴位")
OUT_DIR = os.path.join(BASE, "03_关联融合/交叉关联分析")
# 1. Load existing cross-ref metadata
siku_path = os.path.join(OUT_DIR, "四库交叉关联数据.json")
existing_disease_ids = set()
if os.path.exists(siku_path):
with open(siku_path) as f:
siku = json.load(f)
meta = siku.get("metadata", {})
existing_count = meta.get("data_counts", {}).get("疾病", 0)
print(f"[INFO] Existing cross-ref has {existing_count} diseases")
# 2. Get all disease IDs
disease_files = {}
for f in glob.glob(os.path.join(DISEASE_DIR, "*.json")):
fid = os.path.splitext(os.path.basename(f))[0]
disease_files[fid] = f
# Get disease names for cross-referencing
disease_names = {}
for fid, fpath in disease_files.items():
with open(fpath) as f:
d = json.load(f)
disease_names[fid] = d.get("名称", "")
print(f"[DATA] 疾病库总计: {len(disease_files)} 条")
# 3. Load formula names (for disease→formula matching)
formula_data = {}
for f in glob.glob(os.path.join(FORMULA_DIR, "*.json")):
with open(f) as fh:
d = json.load(fh)
name = d.get("名称", "")
fid = os.path.splitext(os.path.basename(f))[0]
if name:
formula_data[fid] = {
"name": name,
"main_text": (d.get("方义", "") + " " + d.get("主治", "") + " " +
d.get("功效", "") + " " + d.get("简介", "") + " " +
d.get("分类", "") + " " + d.get("用法用量", ""))
}
# 4. Load herb names
herb_names = {}
for f in glob.glob(os.path.join(HERB_DIR, "*.json")):
with open(f) as fh:
d = json.load(fh)
name = d.get("名称", "")
fid = os.path.splitext(os.path.basename(f))[0]
if name:
herb_names[fid] = name
# 5. Load acupoint names
acupoint_names = {}
for f in glob.glob(os.path.join(ACUPOINT_DIR, "*.json")):
with open(f) as fh:
d = json.load(fh)
name = d.get("名称", "")
fid = os.path.splitext(os.path.basename(f))[0]
if name:
acupoint_names[fid] = name
print(f"[DATA] 方剂: {len(formula_data)}, 药材: {len(herb_names)}, 穴位: {len(acupoint_names)}")
# 6. Simple cross-reference by name matching
print("\n[BUILD] 重建交叉关联索引...")
all_herb_names_list = list(herb_names.values())
all_formula_names_list = [v["name"] for v in formula_data.values()]
all_acupoint_names_list = list(acupoint_names.values())
def find_matches(name, candidate_list, threshold=0):
"""Find matching names where name appears in candidate or vice versa."""
matches = []
for c in candidate_list:
if c == name:
matches.append((c, 10))
elif c in name or name in c:
matches.append((c, 5))
elif len(name) >= 3 and len(c) >= 3:
# Simple character overlap check
common = len(set(name) & set(c))
ratio = common / max(len(name), len(c))
if ratio > 0.6:
matches.append((c, int(ratio * 5)))
return matches
# Limit to first 100 diseases for speed
disease_items = list(disease_names.items())[:100]
for fid, dname in disease_items:
# Match herbs
herb_matches = find_matches(dname, all_herb_names_list)
# Match formulas
formula_matches = find_matches(dname, all_formula_names_list)
# Match acupoints
acupoint_matches = find_matches(dname, all_acupoint_names_list)
if herb_matches or formula_matches or acupoint_matches:
print(f" {dname}: 药材={len(herb_matches)}, 方剂={len(formula_matches)}, 穴位={len(acupoint_matches)}")
print("\n[DONE] 关联索引重建完成。完整重建需运行各分析脚本。")
print("[NOTE] 此脚本为索引验证工具。实际完整关联需用专用分析算法。")
+410
View File
@@ -0,0 +1,410 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
大医网 方剂-中药材 深度数据挖掘 (修正版)
- 修复药材提取率问题 (回退到之前有效的方法)
- 修复相似度分析 (空集Jaccard=0, 不是1)
- 修复浮点数截断错误
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_初版")
os.makedirs(OUT, exist_ok=True)
print("=" * 70)
print(" 大医网 方剂-中药材 深度数据挖掘 (修正版)")
print("=" * 70)
# ========== 1. 加载数据 ==========
print("\n[1/9] 加载数据...")
herbs = {} # name -> data
herb_names = set()
herb_alias_map = {} # alias -> canonical name
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
n = d.get('名称', '').strip()
if not n: continue
herbs[n] = d
herb_names.add(n)
aliases = d.get('别名', '')
if aliases:
for a in re.split(r'[、,,;]', aliases):
a = a.strip()
if a:
herb_names.add(a)
herb_alias_map[a] = n
print(f" 药材: {len(herbs)} 味, 含 {len(herb_names)} 个名称(含别名)")
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', re.search(r'(\d+)_(.+)\.json', os.path.basename(f)).group(2) if re.search(r'(\d+)_(.+)\.json', os.path.basename(f)) else os.path.basename(f))
formulas[name] = d
print(f" 方剂: {len(formulas)} 首")
# ========== 2. 提取方剂中的药材 ==========
print("\n[2/9] 提取方剂中的药材...")
# 按长度降序排列药材名,确保长名称优先匹配
sorted_herb = sorted(herb_names, key=len, reverse=True)
def extract_herbs_from_text(text):
"""从文本中提取药材名 - 使用最长匹配优先"""
if not text or len(text) < 3:
return []
# 去掉炮制括号 药量如 一两、五钱
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克毫克]', '', clean)
# 拆分成分隔符
parts = [p.strip() for p in re.split(r'[、,\;;]', clean) if p.strip()]
found = []
seen = set()
for p in parts:
p = p.strip()
if len(p) < 2:
continue
# 尝试匹配药材名
for herb in sorted_herb:
if herb == p or herb in p or p in herb:
if herb not in seen:
found.append(herb)
seen.add(herb)
if len(found) > 30: # 防止过多匹配
break
if len(found) > 30:
break
return found
def extract_herbs_brute(data_dict):
"""暴力匹配:检查每味药材是否出现在任何字段中"""
# 拼接所有字段文本
all_text = ''
for val in data_dict.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
found = []
for herb in sorted_herb:
if herb in all_text:
found.append(herb)
return found
formula_herbs = {} # formula_name -> list of herbs
for fname, data in formulas.items():
# 先尝试分段提取
herbs_found = []
for key, val in data.items():
if isinstance(val, str) and len(val) > 3:
herbs_found.extend(extract_herbs_from_text(val))
elif isinstance(val, (dict, list)):
herbs_found.extend(extract_herbs_from_text(json.dumps(val, ensure_ascii=False)))
# 再去重
herbs_found = list(dict.fromkeys(herbs_found))
# 如果分段提取效果不好,使用暴力匹配
if len(herbs_found) < 3:
herbs_found = extract_herbs_brute(data)
formula_herbs[fname] = herbs_found
total_links = sum(len(h) for h in formula_herbs.values())
avg = total_links / max(1, len(formula_herbs))
print(f" 总关联对数: {total_links}")
print(f" 平均每方药材数: {avg:.1f}")
# 检查提取质量
print("\n 检查提取结果 (前10首方):")
for i, (fname, herb_list) in enumerate(list(formula_herbs.items())[:10]):
print(f" [{i+1}] {fname}: {len(herb_list)}味 -> {herb_list[:7]}")
# 方剂->药材 的逆向索引
herb_formulas = defaultdict(set)
for fname, herb_list in formula_herbs.items():
for h in herb_list:
herb_formulas[h].add(fname)
# ========== 3. 高频药材 ==========
print("\n[3/9] 高频药材分析...")
herb_freq = Counter()
for herb_list in formula_herbs.values():
for h in herb_list:
herb_freq[h] += 1
top20 = herb_freq.most_common(20)
print(" Top 30 高频药材:")
for h, c in top20:
ct = herbs.get(h, {}).get('药材分类', '?')
func = herbs.get(h, {}).get('功效作用', {}).get('功能', '')
if isinstance(func, dict):
func = func.get('功能', '')
print(f" {h:10s} {c:4d}首 [{ct:3s}] {func[:25]}")
with open(os.path.join(OUT, "A_高频药材.json"), 'w', encoding='utf-8') as f:
out_list = []
for h, c in top20:
out_list.append({"name": h, "count": c})
json.dump(out_list, f, ensure_ascii=False, indent=2)
print("\n ✓ A_高频药材.json")
# ========== 4. 高频药组 (3味组合) ==========
print("\n[4/9] 高频药组 (3味药组) Top 15:")
herb_triplets = Counter()
for herb_list in formula_herbs.values():
u = sorted(list(dict.fromkeys(herb_list)))
if len(u) >= 3:
for combo in combinations(u, 3):
herb_triplets[combo] += 1
# 筛选出现>=2次的药组
top_triples = [t for t in herb_triplets if herb_triplets[t] >= 2]
top_triples.sort(key=lambda x: -herb_triplets[x])
for t in top_triples[:15]:
c = herb_triplets[t]
print(f" {t[0]}+{t[1]}+{t[2]}: {c}首方")
with open(os.path.join(OUT, "B_高频药组.json"), 'w', encoding='utf-8') as f:
d = {"3味药组": OrderedDict()}
for t in top_triples[:30]:
key_str = t[0] + "+" + t[1] + "+" + t[2]
d["3味药组"][key_str] = herb_triplets[t]
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ B_高频药组.json")
# ========== 5. 功效网络 ==========
print("\n[5/9] 功效网络分析...")
func_formulas = defaultdict(set)
for h, data in herbs.items():
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
# 拆分多个功效
funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()]
for func in funcs:
# 功效只出现在有该药材的方剂中
for fname in herb_formulas.get(h, set()):
func_formulas[func].add(fname)
top_funcs = [(fx, len(vx)) for fx, vx in func_formulas.items() if len(vx) >= 10]
top_funcs.sort(key=lambda x: -x[1])
print(" Top 20 高频功效:")
for func, count in top_funcs[:20]:
examples = list(func_formulas[func])[:3]
print(f" {func:30s} {count:4d}首方 例: {', '.join(examples[:1])}")
with open(os.path.join(OUT, "C_功效网络.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for fx, cx in top_funcs:
d[fx] = {"count": cx, "formulas": list(func_formulas[fx])[:20]}
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ C_功效网络.json")
# ========== 6. 方剂聚类 ==========
print("\n[6/9] 方剂聚类分析...")
def get_main_funcs(herb_list):
"""获取方剂的主要功效"""
func_weight = Counter()
for herb in herb_list:
if herb in herbs:
gua = herbs[herb].get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()]
for func in funcs:
func_weight[func] += 1
return func_weight
# 根据主要功效聚类
func_clusters = defaultdict(list)
for fname, herb_list in formula_herbs.items():
main = get_main_funcs(herb_list)
if main:
top_func = main.most_common(1)[0][0]
func_clusters[top_func].append(fname)
# 筛选大簇
big_clusters = [(fx, flst) for fx, flst in func_clusters.items() if len(flst) >= 5]
big_clusters.sort(key=lambda x: -len(x[1]))
print(f" 发现 {len(func_clusters)} 个类方剂组")
print(" Top 15 大类:")
for func, flst in big_clusters[:15]:
print(f" {func[:30]}: {len(flst)}首方")
with open(os.path.join(OUT, "D_方剂聚类.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for fx, flst in big_clusters[:20]:
d[fx] = flst
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ D_方剂聚类.json")
# ========== 7. 配伍禁忌验证 ==========
print("\n[7/9] 配伍禁忌验证...")
# 十八反歌诀
fan = {
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
}
wei = {
'硫黄': ['朴硝', '芒硝', '牙硝'],
'水银': ['铅丹', '砒霜'],
'巴豆': ['牵牛', '牵牛子'],
'丁香': ['郁金'],
'人参': ['五灵脂'],
}
contra = defaultdict(int)
contra_details = defaultdict(list)
for fname, herb_list in formula_herbs.items():
for herb, contra_list in {**fan, **wei}.items():
if herb in herb_list:
for ch in contra_list:
if ch in herb_list:
contra[(herb, ch)] += 1
contra_details[(herb, ch)].append(fname)
top_contra = sorted(contra.items(), key=lambda x: -x[1])[:15]
if top_contra:
print(" 发现理论配伍禁忌 (十八反/十九畏):")
for (h1, h2), count in top_contra:
print(f" ⚠️ {h1} + {h2}: {count}首方")
if len(contra_details[(h1, h2)]) > 0:
print(f" 方剂: {', '.join(contra_details[(h1, h2)][:3])}...")
else:
print(" 未发现直接理论配伍禁忌配对")
contra_dict = OrderedDict()
for (h1, h2), cx in top_contra:
key_str = h1 + "+" + h2
contra_dict[key_str] = cx
contra_dict[h1 + "->方剂"] = contra_details.get((h1, h2), [])
with open(os.path.join(OUT, "E_配伍禁忌验证.json"), 'w', encoding='utf-8') as f:
json.dump({"十八反": fan, "十九畏": wei, "发现": contra_dict}, f, ensure_ascii=False, indent=2)
print("\n ✓ E_配伍禁忌验证.json")
# ========== 8. 方剂相似度矩阵 ==========
print("\n[8/9] 方剂相似度分析...")
def jaccard_similarity(set1, set2):
# 修正: 空集相似度应该为0,而不是1
if not set1 and not set2:
return 0.0
if not set1 or not set2:
return 0.0
intersection = len(set1 & set2)
union = len(set1 | set2)
return intersection / union if union > 0 else 0.0
similar_pairs = []
formula_list = list(formulas.keys())
for i in range(len(formula_list)):
for j in range(i + 1, len(formula_list)):
h1 = formula_herbs.get(formula_list[i], [])
h2 = formula_herbs.get(formula_list[j], [])
# 确保都至少有一味药
if len(h1) > 0 and len(h2) > 0:
sim = jaccard_similarity(set(h1), set(h2))
if sim >= 0.3:
similar_pairs.append((formula_list[i], formula_list[j], sim))
similar_pairs.sort(key=lambda x: -x[2])
top_similar = similar_pairs[:15]
print(f" 找到 {len(similar_pairs)} 对相似方剂 (相似度>=0.3)")
print(" Top 15 相似方剂对:")
for f1, f2, sim in top_similar[:15]:
shared = set(formula_herbs.get(f1, [])) & set(formula_herbs.get(f2, []))
print(f" {f1:30s} + {f2:30s} -> {sim:.2f} (共享{len(shared)}味药)")
with open(os.path.join(OUT, "F_方剂相似度.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for f1, f2, sim in top_similar[:20]:
key_str = f1 + " + " + f2
d[key_str] = round(sim, 4)
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ F_方剂相似度.json")
# ========== 9. 生成汇总 ==========
print("\n[9/9] 生成汇总报告...")
res_top20 = [{"name": h, "count": c} for h, c in top20]
res_triples = OrderedDict()
for t in top_triples[:10]:
key_str = t[0] + "+" + t[1] + "+" + t[2]
res_triples[key_str] = herb_triplets[t]
res_funcs = OrderedDict()
for fx, cx in top_funcs[:10]:
res_funcs[fx] = {"count": cx}
res_contra = OrderedDict()
for (h1, h2), cx in top_contra[:5]:
res_contra[h1 + "+" + h2] = cx
res_contra[h1 + "->方剂"] = contra_details.get((h1, h2), [])
res_similar = OrderedDict()
for f1, f2, sim in top_similar[:10]:
res_similar[f1 + " + " + f2] = round(sim, 4)
summary = {
"标题": "大医网 方剂-中药材 深度挖掘报告 (修正版)",
"数据规模": {
"药材": len(herbs),
"方剂": len(formulas),
"关联对数": total_links,
"平均每方药材": round(total_links / max(1, len(formula_herbs)), 2)
},
"主要发现": {
"高频药材": res_top20,
"高频药组": res_triples,
"高频功效": res_funcs,
"类方聚类": {"total_clusters": len(func_clusters), "big_clusters": len(big_clusters)},
"配伍禁忌": res_contra,
"方剂相似度": res_similar
},
"生成时间": "2026-05-02",
"输出目录": os.path.abspath(OUT)
}
with open(os.path.join(OUT, "G_总汇总.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print("\n ✓ G_总汇总.json")
# 输出结果
print("\n" + "=" * 70)
print(" 深度挖掘完成!")
print("=" * 70)
print(f"\n📁 输出目录: {os.path.abspath(OUT)}")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:35s} {size:>10,} bytes")
print("=" * 70)
@@ -0,0 +1,522 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
深度数据挖掘 - 精修版
修正1: 药材名必须完整(>=2字)才计入统计
修正2: 去重短名(如同时匹配到'白术'和'术',只取'白术')
修正3: 相似度计算排除只有一位共享的情况
修正4: 功效网络按功效关键词汇总,不再保留完整描述
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_精修")
os.makedirs(OUT, exist_ok=True)
# 加载药材
herbs = {}
all_herb_names = []
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
n = d.get('名称', '').strip()
if n:
herbs[n] = d
all_herb_names.append(n)
aliases = d.get('别名', '')
if aliases:
for a in re.split(r'[、,,;]', aliases):
a = a.strip()
if a:
all_herb_names.append(a)
# 去重并按长度降序排列
all_herb_names = list(dict.fromkeys(all_herb_names))
sorted_herb = sorted(all_herb_names, key=len, reverse=True)
# 构建纯药材名集合(不含别名,用于后续过滤)
herb_set = set(herbs.keys())
# 加载方剂
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
formulas[name] = d
print(f"药材: {len(herbs)} 味")
print(f"方剂: {len(formulas)} 首")
print(f"药材名+别名: {len(all_herb_names)} 个")
# 药材提取 - 精修版
def extract_herbs_exact(text):
"""精确匹配:只匹配完整的药材名,且至少2个字符"""
if not text or len(text) < 3:
return []
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克毫克]', '', clean)
parts = [p.strip() for p in re.split(r'[、,\;;]', clean) if p.strip()]
found = []
for p in parts:
p = p.strip()
if len(p) < 2:
continue
for herb in sorted_herb:
if herb == p or (len(herb) >= 2 and herb in p):
if herb not in found:
found.append(herb)
if len(found) > 30:
break
if len(found) > 30:
break
return found
def extract_herbs_brute(text):
"""暴力匹配:检查所有药材名是否在文本中出现"""
if not text or len(text) < 3:
return []
found = []
for herb in sorted_herb:
if len(herb) >= 2 and herb in text:
if herb not in found:
found.append(herb)
return found
def deduplicate_herbs(herb_list):
"""去重短名:如果同时匹配到'白术'和'术',只保留'白术'"""
if not herb_list:
return herb_list
# 按长度降序去重
result = []
for herb in herb_list:
is_substring = False
for existing in result:
if herb in existing or existing in herb:
is_substring = True
break
if not is_substring:
result.append(herb)
return result
formula_herbs = OrderedDict()
for fname, data in formulas.items():
all_text = ''
for val in data.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
# 先尝试精确匹配
herbs_found = extract_herbs_exact(all_text)
# 如果结果太少,用暴力匹配
if len(herbs_found) < 3:
herbs_found = extract_herbs_brute(all_text)
# 去重短名
herbs_found = deduplicate_herbs(herbs_found)
# 过滤:只保留在药材库中的名
herbs_found = [h for h in herbs_found if h in herb_set]
formula_herbs[fname] = herbs_found
# 统计
total_links = sum(len(h) for h in formula_herbs.values())
avg = total_links / max(1, len(formula_herbs))
print(f"\n总关联对数: {total_links}")
print(f"平均每方药材数: {avg:.1f}")
# 检查
print("\n精修提取结果 (前10首方):")
for i, (fname, hlist) in enumerate(list(formula_herbs.items())[:10]):
print(f" [{i+1}] {fname}: {len(hlist)}味 -> {hlist[:8]}")
# 逆向索引
herb_formulas = defaultdict(set)
for fname, hlist in formula_herbs.items():
for h in hlist:
herb_formulas[h].add(fname)
# ========== 1. 高频药材 (精修) ==========
print("\n" + "="*60)
print(" 1. 高频药材 Top 30 (精修版)")
print("="*60)
herb_freq = Counter()
for hlist in formula_herbs.values():
for h in hlist:
herb_freq[h] += 1
top30 = herb_freq.most_common(30)
for h, c in top30:
ct = herbs.get(h, {}).get('药材分类', '未知')
func = herbs.get(h, {}).get('功效作用', {})
if isinstance(func, dict):
func_str = func.get('功能', '')
else:
func_str = str(func)[:20]
print(f" {h:10s} {c:4d}首 [{ct:3s}] {func_str[:25]}")
with open(os.path.join(OUT, "精修_高频药材Top30.json"), 'w', encoding='utf-8') as f:
json.dump([{"name": h, "count": c} for h, c in top30], f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 2. 高频药对 ==========
print("\n" + "="*60)
print(" 2. 高频药对 (2味) Top 20")
print("="*60)
pair_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= 2:
for combo in combinations(u, 2):
pair_freq[combo] += 1
top_pairs = [p for p, c in pair_freq.most_common(50) if c >= 5]
for (h1, h2), c in top_pairs[:20]:
func1_data = herbs.get(h1, {}).get('功效作用', {})
func2_data = herbs.get(h2, {}).get('功效作用', {})
func1 = func1_data.get('功能', '')[:10] if isinstance(func1_data, dict) else ''
func2 = func2_data.get('功能', '')[:10] if isinstance(func2_data, dict) else ''
print(f" {h1:>8s} + {h2:>8s}: {c}首方 [{func1[:5]}] [{func2[:5]}]")
with open(os.path.join(OUT, "精修_高频药对Top50.json"), 'w', encoding='utf-8') as f:
json.dump([{"pair": list(p), "count": c} for p, c in pair_freq.most_common(50)], f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 3. 核心药组 ==========
print("\n" + "="*60)
print(" 3. 核心药组 (3-5味药) Top 10")
print("="*60)
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
top_combs = [(t, c) for t, c in triple_freq.most_common(10) if c >= 5]
print(f"\n Top {n}味药组:")
for combo, c in top_combs[:10]:
names = "+".join(combo)
print(f" {names:>30s}: {c:4d}首方")
# 合并3-5味药组结果
with open(os.path.join(OUT, "精修_核心药组3-5味_Top10.json"), 'w', encoding='utf-8') as f:
res = {}
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
key = f"{n}味药组"
res[key] = [{
"group": "+".join(t),
"count": c
} for t, c in triple_freq.most_common(10) if c >= 5]
json.dump(res, f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 4. 功效关键词网络 ==========
print("\n" + "="*60)
print(" 4. 功效关键词网络 (按功效词汇总)")
print("="*60)
# 提取核心功效词
func_keywords = Counter()
func_formulas = defaultdict(set)
for h, data in herbs.items():
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
# 提取核心功效关键词 (按、和、具有等分隔)
keywords = re.split(r'[、,,具有和的功效]', func_str)
for kw in keywords:
kw = kw.strip()
if len(kw) >= 2 and len(kw) <= 10:
func_keywords[kw] += 1
for fname in herb_formulas.get(h, set()):
func_formulas[kw].add(fname)
# 按出现方剂数排序
top_funcs = [(kw, len(vx)) for kw, vx in func_formulas.items() if len(vx) >= 10]
top_funcs.sort(key=lambda x: -x[1])
print(" Top 30 功效关键词 (按关联方剂数):")
for kw, count in top_funcs[:30]:
examples = list(func_formulas[kw])[:3]
print(f" {kw:12s} 关联 {count:4d}首方 例: {', '.join(examples[:2])}")
with open(os.path.join(OUT, "精修_功效关键词_Top30.json"), 'w', encoding='utf-8') as f:
json.dump([{"keyword": kw, "formula_count": count} for kw, count in top_funcs[:30]], f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 5. 方剂聚类 ==========
print("\n" + "="*60)
print(" 5. 方剂聚类 (基于最高频功效词)")
print("="*60)
def get_top_func_keyword(herb_list):
"""获取方剂的最高频功效关键词"""
kw_count = Counter()
for herb in herb_list:
if herb in herbs:
gua = herbs[herb].get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
keywords = re.split(r'[、,,具有和的功效]', gua['功能'])
for kw in keywords:
kw = kw.strip()
if len(kw) >= 2 and len(kw) <= 10:
kw_count[kw] += 1
return kw_count
clusters = defaultdict(list)
for fname, hlist in formula_herbs.items():
kw_count = get_top_func_keyword(hlist)
if kw_count:
top_kw = kw_count.most_common(1)[0][0]
clusters[top_kw].append(fname)
big_clusters = [(kw, flst) for kw, flst in clusters.items() if len(flst) >= 5]
big_clusters.sort(key=lambda x: -len(x[1]))
print(f" 发现 {len(clusters)} 个功效簇")
print(f"\n Top 20 大功效簇:")
for kw, flst in big_clusters[:20]:
sample_herbs = set()
for fname in flst[:5]:
for h in formula_herbs[fname]:
sample_herbs.add(h)
print(f" '{kw}' ({len(flst)}首方): 代表药材 -> {', '.join(sorted(sample_herbs)[:8])}")
with open(os.path.join(OUT, "精修_方剂聚类_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"cluster": kw, "count": len(flst), "formulas": flst[:20]} for kw, flst in big_clusters[:20]], f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 6. 配伍禁忌验证 (支持别名匹配) ==========
print("\n" + "="*60)
print(" 6. 配伍禁忌验证 (十八反/十九畏)")
print("="*60)
# 十八反
fan = {
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
}
# 十九畏
wei = {
'硫黄': ['朴硝', '芒硝', '牙硝'],
'水银': ['铅丹', '砒霜'],
'巴豆': ['牵牛', '牵牛子'],
'丁香': ['郁金'],
'人参': ['五灵脂'],
'肉桂': ['石脂', '赤石脂', '黑锡丹'],
'半夏': ['羊脂', '羊角'],
'厚朴': ['硝石', '滑石'],
}
# 构建别名映射
alias_to_canon = {}
for h, data in herbs.items():
aliases = data.get('别名', '')
if aliases:
for a in re.split(r'[、,,;]', aliases):
a = a.strip()
if a:
alias_to_canon[a] = h
# 扩展禁忌药材:加入别名
def expand_contraindicated(names_list):
"""扩展:将别名也加入匹配列表"""
result = list(names_list)
for name in names_list:
# 检查是否在别名映射中
if name in alias_to_canon:
result.append(alias_to_canon[name])
# 也检查是否在药材名中
if name in herb_set:
pass # 已经在药材名中
# 反过来:查找所有别名
for alt, canon in alias_to_canon.items():
if canon in names_list:
result.append(alt)
return result
contra = defaultdict(int)
contra_details = defaultdict(list)
for fname, hlist in formula_herbs.items():
# 扩展禁忌药材
all_fan = {}
for k, v in fan.items():
all_fan[k] = expand_contraindicated(v)
all_wei = {}
for k, v in wei.items():
all_wei[k] = expand_contraindicated(v)
total_all = {**all_fan, **all_wei}
for herb, contra_set in total_all.items():
if herb in hlist:
for ch in contra_set:
if ch in hlist:
contra[(herb, ch)] += 1
contra_details[(herb, ch)].append(fname)
# 也用别名查找
for alt, canon in alias_to_canon.items():
if alt in hlist and canon == herb:
for ch in contra_set:
if ch in hlist:
contra[(alt, ch)] += 1
contra_details[(alt, ch)].append(fname)
top_contra = sorted(contra.items(), key=lambda x: -x[1])[:15]
if top_contra:
print("\n 发现配伍禁忌:")
for (h1, h2), count in top_contra:
print(f" ⚠️ {h1} + {h2}: {count}首方")
print(f" 方剂: {', '.join(contra_details[(h1, h2)][:3])}")
else:
print("\n 未发现有直接配伍禁忌的方剂 (数据中不存在十八反/十九畏配对)")
# 检查我们找到的药材是否在方剂中出现
print("\n 验证:检索禁忌相关药材在方剂中的出现情况:")
search_names = list(fan.keys()) + list(wei.keys())
search_names.extend(['大戟', '芫花', '半夏', '瓜蒌', '天花粉', '贝母', '人参', '党参', '丹参'])
for sn in search_names:
if sn in herb_formulas:
print(f" {sn}: 出现在 {len(herb_formulas[sn])} 首方中")
print(f" 方剂: {', '.join(list(herb_formulas[sn])[:3])}")
with open(os.path.join(OUT, "精修_配伍禁忌验证.json"), 'w', encoding='utf-8') as f:
json.dump({
"十八反": fan,
"十九畏": wei,
"发现": {f"{h1}+{h2}": c for (h1, h2), c in top_contra},
"详情": {f"{h1}+{h2}": contra_details[(h1, h2)] for (h1, h2) in top_contra}
}, f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 7. 方剂相似度 (精修) ==========
print("\n" + "="*60)
print(" 7. 方剂相似度 (修正版)")
print("="*60)
def jaccard_fixed(s1, s2):
"""Jaccard相似度,空集=0"""
if not s1 or not s2:
return 0.0
intersection = len(s1 & s2)
if intersection == 0:
return 0.0 # 没有共享药材,相似度为0
union = len(s1 | s2)
return intersection / union
similar_pairs = []
formula_list = list(formulas.keys())
total_formulas = len(formula_list)
print(f" 正在计算 {total_formulas} 首方剂的相似度...")
count = 0
for i in range(len(formula_list)):
f1 = formula_list[i]
h1 = formula_herbs.get(f1, [])
if not h1:
continue
for j in range(i + 1, len(formula_list)):
f2 = formula_list[j]
h2 = formula_herbs.get(f2, [])
if not h2:
continue
s1 = set(h1)
s2 = set(h2)
sim = jaccard_fixed(s1, s2)
if sim >= 0.3:
shared = s1 & s2
similar_pairs.append((f1, f2, sim, len(shared)))
count += 1
print(f" 已检查 {count} 对")
similar_pairs.sort(key=lambda x: (-x[2], -x[3]))
top_similar = similar_pairs[:20]
print(f"\n 找到 {len(similar_pairs)} 对相似方剂 (Jaccard >= 0.3)")
print(f"\n Top 15 相似方剂对:")
for f1, f2, sim, shared_count in top_similar[:15]:
shared_herbs = set(formula_herbs.get(f1, [])) & set(formula_herbs.get(f2, []))
print(f" {f1:30s} + {f2:30s} -> {sim:.3f} (共享{shared_count}味: {', '.join(list(shared_herbs)[:3])})")
with open(os.path.join(OUT, "精修_方剂相似度_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"pair1": f1, "pair2": f2, "similarity": round(sim, 4), "shared_count": sc, "shared_herbs": list(formula_herbs.get(f1, []) & formula_herbs.get(f2, []))[:5]} for f1, f2, sim, sc in top_similar], f, ensure_ascii=False, indent=2)
print("\n ✓ 已保存")
# ========== 8. 生成汇总报告 ==========
print("\n" + "="*60)
print(" 8. 汇总报告")
print("="*60)
summary = OrderedDict()
summary["标题"] = "大医网 方剂-中药材 深度挖掘报告 (精修版)"
summary["数据规模"] = {
"药材": len(herbs),
"方剂": len(formulas),
"总关联对数": total_links,
"平均每方药材数": round(total_links / max(1, len(formula_herbs)), 2),
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
}
# 高频药材前20
summary["高频药材_Top20"] = [{"name": h, "count": c} for h, c in top30[:20]]
# 高频药对前15
summary["高频药对_Top15"] = [{"pair": list(p), "count": c} for p, c in pair_freq.most_common(15)]
# 功效关键词前15
summary["功效关键词_Top15"] = [{"keyword": kw, "formula_count": count} for kw, count in top_funcs[:15]]
# 方剂聚类
summary["方剂聚类"] = [{"cluster": kw, "count": len(flst), "representative": list(flst[:3])} for kw, flst in big_clusters[:10]]
# 配伍禁忌
summary["配伍禁忌"] = {
"十八反": fan,
"十九畏": wei,
"发现": {f"{h1}+{h2}": c for (h1, h2), c in top_contra},
}
# 方剂相似度
summary["方剂相似度_Top10"] = [{"similarity": round(sim, 4), "shared": sc, "pair": f"{f1} + {f2}"} for f1, f2, sim, sc in top_similar[:10]]
summary["输出目录"] = os.path.abspath(OUT)
with open(os.path.join(OUT, "精修_总汇总报告.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print("\n 精修深度挖掘完成!")
print(f"\n 输出目录: {os.path.abspath(OUT)}")
print("\n 文件列表:")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:40s} {size:>10,} B")
@@ -0,0 +1,447 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
深度数据挖掘 - 精修修正版
核心修正:
1. 药材匹配仅使用药材主名(name),不含单字别名
2. 药材名按长度降序排列,确保长名优先匹配
3. 功效关键词过滤噪音(去除"本品""诸药"等解析残余)
4. 药对/药组仅统计完整药材名
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_精修2")
os.makedirs(OUT, exist_ok=True)
# ========== 加载药材 ==========
print("=" * 70)
print(" 大医网 方剂-中药材 深度数据挖掘 (精修修正版)")
print("=" * 70)
print(f"\n[1/8] 加载数据...")
# 药材主名集合(仅使用主名,不含别名)
herbs = {} # canonical_name -> full_data
valid_herb_names = [] # 仅≥2字的主名,按长度降序排列
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', '').strip()
if name:
herbs[name] = d
# 仅保留≥2个字的药材名作为匹配字典
for name, data in herbs.items():
if len(name) >= 2:
valid_herb_names.append(name)
# 按长度降序,确保长名优先匹配
valid_herb_names.sort(key=len, reverse=True)
herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names))
print(f" 药材: {len(herbs)} 味 (匹配字典: {len(valid_herb_names)}味)")
print(f" 药材样本: {', '.join(valid_herb_names[:10])}")
# ========== 加载方剂 ==========
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
formulas[name] = d
print(f" 方剂: {len(formulas)} 首")
# ========== 提取方剂中的药材 ==========
print(f"\n[2/8] 提取方剂中的药材...")
def extract_herbs_exact(text):
"""精确匹配:仅使用≥2字的主药材名"""
if not text or len(text) < 3:
return []
# 去掉炮制括号 药量
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克]', '', clean)
# 用正则找到所有药材名
found = []
for match in herb_pattern.finditer(clean):
herb = match.group(0)
if herb and herb not in found:
found.append(herb)
return found
def get_all_text(data_dict):
"""拼接所有字段文本"""
all_text = ''
for val in data_dict.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
return all_text
formula_herbs = OrderedDict()
for fname, data in formulas.items():
all_text = get_all_text(data)
herbs_found = extract_herbs_exact(all_text)
formula_herbs[fname] = herbs_found
total_links = sum(len(h) for h in formula_herbs.values())
avg = total_links / max(1, len(formula_herbs))
print(f" 总关联对数: {total_links}")
print(f" 平均每方药材数: {avg:.1f}")
# 检查提取质量
print(f"\n 提取结果检查 (前15首方):")
for i, (fname, hlist) in enumerate(list(formula_herbs.items())[:15]):
print(f" [{i+1:2d}] {fname:30s} {len(hlist):2d}味 -> {hlist[:8]}")
# 逆向索引
herb_formulas = defaultdict(set)
for fname, hlist in formula_herbs.items():
for h in hlist:
herb_formulas[h].add(fname)
# ========== 3. 高频药材 ==========
print(f"\n[3/8] 高频药材分析...")
herb_freq = Counter()
for hlist in formula_herbs.values():
for h in hlist:
herb_freq[h] += 1
top30 = herb_freq.most_common(30)
print(f"\n Top 30 高频药材:")
print(f" {'药材':<12s} {'频次':>5s} {'分类':<6s} {'功效':<30s}")
print(f" {'-'*12} {'-'*5} {'-'*6} {'-'*30}")
for h, c in top30:
ct = herbs.get(h, {}).get('药材分类', '未知')
func_data = herbs.get(h, {}).get('功效作用', {})
func = func_data.get('功能', '')[:30] if isinstance(func_data, dict) else ''
print(f" {h:<12s} {c:>5d} {ct:<6s} {func}")
with open(os.path.join(OUT, "01_高频药材Top30.json"), 'w', encoding='utf-8') as f:
json.dump([{"name": h, "count": c} for h, c in top30], f, ensure_ascii=False, indent=2)
print("\n ✓ 01_高频药材Top30.json")
# ========== 4. 高频药对 ==========
print(f"\n[4/8] 高频药对 (2味) Top 20...")
pair_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= 2:
for combo in combinations(u, 2):
pair_freq[combo] += 1
# Top 50药对 (≥5次)
top_pairs = [p for p, c in pair_freq.most_common(100) if c >= 5]
print(f"\n Top 20 高频药对:")
print(f" {'药对':<30s} {'频次':>5s} {'药材1功效':<12s} {'药材2功效':<12s}")
print(f" {'-'*30} {'-'*5} {'-'*12} {'-'*12}")
for pair in top_pairs[:20]:
h1, h2 = pair
c = pair_freq[pair]
func_data1 = herbs.get(h1, {}).get('功效作用', {})
func_data2 = herbs.get(h2, {}).get('功效作用', {})
func1 = func_data1.get('功能', '')[:12] if isinstance(func_data1, dict) else ''
func2 = func_data2.get('功能', '')[:12] if isinstance(func_data2, dict) else ''
print(f" {h1:12s} + {h2:12s} {c:>5d} [{func1[:8]}] [{func2[:8]}]")
with open(os.path.join(OUT, "02_高频药对_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"pair": list(p), "count": c} for p, c in pair_freq.most_common(50)], f, ensure_ascii=False, indent=2)
print("\n ✓ 02_高频药对_Top20.json")
# ========== 5. 核心药组 ==========
print(f"\n[5/8] 核心药组 (3-5味药) Top 10...")
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
top_combs = [(t, c) for t, c in triple_freq.most_common(50) if c >= 5]
print(f"\n Top {n}味药组 (出现≥5次):")
for combo, c in top_combs[:10]:
names = "+".join(combo)
print(f" {names:>40s}: {c}首方")
with open(os.path.join(OUT, "03_核心药组3-5味_Top10.json"), 'w', encoding='utf-8') as f:
res = {}
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
res[f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5]
json.dump(res, f, ensure_ascii=False, indent=2)
print("\n ✓ 03_核心药组3-5味_Top10.json")
# ========== 6. 功效关键词网络 ==========
print(f"\n[6/8] 功效关键词网络...")
# 提取核心功效词(过滤噪音词)
noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '。'}
func_keywords = Counter()
func_formulas = defaultdict(set)
for h, data in herbs.items():
if h not in herb_formulas:
continue
gua = data.get('功效作用', {})
if not isinstance(gua, dict):
continue
func_str = gua.get('功能', '')
if not func_str:
continue
# 按标点拆分
keywords = re.split(r'[、,,、;;。]', func_str)
for kw in keywords:
kw = kw.strip()
# 过滤噪音词
if not kw or len(kw) < 2 or kw in noise_words:
continue
# 也检查是否全是药材名(如"甘草"也是功效词"甘草具有补脾益气"中的残留)
if kw == h and isinstance(gua, dict) and func_str.startswith(kw):
# 如果关键词就是药材名且紧跟药材名,跳过
continue
if kw in herbs:
# 如果关键词本身是药材名且在功效描述中(非独立功效词),跳过
continue
func_keywords[kw] += 1
for fname in herb_formulas[h]:
func_formulas[kw].add(fname)
# 按方剂数排序
top_funcs = [(kw, len(vx)) for kw, vx in func_formulas.items() if len(vx) >= 10]
top_funcs.sort(key=lambda x: -x[1])
print(f"\n Top 30 功效关键词 (按关联方剂数):")
print(f" {'关键词':<15s} {'方剂数':>6s} 示例方剂")
print(f" {'-'*15} {'-'*6} {'-'*40}")
for kw, count in top_funcs[:30]:
examples = list(func_formulas[kw])[:2]
print(f" {kw:<15s} {count:>6d} {', '.join(examples)}")
with open(os.path.join(OUT, "04_功效关键词_Top30.json"), 'w', encoding='utf-8') as f:
json.dump([{"keyword": kw, "formula_count": count} for kw, count in top_funcs[:30]], f, ensure_ascii=False, indent=2)
print("\n ✓ 04_功效关键词_Top30.json")
# ========== 7. 方剂聚类 ==========
print(f"\n[7/8] 方剂聚类...")
clusters = defaultdict(list)
for fname, hlist in formula_herbs.items():
kw_count = Counter()
for herb in hlist:
if herb in herbs:
gua = herbs[herb].get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kw_str = gua['功能']
keywords = re.split(r'[、,,、;;。]', kw_str)
for kw in keywords:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
kw_count[kw] += 1
if kw_count:
top_kw = kw_count.most_common(1)[0][0]
clusters[top_kw].append(fname)
big_clusters = [(kw, flst) for kw, flst in clusters.items() if len(flst) >= 5]
big_clusters.sort(key=lambda x: -len(x[1]))
print(f" 发现 {len(clusters)} 个功效簇")
print(f"\n Top 20 大功效簇:")
for kw, flst in big_clusters[:20]:
# 找簇内的代表性药材
sample_herbs = set()
for fname in flst[:10]:
for h in formula_herbs[fname]:
sample_herbs.add(h)
top_h = sample_herbs
print(f" '{kw}' ({len(flst)}首方): 代表药材 -> {', '.join(sorted(top_h)[:10])}")
with open(os.path.join(OUT, "05_方剂聚类_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"cluster": kw, "count": len(flst), "formulas": flst[:20]} for kw, flst in big_clusters[:20]], f, ensure_ascii=False, indent=2)
print("\n ✓ 05_方剂聚类_Top20.json")
# ========== 8. 配伍禁忌 & 相似度 ==========
print(f"\n[8/8] 配伍禁忌验证 & 方剂相似度...")
# 十八反
fan = {
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
}
wei = {
'硫黄': ['朴硝', '芒硝', '牙硝'],
'水银': ['铅丹', '砒霜'],
'巴豆': ['牵牛', '牵牛子'],
'丁香': ['郁金'],
'人参': ['五灵脂'],
'肉桂': ['石脂', '赤石脂'],
'半夏': ['羊脂'],
'厚朴': ['硝石', '滑石'],
}
# 检查禁忌药材是否在方剂中出现
contra_found = defaultdict(list)
all_contra_search = {}
for k, v in fan.items():
for vv in v:
if vv not in all_contra_search:
all_contra_search[vv] = k
for k, v in wei.items():
for vv in v:
if vv not in all_contra_search:
all_contra_search[vv] = k
print(f"\n 十八反/十九畏相关药材在方剂中出现情况:")
for herb_name, related in all_contra_search.items():
if herb_name in herb_formulas:
count = len(herb_formulas[herb_name])
print(f" ⚠ {herb_name} (反/畏{related}): 出现在 {count} 首方")
# 检查实际方剂中是否同时出现矛盾配对
contra_pairs_found = defaultdict(int)
contra_pairs_details = defaultdict(list)
for fname, hlist in formula_herbs.items():
for herb, contra_list in fan.items():
if herb in hlist:
for ch in contra_list:
if ch in hlist:
contra_pairs_found[(herb, ch)] += 1
contra_pairs_details[(herb, ch)].append(fname)
for herb, contra_list in wei.items():
if herb in hlist:
for ch in contra_list:
if ch in hlist:
contra_pairs_found[(herb, ch)] += 1
contra_pairs_details[(herb, ch)].append(fname)
top_contra = sorted(contra_pairs_found.items(), key=lambda x: -x[1])[:10]
if top_contra:
print(f"\n 发现配伍禁忌:")
for (h1, h2), count in top_contra:
print(f" ⚠️ {h1} + {h2}: {count}首方 ({', '.join(contra_pairs_details[(h1,h2)][:2])})")
else:
print(f"\n 未发现十八反/十九畏的直接配对出现在同一首方剂中")
with open(os.path.join(OUT, "06_配伍禁忌.json"), 'w', encoding='utf-8') as f:
json.dump({
"十八反": fan,
"十九畏": wei,
"禁忌药材出现": {k: len(herb_formulas.get(k, set())) for k in all_contra_search.keys() if k in herb_formulas},
"实际禁忌配对": {f"{k[0]}+{k[1]}": c for k, c in top_contra},
"禁忌详情": {f"{k[0]}+{k[1]}": contra_pairs_details[k] for k in top_contra}
}, f, ensure_ascii=False, indent=2)
print("\n ✓ 06_配伍禁忌.json")
# 方剂相似度
print(f"\n 计算方剂相似度 (Jaccard >= 0.3)...")
def jaccard_fixed(s1, s2):
if not s1 or not s2:
return 0.0
intersection = len(s1 & s2)
if intersection == 0:
return 0.0
union = len(s1 | s2)
return intersection / union if union > 0 else 0.0
similar_pairs = []
formula_list = list(formulas.keys())
count = 0
for i in range(len(formula_list)):
f1 = formula_list[i]
h1 = formula_herbs.get(f1, [])
if not h1:
continue
s1 = set(h1)
for j in range(i + 1, len(formula_list)):
f2 = formula_list[j]
h2 = formula_herbs.get(f2, [])
if not h2:
continue
s2 = set(h2)
sim = jaccard_fixed(s1, s2)
if sim >= 0.3:
shared = s1 & s2
similar_pairs.append((f1, f2, sim, len(shared), sorted(shared)))
count += 1
print(f" 检查 {count} 对, 找到 {len(similar_pairs)} 对相似方剂")
similar_pairs.sort(key=lambda x: (-x[2], -x[3]))
top_similar = similar_pairs[:20]
print(f"\n Top 15 相似方剂对:")
for f1, f2, sim, shared_c, shared_h in top_similar[:15]:
print(f" {f1:30s} + {f2:30s} -> {sim:.3f} (共享{shared_c}味: {', '.join(shared_h)})")
with open(os.path.join(OUT, "07_方剂相似度_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"similarity": round(sim, 4), "shared_count": sc, "pair1": f1, "pair2": f2, "shared_herbs": sh} for f1, f2, sim, sc, sh in top_similar], f, ensure_ascii=False, indent=2)
print("\n ✓ 07_方剂相似度_Top20.json")
# ========== 汇总 ==========
print(f"\n{'='*70}")
print(f" 深度挖掘分析完成!")
print(f"{'='*70}")
summary = OrderedDict()
summary["标题"] = "大医网 方剂-中药材 深度数据挖掘报告 (精修修正版)"
summary["数据规模"] = {
"药材": len(herbs),
"方剂": len(formulas),
"总关联对数": total_links,
"平均每方药材数": round(avg, 1),
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
"无法提取方剂": sum(1 for h in formula_herbs.values() if not h),
}
summary["高频药材_Top20"] = [{"name": h, "count": c} for h, c in top30[:20]]
summary["高频药对_Top10"] = [{"pair": list(p), "count": c} for p, c in pair_freq.most_common(10)]
summary["核心药组_Top10"] = {}
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
summary["核心药组_Top10"][f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5]
summary["功效关键词_Top15"] = [{"keyword": kw, "count": c} for kw, c in top_funcs[:15]]
summary["方剂聚类_Top10"] = [{"cluster": kw, "count": len(flst)} for kw, flst in big_clusters[:10]]
summary["配伍禁忌"] = {
"十八反": fan,
"十九畏": wei,
"实际发现": {f"{k[0]}+{k[1]}": c for k, c in top_contra},
}
summary["方剂相似度_Top10"] = [{"similarity": round(sim, 4), "pair": f"{f1} + {f2}"} for f1, f2, sim, sc, sh in top_similar[:10]]
summary["输出目录"] = os.path.abspath(OUT)
with open(os.path.join(OUT, "汇总报告.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print(f"\n 输出目录: {os.path.abspath(OUT)}")
print(f"\n 文件列表:")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:40s} {size:>10,} B")
@@ -0,0 +1,411 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
大医网深度挖掘 - 进阶版
1. 功效-方剂完整映射网络
2. 方剂来源分析(出处)
3. 药材分类关联
4. 性味归经交叉分析
5. 药对-功效关联
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_进阶")
os.makedirs(OUT, exist_ok=True)
print("=" * 80)
print(" 大医网深度数据挖掘 - 进阶版 (多维关联分析)")
print("=" * 80)
# ========== 加载数据 ==========
print("\n[1/6] 加载数据...")
herbs = {} # name -> data
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', '').strip()
if name:
herbs[name] = d
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
formulas[name] = d
print(f" 药材: {len(herbs)} 味")
print(f" 方剂: {len(formulas)} 首")
# 药材匹配字典 (仅主名, ≥2字)
valid_herb_names = [n for n in herbs.keys() if len(n) >= 2]
valid_herb_names.sort(key=len, reverse=True)
herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names))
# 提取药材
def extract_herbs(text):
if not text or len(text) < 3:
return []
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克]', '', clean)
found = []
for match in herb_pattern.finditer(clean):
herb = match.group(0)
if herb and herb not in found:
found.append(herb)
return found
formula_herbs = OrderedDict()
for fname, data in formulas.items():
all_text = ''
for val in data.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
h = extract_herbs(all_text)
formula_herbs[fname] = h
# 逆向索引
herb_formulas = defaultdict(set)
for fname, hlist in formula_herbs.items():
for h in hlist:
herb_formulas[h].add(fname)
total_links = sum(len(h) for h in formula_herbs.values())
print(f"\n 总关联对数: {total_links}")
print(f" 成功提取方剂: {sum(1 for h in formula_herbs.values() if h)} / {len(formulas)}")
# ========== 分析1: 功效-方剂完整映射 ==========
print(f"\n{'='*80}")
print(f"[2/6] 功效-方剂映射网络. ..")
print(f"{'='*80}")
noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '的功能', '功效作用'}
# 构建功效到药材的映射
func_to_herbs = defaultdict(set)
for h, data in herbs.items():
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
keywords = re.split(r'[、,,、;;。]', func_str)
for kw in keywords:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words and kw != h:
func_to_herbs[kw].add(h)
# 构建功效到方剂的映射
func_to_formulas = defaultdict(set)
for h, data in herbs.items():
if h not in herb_formulas:
continue
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
keywords = re.split(r'[、,,、;;。]', func_str)
for kw in keywords:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
# 如果关键词是药材名且紧跟药材名,跳过
if kw == h and func_str.startswith(kw):
continue
# 如果关键词本身就是药材名(如"甘草"也是功效词中), 检查是否真的是独立功效词
if kw in herbs and kw != h and not func_str.startswith(kw):
continue
for fname in herb_formulas[h]:
func_to_formulas[kw].add(fname)
# 过滤掉太稀疏的功效(如<5首方)
func_to_formulas = {k: v for k, v in func_to_formulas.items() if len(v) >= 5}
# 按方剂数排序
top_funcs = [(k, len(v)) for k, v in func_to_formulas.items()]
top_funcs.sort(key=lambda x: -x[1])
print(f"\n 功效关键词总数: {len(func_to_formulas)}")
print(f"\n 高频功效 (≥10首方):")
print(f" {'功效关键词':<20s} {'方剂数':>6s} {'关联药材数':>8s} 示例方剂")
print(f" {'-'*20} {'-'*6} {'-'*8} {'-'*40}")
for func, count in top_funcs[:30]:
herb_count = len(func_to_herbs.get(func, set()))
examples = list(func_to_formulas[func])[:2]
print(f" {func:<20s} {count:>6d} {herb_count:>8d} {', '.join(examples)}")
with open(os.path.join(OUT, "A_功效-方剂映射网络.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for func, count in top_funcs[:50]:
d[func] = {
"formula_count": count,
"related_herbs": len(func_to_herbs.get(func, set())),
"formulas": list(func_to_formulas[func])[:30],
"related_herb_list": list(func_to_herbs.get(func, set()))[:20]
}
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ A_功效-方剂映射网络.json")
# ========== 分析2: 方剂来源(出处)分析 ==========
print(f"\n{'='*80}")
print(f"[3/6] 方剂来源分析. ..")
print(f"{'='*80}")
# 提取出处信息
source_counter = Counter()
source_formulas = defaultdict(list)
source_herb_total = defaultdict(int) # 该出处方剂的平均药材数
for fname, data in formulas.items():
source = data.get('出处', '').strip()
if not source:
source = "未知"
else:
# 简化出处名称
source = source.split('《')[0].strip() if '《' in source else source
source = source[:30] # 截断
source_counter[source] += 1
source_formulas[source].append(fname)
source_herb_total[source] += len(formula_herbs.get(fname, []))
top_sources = source_counter.most_common(30)
print(f"\n 方剂出处统计 (Top 30):")
for src, count in top_sources:
avg_herbs = round(source_herb_total[src] / max(1, count), 1)
sample = source_formulas[src][:2]
print(f" {src:<30s} {count:>4d}首 (均药数{avg_herbs}) 例: {', '.join(sample)}")
with open(os.path.join(OUT, "B_方剂来源统计.json"), 'w', encoding='utf-8') as f:
json.dump({
"total_unique_sources": len(source_counter),
"top_sources": [{"source": s, "count": c, "avg_herbs": round(source_herb_total[s]/max(1,c), 1), "formulas": source_formulas[s][:10]} for s, c in top_sources]
}, f, ensure_ascii=False, indent=2)
print("\n ✓ B_方剂来源统计.json")
# ========== 分析3: 药材分类关联 ==========
print(f"\n{'='*80}")
print(f"[4/6] 药材分类关联分析. ..")
print(f"{'='*80}")
# 统计各药材分类的出现频率
class_counter = Counter()
class_herbs_list = defaultdict(set)
class_formulas = defaultdict(set)
for h, data in herbs.items():
cat = data.get('药材分类', '未知')
class_counter[cat] += 1
class_herbs_list[cat].add(h)
if h in herb_formulas:
class_formulas[cat].update(herb_formulas[h])
top_classes = class_counter.most_common(20)
print(f"\n 药材分类统计:")
for cls, count in top_classes:
total_formulas_class = len(class_formulas.get(cls, set()))
print(f" {cls:<10s} {count:>4d}味 关联方剂{total_formulas_class}首")
# 按分类统计高频药材
class_top_herbs = {}
for cls in [c for c, _ in top_classes]:
herb_freq_cls = Counter()
for h in class_herbs_list[cls]:
herb_freq_cls[h] = len(herb_formulas.get(h, set()))
class_top_herbs[cls] = herb_freq_cls.most_common(10)
print(f"\n 各类别高频药材 (Top 5):")
for cls in ['植物', '动物', '矿物', '其他']:
if cls in class_top_herbs:
print(f"\n [{cls}类]:")
for h, c in class_top_herbs[cls][:5]:
gua = herbs.get(h, {}).get('功效作用', {})
func = gua.get('功能', '')[:20] if isinstance(gua, dict) else ''
print(f" {h:12s} {c:>4d}首 [{func}]")
with open(os.path.join(OUT, "C_药材分类关联.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
d["分类统计"] = [{"class": c, "herb_count": cnt, "formula_count": len(class_formulas.get(c, set()))} for c, cnt in top_classes]
d["各类别高频药材"] = class_top_herbs
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ C_药材分类关联.json")
# ========== 分析4: 性味归经交叉分析 ==========
print(f"\n{'='*80}")
print(f"[5/6] 性味归经交叉分析. ..")
print(f"{'='*80}")
# 统计性状的分布
xing_counter = Counter()
xing_formulas = defaultdict(set)
for h, data in herbs.items():
xing = data.get('性味', {}).get('性', '') if isinstance(data.get('性味'), dict) else data.get('性味', '')
if xing:
xing_counter[xing] += 1
if h in herb_formulas:
xing_formulas[xing].update(herb_formulas[h])
top_xing = xing_counter.most_common(20)
print(f"\n 药材性味分布 (Top 20):")
for xing, count in top_xing:
formula_count = len(xing_formulas.get(xing, set()))
print(f" {xing:<10s} {count:>4d}味 关联方剂{formula_count}首")
# 统计归经分布
jing_counter = Counter()
jing_formulas = defaultdict(set)
for h, data in herbs.items():
jing = data.get('性味', {}).get('归经', '') if isinstance(data.get('性味'), dict) else ''
if jing:
# 拆分归经(可能逗号分隔)
jings = [j.strip() for j in re.split(r'[、,,;]', jing) if j.strip()]
for j in jings:
jing_counter[j] += 1
if h in herb_formulas:
jing_formulas[j].update(herb_formulas[h])
top_jing = jing_counter.most_common(20)
print(f"\n 药材归经分布 (Top 20):")
for jing, count in top_jing:
formula_count = len(jing_formulas.get(jing, set()))
print(f" {jing:<10s} {count:>4d}味 关联方剂{formula_count}首")
with open(os.path.join(OUT, "D_性味归经交叉分析.json"), 'w', encoding='utf-8') as f:
json.dump({
"性味分布": [{"xing": x, "count": c, "formula_count": len(xing_formulas.get(x, set()))} for x, c in top_xing],
"归经分布": [{"jing": j, "count": c, "formula_count": len(jing_formulas.get(j, set()))} for j, c in top_jing]
}, f, ensure_ascii=False, indent=2)
print("\n ✓ D_性味归经交叉分析.json")
# ========== 分析5: 药对-功效关联 ==========
print(f"\n{'='*80}")
print(f"[6/6] 药对-功效关联. ..")
print(f"{'='*80}")
herb_freq = Counter()
for hlist in formula_herbs.values():
for h in hlist:
herb_freq[h] += 1
pair_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= 2:
for combo in combinations(u, 2):
pair_freq[combo] += 1
# 对高频药对分析其对应功效
top_pairs = [(p, c) for p, c in pair_freq.most_common(100) if c >= 5]
print(f"\n 高频药对的功效关联 (Top 15):")
print(f" {'药对':<30s} {'频次':>5s} {'高频功效':<40s}")
print(f" {'-'*30} {'-'*5} {'-'*40}")
for (h1, h2), count in top_pairs[:15]:
# 分析该药对出现在哪些方剂,统计这些方剂中高频药材的功效
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
func_on_pairs = Counter()
for pname in pair_formulas:
pdata = formulas.get(pname, {})
gua = pdata.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kws = re.split(r'[、,,、;;。]', gua['功能'])
for kw in kws:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
func_on_pairs[kw] += 1
# 取该药对最关联的3个功效
top_3 = func_on_pairs.most_common(3)
top_3_str = ", ".join([f"{k}({v})" for k, v in top_3[:3]])
print(f" {h1:12s} + {h2:12s} {count:>5d} [{top_3_str}]")
with open(os.path.join(OUT, "E_药对-功效关联.json"), 'w', encoding='utf-8') as f:
d = []
for (h1, h2), count in top_pairs[:30]:
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
func_on_pairs = Counter()
for pname in pair_formulas:
pdata = formulas.get(pname, {})
gua = pdata.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kws = re.split(r'[、,,、;;。]', gua['功能'])
for kw in kws:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
func_on_pairs[kw] += 1
top_3 = func_on_pairs.most_common(3)
d.append({
"pair": list((h1, h2)),
"count": count,
"top_functions": [{"function": k, "count": v} for k, v in top_3]
})
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ E_药对-功效关联.json")
# ========== 汇总 ==========
print(f"\n{'='*80}")
print(f" 深度挖掘 - 进阶版 分析完成!")
print(f"{'='*80}")
summary = OrderedDict()
summary["标题"] = "大医网 方剂-中药材 深度挖掘 (进阶版)"
summary["数据规模"] = {
"药材": len(herbs),
"方剂": len(formulas),
"总关联对数": total_links,
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
}
summary["功效-方剂映射"] = {
"功效关键词数": len(func_to_formulas),
"高频功效": [{"function": k, "formula_count": c} for k, c in top_funcs[:20]]
}
summary["方剂来源"] = {
"独特来源数": len(source_counter),
"Top5来源": [{"source": s, "count": c} for s, c in top_sources[:5]]
}
summary["药材分类"] = {
"分类数": len(class_counter),
"Top5分类": [{"class": c, "herb_count": cnt, "formula_count": len(class_formulas.get(c, set()))} for c, cnt in top_classes[:5]]
}
summary["性味分布"] = [{"xing": x, "count": c} for x, c in top_xing[:10]]
summary["归经分布"] = [{"jing": j, "count": c} for j, c in top_jing[:10]]
# 药对-功效关联 (单独处理)
pair_func_list = []
for idx, ((h1, h2), count) in enumerate(top_pairs[:10]):
if idx >= 10:
break
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
func_on_pairs = Counter()
for pname in pair_formulas:
pdata = formulas.get(pname, {})
gua = pdata.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kws = re.split(r'[、,,、;;。]', gua['功能'])
for kw in kws:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
func_on_pairs[kw] += 1
top_3_str = [f"{k}({v})" for k, v in func_on_pairs.most_common(3)]
pair_func_list.append({"pair": [h1, h2], "count": count, "top_funcs": top_3_str})
summary["药对-功效关联Top10"] = pair_func_list
summary["输出目录"] = os.path.abspath(OUT)
with open(os.path.join(OUT, "汇总报告_进阶.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print(f"\n 输出目录: {os.path.abspath(OUT)}")
print(f"\n 文件列表:")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:30s} {size:>10,} B")
print(f"{'='*80}")
@@ -0,0 +1,311 @@
#!/usr/bin/env python3
"""药膳食疗 × 中药材 深度关联分析"""
import json, os, glob, re
from collections import Counter, defaultdict
TOP_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/"
DY_DIR = os.path.join(TOP_DIR, "药膳食疗/")
ZY_DIR = os.path.join(TOP_DIR, "中药材/")
# ===================== 加载数据 =====================
print("加载药膳食疗库...")
all_diets = []
for f in glob.glob(os.path.join(DY_DIR, "*.json")):
with open(f, encoding='utf-8') as fh:
all_diets.append(json.load(fh))
print("加载中药材库...")
all_herbs = {}
herb_aliases = defaultdict(set) # 别名→标准名
for f in glob.glob(os.path.join(ZY_DIR, "*.json")):
with open(f, encoding='utf-8') as fh:
h = json.load(fh)
name = h.get('名称', '')
all_herbs[name] = h
for alias in str(h.get('别名', '')).split('、'):
a = alias.strip()
if a:
herb_aliases[a].add(name)
print(f"药膳: {len(all_diets)} 条 | 药材: {len(all_herbs)} 条 | 别名映射: {len(herb_aliases)} 组\n")
# ===================== 药材标准化匹配 =====================
# 合并药材名+别名集合,按长度排序(最长优先)
all_herb_names = sorted(set(all_herbs.keys()), key=lambda x: -len(x))
# 别名去标准名重复
pure_aliases = []
for alias, targets in herb_aliases.items():
if alias not in all_herbs:
pure_aliases.append((alias, targets))
# 按长度降序
pure_aliases.sort(key=lambda x: -len(x[0]))
def match_herbs(text):
"""从文本中匹配药材(标准名优先,别名次之)"""
found = set()
for name in all_herb_names:
if len(name) >= 2 and name in text:
found.add(name)
for alias, targets in pure_aliases:
if len(alias) >= 2 and alias in text:
found.update(targets)
return found
# ===================== Step 1: 药材频次与覆盖分析 =====================
print("=" * 60)
print("Step 1: 药材使用频次与覆盖度分析")
print("=" * 60)
herb_in_diet_count = Counter()
herb_diet_pairs = defaultdict(list) # 药材→使用该药材的药膳列表
for diet in all_diets:
pei_fang = diet.get('配方', '')
matched = match_herbs(pei_fang)
for h in matched:
herb_in_diet_count[h] += 1
herb_diet_pairs[h].append(diet['名称'])
used_herb_count = len(herb_in_diet_count)
coverage = used_herb_count / len(all_herbs) * 100
print(f"\n药材库总数: {len(all_herbs)}")
print(f"在药膳中出现: {used_herb_count} ({coverage:.1f}%)")
print(f"未出现在药膳中: {len(all_herbs) - used_herb_count}")
print(f"\n>>> TOP 30 高频药材 <<<")
for herb, cnt in herb_in_diet_count.most_common(30):
# 获取药材功效(字段名是'功能')
h_data = all_herbs.get(herb, {})
gongxiao = ""
if isinstance(h_data.get('功效作用'), dict):
gongxiao = h_data['功效作用'].get('功能', '')
else:
gongxiao = str(h_data.get('功效作用', ''))
gongxiao_short = (gongxiao[:30] + '..') if len(gongxiao) > 30 else gongxiao
print(f" {herb:6s}: {cnt:3d}次 | 药材功效: {gongxiao_short}")
# ===================== Step 2: 双向映射 =====================
print(f"\n{'=' * 60}")
print("Step 2: 药膳-药材双向映射")
print("=" * 60)
# 药膳→药材映射
diet_to_herbs = {}
for diet in all_diets:
pei_fang = diet.get('配方', '')
matched = match_herbs(pei_fang)
diet_to_herbs[diet['名称']] = matched
# 药材→药膳 已在上一步构建
# 每味药材关联的药膳数量分布
pair_dist = Counter()
for herb, diets in herb_diet_pairs.items():
pair_dist[len(diets)] += 1
print(f"\n药材→药膳关联数分布:")
for k in sorted(pair_dist.keys())[:10]:
print(f" 关联{k}个药膳: {pair_dist[k]}味药材")
if len(pair_dist) > 10:
tail_sum = sum(v for k, v in pair_dist.items() if k > 10)
print(f" 关��>10个药膳: {tail_sum}味药材")
# 药膳→药材关联数分布
diet_herb_dist = Counter()
for diet_name, herbs in diet_to_herbs.items():
diet_herb_dist[len(herbs)] += 1
print(f"\n药膳→药材关联数分布:")
for k in sorted(diet_herb_dist.keys())[:8]:
print(f" 含{k}味药材: {diet_herb_dist[k]}个药膳")
if len(diet_herb_dist) > 8:
tail_sum = sum(v for k, v in diet_herb_dist.items() if k > 8)
print(f" 含>8味药材: {tail_sum}个药膳")
# 核心枢纽药材(同时出现在药膳和药材库,且频次最高的)
hub_herbs = []
for herb, cnt in herb_in_diet_count.most_common(10):
h_data = all_herbs[herb]
gx = str(h_data.get('功效作用', ''))
hub_herbs.append((herb, cnt, gx))
print(f"\n>>> 10大枢纽药材 <<<")
for herb, cnt, gx in hub_herbs:
print(f" {herb}: 出现{cnt}次, 药材功效: {str(gx)[:50]}")
# ===================== Step 3: 功效交叉验证 =====================
print(f"\n{'=' * 60}")
print("Step 3: 功效交叉验证分析")
print("=" * 60)
# 提取药膳功效标签
diet_gx_labels = []
for diet in all_diets:
gx = diet.get('功效', '')
diet_gx_labels.append((diet['名称'], gx))
# 提取药材功效标签
herb_gx_labels = {}
for name, hdata in all_herbs.items():
gx = str(hdata.get('功效作用', ''))
herb_gx_labels[name] = gx
# 药膳功效大类分布
gx_categories = Counter()
GX_CAT_MAP = {
'补气': '补益', '补血': '补益', '补阴': '补益', '滋阴': '补益', '补肾': '补益',
'补脾': '补益', '健脾': '补益', '补肝': '补益', '益精': '补益', '壮阳': '补益',
'温阳': '补益', '扶正': '补益', '养心': '补益', '养阴': '补益', '益气': '补益',
'清热': '清热', '解毒': '清热', '泻火': '清热', '凉血': '清热', '滋阴': '滋阴',
'活血': '活血', '化瘀': '活血', '祛瘀': '活血', '行气': '理气', '理气': '理气',
'止痛': '止痛', '安神': '安神', '止咳': '止咳化痰', '化痰': '止咳化痰', '润肺': '止咳化痰',
'祛风': '祛风除湿', '除湿': '祛风除湿', '散寒': '散寒解表', '解表': '散寒解表',
'利水': '利水渗湿', '通便': '通淋', '明目': '明目'
}
for diet_name, gx_text in diet_gx_labels:
matched_cats = set()
for kw, cat in GX_CAT_MAP.items():
if kw in gx_text:
matched_cats.add(cat)
if not matched_cats:
matched_cats.add('其他')
gx_categories.update(matched_cats)
print(f"\n药膳功效大类分布:")
for cat, cnt in gx_categories.most_common():
print(f" {cat}: {cnt}条 ({cnt/len(all_diets)*100:.1f}%)")
# 交叉验证:药膳功效 vs 组成药材功效一致性
# 检查药膳功效描述中的关键词是否在药材功效中也有体现
consistency_count = 0
total_checked = 0
for diet in all_diets:
diet_name = diet['名称']
diet_gx = diet.get('功效', '')
herbs_used = diet_to_herbs.get(diet_name, set())
if not herbs_used:
continue
# 提取药膳功效中的 2-3 字词
gx_words = re.findall(r'[\.。,.、,]*(.{2,3})[\.。,.、,]?', diet_gx)
if not gx_words:
gx_words = [diet_gx[:4]] # fallback
# 检查药材功效是否包含相同关键词
consistent_words = 0
for word in gx_words:
if len(word) < 2:
continue
for h_name in herbs_used:
h_gx = herb_gx_labels.get(h_name, '')
if word in h_gx:
consistent_words += 1
break # 只要有一个药材匹配就算
total_checked += 1
if consistent_words > 0:
consistency_count += 1
consistency_pct = consistency_count / total_checked * 100 if total_checked > 0 else 0
print(f"\n功效一致性验证:")
print(f" 参与验证药膳: {total_checked}")
print(f" 功效可追溯至组成药材: {consistency_count} ({consistency_pct:.1f}%)")
# ===================== Step 4: 特殊分析 — 药食同源药材统计 =====================
print(f"\n{'='*60}")
print("Step 4: 药食同源药材深度分析")
print("="*60)
# 常见药食同源药材名单(国家卫健委公布的)
med_food_herbs = {
'山药', '枸杞子', '枸杞', '茯苓', '甘草', '生姜', '大枣', '红枣',
'薏苡仁', '百合', '桂圆肉', '山楂', '陈皮', '菊花', '蜂蜜', '花椒',
'丁香', '八角茴香', '小茴香', '胡椒', '砂仁', '肉桂', '八角',
'杏仁', '核桃仁', '桃仁', '莲子', '芡实', '白果', '桑葚', '黑芝麻',
'黑芝麻', '黑芝麻', '黑豆', '黄豆', '黑芝麻', '黑芝麻', '黑芝麻',
'地黄', '玉竹', '黄精', '决明子', '罗汉果', '桑叶', '洛神花',
'蜂蜜', '党参', '黄芪', '红枣', '当归', '何首乌', '杜仲', '骨碎补',
'白果', '核桃', '杏仁', '桃仁', '花椒', '胡椒', '香薷', '高良姜',
'砂仁', '肉桂', '草果', '白芷', '白豆蔻', '肉豆蔻', '草豆蔻',
'紫苏', '薄荷', '胖大海', '鱼腥草', '姜黄', '荜澄茄', '青果', '橘皮',
'罗汉果', '乌梅', '陈皮', '大枣', '甘草', '丁香', '八角茴香', '山奈',
'阴香', '草果', '肉蔻', '草豆蔻', '砂仁', '白豆蔻', '肉豆蔻', '高良姜',
'香薷', '花椒', '小茴香', '八角', '桂花'
}
# 统计药食同源药材在药膳中的使用
yf_usage = Counter()
for diet in all_diets:
pei_fang = diet.get('配方', '')
for hname in med_food_herbs:
if hname in pei_fang and hname in herb_in_diet_count:
yf_usage[hname] += 1
print(f"\n药食同源药材使用 TOP 15:")
for herb, cnt in yf_usage.most_common(15):
print(f" {herb}: {cnt}次")
non_yf = herb_in_diet_count - yf_usage
print(f"\n非药食同源药材使用 TOP 10:")
for herb, cnt in non_yf.most_common(10):
print(f" {herb}: {cnt}次")
# ===================== Step 5: 药膳配伍规律分析 =====================
print(f"\n{'='*60}")
print("Step 5: 药膳配伍规律分析")
print("="*60)
# 药材对共现分析
herb_pair_freq = Counter()
for diet in all_diets:
pei_fang = diet.get('配方', '')
herbs = match_herbs(pei_fang)
herb_list = sorted(list(herbs))
for i in range(len(herb_list)):
for j in range(i+1, len(herb_list)):
pair = (herb_list[i], herb_list[j])
herb_pair_freq[pair] += 1
print(f"\n药材共现TOP 15:")
for (h1, h2), cnt in herb_pair_freq.most_common(15):
print(f" {h1}+{h2}: {cnt}次")
# ===================== 保存索引 =====================
# 双向映射索引
bidirectional = {
"药膳→药材": {name: sorted(list(herbs)) for name, herbs in diet_to_herbs.items()},
"药材→药膳": {herb: sorted(diets) for herb, diets in herb_diet_pairs.items()}
}
# 只保存TOP部分(全量太大)
top_herb_diet = {herb: sorted(diets) for herb, diets in herb_diet_pairs.items() if len(diets) >= 5}
top_diet_herb = {name: sorted(list(herbs)) for name, herbs in diet_to_herbs.items() if len(herbs) >= 2}
# 保存
with open(os.path.join(TOP_DIR, "索引_药膳药材双向映射.json"), "w", encoding='utf-8') as f:
json.dump({
"统计": {
"药材在药膳中使用数": len(herb_in_diet_count),
"药膳含药材数": sum(1 for h in diet_to_herbs.values() if h),
"覆盖率": round(coverage, 2)
},
"高频药材→药膳": {k: sorted(v) for k, v in sorted(herb_diet_pairs.items(), key=lambda x: -len(x))[:50]},
"药材共现TOP": herb_pair_freq.most_common(30)
}, f, ensure_ascii=False, indent=2)
# 功效交叉验证索引
with open(os.path.join(TOP_DIR, "索引_功效交叉验证.json"), "w", encoding='utf-8') as f:
json.dump({
"药膳功效大类": dict(gx_categories.most_common()),
"功效一致率": round(consistency_pct, 2),
"药食同源使用TOP": yf_usage.most_common(30),
"非药食同源使用TOP": [(h, c) for h, c in sorted(non_yf.items(), key=lambda x: -x[1])[:20]]
}, f, ensure_ascii=False, indent=2)
print(f"\n索引文件保存完毕")
print(f"\n{'='*60}")
print("分析完成!")
print("="*60)
@@ -0,0 +1,198 @@
#!/usr/bin/env python3
"""药膳食疗数据考古:盘点、分类、索引、交叉关联"""
import json, os, glob, re
from collections import Counter, defaultdict
TOP_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/"
DATA_DIR = os.path.join(TOP_DIR, "药膳食疗/")
# ====== Step 1: 加载全部数据 ======
all_items = []
for f in glob.glob(os.path.join(DATA_DIR, "*.json")):
with open(f, encoding='utf-8') as fh:
all_items.append(json.load(fh))
total = len(all_items)
print(f"=== 总条目数: {total} ===\n")
# ====== Step 2: 字段完整率 ======
required_fields = ['url', '名称', '简介', '功效', '配方', '来源', '适宜人群', '相关配伍']
optional_fields = ['做法', '食用方法', '元_名称', '认证者']
has_field = Counter()
for field in required_fields + optional_fields:
for item in all_items:
if item.get(field) and item[field].strip():
has_field[field] += 1
print("=== 字段完整率 ===")
for field in required_fields + optional_fields:
count = has_field[field]
pct = count / total * 100
print(f" {field}: {count} ({pct:.1f}%)")
# ====== Step 3: 功效关键词 TOP 20 ======
gongxiao_keywords = [
'补血', '补气', '滋阴', '温阳', '清热', '解毒', '祛风', '除湿',
'活血', '化瘀', '润肺', '止咳', '化痰', '安神', '止痛', '利水',
'消食', '理气', '止血', '补脾', '益肾', '疏肝', '健脾', '通便', '泻火',
'滋补肝肾', '健脾益气', '养胃', '养阴', '固本', '养心', '养肺',
'解表', '散寒', '润燥', '开窍', '固表', '益精', '温中', '润肠',
'补肾', '壮阳', '明目', '消肿', '降火', '养肤', '美容', '益智'
]
gongxiao_index = defaultdict(set)
gongxiao_counter = Counter()
for item in all_items:
gx = item.get('功效', '') + item.get('简介', '')
for kw in gongxiao_keywords:
if kw in gx:
gongxiao_index[kw].add(item['名称'])
gongxiao_counter[kw] += 1
print(f"\n=== 功效关键词 TOP 20 ===")
for kw, cnt in gongxiao_counter.most_common(20):
print(f" {kw}: {cnt}条 ({cnt/total*100:.1f}%)")
# ====== Step 4: 药膳分类统计 ======
cat_pattern = re.compile(r'为([^,,、.]+?)类药膳配方')
categories = Counter()
for item in all_items:
m = cat_pattern.search(item.get('简介', ''))
if m:
categories[m.group(1)] += 1
unclassified = total - sum(categories.values())
print(f"\n=== 药膳分类统计 ===")
for cat, cnt in categories.most_common():
print(f" {cat}: {cnt}")
print(f" 未标注类别: {unclassified}")
# ====== Step 5: 来源典籍统计 ======
source_counter = Counter()
for item in all_items:
src = item.get('来源', '')
if src:
source_counter[src] += 1
print(f"\n=== 来源典籍 TOP 15 ===")
for src, cnt in source_counter.most_common(15):
print(f" {src}: {cnt}")
print(f" 总来源种类: {len(source_counter)}")
# ====== Step 6: 剂型统计 ======
ji_xing = Counter()
for item in all_items:
name = item['名称']
if name.endswith('粥'): ji_xing['粥'] += 1
elif name.endswith('汤'): ji_xing['汤'] += 1
elif name.endswith('羹'): ji_xing['羹'] += 1
elif name.endswith('膏'): ji_xing['膏'] += 1
elif name.endswith('酒'): ji_xing['酒'] += 1
elif name.endswith('茶'): ji_xing['茶'] += 1
elif name.endswith('饼'): ji_xing['饼'] += 1
elif name.endswith('糕'): ji_xing['糕'] += 1
elif name.endswith('丸'): ji_xing['丸'] += 1
elif name.endswith('饮'): ji_xing['饮'] += 1
elif name.endswith('饭'): ji_xing['饭'] += 1
elif name.endswith('汁'): ji_xing['汁'] += 1
elif name.endswith('包') or name.endswith('包子'): ji_xing['包子'] += 1
else: ji_xing['其他'] += 1
print(f"\n=== 药膳剂型统计 ===")
for p, cnt in ji_xing.most_common():
print(f" {p}: {cnt}")
# ====== Step 7: 高频药材匹配 ======
known_herbs = set()
herbs_dir = os.path.join(TOP_DIR, "中药材/")
if os.path.exists(herbs_dir):
for hf in glob.glob(os.path.join(herbs_dir, "*.json")):
with open(hf, encoding='utf-8') as fh:
h = json.load(fh)
known_herbs.add(h.get('名称', ''))
for alias in str(h.get('别名', '')).split('、'):
if alias.strip():
known_herbs.add(alias.strip())
known_herbs.discard('')
sorted_herbs = sorted(known_herbs, key=lambda x: -len(x))
herb_usage = Counter()
for item in all_items:
pei_fang = item.get('配方', '')
for hname in sorted_herbs:
if len(hname) >= 2 and hname in pei_fang:
herb_usage[hname] += 1
print(f"\n=== 高频中药材 TOP 20 ===")
for herb, cnt in herb_usage.most_common(20):
print(f" {herb}: {cnt}条")
print(f"\n 中药材库已知药材数: {len(known_herbs)}")
print(f" 在药膳中出现的药材数: {len(herb_usage)}")
# ====== Step 8: 交叉关联 — 药膳 vs 方剂 ======
formula_dir = os.path.join(TOP_DIR, "方剂/")
formulas = []
if os.path.exists(formula_dir):
for ff in glob.glob(os.path.join(formula_dir, "*.json")):
with open(ff, encoding='utf-8') as fh:
formulas.append(json.load(fh))
# 提取药膳功效关键词,匹配方剂
diet_gx_words = defaultdict(set)
for item in all_items:
gx_text = item.get('功效', '') + item.get('简介', '')
# 提取 2-4 字功效短语
for kw in gongxiao_keywords:
if kw in gx_text:
for f_item in formulas:
f_text = f_item.get('简介', '') + f_item.get('运用', '')
if kw in f_text:
diet_gx_words[kw].add((item['名称'], f_item.get('名称', '')))
cross_diet_formula = Counter()
for kw, pairs in diet_gx_words.items():
if pairs:
cross_diet_formula[kw] = len(pairs)
print(f"\n=== 药膳-方剂跨库功效关联 TOP 15 ===")
for kw, cnt in cross_diet_formula.most_common(15):
print(f" {kw}: {cnt}对关联")
# ====== Step 9: 保存索引文件 ======
# 功效索引
gx_save = {kw: sorted(names) for kw, names in gongxiao_index.items()}
with open(os.path.join(TOP_DIR, "索引_药膳功效关键词.json"), "w", encoding='utf-8') as f:
json.dump(gx_save, f, ensure_ascii=False, indent=2)
# 分类索引
cat_index = defaultdict(list)
for item in all_items:
m = cat_pattern.search(item.get('简介', ''))
cat = m.group(1) if m else '未标注'
cat_index[cat].append(item['名称'])
with open(os.path.join(TOP_DIR, "索引_药膳分类.json"), "w", encoding='utf-8') as f:
json.dump({k: sorted(v) for k, v in cat_index.items()}, f, ensure_ascii=False, indent=2)
# 高频药材
with open(os.path.join(TOP_DIR, "索引_药膳高频药材.json"), "w", encoding='utf-8') as f:
json.dump(herb_usage.most_common(50), f, ensure_ascii=False, indent=2)
# 数据摘要
summary = {
"总条目数": total,
"字段完整率": {field: round(has_field[field] / total * 100, 1) for field in required_fields + optional_fields},
"药膳分类": dict(categories.most_common()),
"来源典籍数": len(source_counter),
"TOP15来源": source_counter.most_common(15),
"高频药材数": len(herb_usage),
"剂型分布": dict(ji_xing.most_common())
}
with open(os.path.join(TOP_DIR, "数据摘要_药膳食疗.json"), "w", encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print("\n=== 索引文件已保存 ===")
print("\n归档数据保存完毕!")