448 lines
17 KiB
Python
448 lines
17 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
深度数据挖掘 - 精修修正版
|
||
核心修正:
|
||
1. 药材匹配仅使用药材主名(name),不含单字别名
|
||
2. 药材名按长度降序排列,确保长名优先匹配
|
||
3. 功效关键词过滤噪音(去除"本品""诸药"等解析残余)
|
||
4. 药对/药组仅统计完整药材名
|
||
"""
|
||
import json
|
||
import os
|
||
import re
|
||
import glob
|
||
from collections import Counter, defaultdict, OrderedDict
|
||
from itertools import combinations
|
||
|
||
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
|
||
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_精修2")
|
||
os.makedirs(OUT, exist_ok=True)
|
||
|
||
# ========== 加载药材 ==========
|
||
print("=" * 70)
|
||
print(" 大医网 方剂-中药材 深度数据挖掘 (精修修正版)")
|
||
print("=" * 70)
|
||
print(f"\n[1/8] 加载数据...")
|
||
|
||
# 药材主名集合(仅使用主名,不含别名)
|
||
herbs = {} # canonical_name -> full_data
|
||
valid_herb_names = [] # 仅≥2字的主名,按长度降序排列
|
||
|
||
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
d = json.load(fh)
|
||
name = d.get('名称', '').strip()
|
||
if name:
|
||
herbs[name] = d
|
||
|
||
# 仅保留≥2个字的药材名作为匹配字典
|
||
for name, data in herbs.items():
|
||
if len(name) >= 2:
|
||
valid_herb_names.append(name)
|
||
|
||
# 按长度降序,确保长名优先匹配
|
||
valid_herb_names.sort(key=len, reverse=True)
|
||
herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names))
|
||
|
||
print(f" 药材: {len(herbs)} 味 (匹配字典: {len(valid_herb_names)}味)")
|
||
print(f" 药材样本: {', '.join(valid_herb_names[:10])}")
|
||
|
||
# ========== 加载方剂 ==========
|
||
formulas = OrderedDict()
|
||
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
d = json.load(fh)
|
||
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
|
||
formulas[name] = d
|
||
|
||
print(f" 方剂: {len(formulas)} 首")
|
||
|
||
# ========== 提取方剂中的药材 ==========
|
||
print(f"\n[2/8] 提取方剂中的药材...")
|
||
|
||
def extract_herbs_exact(text):
|
||
"""精确匹配:仅使用≥2字的主药材名"""
|
||
if not text or len(text) < 3:
|
||
return []
|
||
|
||
# 去掉炮制括号 药量
|
||
clean = re.sub(r'[((][^())]*[))]', '', text)
|
||
clean = re.sub(r'\d+[两钱毫升克]', '', clean)
|
||
|
||
# 用正则找到所有药材名
|
||
found = []
|
||
for match in herb_pattern.finditer(clean):
|
||
herb = match.group(0)
|
||
if herb and herb not in found:
|
||
found.append(herb)
|
||
|
||
return found
|
||
|
||
def get_all_text(data_dict):
|
||
"""拼接所有字段文本"""
|
||
all_text = ''
|
||
for val in data_dict.values():
|
||
if isinstance(val, str):
|
||
all_text += val
|
||
elif isinstance(val, (dict, list)):
|
||
all_text += json.dumps(val, ensure_ascii=False)
|
||
return all_text
|
||
|
||
formula_herbs = OrderedDict()
|
||
for fname, data in formulas.items():
|
||
all_text = get_all_text(data)
|
||
herbs_found = extract_herbs_exact(all_text)
|
||
formula_herbs[fname] = herbs_found
|
||
|
||
total_links = sum(len(h) for h in formula_herbs.values())
|
||
avg = total_links / max(1, len(formula_herbs))
|
||
print(f" 总关联对数: {total_links}")
|
||
print(f" 平均每方药材数: {avg:.1f}")
|
||
|
||
# 检查提取质量
|
||
print(f"\n 提取结果检查 (前15首方):")
|
||
for i, (fname, hlist) in enumerate(list(formula_herbs.items())[:15]):
|
||
print(f" [{i+1:2d}] {fname:30s} {len(hlist):2d}味 -> {hlist[:8]}")
|
||
|
||
# 逆向索引
|
||
herb_formulas = defaultdict(set)
|
||
for fname, hlist in formula_herbs.items():
|
||
for h in hlist:
|
||
herb_formulas[h].add(fname)
|
||
|
||
# ========== 3. 高频药材 ==========
|
||
print(f"\n[3/8] 高频药材分析...")
|
||
herb_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
for h in hlist:
|
||
herb_freq[h] += 1
|
||
|
||
top30 = herb_freq.most_common(30)
|
||
print(f"\n Top 30 高频药材:")
|
||
print(f" {'药材':<12s} {'频次':>5s} {'分类':<6s} {'功效':<30s}")
|
||
print(f" {'-'*12} {'-'*5} {'-'*6} {'-'*30}")
|
||
for h, c in top30:
|
||
ct = herbs.get(h, {}).get('药材分类', '未知')
|
||
func_data = herbs.get(h, {}).get('功效作用', {})
|
||
func = func_data.get('功能', '')[:30] if isinstance(func_data, dict) else ''
|
||
print(f" {h:<12s} {c:>5d} {ct:<6s} {func}")
|
||
|
||
with open(os.path.join(OUT, "01_高频药材Top30.json"), 'w', encoding='utf-8') as f:
|
||
json.dump([{"name": h, "count": c} for h, c in top30], f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 01_高频药材Top30.json")
|
||
|
||
# ========== 4. 高频药对 ==========
|
||
print(f"\n[4/8] 高频药对 (2味) Top 20...")
|
||
pair_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
u = sorted(list(set(hlist)))
|
||
if len(u) >= 2:
|
||
for combo in combinations(u, 2):
|
||
pair_freq[combo] += 1
|
||
|
||
# Top 50药对 (≥5次)
|
||
top_pairs = [p for p, c in pair_freq.most_common(100) if c >= 5]
|
||
print(f"\n Top 20 高频药对:")
|
||
print(f" {'药对':<30s} {'频次':>5s} {'药材1功效':<12s} {'药材2功效':<12s}")
|
||
print(f" {'-'*30} {'-'*5} {'-'*12} {'-'*12}")
|
||
for pair in top_pairs[:20]:
|
||
h1, h2 = pair
|
||
c = pair_freq[pair]
|
||
func_data1 = herbs.get(h1, {}).get('功效作用', {})
|
||
func_data2 = herbs.get(h2, {}).get('功效作用', {})
|
||
func1 = func_data1.get('功能', '')[:12] if isinstance(func_data1, dict) else ''
|
||
func2 = func_data2.get('功能', '')[:12] if isinstance(func_data2, dict) else ''
|
||
print(f" {h1:12s} + {h2:12s} {c:>5d} [{func1[:8]}] [{func2[:8]}]")
|
||
|
||
with open(os.path.join(OUT, "02_高频药对_Top20.json"), 'w', encoding='utf-8') as f:
|
||
json.dump([{"pair": list(p), "count": c} for p, c in pair_freq.most_common(50)], f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 02_高频药对_Top20.json")
|
||
|
||
# ========== 5. 核心药组 ==========
|
||
print(f"\n[5/8] 核心药组 (3-5味药) Top 10...")
|
||
|
||
for n in [3, 4, 5]:
|
||
triple_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
u = sorted(list(set(hlist)))
|
||
if len(u) >= n:
|
||
for combo in combinations(u, n):
|
||
triple_freq[combo] += 1
|
||
|
||
top_combs = [(t, c) for t, c in triple_freq.most_common(50) if c >= 5]
|
||
print(f"\n Top {n}味药组 (出现≥5次):")
|
||
for combo, c in top_combs[:10]:
|
||
names = "+".join(combo)
|
||
print(f" {names:>40s}: {c}首方")
|
||
|
||
with open(os.path.join(OUT, "03_核心药组3-5味_Top10.json"), 'w', encoding='utf-8') as f:
|
||
res = {}
|
||
for n in [3, 4, 5]:
|
||
triple_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
u = sorted(list(set(hlist)))
|
||
if len(u) >= n:
|
||
for combo in combinations(u, n):
|
||
triple_freq[combo] += 1
|
||
res[f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5]
|
||
json.dump(res, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 03_核心药组3-5味_Top10.json")
|
||
|
||
# ========== 6. 功效关键词网络 ==========
|
||
print(f"\n[6/8] 功效关键词网络...")
|
||
|
||
# 提取核心功效词(过滤噪音词)
|
||
noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '。'}
|
||
|
||
func_keywords = Counter()
|
||
func_formulas = defaultdict(set)
|
||
|
||
for h, data in herbs.items():
|
||
if h not in herb_formulas:
|
||
continue
|
||
gua = data.get('功效作用', {})
|
||
if not isinstance(gua, dict):
|
||
continue
|
||
func_str = gua.get('功能', '')
|
||
if not func_str:
|
||
continue
|
||
|
||
# 按标点拆分
|
||
keywords = re.split(r'[、,,、;;。]', func_str)
|
||
for kw in keywords:
|
||
kw = kw.strip()
|
||
# 过滤噪音词
|
||
if not kw or len(kw) < 2 or kw in noise_words:
|
||
continue
|
||
# 也检查是否全是药材名(如"甘草"也是功效词"甘草具有补脾益气"中的残留)
|
||
if kw == h and isinstance(gua, dict) and func_str.startswith(kw):
|
||
# 如果关键词就是药材名且紧跟药材名,跳过
|
||
continue
|
||
if kw in herbs:
|
||
# 如果关键词本身是药材名且在功效描述中(非独立功效词),跳过
|
||
continue
|
||
|
||
func_keywords[kw] += 1
|
||
for fname in herb_formulas[h]:
|
||
func_formulas[kw].add(fname)
|
||
|
||
# 按方剂数排序
|
||
top_funcs = [(kw, len(vx)) for kw, vx in func_formulas.items() if len(vx) >= 10]
|
||
top_funcs.sort(key=lambda x: -x[1])
|
||
|
||
print(f"\n Top 30 功效关键词 (按关联方剂数):")
|
||
print(f" {'关键词':<15s} {'方剂数':>6s} 示例方剂")
|
||
print(f" {'-'*15} {'-'*6} {'-'*40}")
|
||
for kw, count in top_funcs[:30]:
|
||
examples = list(func_formulas[kw])[:2]
|
||
print(f" {kw:<15s} {count:>6d} {', '.join(examples)}")
|
||
|
||
with open(os.path.join(OUT, "04_功效关键词_Top30.json"), 'w', encoding='utf-8') as f:
|
||
json.dump([{"keyword": kw, "formula_count": count} for kw, count in top_funcs[:30]], f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 04_功效关键词_Top30.json")
|
||
|
||
# ========== 7. 方剂聚类 ==========
|
||
print(f"\n[7/8] 方剂聚类...")
|
||
|
||
clusters = defaultdict(list)
|
||
for fname, hlist in formula_herbs.items():
|
||
kw_count = Counter()
|
||
for herb in hlist:
|
||
if herb in herbs:
|
||
gua = herbs[herb].get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
kw_str = gua['功能']
|
||
keywords = re.split(r'[、,,、;;。]', kw_str)
|
||
for kw in keywords:
|
||
kw = kw.strip()
|
||
if kw and len(kw) >= 2 and kw not in noise_words:
|
||
kw_count[kw] += 1
|
||
if kw_count:
|
||
top_kw = kw_count.most_common(1)[0][0]
|
||
clusters[top_kw].append(fname)
|
||
|
||
big_clusters = [(kw, flst) for kw, flst in clusters.items() if len(flst) >= 5]
|
||
big_clusters.sort(key=lambda x: -len(x[1]))
|
||
|
||
print(f" 发现 {len(clusters)} 个功效簇")
|
||
print(f"\n Top 20 大功效簇:")
|
||
for kw, flst in big_clusters[:20]:
|
||
# 找簇内的代表性药材
|
||
sample_herbs = set()
|
||
for fname in flst[:10]:
|
||
for h in formula_herbs[fname]:
|
||
sample_herbs.add(h)
|
||
top_h = sample_herbs
|
||
print(f" '{kw}' ({len(flst)}首方): 代表药材 -> {', '.join(sorted(top_h)[:10])}")
|
||
|
||
with open(os.path.join(OUT, "05_方剂聚类_Top20.json"), 'w', encoding='utf-8') as f:
|
||
json.dump([{"cluster": kw, "count": len(flst), "formulas": flst[:20]} for kw, flst in big_clusters[:20]], f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 05_方剂聚类_Top20.json")
|
||
|
||
# ========== 8. 配伍禁忌 & 相似度 ==========
|
||
print(f"\n[8/8] 配伍禁忌验证 & 方剂相似度...")
|
||
|
||
# 十八反
|
||
fan = {
|
||
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
|
||
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
|
||
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
|
||
}
|
||
wei = {
|
||
'硫黄': ['朴硝', '芒硝', '牙硝'],
|
||
'水银': ['铅丹', '砒霜'],
|
||
'巴豆': ['牵牛', '牵牛子'],
|
||
'丁香': ['郁金'],
|
||
'人参': ['五灵脂'],
|
||
'肉桂': ['石脂', '赤石脂'],
|
||
'半夏': ['羊脂'],
|
||
'厚朴': ['硝石', '滑石'],
|
||
}
|
||
|
||
# 检查禁忌药材是否在方剂中出现
|
||
contra_found = defaultdict(list)
|
||
all_contra_search = {}
|
||
for k, v in fan.items():
|
||
for vv in v:
|
||
if vv not in all_contra_search:
|
||
all_contra_search[vv] = k
|
||
for k, v in wei.items():
|
||
for vv in v:
|
||
if vv not in all_contra_search:
|
||
all_contra_search[vv] = k
|
||
|
||
print(f"\n 十八反/十九畏相关药材在方剂中出现情况:")
|
||
for herb_name, related in all_contra_search.items():
|
||
if herb_name in herb_formulas:
|
||
count = len(herb_formulas[herb_name])
|
||
print(f" ⚠ {herb_name} (反/畏{related}): 出现在 {count} 首方")
|
||
|
||
# 检查实际方剂中是否同时出现矛盾配对
|
||
contra_pairs_found = defaultdict(int)
|
||
contra_pairs_details = defaultdict(list)
|
||
|
||
for fname, hlist in formula_herbs.items():
|
||
for herb, contra_list in fan.items():
|
||
if herb in hlist:
|
||
for ch in contra_list:
|
||
if ch in hlist:
|
||
contra_pairs_found[(herb, ch)] += 1
|
||
contra_pairs_details[(herb, ch)].append(fname)
|
||
for herb, contra_list in wei.items():
|
||
if herb in hlist:
|
||
for ch in contra_list:
|
||
if ch in hlist:
|
||
contra_pairs_found[(herb, ch)] += 1
|
||
contra_pairs_details[(herb, ch)].append(fname)
|
||
|
||
top_contra = sorted(contra_pairs_found.items(), key=lambda x: -x[1])[:10]
|
||
if top_contra:
|
||
print(f"\n 发现配伍禁忌:")
|
||
for (h1, h2), count in top_contra:
|
||
print(f" ⚠️ {h1} + {h2}: {count}首方 ({', '.join(contra_pairs_details[(h1,h2)][:2])})")
|
||
else:
|
||
print(f"\n 未发现十八反/十九畏的直接配对出现在同一首方剂中")
|
||
|
||
with open(os.path.join(OUT, "06_配伍禁忌.json"), 'w', encoding='utf-8') as f:
|
||
json.dump({
|
||
"十八反": fan,
|
||
"十九畏": wei,
|
||
"禁忌药材出现": {k: len(herb_formulas.get(k, set())) for k in all_contra_search.keys() if k in herb_formulas},
|
||
"实际禁忌配对": {f"{k[0]}+{k[1]}": c for k, c in top_contra},
|
||
"禁忌详情": {f"{k[0]}+{k[1]}": contra_pairs_details[k] for k in top_contra}
|
||
}, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 06_配伍禁忌.json")
|
||
|
||
# 方剂相似度
|
||
print(f"\n 计算方剂相似度 (Jaccard >= 0.3)...")
|
||
|
||
def jaccard_fixed(s1, s2):
|
||
if not s1 or not s2:
|
||
return 0.0
|
||
intersection = len(s1 & s2)
|
||
if intersection == 0:
|
||
return 0.0
|
||
union = len(s1 | s2)
|
||
return intersection / union if union > 0 else 0.0
|
||
|
||
similar_pairs = []
|
||
formula_list = list(formulas.keys())
|
||
count = 0
|
||
for i in range(len(formula_list)):
|
||
f1 = formula_list[i]
|
||
h1 = formula_herbs.get(f1, [])
|
||
if not h1:
|
||
continue
|
||
s1 = set(h1)
|
||
for j in range(i + 1, len(formula_list)):
|
||
f2 = formula_list[j]
|
||
h2 = formula_herbs.get(f2, [])
|
||
if not h2:
|
||
continue
|
||
s2 = set(h2)
|
||
sim = jaccard_fixed(s1, s2)
|
||
if sim >= 0.3:
|
||
shared = s1 & s2
|
||
similar_pairs.append((f1, f2, sim, len(shared), sorted(shared)))
|
||
count += 1
|
||
|
||
print(f" 检查 {count} 对, 找到 {len(similar_pairs)} 对相似方剂")
|
||
similar_pairs.sort(key=lambda x: (-x[2], -x[3]))
|
||
top_similar = similar_pairs[:20]
|
||
|
||
print(f"\n Top 15 相似方剂对:")
|
||
for f1, f2, sim, shared_c, shared_h in top_similar[:15]:
|
||
print(f" {f1:30s} + {f2:30s} -> {sim:.3f} (共享{shared_c}味: {', '.join(shared_h)})")
|
||
|
||
with open(os.path.join(OUT, "07_方剂相似度_Top20.json"), 'w', encoding='utf-8') as f:
|
||
json.dump([{"similarity": round(sim, 4), "shared_count": sc, "pair1": f1, "pair2": f2, "shared_herbs": sh} for f1, f2, sim, sc, sh in top_similar], f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ 07_方剂相似度_Top20.json")
|
||
|
||
# ========== 汇总 ==========
|
||
print(f"\n{'='*70}")
|
||
print(f" 深度挖掘分析完成!")
|
||
print(f"{'='*70}")
|
||
|
||
summary = OrderedDict()
|
||
summary["标题"] = "大医网 方剂-中药材 深度数据挖掘报告 (精修修正版)"
|
||
summary["数据规模"] = {
|
||
"药材": len(herbs),
|
||
"方剂": len(formulas),
|
||
"总关联对数": total_links,
|
||
"平均每方药材数": round(avg, 1),
|
||
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
|
||
"无法提取方剂": sum(1 for h in formula_herbs.values() if not h),
|
||
}
|
||
summary["高频药材_Top20"] = [{"name": h, "count": c} for h, c in top30[:20]]
|
||
summary["高频药对_Top10"] = [{"pair": list(p), "count": c} for p, c in pair_freq.most_common(10)]
|
||
summary["核心药组_Top10"] = {}
|
||
for n in [3, 4, 5]:
|
||
triple_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
u = sorted(list(set(hlist)))
|
||
if len(u) >= n:
|
||
for combo in combinations(u, n):
|
||
triple_freq[combo] += 1
|
||
summary["核心药组_Top10"][f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5]
|
||
|
||
summary["功效关键词_Top15"] = [{"keyword": kw, "count": c} for kw, c in top_funcs[:15]]
|
||
summary["方剂聚类_Top10"] = [{"cluster": kw, "count": len(flst)} for kw, flst in big_clusters[:10]]
|
||
summary["配伍禁忌"] = {
|
||
"十八反": fan,
|
||
"十九畏": wei,
|
||
"实际发现": {f"{k[0]}+{k[1]}": c for k, c in top_contra},
|
||
}
|
||
summary["方剂相似度_Top10"] = [{"similarity": round(sim, 4), "pair": f"{f1} + {f2}"} for f1, f2, sim, sc, sh in top_similar[:10]]
|
||
summary["输出目录"] = os.path.abspath(OUT)
|
||
|
||
with open(os.path.join(OUT, "汇总报告.json"), 'w', encoding='utf-8') as f:
|
||
json.dump(summary, f, ensure_ascii=False, indent=2)
|
||
|
||
print(f"\n 输出目录: {os.path.abspath(OUT)}")
|
||
print(f"\n 文件列表:")
|
||
for fn in sorted(os.listdir(OUT)):
|
||
fp = os.path.join(OUT, fn)
|
||
size = os.path.getsize(fp)
|
||
print(f" {fn:40s} {size:>10,} B")
|