Files
health/大医网/05_脚本工具/深度挖掘精修2.py
T

448 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
深度数据挖掘 - 精修修正版
核心修正:
1. 药材匹配仅使用药材主名(name),不含单字别名
2. 药材名按长度降序排列,确保长名优先匹配
3. 功效关键词过滤噪音(去除"本品""诸药"等解析残余)
4. 药对/药组仅统计完整药材名
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_精修2")
os.makedirs(OUT, exist_ok=True)
# ========== 加载药材 ==========
print("=" * 70)
print(" 大医网 方剂-中药材 深度数据挖掘 (精修修正版)")
print("=" * 70)
print(f"\n[1/8] 加载数据...")
# 药材主名集合(仅使用主名,不含别名)
herbs = {} # canonical_name -> full_data
valid_herb_names = [] # 仅≥2字的主名,按长度降序排列
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', '').strip()
if name:
herbs[name] = d
# 仅保留≥2个字的药材名作为匹配字典
for name, data in herbs.items():
if len(name) >= 2:
valid_herb_names.append(name)
# 按长度降序,确保长名优先匹配
valid_herb_names.sort(key=len, reverse=True)
herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names))
print(f" 药材: {len(herbs)} 味 (匹配字典: {len(valid_herb_names)}味)")
print(f" 药材样本: {', '.join(valid_herb_names[:10])}")
# ========== 加载方剂 ==========
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
formulas[name] = d
print(f" 方剂: {len(formulas)} 首")
# ========== 提取方剂中的药材 ==========
print(f"\n[2/8] 提取方剂中的药材...")
def extract_herbs_exact(text):
"""精确匹配:仅使用≥2字的主药材名"""
if not text or len(text) < 3:
return []
# 去掉炮制括号 药量
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克]', '', clean)
# 用正则找到所有药材名
found = []
for match in herb_pattern.finditer(clean):
herb = match.group(0)
if herb and herb not in found:
found.append(herb)
return found
def get_all_text(data_dict):
"""拼接所有字段文本"""
all_text = ''
for val in data_dict.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
return all_text
formula_herbs = OrderedDict()
for fname, data in formulas.items():
all_text = get_all_text(data)
herbs_found = extract_herbs_exact(all_text)
formula_herbs[fname] = herbs_found
total_links = sum(len(h) for h in formula_herbs.values())
avg = total_links / max(1, len(formula_herbs))
print(f" 总关联对数: {total_links}")
print(f" 平均每方药材数: {avg:.1f}")
# 检查提取质量
print(f"\n 提取结果检查 (前15首方):")
for i, (fname, hlist) in enumerate(list(formula_herbs.items())[:15]):
print(f" [{i+1:2d}] {fname:30s} {len(hlist):2d}味 -> {hlist[:8]}")
# 逆向索引
herb_formulas = defaultdict(set)
for fname, hlist in formula_herbs.items():
for h in hlist:
herb_formulas[h].add(fname)
# ========== 3. 高频药材 ==========
print(f"\n[3/8] 高频药材分析...")
herb_freq = Counter()
for hlist in formula_herbs.values():
for h in hlist:
herb_freq[h] += 1
top30 = herb_freq.most_common(30)
print(f"\n Top 30 高频药材:")
print(f" {'药材':<12s} {'频次':>5s} {'分类':<6s} {'功效':<30s}")
print(f" {'-'*12} {'-'*5} {'-'*6} {'-'*30}")
for h, c in top30:
ct = herbs.get(h, {}).get('药材分类', '未知')
func_data = herbs.get(h, {}).get('功效作用', {})
func = func_data.get('功能', '')[:30] if isinstance(func_data, dict) else ''
print(f" {h:<12s} {c:>5d} {ct:<6s} {func}")
with open(os.path.join(OUT, "01_高频药材Top30.json"), 'w', encoding='utf-8') as f:
json.dump([{"name": h, "count": c} for h, c in top30], f, ensure_ascii=False, indent=2)
print("\n ✓ 01_高频药材Top30.json")
# ========== 4. 高频药对 ==========
print(f"\n[4/8] 高频药对 (2味) Top 20...")
pair_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= 2:
for combo in combinations(u, 2):
pair_freq[combo] += 1
# Top 50药对 (≥5次)
top_pairs = [p for p, c in pair_freq.most_common(100) if c >= 5]
print(f"\n Top 20 高频药对:")
print(f" {'药对':<30s} {'频次':>5s} {'药材1功效':<12s} {'药材2功效':<12s}")
print(f" {'-'*30} {'-'*5} {'-'*12} {'-'*12}")
for pair in top_pairs[:20]:
h1, h2 = pair
c = pair_freq[pair]
func_data1 = herbs.get(h1, {}).get('功效作用', {})
func_data2 = herbs.get(h2, {}).get('功效作用', {})
func1 = func_data1.get('功能', '')[:12] if isinstance(func_data1, dict) else ''
func2 = func_data2.get('功能', '')[:12] if isinstance(func_data2, dict) else ''
print(f" {h1:12s} + {h2:12s} {c:>5d} [{func1[:8]}] [{func2[:8]}]")
with open(os.path.join(OUT, "02_高频药对_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"pair": list(p), "count": c} for p, c in pair_freq.most_common(50)], f, ensure_ascii=False, indent=2)
print("\n ✓ 02_高频药对_Top20.json")
# ========== 5. 核心药组 ==========
print(f"\n[5/8] 核心药组 (3-5味药) Top 10...")
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
top_combs = [(t, c) for t, c in triple_freq.most_common(50) if c >= 5]
print(f"\n Top {n}味药组 (出现≥5次):")
for combo, c in top_combs[:10]:
names = "+".join(combo)
print(f" {names:>40s}: {c}首方")
with open(os.path.join(OUT, "03_核心药组3-5味_Top10.json"), 'w', encoding='utf-8') as f:
res = {}
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
res[f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5]
json.dump(res, f, ensure_ascii=False, indent=2)
print("\n ✓ 03_核心药组3-5味_Top10.json")
# ========== 6. 功效关键词网络 ==========
print(f"\n[6/8] 功效关键词网络...")
# 提取核心功效词(过滤噪音词)
noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '。'}
func_keywords = Counter()
func_formulas = defaultdict(set)
for h, data in herbs.items():
if h not in herb_formulas:
continue
gua = data.get('功效作用', {})
if not isinstance(gua, dict):
continue
func_str = gua.get('功能', '')
if not func_str:
continue
# 按标点拆分
keywords = re.split(r'[、,,、;;。]', func_str)
for kw in keywords:
kw = kw.strip()
# 过滤噪音词
if not kw or len(kw) < 2 or kw in noise_words:
continue
# 也检查是否全是药材名(如"甘草"也是功效词"甘草具有补脾益气"中的残留)
if kw == h and isinstance(gua, dict) and func_str.startswith(kw):
# 如果关键词就是药材名且紧跟药材名,跳过
continue
if kw in herbs:
# 如果关键词本身是药材名且在功效描述中(非独立功效词),跳过
continue
func_keywords[kw] += 1
for fname in herb_formulas[h]:
func_formulas[kw].add(fname)
# 按方剂数排序
top_funcs = [(kw, len(vx)) for kw, vx in func_formulas.items() if len(vx) >= 10]
top_funcs.sort(key=lambda x: -x[1])
print(f"\n Top 30 功效关键词 (按关联方剂数):")
print(f" {'关键词':<15s} {'方剂数':>6s} 示例方剂")
print(f" {'-'*15} {'-'*6} {'-'*40}")
for kw, count in top_funcs[:30]:
examples = list(func_formulas[kw])[:2]
print(f" {kw:<15s} {count:>6d} {', '.join(examples)}")
with open(os.path.join(OUT, "04_功效关键词_Top30.json"), 'w', encoding='utf-8') as f:
json.dump([{"keyword": kw, "formula_count": count} for kw, count in top_funcs[:30]], f, ensure_ascii=False, indent=2)
print("\n ✓ 04_功效关键词_Top30.json")
# ========== 7. 方剂聚类 ==========
print(f"\n[7/8] 方剂聚类...")
clusters = defaultdict(list)
for fname, hlist in formula_herbs.items():
kw_count = Counter()
for herb in hlist:
if herb in herbs:
gua = herbs[herb].get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kw_str = gua['功能']
keywords = re.split(r'[、,,、;;。]', kw_str)
for kw in keywords:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
kw_count[kw] += 1
if kw_count:
top_kw = kw_count.most_common(1)[0][0]
clusters[top_kw].append(fname)
big_clusters = [(kw, flst) for kw, flst in clusters.items() if len(flst) >= 5]
big_clusters.sort(key=lambda x: -len(x[1]))
print(f" 发现 {len(clusters)} 个功效簇")
print(f"\n Top 20 大功效簇:")
for kw, flst in big_clusters[:20]:
# 找簇内的代表性药材
sample_herbs = set()
for fname in flst[:10]:
for h in formula_herbs[fname]:
sample_herbs.add(h)
top_h = sample_herbs
print(f" '{kw}' ({len(flst)}首方): 代表药材 -> {', '.join(sorted(top_h)[:10])}")
with open(os.path.join(OUT, "05_方剂聚类_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"cluster": kw, "count": len(flst), "formulas": flst[:20]} for kw, flst in big_clusters[:20]], f, ensure_ascii=False, indent=2)
print("\n ✓ 05_方剂聚类_Top20.json")
# ========== 8. 配伍禁忌 & 相似度 ==========
print(f"\n[8/8] 配伍禁忌验证 & 方剂相似度...")
# 十八反
fan = {
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
}
wei = {
'硫黄': ['朴硝', '芒硝', '牙硝'],
'水银': ['铅丹', '砒霜'],
'巴豆': ['牵牛', '牵牛子'],
'丁香': ['郁金'],
'人参': ['五灵脂'],
'肉桂': ['石脂', '赤石脂'],
'半夏': ['羊脂'],
'厚朴': ['硝石', '滑石'],
}
# 检查禁忌药材是否在方剂中出现
contra_found = defaultdict(list)
all_contra_search = {}
for k, v in fan.items():
for vv in v:
if vv not in all_contra_search:
all_contra_search[vv] = k
for k, v in wei.items():
for vv in v:
if vv not in all_contra_search:
all_contra_search[vv] = k
print(f"\n 十八反/十九畏相关药材在方剂中出现情况:")
for herb_name, related in all_contra_search.items():
if herb_name in herb_formulas:
count = len(herb_formulas[herb_name])
print(f" ⚠ {herb_name} (反/畏{related}): 出现在 {count} 首方")
# 检查实际方剂中是否同时出现矛盾配对
contra_pairs_found = defaultdict(int)
contra_pairs_details = defaultdict(list)
for fname, hlist in formula_herbs.items():
for herb, contra_list in fan.items():
if herb in hlist:
for ch in contra_list:
if ch in hlist:
contra_pairs_found[(herb, ch)] += 1
contra_pairs_details[(herb, ch)].append(fname)
for herb, contra_list in wei.items():
if herb in hlist:
for ch in contra_list:
if ch in hlist:
contra_pairs_found[(herb, ch)] += 1
contra_pairs_details[(herb, ch)].append(fname)
top_contra = sorted(contra_pairs_found.items(), key=lambda x: -x[1])[:10]
if top_contra:
print(f"\n 发现配伍禁忌:")
for (h1, h2), count in top_contra:
print(f" ⚠️ {h1} + {h2}: {count}首方 ({', '.join(contra_pairs_details[(h1,h2)][:2])})")
else:
print(f"\n 未发现十八反/十九畏的直接配对出现在同一首方剂中")
with open(os.path.join(OUT, "06_配伍禁忌.json"), 'w', encoding='utf-8') as f:
json.dump({
"十八反": fan,
"十九畏": wei,
"禁忌药材出现": {k: len(herb_formulas.get(k, set())) for k in all_contra_search.keys() if k in herb_formulas},
"实际禁忌配对": {f"{k[0]}+{k[1]}": c for k, c in top_contra},
"禁忌详情": {f"{k[0]}+{k[1]}": contra_pairs_details[k] for k in top_contra}
}, f, ensure_ascii=False, indent=2)
print("\n ✓ 06_配伍禁忌.json")
# 方剂相似度
print(f"\n 计算方剂相似度 (Jaccard >= 0.3)...")
def jaccard_fixed(s1, s2):
if not s1 or not s2:
return 0.0
intersection = len(s1 & s2)
if intersection == 0:
return 0.0
union = len(s1 | s2)
return intersection / union if union > 0 else 0.0
similar_pairs = []
formula_list = list(formulas.keys())
count = 0
for i in range(len(formula_list)):
f1 = formula_list[i]
h1 = formula_herbs.get(f1, [])
if not h1:
continue
s1 = set(h1)
for j in range(i + 1, len(formula_list)):
f2 = formula_list[j]
h2 = formula_herbs.get(f2, [])
if not h2:
continue
s2 = set(h2)
sim = jaccard_fixed(s1, s2)
if sim >= 0.3:
shared = s1 & s2
similar_pairs.append((f1, f2, sim, len(shared), sorted(shared)))
count += 1
print(f" 检查 {count} 对, 找到 {len(similar_pairs)} 对相似方剂")
similar_pairs.sort(key=lambda x: (-x[2], -x[3]))
top_similar = similar_pairs[:20]
print(f"\n Top 15 相似方剂对:")
for f1, f2, sim, shared_c, shared_h in top_similar[:15]:
print(f" {f1:30s} + {f2:30s} -> {sim:.3f} (共享{shared_c}味: {', '.join(shared_h)})")
with open(os.path.join(OUT, "07_方剂相似度_Top20.json"), 'w', encoding='utf-8') as f:
json.dump([{"similarity": round(sim, 4), "shared_count": sc, "pair1": f1, "pair2": f2, "shared_herbs": sh} for f1, f2, sim, sc, sh in top_similar], f, ensure_ascii=False, indent=2)
print("\n ✓ 07_方剂相似度_Top20.json")
# ========== 汇总 ==========
print(f"\n{'='*70}")
print(f" 深度挖掘分析完成!")
print(f"{'='*70}")
summary = OrderedDict()
summary["标题"] = "大医网 方剂-中药材 深度数据挖掘报告 (精修修正版)"
summary["数据规模"] = {
"药材": len(herbs),
"方剂": len(formulas),
"总关联对数": total_links,
"平均每方药材数": round(avg, 1),
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
"无法提取方剂": sum(1 for h in formula_herbs.values() if not h),
}
summary["高频药材_Top20"] = [{"name": h, "count": c} for h, c in top30[:20]]
summary["高频药对_Top10"] = [{"pair": list(p), "count": c} for p, c in pair_freq.most_common(10)]
summary["核心药组_Top10"] = {}
for n in [3, 4, 5]:
triple_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= n:
for combo in combinations(u, n):
triple_freq[combo] += 1
summary["核心药组_Top10"][f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5]
summary["功效关键词_Top15"] = [{"keyword": kw, "count": c} for kw, c in top_funcs[:15]]
summary["方剂聚类_Top10"] = [{"cluster": kw, "count": len(flst)} for kw, flst in big_clusters[:10]]
summary["配伍禁忌"] = {
"十八反": fan,
"十九畏": wei,
"实际发现": {f"{k[0]}+{k[1]}": c for k, c in top_contra},
}
summary["方剂相似度_Top10"] = [{"similarity": round(sim, 4), "pair": f"{f1} + {f2}"} for f1, f2, sim, sc, sh in top_similar[:10]]
summary["输出目录"] = os.path.abspath(OUT)
with open(os.path.join(OUT, "汇总报告.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print(f"\n 输出目录: {os.path.abspath(OUT)}")
print(f"\n 文件列表:")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:40s} {size:>10,} B")