Files
health/大医网/05_脚本工具/深度挖掘6.py
T

411 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
大医网 方剂-中药材 深度数据挖掘 (修正版)
- 修复药材提取率问题 (回退到之前有效的方法)
- 修复相似度分析 (空集Jaccard=0, 不是1)
- 修复浮点数截断错误
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_初版")
os.makedirs(OUT, exist_ok=True)
print("=" * 70)
print(" 大医网 方剂-中药材 深度数据挖掘 (修正版)")
print("=" * 70)
# ========== 1. 加载数据 ==========
print("\n[1/9] 加载数据...")
herbs = {} # name -> data
herb_names = set()
herb_alias_map = {} # alias -> canonical name
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
n = d.get('名称', '').strip()
if not n: continue
herbs[n] = d
herb_names.add(n)
aliases = d.get('别名', '')
if aliases:
for a in re.split(r'[、,,;]', aliases):
a = a.strip()
if a:
herb_names.add(a)
herb_alias_map[a] = n
print(f" 药材: {len(herbs)} 味, 含 {len(herb_names)} 个名称(含别名)")
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', re.search(r'(\d+)_(.+)\.json', os.path.basename(f)).group(2) if re.search(r'(\d+)_(.+)\.json', os.path.basename(f)) else os.path.basename(f))
formulas[name] = d
print(f" 方剂: {len(formulas)} 首")
# ========== 2. 提取方剂中的药材 ==========
print("\n[2/9] 提取方剂中的药材...")
# 按长度降序排列药材名,确保长名称优先匹配
sorted_herb = sorted(herb_names, key=len, reverse=True)
def extract_herbs_from_text(text):
"""从文本中提取药材名 - 使用最长匹配优先"""
if not text or len(text) < 3:
return []
# 去掉炮制括号 药量如 一两、五钱
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克毫克]', '', clean)
# 拆分成分隔符
parts = [p.strip() for p in re.split(r'[、,\;;]', clean) if p.strip()]
found = []
seen = set()
for p in parts:
p = p.strip()
if len(p) < 2:
continue
# 尝试匹配药材名
for herb in sorted_herb:
if herb == p or herb in p or p in herb:
if herb not in seen:
found.append(herb)
seen.add(herb)
if len(found) > 30: # 防止过多匹配
break
if len(found) > 30:
break
return found
def extract_herbs_brute(data_dict):
"""暴力匹配:检查每味药材是否出现在任何字段中"""
# 拼接所有字段文本
all_text = ''
for val in data_dict.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
found = []
for herb in sorted_herb:
if herb in all_text:
found.append(herb)
return found
formula_herbs = {} # formula_name -> list of herbs
for fname, data in formulas.items():
# 先尝试分段提取
herbs_found = []
for key, val in data.items():
if isinstance(val, str) and len(val) > 3:
herbs_found.extend(extract_herbs_from_text(val))
elif isinstance(val, (dict, list)):
herbs_found.extend(extract_herbs_from_text(json.dumps(val, ensure_ascii=False)))
# 再去重
herbs_found = list(dict.fromkeys(herbs_found))
# 如果分段提取效果不好,使用暴力匹配
if len(herbs_found) < 3:
herbs_found = extract_herbs_brute(data)
formula_herbs[fname] = herbs_found
total_links = sum(len(h) for h in formula_herbs.values())
avg = total_links / max(1, len(formula_herbs))
print(f" 总关联对数: {total_links}")
print(f" 平均每方药材数: {avg:.1f}")
# 检查提取质量
print("\n 检查提取结果 (前10首方):")
for i, (fname, herb_list) in enumerate(list(formula_herbs.items())[:10]):
print(f" [{i+1}] {fname}: {len(herb_list)}味 -> {herb_list[:7]}")
# 方剂->药材 的逆向索引
herb_formulas = defaultdict(set)
for fname, herb_list in formula_herbs.items():
for h in herb_list:
herb_formulas[h].add(fname)
# ========== 3. 高频药材 ==========
print("\n[3/9] 高频药材分析...")
herb_freq = Counter()
for herb_list in formula_herbs.values():
for h in herb_list:
herb_freq[h] += 1
top20 = herb_freq.most_common(20)
print(" Top 30 高频药材:")
for h, c in top20:
ct = herbs.get(h, {}).get('药材分类', '?')
func = herbs.get(h, {}).get('功效作用', {}).get('功能', '')
if isinstance(func, dict):
func = func.get('功能', '')
print(f" {h:10s} {c:4d}首 [{ct:3s}] {func[:25]}")
with open(os.path.join(OUT, "A_高频药材.json"), 'w', encoding='utf-8') as f:
out_list = []
for h, c in top20:
out_list.append({"name": h, "count": c})
json.dump(out_list, f, ensure_ascii=False, indent=2)
print("\n ✓ A_高频药材.json")
# ========== 4. 高频药组 (3味组合) ==========
print("\n[4/9] 高频药组 (3味药组) Top 15:")
herb_triplets = Counter()
for herb_list in formula_herbs.values():
u = sorted(list(dict.fromkeys(herb_list)))
if len(u) >= 3:
for combo in combinations(u, 3):
herb_triplets[combo] += 1
# 筛选出现>=2次的药组
top_triples = [t for t in herb_triplets if herb_triplets[t] >= 2]
top_triples.sort(key=lambda x: -herb_triplets[x])
for t in top_triples[:15]:
c = herb_triplets[t]
print(f" {t[0]}+{t[1]}+{t[2]}: {c}首方")
with open(os.path.join(OUT, "B_高频药组.json"), 'w', encoding='utf-8') as f:
d = {"3味药组": OrderedDict()}
for t in top_triples[:30]:
key_str = t[0] + "+" + t[1] + "+" + t[2]
d["3味药组"][key_str] = herb_triplets[t]
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ B_高频药组.json")
# ========== 5. 功效网络 ==========
print("\n[5/9] 功效网络分析...")
func_formulas = defaultdict(set)
for h, data in herbs.items():
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
# 拆分多个功效
funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()]
for func in funcs:
# 功效只出现在有该药材的方剂中
for fname in herb_formulas.get(h, set()):
func_formulas[func].add(fname)
top_funcs = [(fx, len(vx)) for fx, vx in func_formulas.items() if len(vx) >= 10]
top_funcs.sort(key=lambda x: -x[1])
print(" Top 20 高频功效:")
for func, count in top_funcs[:20]:
examples = list(func_formulas[func])[:3]
print(f" {func:30s} {count:4d}首方 例: {', '.join(examples[:1])}")
with open(os.path.join(OUT, "C_功效网络.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for fx, cx in top_funcs:
d[fx] = {"count": cx, "formulas": list(func_formulas[fx])[:20]}
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ C_功效网络.json")
# ========== 6. 方剂聚类 ==========
print("\n[6/9] 方剂聚类分析...")
def get_main_funcs(herb_list):
"""获取方剂的主要功效"""
func_weight = Counter()
for herb in herb_list:
if herb in herbs:
gua = herbs[herb].get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()]
for func in funcs:
func_weight[func] += 1
return func_weight
# 根据主要功效聚类
func_clusters = defaultdict(list)
for fname, herb_list in formula_herbs.items():
main = get_main_funcs(herb_list)
if main:
top_func = main.most_common(1)[0][0]
func_clusters[top_func].append(fname)
# 筛选大簇
big_clusters = [(fx, flst) for fx, flst in func_clusters.items() if len(flst) >= 5]
big_clusters.sort(key=lambda x: -len(x[1]))
print(f" 发现 {len(func_clusters)} 个类方剂组")
print(" Top 15 大类:")
for func, flst in big_clusters[:15]:
print(f" {func[:30]}: {len(flst)}首方")
with open(os.path.join(OUT, "D_方剂聚类.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for fx, flst in big_clusters[:20]:
d[fx] = flst
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ D_方剂聚类.json")
# ========== 7. 配伍禁忌验证 ==========
print("\n[7/9] 配伍禁忌验证...")
# 十八反歌诀
fan = {
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
}
wei = {
'硫黄': ['朴硝', '芒硝', '牙硝'],
'水银': ['铅丹', '砒霜'],
'巴豆': ['牵牛', '牵牛子'],
'丁香': ['郁金'],
'人参': ['五灵脂'],
}
contra = defaultdict(int)
contra_details = defaultdict(list)
for fname, herb_list in formula_herbs.items():
for herb, contra_list in {**fan, **wei}.items():
if herb in herb_list:
for ch in contra_list:
if ch in herb_list:
contra[(herb, ch)] += 1
contra_details[(herb, ch)].append(fname)
top_contra = sorted(contra.items(), key=lambda x: -x[1])[:15]
if top_contra:
print(" 发现理论配伍禁忌 (十八反/十九畏):")
for (h1, h2), count in top_contra:
print(f" ⚠️ {h1} + {h2}: {count}首方")
if len(contra_details[(h1, h2)]) > 0:
print(f" 方剂: {', '.join(contra_details[(h1, h2)][:3])}...")
else:
print(" 未发现直接理论配伍禁忌配对")
contra_dict = OrderedDict()
for (h1, h2), cx in top_contra:
key_str = h1 + "+" + h2
contra_dict[key_str] = cx
contra_dict[h1 + "->方剂"] = contra_details.get((h1, h2), [])
with open(os.path.join(OUT, "E_配伍禁忌验证.json"), 'w', encoding='utf-8') as f:
json.dump({"十八反": fan, "十九畏": wei, "发现": contra_dict}, f, ensure_ascii=False, indent=2)
print("\n ✓ E_配伍禁忌验证.json")
# ========== 8. 方剂相似度矩阵 ==========
print("\n[8/9] 方剂相似度分析...")
def jaccard_similarity(set1, set2):
# 修正: 空集相似度应该为0,而不是1
if not set1 and not set2:
return 0.0
if not set1 or not set2:
return 0.0
intersection = len(set1 & set2)
union = len(set1 | set2)
return intersection / union if union > 0 else 0.0
similar_pairs = []
formula_list = list(formulas.keys())
for i in range(len(formula_list)):
for j in range(i + 1, len(formula_list)):
h1 = formula_herbs.get(formula_list[i], [])
h2 = formula_herbs.get(formula_list[j], [])
# 确保都至少有一味药
if len(h1) > 0 and len(h2) > 0:
sim = jaccard_similarity(set(h1), set(h2))
if sim >= 0.3:
similar_pairs.append((formula_list[i], formula_list[j], sim))
similar_pairs.sort(key=lambda x: -x[2])
top_similar = similar_pairs[:15]
print(f" 找到 {len(similar_pairs)} 对相似方剂 (相似度>=0.3)")
print(" Top 15 相似方剂对:")
for f1, f2, sim in top_similar[:15]:
shared = set(formula_herbs.get(f1, [])) & set(formula_herbs.get(f2, []))
print(f" {f1:30s} + {f2:30s} -> {sim:.2f} (共享{len(shared)}味药)")
with open(os.path.join(OUT, "F_方剂相似度.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for f1, f2, sim in top_similar[:20]:
key_str = f1 + " + " + f2
d[key_str] = round(sim, 4)
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ F_方剂相似度.json")
# ========== 9. 生成汇总 ==========
print("\n[9/9] 生成汇总报告...")
res_top20 = [{"name": h, "count": c} for h, c in top20]
res_triples = OrderedDict()
for t in top_triples[:10]:
key_str = t[0] + "+" + t[1] + "+" + t[2]
res_triples[key_str] = herb_triplets[t]
res_funcs = OrderedDict()
for fx, cx in top_funcs[:10]:
res_funcs[fx] = {"count": cx}
res_contra = OrderedDict()
for (h1, h2), cx in top_contra[:5]:
res_contra[h1 + "+" + h2] = cx
res_contra[h1 + "->方剂"] = contra_details.get((h1, h2), [])
res_similar = OrderedDict()
for f1, f2, sim in top_similar[:10]:
res_similar[f1 + " + " + f2] = round(sim, 4)
summary = {
"标题": "大医网 方剂-中药材 深度挖掘报告 (修正版)",
"数据规模": {
"药材": len(herbs),
"方剂": len(formulas),
"关联对数": total_links,
"平均每方药材": round(total_links / max(1, len(formula_herbs)), 2)
},
"主要发现": {
"高频药材": res_top20,
"高频药组": res_triples,
"高频功效": res_funcs,
"类方聚类": {"total_clusters": len(func_clusters), "big_clusters": len(big_clusters)},
"配伍禁忌": res_contra,
"方剂相似度": res_similar
},
"生成时间": "2026-05-02",
"输出目录": os.path.abspath(OUT)
}
with open(os.path.join(OUT, "G_总汇总.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print("\n ✓ G_总汇总.json")
# 输出结果
print("\n" + "=" * 70)
print(" 深度挖掘完成!")
print("=" * 70)
print(f"\n📁 输出目录: {os.path.abspath(OUT)}")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:35s} {size:>10,} bytes")
print("=" * 70)