411 lines
14 KiB
Python
411 lines
14 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
大医网 方剂-中药材 深度数据挖掘 (修正版)
|
||
- 修复药材提取率问题 (回退到之前有效的方法)
|
||
- 修复相似度分析 (空集Jaccard=0, 不是1)
|
||
- 修复浮点数截断错误
|
||
"""
|
||
import json
|
||
import os
|
||
import re
|
||
import glob
|
||
from collections import Counter, defaultdict, OrderedDict
|
||
from itertools import combinations
|
||
|
||
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
|
||
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_初版")
|
||
os.makedirs(OUT, exist_ok=True)
|
||
|
||
print("=" * 70)
|
||
print(" 大医网 方剂-中药材 深度数据挖掘 (修正版)")
|
||
print("=" * 70)
|
||
|
||
# ========== 1. 加载数据 ==========
|
||
print("\n[1/9] 加载数据...")
|
||
|
||
herbs = {} # name -> data
|
||
herb_names = set()
|
||
herb_alias_map = {} # alias -> canonical name
|
||
|
||
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
d = json.load(fh)
|
||
n = d.get('名称', '').strip()
|
||
if not n: continue
|
||
herbs[n] = d
|
||
herb_names.add(n)
|
||
aliases = d.get('别名', '')
|
||
if aliases:
|
||
for a in re.split(r'[、,,;]', aliases):
|
||
a = a.strip()
|
||
if a:
|
||
herb_names.add(a)
|
||
herb_alias_map[a] = n
|
||
|
||
print(f" 药材: {len(herbs)} 味, 含 {len(herb_names)} 个名称(含别名)")
|
||
|
||
formulas = OrderedDict()
|
||
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
d = json.load(fh)
|
||
name = d.get('名称', re.search(r'(\d+)_(.+)\.json', os.path.basename(f)).group(2) if re.search(r'(\d+)_(.+)\.json', os.path.basename(f)) else os.path.basename(f))
|
||
formulas[name] = d
|
||
|
||
print(f" 方剂: {len(formulas)} 首")
|
||
|
||
# ========== 2. 提取方剂中的药材 ==========
|
||
print("\n[2/9] 提取方剂中的药材...")
|
||
|
||
# 按长度降序排列药材名,确保长名称优先匹配
|
||
sorted_herb = sorted(herb_names, key=len, reverse=True)
|
||
|
||
def extract_herbs_from_text(text):
|
||
"""从文本中提取药材名 - 使用最长匹配优先"""
|
||
if not text or len(text) < 3:
|
||
return []
|
||
|
||
# 去掉炮制括号 药量如 一两、五钱
|
||
clean = re.sub(r'[((][^())]*[))]', '', text)
|
||
clean = re.sub(r'\d+[两钱毫升克毫克]', '', clean)
|
||
|
||
# 拆分成分隔符
|
||
parts = [p.strip() for p in re.split(r'[、,\;;]', clean) if p.strip()]
|
||
|
||
found = []
|
||
seen = set()
|
||
for p in parts:
|
||
p = p.strip()
|
||
if len(p) < 2:
|
||
continue
|
||
# 尝试匹配药材名
|
||
for herb in sorted_herb:
|
||
if herb == p or herb in p or p in herb:
|
||
if herb not in seen:
|
||
found.append(herb)
|
||
seen.add(herb)
|
||
if len(found) > 30: # 防止过多匹配
|
||
break
|
||
if len(found) > 30:
|
||
break
|
||
|
||
return found
|
||
|
||
def extract_herbs_brute(data_dict):
|
||
"""暴力匹配:检查每味药材是否出现在任何字段中"""
|
||
# 拼接所有字段文本
|
||
all_text = ''
|
||
for val in data_dict.values():
|
||
if isinstance(val, str):
|
||
all_text += val
|
||
elif isinstance(val, (dict, list)):
|
||
all_text += json.dumps(val, ensure_ascii=False)
|
||
|
||
found = []
|
||
for herb in sorted_herb:
|
||
if herb in all_text:
|
||
found.append(herb)
|
||
|
||
return found
|
||
|
||
formula_herbs = {} # formula_name -> list of herbs
|
||
for fname, data in formulas.items():
|
||
# 先尝试分段提取
|
||
herbs_found = []
|
||
for key, val in data.items():
|
||
if isinstance(val, str) and len(val) > 3:
|
||
herbs_found.extend(extract_herbs_from_text(val))
|
||
elif isinstance(val, (dict, list)):
|
||
herbs_found.extend(extract_herbs_from_text(json.dumps(val, ensure_ascii=False)))
|
||
|
||
# 再去重
|
||
herbs_found = list(dict.fromkeys(herbs_found))
|
||
|
||
# 如果分段提取效果不好,使用暴力匹配
|
||
if len(herbs_found) < 3:
|
||
herbs_found = extract_herbs_brute(data)
|
||
|
||
formula_herbs[fname] = herbs_found
|
||
|
||
total_links = sum(len(h) for h in formula_herbs.values())
|
||
avg = total_links / max(1, len(formula_herbs))
|
||
print(f" 总关联对数: {total_links}")
|
||
print(f" 平均每方药材数: {avg:.1f}")
|
||
|
||
# 检查提取质量
|
||
print("\n 检查提取结果 (前10首方):")
|
||
for i, (fname, herb_list) in enumerate(list(formula_herbs.items())[:10]):
|
||
print(f" [{i+1}] {fname}: {len(herb_list)}味 -> {herb_list[:7]}")
|
||
|
||
# 方剂->药材 的逆向索引
|
||
herb_formulas = defaultdict(set)
|
||
for fname, herb_list in formula_herbs.items():
|
||
for h in herb_list:
|
||
herb_formulas[h].add(fname)
|
||
|
||
# ========== 3. 高频药材 ==========
|
||
print("\n[3/9] 高频药材分析...")
|
||
herb_freq = Counter()
|
||
for herb_list in formula_herbs.values():
|
||
for h in herb_list:
|
||
herb_freq[h] += 1
|
||
top20 = herb_freq.most_common(20)
|
||
|
||
print(" Top 30 高频药材:")
|
||
for h, c in top20:
|
||
ct = herbs.get(h, {}).get('药材分类', '?')
|
||
func = herbs.get(h, {}).get('功效作用', {}).get('功能', '')
|
||
if isinstance(func, dict):
|
||
func = func.get('功能', '')
|
||
print(f" {h:10s} {c:4d}首 [{ct:3s}] {func[:25]}")
|
||
|
||
with open(os.path.join(OUT, "A_高频药材.json"), 'w', encoding='utf-8') as f:
|
||
out_list = []
|
||
for h, c in top20:
|
||
out_list.append({"name": h, "count": c})
|
||
json.dump(out_list, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ A_高频药材.json")
|
||
|
||
# ========== 4. 高频药组 (3味组合) ==========
|
||
print("\n[4/9] 高频药组 (3味药组) Top 15:")
|
||
herb_triplets = Counter()
|
||
for herb_list in formula_herbs.values():
|
||
u = sorted(list(dict.fromkeys(herb_list)))
|
||
if len(u) >= 3:
|
||
for combo in combinations(u, 3):
|
||
herb_triplets[combo] += 1
|
||
|
||
# 筛选出现>=2次的药组
|
||
top_triples = [t for t in herb_triplets if herb_triplets[t] >= 2]
|
||
top_triples.sort(key=lambda x: -herb_triplets[x])
|
||
|
||
for t in top_triples[:15]:
|
||
c = herb_triplets[t]
|
||
print(f" {t[0]}+{t[1]}+{t[2]}: {c}首方")
|
||
|
||
with open(os.path.join(OUT, "B_高频药组.json"), 'w', encoding='utf-8') as f:
|
||
d = {"3味药组": OrderedDict()}
|
||
for t in top_triples[:30]:
|
||
key_str = t[0] + "+" + t[1] + "+" + t[2]
|
||
d["3味药组"][key_str] = herb_triplets[t]
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ B_高频药组.json")
|
||
|
||
# ========== 5. 功效网络 ==========
|
||
print("\n[5/9] 功效网络分析...")
|
||
func_formulas = defaultdict(set)
|
||
for h, data in herbs.items():
|
||
gua = data.get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
func_str = gua['功能']
|
||
# 拆分多个功效
|
||
funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()]
|
||
for func in funcs:
|
||
# 功效只出现在有该药材的方剂中
|
||
for fname in herb_formulas.get(h, set()):
|
||
func_formulas[func].add(fname)
|
||
|
||
top_funcs = [(fx, len(vx)) for fx, vx in func_formulas.items() if len(vx) >= 10]
|
||
top_funcs.sort(key=lambda x: -x[1])
|
||
|
||
print(" Top 20 高频功效:")
|
||
for func, count in top_funcs[:20]:
|
||
examples = list(func_formulas[func])[:3]
|
||
print(f" {func:30s} {count:4d}首方 例: {', '.join(examples[:1])}")
|
||
|
||
with open(os.path.join(OUT, "C_功效网络.json"), 'w', encoding='utf-8') as f:
|
||
d = OrderedDict()
|
||
for fx, cx in top_funcs:
|
||
d[fx] = {"count": cx, "formulas": list(func_formulas[fx])[:20]}
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ C_功效网络.json")
|
||
|
||
# ========== 6. 方剂聚类 ==========
|
||
print("\n[6/9] 方剂聚类分析...")
|
||
|
||
def get_main_funcs(herb_list):
|
||
"""获取方剂的主要功效"""
|
||
func_weight = Counter()
|
||
for herb in herb_list:
|
||
if herb in herbs:
|
||
gua = herbs[herb].get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
func_str = gua['功能']
|
||
funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()]
|
||
for func in funcs:
|
||
func_weight[func] += 1
|
||
return func_weight
|
||
|
||
# 根据主要功效聚类
|
||
func_clusters = defaultdict(list)
|
||
for fname, herb_list in formula_herbs.items():
|
||
main = get_main_funcs(herb_list)
|
||
if main:
|
||
top_func = main.most_common(1)[0][0]
|
||
func_clusters[top_func].append(fname)
|
||
|
||
# 筛选大簇
|
||
big_clusters = [(fx, flst) for fx, flst in func_clusters.items() if len(flst) >= 5]
|
||
big_clusters.sort(key=lambda x: -len(x[1]))
|
||
|
||
print(f" 发现 {len(func_clusters)} 个类方剂组")
|
||
print(" Top 15 大类:")
|
||
for func, flst in big_clusters[:15]:
|
||
print(f" {func[:30]}: {len(flst)}首方")
|
||
|
||
with open(os.path.join(OUT, "D_方剂聚类.json"), 'w', encoding='utf-8') as f:
|
||
d = OrderedDict()
|
||
for fx, flst in big_clusters[:20]:
|
||
d[fx] = flst
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ D_方剂聚类.json")
|
||
|
||
# ========== 7. 配伍禁忌验证 ==========
|
||
print("\n[7/9] 配伍禁忌验证...")
|
||
|
||
# 十八反歌诀
|
||
fan = {
|
||
'甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'],
|
||
'乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'],
|
||
'藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'],
|
||
}
|
||
|
||
wei = {
|
||
'硫黄': ['朴硝', '芒硝', '牙硝'],
|
||
'水银': ['铅丹', '砒霜'],
|
||
'巴豆': ['牵牛', '牵牛子'],
|
||
'丁香': ['郁金'],
|
||
'人参': ['五灵脂'],
|
||
}
|
||
|
||
contra = defaultdict(int)
|
||
contra_details = defaultdict(list)
|
||
|
||
for fname, herb_list in formula_herbs.items():
|
||
for herb, contra_list in {**fan, **wei}.items():
|
||
if herb in herb_list:
|
||
for ch in contra_list:
|
||
if ch in herb_list:
|
||
contra[(herb, ch)] += 1
|
||
contra_details[(herb, ch)].append(fname)
|
||
|
||
top_contra = sorted(contra.items(), key=lambda x: -x[1])[:15]
|
||
if top_contra:
|
||
print(" 发现理论配伍禁忌 (十八反/十九畏):")
|
||
for (h1, h2), count in top_contra:
|
||
print(f" ⚠️ {h1} + {h2}: {count}首方")
|
||
if len(contra_details[(h1, h2)]) > 0:
|
||
print(f" 方剂: {', '.join(contra_details[(h1, h2)][:3])}...")
|
||
else:
|
||
print(" 未发现直接理论配伍禁忌配对")
|
||
|
||
contra_dict = OrderedDict()
|
||
for (h1, h2), cx in top_contra:
|
||
key_str = h1 + "+" + h2
|
||
contra_dict[key_str] = cx
|
||
contra_dict[h1 + "->方剂"] = contra_details.get((h1, h2), [])
|
||
|
||
with open(os.path.join(OUT, "E_配伍禁忌验证.json"), 'w', encoding='utf-8') as f:
|
||
json.dump({"十八反": fan, "十九畏": wei, "发现": contra_dict}, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ E_配伍禁忌验证.json")
|
||
|
||
# ========== 8. 方剂相似度矩阵 ==========
|
||
print("\n[8/9] 方剂相似度分析...")
|
||
|
||
def jaccard_similarity(set1, set2):
|
||
# 修正: 空集相似度应该为0,而不是1
|
||
if not set1 and not set2:
|
||
return 0.0
|
||
if not set1 or not set2:
|
||
return 0.0
|
||
intersection = len(set1 & set2)
|
||
union = len(set1 | set2)
|
||
return intersection / union if union > 0 else 0.0
|
||
|
||
similar_pairs = []
|
||
formula_list = list(formulas.keys())
|
||
for i in range(len(formula_list)):
|
||
for j in range(i + 1, len(formula_list)):
|
||
h1 = formula_herbs.get(formula_list[i], [])
|
||
h2 = formula_herbs.get(formula_list[j], [])
|
||
# 确保都至少有一味药
|
||
if len(h1) > 0 and len(h2) > 0:
|
||
sim = jaccard_similarity(set(h1), set(h2))
|
||
if sim >= 0.3:
|
||
similar_pairs.append((formula_list[i], formula_list[j], sim))
|
||
|
||
similar_pairs.sort(key=lambda x: -x[2])
|
||
top_similar = similar_pairs[:15]
|
||
|
||
print(f" 找到 {len(similar_pairs)} 对相似方剂 (相似度>=0.3)")
|
||
print(" Top 15 相似方剂对:")
|
||
for f1, f2, sim in top_similar[:15]:
|
||
shared = set(formula_herbs.get(f1, [])) & set(formula_herbs.get(f2, []))
|
||
print(f" {f1:30s} + {f2:30s} -> {sim:.2f} (共享{len(shared)}味药)")
|
||
|
||
with open(os.path.join(OUT, "F_方剂相似度.json"), 'w', encoding='utf-8') as f:
|
||
d = OrderedDict()
|
||
for f1, f2, sim in top_similar[:20]:
|
||
key_str = f1 + " + " + f2
|
||
d[key_str] = round(sim, 4)
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ F_方剂相似度.json")
|
||
|
||
# ========== 9. 生成汇总 ==========
|
||
print("\n[9/9] 生成汇总报告...")
|
||
|
||
res_top20 = [{"name": h, "count": c} for h, c in top20]
|
||
res_triples = OrderedDict()
|
||
for t in top_triples[:10]:
|
||
key_str = t[0] + "+" + t[1] + "+" + t[2]
|
||
res_triples[key_str] = herb_triplets[t]
|
||
|
||
res_funcs = OrderedDict()
|
||
for fx, cx in top_funcs[:10]:
|
||
res_funcs[fx] = {"count": cx}
|
||
|
||
res_contra = OrderedDict()
|
||
for (h1, h2), cx in top_contra[:5]:
|
||
res_contra[h1 + "+" + h2] = cx
|
||
res_contra[h1 + "->方剂"] = contra_details.get((h1, h2), [])
|
||
|
||
res_similar = OrderedDict()
|
||
for f1, f2, sim in top_similar[:10]:
|
||
res_similar[f1 + " + " + f2] = round(sim, 4)
|
||
|
||
summary = {
|
||
"标题": "大医网 方剂-中药材 深度挖掘报告 (修正版)",
|
||
"数据规模": {
|
||
"药材": len(herbs),
|
||
"方剂": len(formulas),
|
||
"关联对数": total_links,
|
||
"平均每方药材": round(total_links / max(1, len(formula_herbs)), 2)
|
||
},
|
||
"主要发现": {
|
||
"高频药材": res_top20,
|
||
"高频药组": res_triples,
|
||
"高频功效": res_funcs,
|
||
"类方聚类": {"total_clusters": len(func_clusters), "big_clusters": len(big_clusters)},
|
||
"配伍禁忌": res_contra,
|
||
"方剂相似度": res_similar
|
||
},
|
||
"生成时间": "2026-05-02",
|
||
"输出目录": os.path.abspath(OUT)
|
||
}
|
||
|
||
with open(os.path.join(OUT, "G_总汇总.json"), 'w', encoding='utf-8') as f:
|
||
json.dump(summary, f, ensure_ascii=False, indent=2)
|
||
|
||
print("\n ✓ G_总汇总.json")
|
||
|
||
# 输出结果
|
||
print("\n" + "=" * 70)
|
||
print(" 深度挖掘完成!")
|
||
print("=" * 70)
|
||
print(f"\n📁 输出目录: {os.path.abspath(OUT)}")
|
||
for fn in sorted(os.listdir(OUT)):
|
||
fp = os.path.join(OUT, fn)
|
||
size = os.path.getsize(fp)
|
||
print(f" {fn:35s} {size:>10,} bytes")
|
||
print("=" * 70)
|