#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 大医网 方剂-中药材 深度数据挖掘 (修正版) - 修复药材提取率问题 (回退到之前有效的方法) - 修复相似度分析 (空集Jaccard=0, 不是1) - 修复浮点数截断错误 """ import json import os import re import glob from collections import Counter, defaultdict, OrderedDict from itertools import combinations base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网" OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_初版") os.makedirs(OUT, exist_ok=True) print("=" * 70) print(" 大医网 方剂-中药材 深度数据挖掘 (修正版)") print("=" * 70) # ========== 1. 加载数据 ========== print("\n[1/9] 加载数据...") herbs = {} # name -> data herb_names = set() herb_alias_map = {} # alias -> canonical name for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")): with open(f, encoding='utf-8') as fh: d = json.load(fh) n = d.get('名称', '').strip() if not n: continue herbs[n] = d herb_names.add(n) aliases = d.get('别名', '') if aliases: for a in re.split(r'[、,,;]', aliases): a = a.strip() if a: herb_names.add(a) herb_alias_map[a] = n print(f" 药材: {len(herbs)} 味, 含 {len(herb_names)} 个名称(含别名)") formulas = OrderedDict() for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")): with open(f, encoding='utf-8') as fh: d = json.load(fh) name = d.get('名称', re.search(r'(\d+)_(.+)\.json', os.path.basename(f)).group(2) if re.search(r'(\d+)_(.+)\.json', os.path.basename(f)) else os.path.basename(f)) formulas[name] = d print(f" 方剂: {len(formulas)} 首") # ========== 2. 提取方剂中的药材 ========== print("\n[2/9] 提取方剂中的药材...") # 按长度降序排列药材名,确保长名称优先匹配 sorted_herb = sorted(herb_names, key=len, reverse=True) def extract_herbs_from_text(text): """从文本中提取药材名 - 使用最长匹配优先""" if not text or len(text) < 3: return [] # 去掉炮制括号 药量如 一两、五钱 clean = re.sub(r'[((][^())]*[))]', '', text) clean = re.sub(r'\d+[两钱毫升克毫克]', '', clean) # 拆分成分隔符 parts = [p.strip() for p in re.split(r'[、,\;;]', clean) if p.strip()] found = [] seen = set() for p in parts: p = p.strip() if len(p) < 2: continue # 尝试匹配药材名 for herb in sorted_herb: if herb == p or herb in p or p in herb: if herb not in seen: found.append(herb) seen.add(herb) if len(found) > 30: # 防止过多匹配 break if len(found) > 30: break return found def extract_herbs_brute(data_dict): """暴力匹配:检查每味药材是否出现在任何字段中""" # 拼接所有字段文本 all_text = '' for val in data_dict.values(): if isinstance(val, str): all_text += val elif isinstance(val, (dict, list)): all_text += json.dumps(val, ensure_ascii=False) found = [] for herb in sorted_herb: if herb in all_text: found.append(herb) return found formula_herbs = {} # formula_name -> list of herbs for fname, data in formulas.items(): # 先尝试分段提取 herbs_found = [] for key, val in data.items(): if isinstance(val, str) and len(val) > 3: herbs_found.extend(extract_herbs_from_text(val)) elif isinstance(val, (dict, list)): herbs_found.extend(extract_herbs_from_text(json.dumps(val, ensure_ascii=False))) # 再去重 herbs_found = list(dict.fromkeys(herbs_found)) # 如果分段提取效果不好,使用暴力匹配 if len(herbs_found) < 3: herbs_found = extract_herbs_brute(data) formula_herbs[fname] = herbs_found total_links = sum(len(h) for h in formula_herbs.values()) avg = total_links / max(1, len(formula_herbs)) print(f" 总关联对数: {total_links}") print(f" 平均每方药材数: {avg:.1f}") # 检查提取质量 print("\n 检查提取结果 (前10首方):") for i, (fname, herb_list) in enumerate(list(formula_herbs.items())[:10]): print(f" [{i+1}] {fname}: {len(herb_list)}味 -> {herb_list[:7]}") # 方剂->药材 的逆向索引 herb_formulas = defaultdict(set) for fname, herb_list in formula_herbs.items(): for h in herb_list: herb_formulas[h].add(fname) # ========== 3. 高频药材 ========== print("\n[3/9] 高频药材分析...") herb_freq = Counter() for herb_list in formula_herbs.values(): for h in herb_list: herb_freq[h] += 1 top20 = herb_freq.most_common(20) print(" Top 30 高频药材:") for h, c in top20: ct = herbs.get(h, {}).get('药材分类', '?') func = herbs.get(h, {}).get('功效作用', {}).get('功能', '') if isinstance(func, dict): func = func.get('功能', '') print(f" {h:10s} {c:4d}首 [{ct:3s}] {func[:25]}") with open(os.path.join(OUT, "A_高频药材.json"), 'w', encoding='utf-8') as f: out_list = [] for h, c in top20: out_list.append({"name": h, "count": c}) json.dump(out_list, f, ensure_ascii=False, indent=2) print("\n ✓ A_高频药材.json") # ========== 4. 高频药组 (3味组合) ========== print("\n[4/9] 高频药组 (3味药组) Top 15:") herb_triplets = Counter() for herb_list in formula_herbs.values(): u = sorted(list(dict.fromkeys(herb_list))) if len(u) >= 3: for combo in combinations(u, 3): herb_triplets[combo] += 1 # 筛选出现>=2次的药组 top_triples = [t for t in herb_triplets if herb_triplets[t] >= 2] top_triples.sort(key=lambda x: -herb_triplets[x]) for t in top_triples[:15]: c = herb_triplets[t] print(f" {t[0]}+{t[1]}+{t[2]}: {c}首方") with open(os.path.join(OUT, "B_高频药组.json"), 'w', encoding='utf-8') as f: d = {"3味药组": OrderedDict()} for t in top_triples[:30]: key_str = t[0] + "+" + t[1] + "+" + t[2] d["3味药组"][key_str] = herb_triplets[t] json.dump(d, f, ensure_ascii=False, indent=2) print("\n ✓ B_高频药组.json") # ========== 5. 功效网络 ========== print("\n[5/9] 功效网络分析...") func_formulas = defaultdict(set) for h, data in herbs.items(): gua = data.get('功效作用', {}) if isinstance(gua, dict) and gua.get('功能'): func_str = gua['功能'] # 拆分多个功效 funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()] for func in funcs: # 功效只出现在有该药材的方剂中 for fname in herb_formulas.get(h, set()): func_formulas[func].add(fname) top_funcs = [(fx, len(vx)) for fx, vx in func_formulas.items() if len(vx) >= 10] top_funcs.sort(key=lambda x: -x[1]) print(" Top 20 高频功效:") for func, count in top_funcs[:20]: examples = list(func_formulas[func])[:3] print(f" {func:30s} {count:4d}首方 例: {', '.join(examples[:1])}") with open(os.path.join(OUT, "C_功效网络.json"), 'w', encoding='utf-8') as f: d = OrderedDict() for fx, cx in top_funcs: d[fx] = {"count": cx, "formulas": list(func_formulas[fx])[:20]} json.dump(d, f, ensure_ascii=False, indent=2) print("\n ✓ C_功效网络.json") # ========== 6. 方剂聚类 ========== print("\n[6/9] 方剂聚类分析...") def get_main_funcs(herb_list): """获取方剂的主要功效""" func_weight = Counter() for herb in herb_list: if herb in herbs: gua = herbs[herb].get('功效作用', {}) if isinstance(gua, dict) and gua.get('功能'): func_str = gua['功能'] funcs = [fx.strip() for fx in re.split(r'[,。.::]', func_str) if fx.strip()] for func in funcs: func_weight[func] += 1 return func_weight # 根据主要功效聚类 func_clusters = defaultdict(list) for fname, herb_list in formula_herbs.items(): main = get_main_funcs(herb_list) if main: top_func = main.most_common(1)[0][0] func_clusters[top_func].append(fname) # 筛选大簇 big_clusters = [(fx, flst) for fx, flst in func_clusters.items() if len(flst) >= 5] big_clusters.sort(key=lambda x: -len(x[1])) print(f" 发现 {len(func_clusters)} 个类方剂组") print(" Top 15 大类:") for func, flst in big_clusters[:15]: print(f" {func[:30]}: {len(flst)}首方") with open(os.path.join(OUT, "D_方剂聚类.json"), 'w', encoding='utf-8') as f: d = OrderedDict() for fx, flst in big_clusters[:20]: d[fx] = flst json.dump(d, f, ensure_ascii=False, indent=2) print("\n ✓ D_方剂聚类.json") # ========== 7. 配伍禁忌验证 ========== print("\n[7/9] 配伍禁忌验证...") # 十八反歌诀 fan = { '甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'], '乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'], '藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'], } wei = { '硫黄': ['朴硝', '芒硝', '牙硝'], '水银': ['铅丹', '砒霜'], '巴豆': ['牵牛', '牵牛子'], '丁香': ['郁金'], '人参': ['五灵脂'], } contra = defaultdict(int) contra_details = defaultdict(list) for fname, herb_list in formula_herbs.items(): for herb, contra_list in {**fan, **wei}.items(): if herb in herb_list: for ch in contra_list: if ch in herb_list: contra[(herb, ch)] += 1 contra_details[(herb, ch)].append(fname) top_contra = sorted(contra.items(), key=lambda x: -x[1])[:15] if top_contra: print(" 发现理论配伍禁忌 (十八反/十九畏):") for (h1, h2), count in top_contra: print(f" ⚠️ {h1} + {h2}: {count}首方") if len(contra_details[(h1, h2)]) > 0: print(f" 方剂: {', '.join(contra_details[(h1, h2)][:3])}...") else: print(" 未发现直接理论配伍禁忌配对") contra_dict = OrderedDict() for (h1, h2), cx in top_contra: key_str = h1 + "+" + h2 contra_dict[key_str] = cx contra_dict[h1 + "->方剂"] = contra_details.get((h1, h2), []) with open(os.path.join(OUT, "E_配伍禁忌验证.json"), 'w', encoding='utf-8') as f: json.dump({"十八反": fan, "十九畏": wei, "发现": contra_dict}, f, ensure_ascii=False, indent=2) print("\n ✓ E_配伍禁忌验证.json") # ========== 8. 方剂相似度矩阵 ========== print("\n[8/9] 方剂相似度分析...") def jaccard_similarity(set1, set2): # 修正: 空集相似度应该为0,而不是1 if not set1 and not set2: return 0.0 if not set1 or not set2: return 0.0 intersection = len(set1 & set2) union = len(set1 | set2) return intersection / union if union > 0 else 0.0 similar_pairs = [] formula_list = list(formulas.keys()) for i in range(len(formula_list)): for j in range(i + 1, len(formula_list)): h1 = formula_herbs.get(formula_list[i], []) h2 = formula_herbs.get(formula_list[j], []) # 确保都至少有一味药 if len(h1) > 0 and len(h2) > 0: sim = jaccard_similarity(set(h1), set(h2)) if sim >= 0.3: similar_pairs.append((formula_list[i], formula_list[j], sim)) similar_pairs.sort(key=lambda x: -x[2]) top_similar = similar_pairs[:15] print(f" 找到 {len(similar_pairs)} 对相似方剂 (相似度>=0.3)") print(" Top 15 相似方剂对:") for f1, f2, sim in top_similar[:15]: shared = set(formula_herbs.get(f1, [])) & set(formula_herbs.get(f2, [])) print(f" {f1:30s} + {f2:30s} -> {sim:.2f} (共享{len(shared)}味药)") with open(os.path.join(OUT, "F_方剂相似度.json"), 'w', encoding='utf-8') as f: d = OrderedDict() for f1, f2, sim in top_similar[:20]: key_str = f1 + " + " + f2 d[key_str] = round(sim, 4) json.dump(d, f, ensure_ascii=False, indent=2) print("\n ✓ F_方剂相似度.json") # ========== 9. 生成汇总 ========== print("\n[9/9] 生成汇总报告...") res_top20 = [{"name": h, "count": c} for h, c in top20] res_triples = OrderedDict() for t in top_triples[:10]: key_str = t[0] + "+" + t[1] + "+" + t[2] res_triples[key_str] = herb_triplets[t] res_funcs = OrderedDict() for fx, cx in top_funcs[:10]: res_funcs[fx] = {"count": cx} res_contra = OrderedDict() for (h1, h2), cx in top_contra[:5]: res_contra[h1 + "+" + h2] = cx res_contra[h1 + "->方剂"] = contra_details.get((h1, h2), []) res_similar = OrderedDict() for f1, f2, sim in top_similar[:10]: res_similar[f1 + " + " + f2] = round(sim, 4) summary = { "标题": "大医网 方剂-中药材 深度挖掘报告 (修正版)", "数据规模": { "药材": len(herbs), "方剂": len(formulas), "关联对数": total_links, "平均每方药材": round(total_links / max(1, len(formula_herbs)), 2) }, "主要发现": { "高频药材": res_top20, "高频药组": res_triples, "高频功效": res_funcs, "类方聚类": {"total_clusters": len(func_clusters), "big_clusters": len(big_clusters)}, "配伍禁忌": res_contra, "方剂相似度": res_similar }, "生成时间": "2026-05-02", "输出目录": os.path.abspath(OUT) } with open(os.path.join(OUT, "G_总汇总.json"), 'w', encoding='utf-8') as f: json.dump(summary, f, ensure_ascii=False, indent=2) print("\n ✓ G_总汇总.json") # 输出结果 print("\n" + "=" * 70) print(" 深度挖掘完成!") print("=" * 70) print(f"\n📁 输出目录: {os.path.abspath(OUT)}") for fn in sorted(os.listdir(OUT)): fp = os.path.join(OUT, fn) size = os.path.getsize(fp) print(f" {fn:35s} {size:>10,} bytes") print("=" * 70)