#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 深度数据挖掘 - 精修修正版 核心修正: 1. 药材匹配仅使用药材主名(name),不含单字别名 2. 药材名按长度降序排列,确保长名优先匹配 3. 功效关键词过滤噪音(去除"本品""诸药"等解析残余) 4. 药对/药组仅统计完整药材名 """ import json import os import re import glob from collections import Counter, defaultdict, OrderedDict from itertools import combinations base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网" OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_精修2") os.makedirs(OUT, exist_ok=True) # ========== 加载药材 ========== print("=" * 70) print(" 大医网 方剂-中药材 深度数据挖掘 (精修修正版)") print("=" * 70) print(f"\n[1/8] 加载数据...") # 药材主名集合(仅使用主名,不含别名) herbs = {} # canonical_name -> full_data valid_herb_names = [] # 仅≥2字的主名,按长度降序排列 for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")): with open(f, encoding='utf-8') as fh: d = json.load(fh) name = d.get('名称', '').strip() if name: herbs[name] = d # 仅保留≥2个字的药材名作为匹配字典 for name, data in herbs.items(): if len(name) >= 2: valid_herb_names.append(name) # 按长度降序,确保长名优先匹配 valid_herb_names.sort(key=len, reverse=True) herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names)) print(f" 药材: {len(herbs)} 味 (匹配字典: {len(valid_herb_names)}味)") print(f" 药材样本: {', '.join(valid_herb_names[:10])}") # ========== 加载方剂 ========== formulas = OrderedDict() for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")): with open(f, encoding='utf-8') as fh: d = json.load(fh) name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0]) formulas[name] = d print(f" 方剂: {len(formulas)} 首") # ========== 提取方剂中的药材 ========== print(f"\n[2/8] 提取方剂中的药材...") def extract_herbs_exact(text): """精确匹配:仅使用≥2字的主药材名""" if not text or len(text) < 3: return [] # 去掉炮制括号 药量 clean = re.sub(r'[((][^())]*[))]', '', text) clean = re.sub(r'\d+[两钱毫升克]', '', clean) # 用正则找到所有药材名 found = [] for match in herb_pattern.finditer(clean): herb = match.group(0) if herb and herb not in found: found.append(herb) return found def get_all_text(data_dict): """拼接所有字段文本""" all_text = '' for val in data_dict.values(): if isinstance(val, str): all_text += val elif isinstance(val, (dict, list)): all_text += json.dumps(val, ensure_ascii=False) return all_text formula_herbs = OrderedDict() for fname, data in formulas.items(): all_text = get_all_text(data) herbs_found = extract_herbs_exact(all_text) formula_herbs[fname] = herbs_found total_links = sum(len(h) for h in formula_herbs.values()) avg = total_links / max(1, len(formula_herbs)) print(f" 总关联对数: {total_links}") print(f" 平均每方药材数: {avg:.1f}") # 检查提取质量 print(f"\n 提取结果检查 (前15首方):") for i, (fname, hlist) in enumerate(list(formula_herbs.items())[:15]): print(f" [{i+1:2d}] {fname:30s} {len(hlist):2d}味 -> {hlist[:8]}") # 逆向索引 herb_formulas = defaultdict(set) for fname, hlist in formula_herbs.items(): for h in hlist: herb_formulas[h].add(fname) # ========== 3. 高频药材 ========== print(f"\n[3/8] 高频药材分析...") herb_freq = Counter() for hlist in formula_herbs.values(): for h in hlist: herb_freq[h] += 1 top30 = herb_freq.most_common(30) print(f"\n Top 30 高频药材:") print(f" {'药材':<12s} {'频次':>5s} {'分类':<6s} {'功效':<30s}") print(f" {'-'*12} {'-'*5} {'-'*6} {'-'*30}") for h, c in top30: ct = herbs.get(h, {}).get('药材分类', '未知') func_data = herbs.get(h, {}).get('功效作用', {}) func = func_data.get('功能', '')[:30] if isinstance(func_data, dict) else '' print(f" {h:<12s} {c:>5d} {ct:<6s} {func}") with open(os.path.join(OUT, "01_高频药材Top30.json"), 'w', encoding='utf-8') as f: json.dump([{"name": h, "count": c} for h, c in top30], f, ensure_ascii=False, indent=2) print("\n ✓ 01_高频药材Top30.json") # ========== 4. 高频药对 ========== print(f"\n[4/8] 高频药对 (2味) Top 20...") pair_freq = Counter() for hlist in formula_herbs.values(): u = sorted(list(set(hlist))) if len(u) >= 2: for combo in combinations(u, 2): pair_freq[combo] += 1 # Top 50药对 (≥5次) top_pairs = [p for p, c in pair_freq.most_common(100) if c >= 5] print(f"\n Top 20 高频药对:") print(f" {'药对':<30s} {'频次':>5s} {'药材1功效':<12s} {'药材2功效':<12s}") print(f" {'-'*30} {'-'*5} {'-'*12} {'-'*12}") for pair in top_pairs[:20]: h1, h2 = pair c = pair_freq[pair] func_data1 = herbs.get(h1, {}).get('功效作用', {}) func_data2 = herbs.get(h2, {}).get('功效作用', {}) func1 = func_data1.get('功能', '')[:12] if isinstance(func_data1, dict) else '' func2 = func_data2.get('功能', '')[:12] if isinstance(func_data2, dict) else '' print(f" {h1:12s} + {h2:12s} {c:>5d} [{func1[:8]}] [{func2[:8]}]") with open(os.path.join(OUT, "02_高频药对_Top20.json"), 'w', encoding='utf-8') as f: json.dump([{"pair": list(p), "count": c} for p, c in pair_freq.most_common(50)], f, ensure_ascii=False, indent=2) print("\n ✓ 02_高频药对_Top20.json") # ========== 5. 核心药组 ========== print(f"\n[5/8] 核心药组 (3-5味药) Top 10...") for n in [3, 4, 5]: triple_freq = Counter() for hlist in formula_herbs.values(): u = sorted(list(set(hlist))) if len(u) >= n: for combo in combinations(u, n): triple_freq[combo] += 1 top_combs = [(t, c) for t, c in triple_freq.most_common(50) if c >= 5] print(f"\n Top {n}味药组 (出现≥5次):") for combo, c in top_combs[:10]: names = "+".join(combo) print(f" {names:>40s}: {c}首方") with open(os.path.join(OUT, "03_核心药组3-5味_Top10.json"), 'w', encoding='utf-8') as f: res = {} for n in [3, 4, 5]: triple_freq = Counter() for hlist in formula_herbs.values(): u = sorted(list(set(hlist))) if len(u) >= n: for combo in combinations(u, n): triple_freq[combo] += 1 res[f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5] json.dump(res, f, ensure_ascii=False, indent=2) print("\n ✓ 03_核心药组3-5味_Top10.json") # ========== 6. 功效关键词网络 ========== print(f"\n[6/8] 功效关键词网络...") # 提取核心功效词(过滤噪音词) noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '。'} func_keywords = Counter() func_formulas = defaultdict(set) for h, data in herbs.items(): if h not in herb_formulas: continue gua = data.get('功效作用', {}) if not isinstance(gua, dict): continue func_str = gua.get('功能', '') if not func_str: continue # 按标点拆分 keywords = re.split(r'[、,,、;;。]', func_str) for kw in keywords: kw = kw.strip() # 过滤噪音词 if not kw or len(kw) < 2 or kw in noise_words: continue # 也检查是否全是药材名(如"甘草"也是功效词"甘草具有补脾益气"中的残留) if kw == h and isinstance(gua, dict) and func_str.startswith(kw): # 如果关键词就是药材名且紧跟药材名,跳过 continue if kw in herbs: # 如果关键词本身是药材名且在功效描述中(非独立功效词),跳过 continue func_keywords[kw] += 1 for fname in herb_formulas[h]: func_formulas[kw].add(fname) # 按方剂数排序 top_funcs = [(kw, len(vx)) for kw, vx in func_formulas.items() if len(vx) >= 10] top_funcs.sort(key=lambda x: -x[1]) print(f"\n Top 30 功效关键词 (按关联方剂数):") print(f" {'关键词':<15s} {'方剂数':>6s} 示例方剂") print(f" {'-'*15} {'-'*6} {'-'*40}") for kw, count in top_funcs[:30]: examples = list(func_formulas[kw])[:2] print(f" {kw:<15s} {count:>6d} {', '.join(examples)}") with open(os.path.join(OUT, "04_功效关键词_Top30.json"), 'w', encoding='utf-8') as f: json.dump([{"keyword": kw, "formula_count": count} for kw, count in top_funcs[:30]], f, ensure_ascii=False, indent=2) print("\n ✓ 04_功效关键词_Top30.json") # ========== 7. 方剂聚类 ========== print(f"\n[7/8] 方剂聚类...") clusters = defaultdict(list) for fname, hlist in formula_herbs.items(): kw_count = Counter() for herb in hlist: if herb in herbs: gua = herbs[herb].get('功效作用', {}) if isinstance(gua, dict) and gua.get('功能'): kw_str = gua['功能'] keywords = re.split(r'[、,,、;;。]', kw_str) for kw in keywords: kw = kw.strip() if kw and len(kw) >= 2 and kw not in noise_words: kw_count[kw] += 1 if kw_count: top_kw = kw_count.most_common(1)[0][0] clusters[top_kw].append(fname) big_clusters = [(kw, flst) for kw, flst in clusters.items() if len(flst) >= 5] big_clusters.sort(key=lambda x: -len(x[1])) print(f" 发现 {len(clusters)} 个功效簇") print(f"\n Top 20 大功效簇:") for kw, flst in big_clusters[:20]: # 找簇内的代表性药材 sample_herbs = set() for fname in flst[:10]: for h in formula_herbs[fname]: sample_herbs.add(h) top_h = sample_herbs print(f" '{kw}' ({len(flst)}首方): 代表药材 -> {', '.join(sorted(top_h)[:10])}") with open(os.path.join(OUT, "05_方剂聚类_Top20.json"), 'w', encoding='utf-8') as f: json.dump([{"cluster": kw, "count": len(flst), "formulas": flst[:20]} for kw, flst in big_clusters[:20]], f, ensure_ascii=False, indent=2) print("\n ✓ 05_方剂聚类_Top20.json") # ========== 8. 配伍禁忌 & 相似度 ========== print(f"\n[8/8] 配伍禁忌验证 & 方剂相似度...") # 十八反 fan = { '甘草': ['大戟', '芫花', '甘遂', '京大戟', '红大戟'], '乌头': ['半夏', '瓜蒌', '天花粉', '贝母', '平贝母', '川贝母', '浙贝母', '白蔹', '白及'], '藜芦': ['人参', '党参', '丹参', '玄参', '沙参', '苦参', '细辛', '白芍', '赤芍'], } wei = { '硫黄': ['朴硝', '芒硝', '牙硝'], '水银': ['铅丹', '砒霜'], '巴豆': ['牵牛', '牵牛子'], '丁香': ['郁金'], '人参': ['五灵脂'], '肉桂': ['石脂', '赤石脂'], '半夏': ['羊脂'], '厚朴': ['硝石', '滑石'], } # 检查禁忌药材是否在方剂中出现 contra_found = defaultdict(list) all_contra_search = {} for k, v in fan.items(): for vv in v: if vv not in all_contra_search: all_contra_search[vv] = k for k, v in wei.items(): for vv in v: if vv not in all_contra_search: all_contra_search[vv] = k print(f"\n 十八反/十九畏相关药材在方剂中出现情况:") for herb_name, related in all_contra_search.items(): if herb_name in herb_formulas: count = len(herb_formulas[herb_name]) print(f" ⚠ {herb_name} (反/畏{related}): 出现在 {count} 首方") # 检查实际方剂中是否同时出现矛盾配对 contra_pairs_found = defaultdict(int) contra_pairs_details = defaultdict(list) for fname, hlist in formula_herbs.items(): for herb, contra_list in fan.items(): if herb in hlist: for ch in contra_list: if ch in hlist: contra_pairs_found[(herb, ch)] += 1 contra_pairs_details[(herb, ch)].append(fname) for herb, contra_list in wei.items(): if herb in hlist: for ch in contra_list: if ch in hlist: contra_pairs_found[(herb, ch)] += 1 contra_pairs_details[(herb, ch)].append(fname) top_contra = sorted(contra_pairs_found.items(), key=lambda x: -x[1])[:10] if top_contra: print(f"\n 发现配伍禁忌:") for (h1, h2), count in top_contra: print(f" ⚠️ {h1} + {h2}: {count}首方 ({', '.join(contra_pairs_details[(h1,h2)][:2])})") else: print(f"\n 未发现十八反/十九畏的直接配对出现在同一首方剂中") with open(os.path.join(OUT, "06_配伍禁忌.json"), 'w', encoding='utf-8') as f: json.dump({ "十八反": fan, "十九畏": wei, "禁忌药材出现": {k: len(herb_formulas.get(k, set())) for k in all_contra_search.keys() if k in herb_formulas}, "实际禁忌配对": {f"{k[0]}+{k[1]}": c for k, c in top_contra}, "禁忌详情": {f"{k[0]}+{k[1]}": contra_pairs_details[k] for k in top_contra} }, f, ensure_ascii=False, indent=2) print("\n ✓ 06_配伍禁忌.json") # 方剂相似度 print(f"\n 计算方剂相似度 (Jaccard >= 0.3)...") def jaccard_fixed(s1, s2): if not s1 or not s2: return 0.0 intersection = len(s1 & s2) if intersection == 0: return 0.0 union = len(s1 | s2) return intersection / union if union > 0 else 0.0 similar_pairs = [] formula_list = list(formulas.keys()) count = 0 for i in range(len(formula_list)): f1 = formula_list[i] h1 = formula_herbs.get(f1, []) if not h1: continue s1 = set(h1) for j in range(i + 1, len(formula_list)): f2 = formula_list[j] h2 = formula_herbs.get(f2, []) if not h2: continue s2 = set(h2) sim = jaccard_fixed(s1, s2) if sim >= 0.3: shared = s1 & s2 similar_pairs.append((f1, f2, sim, len(shared), sorted(shared))) count += 1 print(f" 检查 {count} 对, 找到 {len(similar_pairs)} 对相似方剂") similar_pairs.sort(key=lambda x: (-x[2], -x[3])) top_similar = similar_pairs[:20] print(f"\n Top 15 相似方剂对:") for f1, f2, sim, shared_c, shared_h in top_similar[:15]: print(f" {f1:30s} + {f2:30s} -> {sim:.3f} (共享{shared_c}味: {', '.join(shared_h)})") with open(os.path.join(OUT, "07_方剂相似度_Top20.json"), 'w', encoding='utf-8') as f: json.dump([{"similarity": round(sim, 4), "shared_count": sc, "pair1": f1, "pair2": f2, "shared_herbs": sh} for f1, f2, sim, sc, sh in top_similar], f, ensure_ascii=False, indent=2) print("\n ✓ 07_方剂相似度_Top20.json") # ========== 汇总 ========== print(f"\n{'='*70}") print(f" 深度挖掘分析完成!") print(f"{'='*70}") summary = OrderedDict() summary["标题"] = "大医网 方剂-中药材 深度数据挖掘报告 (精修修正版)" summary["数据规模"] = { "药材": len(herbs), "方剂": len(formulas), "总关联对数": total_links, "平均每方药材数": round(avg, 1), "成功提取方剂": sum(1 for h in formula_herbs.values() if h), "无法提取方剂": sum(1 for h in formula_herbs.values() if not h), } summary["高频药材_Top20"] = [{"name": h, "count": c} for h, c in top30[:20]] summary["高频药对_Top10"] = [{"pair": list(p), "count": c} for p, c in pair_freq.most_common(10)] summary["核心药组_Top10"] = {} for n in [3, 4, 5]: triple_freq = Counter() for hlist in formula_herbs.values(): u = sorted(list(set(hlist))) if len(u) >= n: for combo in combinations(u, n): triple_freq[combo] += 1 summary["核心药组_Top10"][f"{n}味药组"] = [{"group": "+".join(t), "count": c} for t, c in triple_freq.most_common(10) if c >= 5] summary["功效关键词_Top15"] = [{"keyword": kw, "count": c} for kw, c in top_funcs[:15]] summary["方剂聚类_Top10"] = [{"cluster": kw, "count": len(flst)} for kw, flst in big_clusters[:10]] summary["配伍禁忌"] = { "十八反": fan, "十九畏": wei, "实际发现": {f"{k[0]}+{k[1]}": c for k, c in top_contra}, } summary["方剂相似度_Top10"] = [{"similarity": round(sim, 4), "pair": f"{f1} + {f2}"} for f1, f2, sim, sc, sh in top_similar[:10]] summary["输出目录"] = os.path.abspath(OUT) with open(os.path.join(OUT, "汇总报告.json"), 'w', encoding='utf-8') as f: json.dump(summary, f, ensure_ascii=False, indent=2) print(f"\n 输出目录: {os.path.abspath(OUT)}") print(f"\n 文件列表:") for fn in sorted(os.listdir(OUT)): fp = os.path.join(OUT, fn) size = os.path.getsize(fp) print(f" {fn:40s} {size:>10,} B")