Files

412 lines
15 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
大医网深度挖掘 - 进阶版
1. 功效-方剂完整映射网络
2. 方剂来源分析(出处)
3. 药材分类关联
4. 性味归经交叉分析
5. 药对-功效关联
"""
import json
import os
import re
import glob
from collections import Counter, defaultdict, OrderedDict
from itertools import combinations
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_进阶")
os.makedirs(OUT, exist_ok=True)
print("=" * 80)
print(" 大医网深度数据挖掘 - 进阶版 (多维关联分析)")
print("=" * 80)
# ========== 加载数据 ==========
print("\n[1/6] 加载数据...")
herbs = {} # name -> data
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', '').strip()
if name:
herbs[name] = d
formulas = OrderedDict()
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
with open(f, encoding='utf-8') as fh:
d = json.load(fh)
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
formulas[name] = d
print(f" 药材: {len(herbs)} 味")
print(f" 方剂: {len(formulas)} 首")
# 药材匹配字典 (仅主名, ≥2字)
valid_herb_names = [n for n in herbs.keys() if len(n) >= 2]
valid_herb_names.sort(key=len, reverse=True)
herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names))
# 提取药材
def extract_herbs(text):
if not text or len(text) < 3:
return []
clean = re.sub(r'[((][^())]*[))]', '', text)
clean = re.sub(r'\d+[两钱毫升克]', '', clean)
found = []
for match in herb_pattern.finditer(clean):
herb = match.group(0)
if herb and herb not in found:
found.append(herb)
return found
formula_herbs = OrderedDict()
for fname, data in formulas.items():
all_text = ''
for val in data.values():
if isinstance(val, str):
all_text += val
elif isinstance(val, (dict, list)):
all_text += json.dumps(val, ensure_ascii=False)
h = extract_herbs(all_text)
formula_herbs[fname] = h
# 逆向索引
herb_formulas = defaultdict(set)
for fname, hlist in formula_herbs.items():
for h in hlist:
herb_formulas[h].add(fname)
total_links = sum(len(h) for h in formula_herbs.values())
print(f"\n 总关联对数: {total_links}")
print(f" 成功提取方剂: {sum(1 for h in formula_herbs.values() if h)} / {len(formulas)}")
# ========== 分析1: 功效-方剂完整映射 ==========
print(f"\n{'='*80}")
print(f"[2/6] 功效-方剂映射网络. ..")
print(f"{'='*80}")
noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '的功能', '功效作用'}
# 构建功效到药材的映射
func_to_herbs = defaultdict(set)
for h, data in herbs.items():
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
keywords = re.split(r'[、,,、;;。]', func_str)
for kw in keywords:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words and kw != h:
func_to_herbs[kw].add(h)
# 构建功效到方剂的映射
func_to_formulas = defaultdict(set)
for h, data in herbs.items():
if h not in herb_formulas:
continue
gua = data.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
func_str = gua['功能']
keywords = re.split(r'[、,,、;;。]', func_str)
for kw in keywords:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
# 如果关键词是药材名且紧跟药材名,跳过
if kw == h and func_str.startswith(kw):
continue
# 如果关键词本身就是药材名(如"甘草"也是功效词中), 检查是否真的是独立功效词
if kw in herbs and kw != h and not func_str.startswith(kw):
continue
for fname in herb_formulas[h]:
func_to_formulas[kw].add(fname)
# 过滤掉太稀疏的功效(如<5首方)
func_to_formulas = {k: v for k, v in func_to_formulas.items() if len(v) >= 5}
# 按方剂数排序
top_funcs = [(k, len(v)) for k, v in func_to_formulas.items()]
top_funcs.sort(key=lambda x: -x[1])
print(f"\n 功效关键词总数: {len(func_to_formulas)}")
print(f"\n 高频功效 (≥10首方):")
print(f" {'功效关键词':<20s} {'方剂数':>6s} {'关联药材数':>8s} 示例方剂")
print(f" {'-'*20} {'-'*6} {'-'*8} {'-'*40}")
for func, count in top_funcs[:30]:
herb_count = len(func_to_herbs.get(func, set()))
examples = list(func_to_formulas[func])[:2]
print(f" {func:<20s} {count:>6d} {herb_count:>8d} {', '.join(examples)}")
with open(os.path.join(OUT, "A_功效-方剂映射网络.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
for func, count in top_funcs[:50]:
d[func] = {
"formula_count": count,
"related_herbs": len(func_to_herbs.get(func, set())),
"formulas": list(func_to_formulas[func])[:30],
"related_herb_list": list(func_to_herbs.get(func, set()))[:20]
}
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ A_功效-方剂映射网络.json")
# ========== 分析2: 方剂来源(出处)分析 ==========
print(f"\n{'='*80}")
print(f"[3/6] 方剂来源分析. ..")
print(f"{'='*80}")
# 提取出处信息
source_counter = Counter()
source_formulas = defaultdict(list)
source_herb_total = defaultdict(int) # 该出处方剂的平均药材数
for fname, data in formulas.items():
source = data.get('出处', '').strip()
if not source:
source = "未知"
else:
# 简化出处名称
source = source.split('《')[0].strip() if '《' in source else source
source = source[:30] # 截断
source_counter[source] += 1
source_formulas[source].append(fname)
source_herb_total[source] += len(formula_herbs.get(fname, []))
top_sources = source_counter.most_common(30)
print(f"\n 方剂出处统计 (Top 30):")
for src, count in top_sources:
avg_herbs = round(source_herb_total[src] / max(1, count), 1)
sample = source_formulas[src][:2]
print(f" {src:<30s} {count:>4d}首 (均药数{avg_herbs}) 例: {', '.join(sample)}")
with open(os.path.join(OUT, "B_方剂来源统计.json"), 'w', encoding='utf-8') as f:
json.dump({
"total_unique_sources": len(source_counter),
"top_sources": [{"source": s, "count": c, "avg_herbs": round(source_herb_total[s]/max(1,c), 1), "formulas": source_formulas[s][:10]} for s, c in top_sources]
}, f, ensure_ascii=False, indent=2)
print("\n ✓ B_方剂来源统计.json")
# ========== 分析3: 药材分类关联 ==========
print(f"\n{'='*80}")
print(f"[4/6] 药材分类关联分析. ..")
print(f"{'='*80}")
# 统计各药材分类的出现频率
class_counter = Counter()
class_herbs_list = defaultdict(set)
class_formulas = defaultdict(set)
for h, data in herbs.items():
cat = data.get('药材分类', '未知')
class_counter[cat] += 1
class_herbs_list[cat].add(h)
if h in herb_formulas:
class_formulas[cat].update(herb_formulas[h])
top_classes = class_counter.most_common(20)
print(f"\n 药材分类统计:")
for cls, count in top_classes:
total_formulas_class = len(class_formulas.get(cls, set()))
print(f" {cls:<10s} {count:>4d}味 关联方剂{total_formulas_class}首")
# 按分类统计高频药材
class_top_herbs = {}
for cls in [c for c, _ in top_classes]:
herb_freq_cls = Counter()
for h in class_herbs_list[cls]:
herb_freq_cls[h] = len(herb_formulas.get(h, set()))
class_top_herbs[cls] = herb_freq_cls.most_common(10)
print(f"\n 各类别高频药材 (Top 5):")
for cls in ['植物', '动物', '矿物', '其他']:
if cls in class_top_herbs:
print(f"\n [{cls}类]:")
for h, c in class_top_herbs[cls][:5]:
gua = herbs.get(h, {}).get('功效作用', {})
func = gua.get('功能', '')[:20] if isinstance(gua, dict) else ''
print(f" {h:12s} {c:>4d}首 [{func}]")
with open(os.path.join(OUT, "C_药材分类关联.json"), 'w', encoding='utf-8') as f:
d = OrderedDict()
d["分类统计"] = [{"class": c, "herb_count": cnt, "formula_count": len(class_formulas.get(c, set()))} for c, cnt in top_classes]
d["各类别高频药材"] = class_top_herbs
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ C_药材分类关联.json")
# ========== 分析4: 性味归经交叉分析 ==========
print(f"\n{'='*80}")
print(f"[5/6] 性味归经交叉分析. ..")
print(f"{'='*80}")
# 统计性状的分布
xing_counter = Counter()
xing_formulas = defaultdict(set)
for h, data in herbs.items():
xing = data.get('性味', {}).get('性', '') if isinstance(data.get('性味'), dict) else data.get('性味', '')
if xing:
xing_counter[xing] += 1
if h in herb_formulas:
xing_formulas[xing].update(herb_formulas[h])
top_xing = xing_counter.most_common(20)
print(f"\n 药材性味分布 (Top 20):")
for xing, count in top_xing:
formula_count = len(xing_formulas.get(xing, set()))
print(f" {xing:<10s} {count:>4d}味 关联方剂{formula_count}首")
# 统计归经分布
jing_counter = Counter()
jing_formulas = defaultdict(set)
for h, data in herbs.items():
jing = data.get('性味', {}).get('归经', '') if isinstance(data.get('性味'), dict) else ''
if jing:
# 拆分归经(可能逗号分隔)
jings = [j.strip() for j in re.split(r'[、,,;]', jing) if j.strip()]
for j in jings:
jing_counter[j] += 1
if h in herb_formulas:
jing_formulas[j].update(herb_formulas[h])
top_jing = jing_counter.most_common(20)
print(f"\n 药材归经分布 (Top 20):")
for jing, count in top_jing:
formula_count = len(jing_formulas.get(jing, set()))
print(f" {jing:<10s} {count:>4d}味 关联方剂{formula_count}首")
with open(os.path.join(OUT, "D_性味归经交叉分析.json"), 'w', encoding='utf-8') as f:
json.dump({
"性味分布": [{"xing": x, "count": c, "formula_count": len(xing_formulas.get(x, set()))} for x, c in top_xing],
"归经分布": [{"jing": j, "count": c, "formula_count": len(jing_formulas.get(j, set()))} for j, c in top_jing]
}, f, ensure_ascii=False, indent=2)
print("\n ✓ D_性味归经交叉分析.json")
# ========== 分析5: 药对-功效关联 ==========
print(f"\n{'='*80}")
print(f"[6/6] 药对-功效关联. ..")
print(f"{'='*80}")
herb_freq = Counter()
for hlist in formula_herbs.values():
for h in hlist:
herb_freq[h] += 1
pair_freq = Counter()
for hlist in formula_herbs.values():
u = sorted(list(set(hlist)))
if len(u) >= 2:
for combo in combinations(u, 2):
pair_freq[combo] += 1
# 对高频药对分析其对应功效
top_pairs = [(p, c) for p, c in pair_freq.most_common(100) if c >= 5]
print(f"\n 高频药对的功效关联 (Top 15):")
print(f" {'药对':<30s} {'频次':>5s} {'高频功效':<40s}")
print(f" {'-'*30} {'-'*5} {'-'*40}")
for (h1, h2), count in top_pairs[:15]:
# 分析该药对出现在哪些方剂,统计这些方剂中高频药材的功效
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
func_on_pairs = Counter()
for pname in pair_formulas:
pdata = formulas.get(pname, {})
gua = pdata.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kws = re.split(r'[、,,、;;。]', gua['功能'])
for kw in kws:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
func_on_pairs[kw] += 1
# 取该药对最关联的3个功效
top_3 = func_on_pairs.most_common(3)
top_3_str = ", ".join([f"{k}({v})" for k, v in top_3[:3]])
print(f" {h1:12s} + {h2:12s} {count:>5d} [{top_3_str}]")
with open(os.path.join(OUT, "E_药对-功效关联.json"), 'w', encoding='utf-8') as f:
d = []
for (h1, h2), count in top_pairs[:30]:
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
func_on_pairs = Counter()
for pname in pair_formulas:
pdata = formulas.get(pname, {})
gua = pdata.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kws = re.split(r'[、,,、;;。]', gua['功能'])
for kw in kws:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
func_on_pairs[kw] += 1
top_3 = func_on_pairs.most_common(3)
d.append({
"pair": list((h1, h2)),
"count": count,
"top_functions": [{"function": k, "count": v} for k, v in top_3]
})
json.dump(d, f, ensure_ascii=False, indent=2)
print("\n ✓ E_药对-功效关联.json")
# ========== 汇总 ==========
print(f"\n{'='*80}")
print(f" 深度挖掘 - 进阶版 分析完成!")
print(f"{'='*80}")
summary = OrderedDict()
summary["标题"] = "大医网 方剂-中药材 深度挖掘 (进阶版)"
summary["数据规模"] = {
"药材": len(herbs),
"方剂": len(formulas),
"总关联对数": total_links,
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
}
summary["功效-方剂映射"] = {
"功效关键词数": len(func_to_formulas),
"高频功效": [{"function": k, "formula_count": c} for k, c in top_funcs[:20]]
}
summary["方剂来源"] = {
"独特来源数": len(source_counter),
"Top5来源": [{"source": s, "count": c} for s, c in top_sources[:5]]
}
summary["药材分类"] = {
"分类数": len(class_counter),
"Top5分类": [{"class": c, "herb_count": cnt, "formula_count": len(class_formulas.get(c, set()))} for c, cnt in top_classes[:5]]
}
summary["性味分布"] = [{"xing": x, "count": c} for x, c in top_xing[:10]]
summary["归经分布"] = [{"jing": j, "count": c} for j, c in top_jing[:10]]
# 药对-功效关联 (单独处理)
pair_func_list = []
for idx, ((h1, h2), count) in enumerate(top_pairs[:10]):
if idx >= 10:
break
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
func_on_pairs = Counter()
for pname in pair_formulas:
pdata = formulas.get(pname, {})
gua = pdata.get('功效作用', {})
if isinstance(gua, dict) and gua.get('功能'):
kws = re.split(r'[、,,、;;。]', gua['功能'])
for kw in kws:
kw = kw.strip()
if kw and len(kw) >= 2 and kw not in noise_words:
func_on_pairs[kw] += 1
top_3_str = [f"{k}({v})" for k, v in func_on_pairs.most_common(3)]
pair_func_list.append({"pair": [h1, h2], "count": count, "top_funcs": top_3_str})
summary["药对-功效关联Top10"] = pair_func_list
summary["输出目录"] = os.path.abspath(OUT)
with open(os.path.join(OUT, "汇总报告_进阶.json"), 'w', encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print(f"\n 输出目录: {os.path.abspath(OUT)}")
print(f"\n 文件列表:")
for fn in sorted(os.listdir(OUT)):
fp = os.path.join(OUT, fn)
size = os.path.getsize(fp)
print(f" {fn:30s} {size:>10,} B")
print(f"{'='*80}")