412 lines
15 KiB
Python
412 lines
15 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
大医网深度挖掘 - 进阶版
|
||
1. 功效-方剂完整映射网络
|
||
2. 方剂来源分析(出处)
|
||
3. 药材分类关联
|
||
4. 性味归经交叉分析
|
||
5. 药对-功效关联
|
||
"""
|
||
import json
|
||
import os
|
||
import re
|
||
import glob
|
||
from collections import Counter, defaultdict, OrderedDict
|
||
from itertools import combinations
|
||
|
||
base = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
|
||
OUT = os.path.join(base, "03_关联融合/深度挖掘迭代/分析_进阶")
|
||
os.makedirs(OUT, exist_ok=True)
|
||
|
||
print("=" * 80)
|
||
print(" 大医网深度数据挖掘 - 进阶版 (多维关联分析)")
|
||
print("=" * 80)
|
||
|
||
# ========== 加载数据 ==========
|
||
print("\n[1/6] 加载数据...")
|
||
|
||
herbs = {} # name -> data
|
||
for f in glob.glob(os.path.join(base, "01_来源数据/中药材", "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
d = json.load(fh)
|
||
name = d.get('名称', '').strip()
|
||
if name:
|
||
herbs[name] = d
|
||
|
||
formulas = OrderedDict()
|
||
for f in glob.glob(os.path.join(base, "01_来源数据/方剂", "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
d = json.load(fh)
|
||
name = d.get('名称', os.path.basename(f).rsplit('_', 1)[0])
|
||
formulas[name] = d
|
||
|
||
print(f" 药材: {len(herbs)} 味")
|
||
print(f" 方剂: {len(formulas)} 首")
|
||
|
||
# 药材匹配字典 (仅主名, ≥2字)
|
||
valid_herb_names = [n for n in herbs.keys() if len(n) >= 2]
|
||
valid_herb_names.sort(key=len, reverse=True)
|
||
herb_pattern = re.compile('|'.join(re.escape(n) for n in valid_herb_names))
|
||
|
||
# 提取药材
|
||
def extract_herbs(text):
|
||
if not text or len(text) < 3:
|
||
return []
|
||
clean = re.sub(r'[((][^())]*[))]', '', text)
|
||
clean = re.sub(r'\d+[两钱毫升克]', '', clean)
|
||
found = []
|
||
for match in herb_pattern.finditer(clean):
|
||
herb = match.group(0)
|
||
if herb and herb not in found:
|
||
found.append(herb)
|
||
return found
|
||
|
||
formula_herbs = OrderedDict()
|
||
for fname, data in formulas.items():
|
||
all_text = ''
|
||
for val in data.values():
|
||
if isinstance(val, str):
|
||
all_text += val
|
||
elif isinstance(val, (dict, list)):
|
||
all_text += json.dumps(val, ensure_ascii=False)
|
||
h = extract_herbs(all_text)
|
||
formula_herbs[fname] = h
|
||
|
||
# 逆向索引
|
||
herb_formulas = defaultdict(set)
|
||
for fname, hlist in formula_herbs.items():
|
||
for h in hlist:
|
||
herb_formulas[h].add(fname)
|
||
|
||
total_links = sum(len(h) for h in formula_herbs.values())
|
||
print(f"\n 总关联对数: {total_links}")
|
||
print(f" 成功提取方剂: {sum(1 for h in formula_herbs.values() if h)} / {len(formulas)}")
|
||
|
||
# ========== 分析1: 功效-方剂完整映射 ==========
|
||
print(f"\n{'='*80}")
|
||
print(f"[2/6] 功效-方剂映射网络. ..")
|
||
print(f"{'='*80}")
|
||
|
||
noise_words = {'本品', '诸药', '的', '具有', '的功效', '功效', '。', '的功能', '功效作用'}
|
||
|
||
# 构建功效到药材的映射
|
||
func_to_herbs = defaultdict(set)
|
||
for h, data in herbs.items():
|
||
gua = data.get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
func_str = gua['功能']
|
||
keywords = re.split(r'[、,,、;;。]', func_str)
|
||
for kw in keywords:
|
||
kw = kw.strip()
|
||
if kw and len(kw) >= 2 and kw not in noise_words and kw != h:
|
||
func_to_herbs[kw].add(h)
|
||
|
||
# 构建功效到方剂的映射
|
||
func_to_formulas = defaultdict(set)
|
||
for h, data in herbs.items():
|
||
if h not in herb_formulas:
|
||
continue
|
||
gua = data.get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
func_str = gua['功能']
|
||
keywords = re.split(r'[、,,、;;。]', func_str)
|
||
for kw in keywords:
|
||
kw = kw.strip()
|
||
if kw and len(kw) >= 2 and kw not in noise_words:
|
||
# 如果关键词是药材名且紧跟药材名,跳过
|
||
if kw == h and func_str.startswith(kw):
|
||
continue
|
||
# 如果关键词本身就是药材名(如"甘草"也是功效词中), 检查是否真的是独立功效词
|
||
if kw in herbs and kw != h and not func_str.startswith(kw):
|
||
continue
|
||
for fname in herb_formulas[h]:
|
||
func_to_formulas[kw].add(fname)
|
||
|
||
# 过滤掉太稀疏的功效(如<5首方)
|
||
func_to_formulas = {k: v for k, v in func_to_formulas.items() if len(v) >= 5}
|
||
|
||
# 按方剂数排序
|
||
top_funcs = [(k, len(v)) for k, v in func_to_formulas.items()]
|
||
top_funcs.sort(key=lambda x: -x[1])
|
||
|
||
print(f"\n 功效关键词总数: {len(func_to_formulas)}")
|
||
print(f"\n 高频功效 (≥10首方):")
|
||
print(f" {'功效关键词':<20s} {'方剂数':>6s} {'关联药材数':>8s} 示例方剂")
|
||
print(f" {'-'*20} {'-'*6} {'-'*8} {'-'*40}")
|
||
for func, count in top_funcs[:30]:
|
||
herb_count = len(func_to_herbs.get(func, set()))
|
||
examples = list(func_to_formulas[func])[:2]
|
||
print(f" {func:<20s} {count:>6d} {herb_count:>8d} {', '.join(examples)}")
|
||
|
||
with open(os.path.join(OUT, "A_功效-方剂映射网络.json"), 'w', encoding='utf-8') as f:
|
||
d = OrderedDict()
|
||
for func, count in top_funcs[:50]:
|
||
d[func] = {
|
||
"formula_count": count,
|
||
"related_herbs": len(func_to_herbs.get(func, set())),
|
||
"formulas": list(func_to_formulas[func])[:30],
|
||
"related_herb_list": list(func_to_herbs.get(func, set()))[:20]
|
||
}
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ A_功效-方剂映射网络.json")
|
||
|
||
# ========== 分析2: 方剂来源(出处)分析 ==========
|
||
print(f"\n{'='*80}")
|
||
print(f"[3/6] 方剂来源分析. ..")
|
||
print(f"{'='*80}")
|
||
|
||
# 提取出处信息
|
||
source_counter = Counter()
|
||
source_formulas = defaultdict(list)
|
||
source_herb_total = defaultdict(int) # 该出处方剂的平均药材数
|
||
|
||
for fname, data in formulas.items():
|
||
source = data.get('出处', '').strip()
|
||
if not source:
|
||
source = "未知"
|
||
else:
|
||
# 简化出处名称
|
||
source = source.split('《')[0].strip() if '《' in source else source
|
||
source = source[:30] # 截断
|
||
|
||
source_counter[source] += 1
|
||
source_formulas[source].append(fname)
|
||
source_herb_total[source] += len(formula_herbs.get(fname, []))
|
||
|
||
top_sources = source_counter.most_common(30)
|
||
print(f"\n 方剂出处统计 (Top 30):")
|
||
for src, count in top_sources:
|
||
avg_herbs = round(source_herb_total[src] / max(1, count), 1)
|
||
sample = source_formulas[src][:2]
|
||
print(f" {src:<30s} {count:>4d}首 (均药数{avg_herbs}) 例: {', '.join(sample)}")
|
||
|
||
with open(os.path.join(OUT, "B_方剂来源统计.json"), 'w', encoding='utf-8') as f:
|
||
json.dump({
|
||
"total_unique_sources": len(source_counter),
|
||
"top_sources": [{"source": s, "count": c, "avg_herbs": round(source_herb_total[s]/max(1,c), 1), "formulas": source_formulas[s][:10]} for s, c in top_sources]
|
||
}, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ B_方剂来源统计.json")
|
||
|
||
# ========== 分析3: 药材分类关联 ==========
|
||
print(f"\n{'='*80}")
|
||
print(f"[4/6] 药材分类关联分析. ..")
|
||
print(f"{'='*80}")
|
||
|
||
# 统计各药材分类的出现频率
|
||
class_counter = Counter()
|
||
class_herbs_list = defaultdict(set)
|
||
class_formulas = defaultdict(set)
|
||
|
||
for h, data in herbs.items():
|
||
cat = data.get('药材分类', '未知')
|
||
class_counter[cat] += 1
|
||
class_herbs_list[cat].add(h)
|
||
if h in herb_formulas:
|
||
class_formulas[cat].update(herb_formulas[h])
|
||
|
||
top_classes = class_counter.most_common(20)
|
||
print(f"\n 药材分类统计:")
|
||
for cls, count in top_classes:
|
||
total_formulas_class = len(class_formulas.get(cls, set()))
|
||
print(f" {cls:<10s} {count:>4d}味 关联方剂{total_formulas_class}首")
|
||
|
||
# 按分类统计高频药材
|
||
class_top_herbs = {}
|
||
for cls in [c for c, _ in top_classes]:
|
||
herb_freq_cls = Counter()
|
||
for h in class_herbs_list[cls]:
|
||
herb_freq_cls[h] = len(herb_formulas.get(h, set()))
|
||
class_top_herbs[cls] = herb_freq_cls.most_common(10)
|
||
|
||
print(f"\n 各类别高频药材 (Top 5):")
|
||
for cls in ['植物', '动物', '矿物', '其他']:
|
||
if cls in class_top_herbs:
|
||
print(f"\n [{cls}类]:")
|
||
for h, c in class_top_herbs[cls][:5]:
|
||
gua = herbs.get(h, {}).get('功效作用', {})
|
||
func = gua.get('功能', '')[:20] if isinstance(gua, dict) else ''
|
||
print(f" {h:12s} {c:>4d}首 [{func}]")
|
||
|
||
with open(os.path.join(OUT, "C_药材分类关联.json"), 'w', encoding='utf-8') as f:
|
||
d = OrderedDict()
|
||
d["分类统计"] = [{"class": c, "herb_count": cnt, "formula_count": len(class_formulas.get(c, set()))} for c, cnt in top_classes]
|
||
d["各类别高频药材"] = class_top_herbs
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ C_药材分类关联.json")
|
||
|
||
# ========== 分析4: 性味归经交叉分析 ==========
|
||
print(f"\n{'='*80}")
|
||
print(f"[5/6] 性味归经交叉分析. ..")
|
||
print(f"{'='*80}")
|
||
|
||
# 统计性状的分布
|
||
xing_counter = Counter()
|
||
xing_formulas = defaultdict(set)
|
||
|
||
for h, data in herbs.items():
|
||
xing = data.get('性味', {}).get('性', '') if isinstance(data.get('性味'), dict) else data.get('性味', '')
|
||
if xing:
|
||
xing_counter[xing] += 1
|
||
if h in herb_formulas:
|
||
xing_formulas[xing].update(herb_formulas[h])
|
||
|
||
top_xing = xing_counter.most_common(20)
|
||
print(f"\n 药材性味分布 (Top 20):")
|
||
for xing, count in top_xing:
|
||
formula_count = len(xing_formulas.get(xing, set()))
|
||
print(f" {xing:<10s} {count:>4d}味 关联方剂{formula_count}首")
|
||
|
||
# 统计归经分布
|
||
jing_counter = Counter()
|
||
jing_formulas = defaultdict(set)
|
||
|
||
for h, data in herbs.items():
|
||
jing = data.get('性味', {}).get('归经', '') if isinstance(data.get('性味'), dict) else ''
|
||
if jing:
|
||
# 拆分归经(可能逗号分隔)
|
||
jings = [j.strip() for j in re.split(r'[、,,;]', jing) if j.strip()]
|
||
for j in jings:
|
||
jing_counter[j] += 1
|
||
if h in herb_formulas:
|
||
jing_formulas[j].update(herb_formulas[h])
|
||
|
||
top_jing = jing_counter.most_common(20)
|
||
print(f"\n 药材归经分布 (Top 20):")
|
||
for jing, count in top_jing:
|
||
formula_count = len(jing_formulas.get(jing, set()))
|
||
print(f" {jing:<10s} {count:>4d}味 关联方剂{formula_count}首")
|
||
|
||
with open(os.path.join(OUT, "D_性味归经交叉分析.json"), 'w', encoding='utf-8') as f:
|
||
json.dump({
|
||
"性味分布": [{"xing": x, "count": c, "formula_count": len(xing_formulas.get(x, set()))} for x, c in top_xing],
|
||
"归经分布": [{"jing": j, "count": c, "formula_count": len(jing_formulas.get(j, set()))} for j, c in top_jing]
|
||
}, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ D_性味归经交叉分析.json")
|
||
|
||
# ========== 分析5: 药对-功效关联 ==========
|
||
print(f"\n{'='*80}")
|
||
print(f"[6/6] 药对-功效关联. ..")
|
||
print(f"{'='*80}")
|
||
|
||
herb_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
for h in hlist:
|
||
herb_freq[h] += 1
|
||
|
||
pair_freq = Counter()
|
||
for hlist in formula_herbs.values():
|
||
u = sorted(list(set(hlist)))
|
||
if len(u) >= 2:
|
||
for combo in combinations(u, 2):
|
||
pair_freq[combo] += 1
|
||
|
||
# 对高频药对分析其对应功效
|
||
top_pairs = [(p, c) for p, c in pair_freq.most_common(100) if c >= 5]
|
||
print(f"\n 高频药对的功效关联 (Top 15):")
|
||
print(f" {'药对':<30s} {'频次':>5s} {'高频功效':<40s}")
|
||
print(f" {'-'*30} {'-'*5} {'-'*40}")
|
||
|
||
for (h1, h2), count in top_pairs[:15]:
|
||
# 分析该药对出现在哪些方剂,统计这些方剂中高频药材的功效
|
||
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
|
||
func_on_pairs = Counter()
|
||
|
||
for pname in pair_formulas:
|
||
pdata = formulas.get(pname, {})
|
||
gua = pdata.get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
kws = re.split(r'[、,,、;;。]', gua['功能'])
|
||
for kw in kws:
|
||
kw = kw.strip()
|
||
if kw and len(kw) >= 2 and kw not in noise_words:
|
||
func_on_pairs[kw] += 1
|
||
|
||
# 取该药对最关联的3个功效
|
||
top_3 = func_on_pairs.most_common(3)
|
||
top_3_str = ", ".join([f"{k}({v})" for k, v in top_3[:3]])
|
||
print(f" {h1:12s} + {h2:12s} {count:>5d} [{top_3_str}]")
|
||
|
||
with open(os.path.join(OUT, "E_药对-功效关联.json"), 'w', encoding='utf-8') as f:
|
||
d = []
|
||
for (h1, h2), count in top_pairs[:30]:
|
||
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
|
||
func_on_pairs = Counter()
|
||
for pname in pair_formulas:
|
||
pdata = formulas.get(pname, {})
|
||
gua = pdata.get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
kws = re.split(r'[、,,、;;。]', gua['功能'])
|
||
for kw in kws:
|
||
kw = kw.strip()
|
||
if kw and len(kw) >= 2 and kw not in noise_words:
|
||
func_on_pairs[kw] += 1
|
||
top_3 = func_on_pairs.most_common(3)
|
||
d.append({
|
||
"pair": list((h1, h2)),
|
||
"count": count,
|
||
"top_functions": [{"function": k, "count": v} for k, v in top_3]
|
||
})
|
||
json.dump(d, f, ensure_ascii=False, indent=2)
|
||
print("\n ✓ E_药对-功效关联.json")
|
||
|
||
# ========== 汇总 ==========
|
||
print(f"\n{'='*80}")
|
||
print(f" 深度挖掘 - 进阶版 分析完成!")
|
||
print(f"{'='*80}")
|
||
|
||
summary = OrderedDict()
|
||
summary["标题"] = "大医网 方剂-中药材 深度挖掘 (进阶版)"
|
||
summary["数据规模"] = {
|
||
"药材": len(herbs),
|
||
"方剂": len(formulas),
|
||
"总关联对数": total_links,
|
||
"成功提取方剂": sum(1 for h in formula_herbs.values() if h),
|
||
}
|
||
summary["功效-方剂映射"] = {
|
||
"功效关键词数": len(func_to_formulas),
|
||
"高频功效": [{"function": k, "formula_count": c} for k, c in top_funcs[:20]]
|
||
}
|
||
summary["方剂来源"] = {
|
||
"独特来源数": len(source_counter),
|
||
"Top5来源": [{"source": s, "count": c} for s, c in top_sources[:5]]
|
||
}
|
||
summary["药材分类"] = {
|
||
"分类数": len(class_counter),
|
||
"Top5分类": [{"class": c, "herb_count": cnt, "formula_count": len(class_formulas.get(c, set()))} for c, cnt in top_classes[:5]]
|
||
}
|
||
summary["性味分布"] = [{"xing": x, "count": c} for x, c in top_xing[:10]]
|
||
summary["归经分布"] = [{"jing": j, "count": c} for j, c in top_jing[:10]]
|
||
# 药对-功效关联 (单独处理)
|
||
pair_func_list = []
|
||
for idx, ((h1, h2), count) in enumerate(top_pairs[:10]):
|
||
if idx >= 10:
|
||
break
|
||
pair_formulas = herb_formulas.get(h1, set()) & herb_formulas.get(h2, set())
|
||
func_on_pairs = Counter()
|
||
for pname in pair_formulas:
|
||
pdata = formulas.get(pname, {})
|
||
gua = pdata.get('功效作用', {})
|
||
if isinstance(gua, dict) and gua.get('功能'):
|
||
kws = re.split(r'[、,,、;;。]', gua['功能'])
|
||
for kw in kws:
|
||
kw = kw.strip()
|
||
if kw and len(kw) >= 2 and kw not in noise_words:
|
||
func_on_pairs[kw] += 1
|
||
top_3_str = [f"{k}({v})" for k, v in func_on_pairs.most_common(3)]
|
||
pair_func_list.append({"pair": [h1, h2], "count": count, "top_funcs": top_3_str})
|
||
|
||
summary["药对-功效关联Top10"] = pair_func_list
|
||
summary["输出目录"] = os.path.abspath(OUT)
|
||
|
||
with open(os.path.join(OUT, "汇总报告_进阶.json"), 'w', encoding='utf-8') as f:
|
||
json.dump(summary, f, ensure_ascii=False, indent=2)
|
||
|
||
print(f"\n 输出目录: {os.path.abspath(OUT)}")
|
||
print(f"\n 文件列表:")
|
||
for fn in sorted(os.listdir(OUT)):
|
||
fp = os.path.join(OUT, fn)
|
||
size = os.path.getsize(fp)
|
||
print(f" {fn:30s} {size:>10,} B")
|
||
print(f"{'='*80}")
|