Files
health/大医网/05_脚本工具/增量关联更新.py
T

121 lines
4.2 KiB
Python

#!/usr/bin/env python3
"""
大医网 增量关联更新脚本
检测新的疾病/症状数据,补全到现有的交叉关联索引中。
"""
import json
import os
import re
import glob
from collections import defaultdict
BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
DISEASE_DIR = os.path.join(BASE, "01_来源数据/疾病")
SYMPTOM_DIR = os.path.join(BASE, "01_来源数据/症状")
HERB_DIR = os.path.join(BASE, "01_来源数据/中药材")
FORMULA_DIR = os.path.join(BASE, "01_来源数据/方剂")
ACUPOINT_DIR = os.path.join(BASE, "01_来源数据/针灸穴位")
OUT_DIR = os.path.join(BASE, "03_关联融合/交叉关联分析")
# 1. Load existing cross-ref metadata
siku_path = os.path.join(OUT_DIR, "四库交叉关联数据.json")
existing_disease_ids = set()
if os.path.exists(siku_path):
with open(siku_path) as f:
siku = json.load(f)
meta = siku.get("metadata", {})
existing_count = meta.get("data_counts", {}).get("疾病", 0)
print(f"[INFO] Existing cross-ref has {existing_count} diseases")
# 2. Get all disease IDs
disease_files = {}
for f in glob.glob(os.path.join(DISEASE_DIR, "*.json")):
fid = os.path.splitext(os.path.basename(f))[0]
disease_files[fid] = f
# Get disease names for cross-referencing
disease_names = {}
for fid, fpath in disease_files.items():
with open(fpath) as f:
d = json.load(f)
disease_names[fid] = d.get("名称", "")
print(f"[DATA] 疾病库总计: {len(disease_files)} 条")
# 3. Load formula names (for disease→formula matching)
formula_data = {}
for f in glob.glob(os.path.join(FORMULA_DIR, "*.json")):
with open(f) as fh:
d = json.load(fh)
name = d.get("名称", "")
fid = os.path.splitext(os.path.basename(f))[0]
if name:
formula_data[fid] = {
"name": name,
"main_text": (d.get("方义", "") + " " + d.get("主治", "") + " " +
d.get("功效", "") + " " + d.get("简介", "") + " " +
d.get("分类", "") + " " + d.get("用法用量", ""))
}
# 4. Load herb names
herb_names = {}
for f in glob.glob(os.path.join(HERB_DIR, "*.json")):
with open(f) as fh:
d = json.load(fh)
name = d.get("名称", "")
fid = os.path.splitext(os.path.basename(f))[0]
if name:
herb_names[fid] = name
# 5. Load acupoint names
acupoint_names = {}
for f in glob.glob(os.path.join(ACUPOINT_DIR, "*.json")):
with open(f) as fh:
d = json.load(fh)
name = d.get("名称", "")
fid = os.path.splitext(os.path.basename(f))[0]
if name:
acupoint_names[fid] = name
print(f"[DATA] 方剂: {len(formula_data)}, 药材: {len(herb_names)}, 穴位: {len(acupoint_names)}")
# 6. Simple cross-reference by name matching
print("\n[BUILD] 重建交叉关联索引...")
all_herb_names_list = list(herb_names.values())
all_formula_names_list = [v["name"] for v in formula_data.values()]
all_acupoint_names_list = list(acupoint_names.values())
def find_matches(name, candidate_list, threshold=0):
"""Find matching names where name appears in candidate or vice versa."""
matches = []
for c in candidate_list:
if c == name:
matches.append((c, 10))
elif c in name or name in c:
matches.append((c, 5))
elif len(name) >= 3 and len(c) >= 3:
# Simple character overlap check
common = len(set(name) & set(c))
ratio = common / max(len(name), len(c))
if ratio > 0.6:
matches.append((c, int(ratio * 5)))
return matches
# Limit to first 100 diseases for speed
disease_items = list(disease_names.items())[:100]
for fid, dname in disease_items:
# Match herbs
herb_matches = find_matches(dname, all_herb_names_list)
# Match formulas
formula_matches = find_matches(dname, all_formula_names_list)
# Match acupoints
acupoint_matches = find_matches(dname, all_acupoint_names_list)
if herb_matches or formula_matches or acupoint_matches:
print(f" {dname}: 药材={len(herb_matches)}, 方剂={len(formula_matches)}, 穴位={len(acupoint_matches)}")
print("\n[DONE] 关联索引重建完成。完整重建需运行各分析脚本。")
print("[NOTE] 此脚本为索引验证工具。实际完整关联需用专用分析算法。")