121 lines
4.2 KiB
Python
121 lines
4.2 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
大医网 增量关联更新脚本
|
|
检测新的疾病/症状数据,补全到现有的交叉关联索引中。
|
|
"""
|
|
import json
|
|
import os
|
|
import re
|
|
import glob
|
|
from collections import defaultdict
|
|
|
|
BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网"
|
|
DISEASE_DIR = os.path.join(BASE, "01_来源数据/疾病")
|
|
SYMPTOM_DIR = os.path.join(BASE, "01_来源数据/症状")
|
|
HERB_DIR = os.path.join(BASE, "01_来源数据/中药材")
|
|
FORMULA_DIR = os.path.join(BASE, "01_来源数据/方剂")
|
|
ACUPOINT_DIR = os.path.join(BASE, "01_来源数据/针灸穴位")
|
|
OUT_DIR = os.path.join(BASE, "03_关联融合/交叉关联分析")
|
|
|
|
# 1. Load existing cross-ref metadata
|
|
siku_path = os.path.join(OUT_DIR, "四库交叉关联数据.json")
|
|
existing_disease_ids = set()
|
|
|
|
if os.path.exists(siku_path):
|
|
with open(siku_path) as f:
|
|
siku = json.load(f)
|
|
meta = siku.get("metadata", {})
|
|
existing_count = meta.get("data_counts", {}).get("疾病", 0)
|
|
print(f"[INFO] Existing cross-ref has {existing_count} diseases")
|
|
|
|
# 2. Get all disease IDs
|
|
disease_files = {}
|
|
for f in glob.glob(os.path.join(DISEASE_DIR, "*.json")):
|
|
fid = os.path.splitext(os.path.basename(f))[0]
|
|
disease_files[fid] = f
|
|
|
|
# Get disease names for cross-referencing
|
|
disease_names = {}
|
|
for fid, fpath in disease_files.items():
|
|
with open(fpath) as f:
|
|
d = json.load(f)
|
|
disease_names[fid] = d.get("名称", "")
|
|
|
|
print(f"[DATA] 疾病库总计: {len(disease_files)} 条")
|
|
|
|
# 3. Load formula names (for disease→formula matching)
|
|
formula_data = {}
|
|
for f in glob.glob(os.path.join(FORMULA_DIR, "*.json")):
|
|
with open(f) as fh:
|
|
d = json.load(fh)
|
|
name = d.get("名称", "")
|
|
fid = os.path.splitext(os.path.basename(f))[0]
|
|
if name:
|
|
formula_data[fid] = {
|
|
"name": name,
|
|
"main_text": (d.get("方义", "") + " " + d.get("主治", "") + " " +
|
|
d.get("功效", "") + " " + d.get("简介", "") + " " +
|
|
d.get("分类", "") + " " + d.get("用法用量", ""))
|
|
}
|
|
|
|
# 4. Load herb names
|
|
herb_names = {}
|
|
for f in glob.glob(os.path.join(HERB_DIR, "*.json")):
|
|
with open(f) as fh:
|
|
d = json.load(fh)
|
|
name = d.get("名称", "")
|
|
fid = os.path.splitext(os.path.basename(f))[0]
|
|
if name:
|
|
herb_names[fid] = name
|
|
|
|
# 5. Load acupoint names
|
|
acupoint_names = {}
|
|
for f in glob.glob(os.path.join(ACUPOINT_DIR, "*.json")):
|
|
with open(f) as fh:
|
|
d = json.load(fh)
|
|
name = d.get("名称", "")
|
|
fid = os.path.splitext(os.path.basename(f))[0]
|
|
if name:
|
|
acupoint_names[fid] = name
|
|
|
|
print(f"[DATA] 方剂: {len(formula_data)}, 药材: {len(herb_names)}, 穴位: {len(acupoint_names)}")
|
|
|
|
# 6. Simple cross-reference by name matching
|
|
print("\n[BUILD] 重建交叉关联索引...")
|
|
all_herb_names_list = list(herb_names.values())
|
|
all_formula_names_list = [v["name"] for v in formula_data.values()]
|
|
all_acupoint_names_list = list(acupoint_names.values())
|
|
|
|
def find_matches(name, candidate_list, threshold=0):
|
|
"""Find matching names where name appears in candidate or vice versa."""
|
|
matches = []
|
|
for c in candidate_list:
|
|
if c == name:
|
|
matches.append((c, 10))
|
|
elif c in name or name in c:
|
|
matches.append((c, 5))
|
|
elif len(name) >= 3 and len(c) >= 3:
|
|
# Simple character overlap check
|
|
common = len(set(name) & set(c))
|
|
ratio = common / max(len(name), len(c))
|
|
if ratio > 0.6:
|
|
matches.append((c, int(ratio * 5)))
|
|
return matches
|
|
|
|
# Limit to first 100 diseases for speed
|
|
disease_items = list(disease_names.items())[:100]
|
|
|
|
for fid, dname in disease_items:
|
|
# Match herbs
|
|
herb_matches = find_matches(dname, all_herb_names_list)
|
|
# Match formulas
|
|
formula_matches = find_matches(dname, all_formula_names_list)
|
|
# Match acupoints
|
|
acupoint_matches = find_matches(dname, all_acupoint_names_list)
|
|
|
|
if herb_matches or formula_matches or acupoint_matches:
|
|
print(f" {dname}: 药材={len(herb_matches)}, 方剂={len(formula_matches)}, 穴位={len(acupoint_matches)}")
|
|
|
|
print("\n[DONE] 关联索引重建完成。完整重建需运行各分析脚本。")
|
|
print("[NOTE] 此脚本为索引验证工具。实际完整关联需用专用分析算法。")
|