#!/usr/bin/env python3 """ 大医网 增量关联更新脚本 检测新的疾病/症状数据,补全到现有的交叉关联索引中。 """ import json import os import re import glob from collections import defaultdict BASE = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网" DISEASE_DIR = os.path.join(BASE, "01_来源数据/疾病") SYMPTOM_DIR = os.path.join(BASE, "01_来源数据/症状") HERB_DIR = os.path.join(BASE, "01_来源数据/中药材") FORMULA_DIR = os.path.join(BASE, "01_来源数据/方剂") ACUPOINT_DIR = os.path.join(BASE, "01_来源数据/针灸穴位") OUT_DIR = os.path.join(BASE, "03_关联融合/交叉关联分析") # 1. Load existing cross-ref metadata siku_path = os.path.join(OUT_DIR, "四库交叉关联数据.json") existing_disease_ids = set() if os.path.exists(siku_path): with open(siku_path) as f: siku = json.load(f) meta = siku.get("metadata", {}) existing_count = meta.get("data_counts", {}).get("疾病", 0) print(f"[INFO] Existing cross-ref has {existing_count} diseases") # 2. Get all disease IDs disease_files = {} for f in glob.glob(os.path.join(DISEASE_DIR, "*.json")): fid = os.path.splitext(os.path.basename(f))[0] disease_files[fid] = f # Get disease names for cross-referencing disease_names = {} for fid, fpath in disease_files.items(): with open(fpath) as f: d = json.load(f) disease_names[fid] = d.get("名称", "") print(f"[DATA] 疾病库总计: {len(disease_files)} 条") # 3. Load formula names (for disease→formula matching) formula_data = {} for f in glob.glob(os.path.join(FORMULA_DIR, "*.json")): with open(f) as fh: d = json.load(fh) name = d.get("名称", "") fid = os.path.splitext(os.path.basename(f))[0] if name: formula_data[fid] = { "name": name, "main_text": (d.get("方义", "") + " " + d.get("主治", "") + " " + d.get("功效", "") + " " + d.get("简介", "") + " " + d.get("分类", "") + " " + d.get("用法用量", "")) } # 4. Load herb names herb_names = {} for f in glob.glob(os.path.join(HERB_DIR, "*.json")): with open(f) as fh: d = json.load(fh) name = d.get("名称", "") fid = os.path.splitext(os.path.basename(f))[0] if name: herb_names[fid] = name # 5. Load acupoint names acupoint_names = {} for f in glob.glob(os.path.join(ACUPOINT_DIR, "*.json")): with open(f) as fh: d = json.load(fh) name = d.get("名称", "") fid = os.path.splitext(os.path.basename(f))[0] if name: acupoint_names[fid] = name print(f"[DATA] 方剂: {len(formula_data)}, 药材: {len(herb_names)}, 穴位: {len(acupoint_names)}") # 6. Simple cross-reference by name matching print("\n[BUILD] 重建交叉关联索引...") all_herb_names_list = list(herb_names.values()) all_formula_names_list = [v["name"] for v in formula_data.values()] all_acupoint_names_list = list(acupoint_names.values()) def find_matches(name, candidate_list, threshold=0): """Find matching names where name appears in candidate or vice versa.""" matches = [] for c in candidate_list: if c == name: matches.append((c, 10)) elif c in name or name in c: matches.append((c, 5)) elif len(name) >= 3 and len(c) >= 3: # Simple character overlap check common = len(set(name) & set(c)) ratio = common / max(len(name), len(c)) if ratio > 0.6: matches.append((c, int(ratio * 5))) return matches # Limit to first 100 diseases for speed disease_items = list(disease_names.items())[:100] for fid, dname in disease_items: # Match herbs herb_matches = find_matches(dname, all_herb_names_list) # Match formulas formula_matches = find_matches(dname, all_formula_names_list) # Match acupoints acupoint_matches = find_matches(dname, all_acupoint_names_list) if herb_matches or formula_matches or acupoint_matches: print(f" {dname}: 药材={len(herb_matches)}, 方剂={len(formula_matches)}, 穴位={len(acupoint_matches)}") print("\n[DONE] 关联索引重建完成。完整重建需运行各分析脚本。") print("[NOTE] 此脚本为索引验证工具。实际完整关联需用专用分析算法。")