199 lines
7.3 KiB
Python
199 lines
7.3 KiB
Python
#!/usr/bin/env python3
|
||
"""药膳食疗数据考古:盘点、分类、索引、交叉关联"""
|
||
import json, os, glob, re
|
||
from collections import Counter, defaultdict
|
||
|
||
TOP_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/"
|
||
DATA_DIR = os.path.join(TOP_DIR, "药膳食疗/")
|
||
|
||
# ====== Step 1: 加载全部数据 ======
|
||
all_items = []
|
||
for f in glob.glob(os.path.join(DATA_DIR, "*.json")):
|
||
with open(f, encoding='utf-8') as fh:
|
||
all_items.append(json.load(fh))
|
||
|
||
total = len(all_items)
|
||
print(f"=== 总条目数: {total} ===\n")
|
||
|
||
# ====== Step 2: 字段完整率 ======
|
||
required_fields = ['url', '名称', '简介', '功效', '配方', '来源', '适宜人群', '相关配伍']
|
||
optional_fields = ['做法', '食用方法', '元_名称', '认证者']
|
||
|
||
has_field = Counter()
|
||
for field in required_fields + optional_fields:
|
||
for item in all_items:
|
||
if item.get(field) and item[field].strip():
|
||
has_field[field] += 1
|
||
|
||
print("=== 字段完整率 ===")
|
||
for field in required_fields + optional_fields:
|
||
count = has_field[field]
|
||
pct = count / total * 100
|
||
print(f" {field}: {count} ({pct:.1f}%)")
|
||
|
||
# ====== Step 3: 功效关键词 TOP 20 ======
|
||
gongxiao_keywords = [
|
||
'补血', '补气', '滋阴', '温阳', '清热', '解毒', '祛风', '除湿',
|
||
'活血', '化瘀', '润肺', '止咳', '化痰', '安神', '止痛', '利水',
|
||
'消食', '理气', '止血', '补脾', '益肾', '疏肝', '健脾', '通便', '泻火',
|
||
'滋补肝肾', '健脾益气', '养胃', '养阴', '固本', '养心', '养肺',
|
||
'解表', '散寒', '润燥', '开窍', '固表', '益精', '温中', '润肠',
|
||
'补肾', '壮阳', '明目', '消肿', '降火', '养肤', '美容', '益智'
|
||
]
|
||
|
||
gongxiao_index = defaultdict(set)
|
||
gongxiao_counter = Counter()
|
||
for item in all_items:
|
||
gx = item.get('功效', '') + item.get('简介', '')
|
||
for kw in gongxiao_keywords:
|
||
if kw in gx:
|
||
gongxiao_index[kw].add(item['名称'])
|
||
gongxiao_counter[kw] += 1
|
||
|
||
print(f"\n=== 功效关键词 TOP 20 ===")
|
||
for kw, cnt in gongxiao_counter.most_common(20):
|
||
print(f" {kw}: {cnt}条 ({cnt/total*100:.1f}%)")
|
||
|
||
# ====== Step 4: 药膳分类统计 ======
|
||
cat_pattern = re.compile(r'为([^,,、.]+?)类药膳配方')
|
||
categories = Counter()
|
||
for item in all_items:
|
||
m = cat_pattern.search(item.get('简介', ''))
|
||
if m:
|
||
categories[m.group(1)] += 1
|
||
|
||
unclassified = total - sum(categories.values())
|
||
|
||
print(f"\n=== 药膳分类统计 ===")
|
||
for cat, cnt in categories.most_common():
|
||
print(f" {cat}: {cnt}")
|
||
print(f" 未标注类别: {unclassified}")
|
||
|
||
# ====== Step 5: 来源典籍统计 ======
|
||
source_counter = Counter()
|
||
for item in all_items:
|
||
src = item.get('来源', '')
|
||
if src:
|
||
source_counter[src] += 1
|
||
|
||
print(f"\n=== 来源典籍 TOP 15 ===")
|
||
for src, cnt in source_counter.most_common(15):
|
||
print(f" {src}: {cnt}")
|
||
print(f" 总来源种类: {len(source_counter)}")
|
||
|
||
# ====== Step 6: 剂型统计 ======
|
||
ji_xing = Counter()
|
||
for item in all_items:
|
||
name = item['名称']
|
||
if name.endswith('粥'): ji_xing['粥'] += 1
|
||
elif name.endswith('汤'): ji_xing['汤'] += 1
|
||
elif name.endswith('羹'): ji_xing['羹'] += 1
|
||
elif name.endswith('膏'): ji_xing['膏'] += 1
|
||
elif name.endswith('酒'): ji_xing['酒'] += 1
|
||
elif name.endswith('茶'): ji_xing['茶'] += 1
|
||
elif name.endswith('饼'): ji_xing['饼'] += 1
|
||
elif name.endswith('糕'): ji_xing['糕'] += 1
|
||
elif name.endswith('丸'): ji_xing['丸'] += 1
|
||
elif name.endswith('饮'): ji_xing['饮'] += 1
|
||
elif name.endswith('饭'): ji_xing['饭'] += 1
|
||
elif name.endswith('汁'): ji_xing['汁'] += 1
|
||
elif name.endswith('包') or name.endswith('包子'): ji_xing['包子'] += 1
|
||
else: ji_xing['其他'] += 1
|
||
|
||
print(f"\n=== 药膳剂型统计 ===")
|
||
for p, cnt in ji_xing.most_common():
|
||
print(f" {p}: {cnt}")
|
||
|
||
# ====== Step 7: 高频药材匹配 ======
|
||
known_herbs = set()
|
||
herbs_dir = os.path.join(TOP_DIR, "中药材/")
|
||
if os.path.exists(herbs_dir):
|
||
for hf in glob.glob(os.path.join(herbs_dir, "*.json")):
|
||
with open(hf, encoding='utf-8') as fh:
|
||
h = json.load(fh)
|
||
known_herbs.add(h.get('名称', ''))
|
||
for alias in str(h.get('别名', '')).split('、'):
|
||
if alias.strip():
|
||
known_herbs.add(alias.strip())
|
||
|
||
known_herbs.discard('')
|
||
sorted_herbs = sorted(known_herbs, key=lambda x: -len(x))
|
||
|
||
herb_usage = Counter()
|
||
for item in all_items:
|
||
pei_fang = item.get('配方', '')
|
||
for hname in sorted_herbs:
|
||
if len(hname) >= 2 and hname in pei_fang:
|
||
herb_usage[hname] += 1
|
||
|
||
print(f"\n=== 高频中药材 TOP 20 ===")
|
||
for herb, cnt in herb_usage.most_common(20):
|
||
print(f" {herb}: {cnt}条")
|
||
|
||
print(f"\n 中药材库已知药材数: {len(known_herbs)}")
|
||
print(f" 在药膳中出现的药材数: {len(herb_usage)}")
|
||
|
||
# ====== Step 8: 交叉关联 — 药膳 vs 方剂 ======
|
||
formula_dir = os.path.join(TOP_DIR, "方剂/")
|
||
formulas = []
|
||
if os.path.exists(formula_dir):
|
||
for ff in glob.glob(os.path.join(formula_dir, "*.json")):
|
||
with open(ff, encoding='utf-8') as fh:
|
||
formulas.append(json.load(fh))
|
||
|
||
# 提取药膳功效关键词,匹配方剂
|
||
diet_gx_words = defaultdict(set)
|
||
for item in all_items:
|
||
gx_text = item.get('功效', '') + item.get('简介', '')
|
||
# 提取 2-4 字功效短语
|
||
for kw in gongxiao_keywords:
|
||
if kw in gx_text:
|
||
for f_item in formulas:
|
||
f_text = f_item.get('简介', '') + f_item.get('运用', '')
|
||
if kw in f_text:
|
||
diet_gx_words[kw].add((item['名称'], f_item.get('名称', '')))
|
||
|
||
cross_diet_formula = Counter()
|
||
for kw, pairs in diet_gx_words.items():
|
||
if pairs:
|
||
cross_diet_formula[kw] = len(pairs)
|
||
|
||
print(f"\n=== 药膳-方剂跨库功效关联 TOP 15 ===")
|
||
for kw, cnt in cross_diet_formula.most_common(15):
|
||
print(f" {kw}: {cnt}对关联")
|
||
|
||
# ====== Step 9: 保存索引文件 ======
|
||
# 功效索引
|
||
gx_save = {kw: sorted(names) for kw, names in gongxiao_index.items()}
|
||
with open(os.path.join(TOP_DIR, "索引_药膳功效关键词.json"), "w", encoding='utf-8') as f:
|
||
json.dump(gx_save, f, ensure_ascii=False, indent=2)
|
||
|
||
# 分类索引
|
||
cat_index = defaultdict(list)
|
||
for item in all_items:
|
||
m = cat_pattern.search(item.get('简介', ''))
|
||
cat = m.group(1) if m else '未标注'
|
||
cat_index[cat].append(item['名称'])
|
||
with open(os.path.join(TOP_DIR, "索引_药膳分类.json"), "w", encoding='utf-8') as f:
|
||
json.dump({k: sorted(v) for k, v in cat_index.items()}, f, ensure_ascii=False, indent=2)
|
||
|
||
# 高频药材
|
||
with open(os.path.join(TOP_DIR, "索引_药膳高频药材.json"), "w", encoding='utf-8') as f:
|
||
json.dump(herb_usage.most_common(50), f, ensure_ascii=False, indent=2)
|
||
|
||
# 数据摘要
|
||
summary = {
|
||
"总条目数": total,
|
||
"字段完整率": {field: round(has_field[field] / total * 100, 1) for field in required_fields + optional_fields},
|
||
"药膳分类": dict(categories.most_common()),
|
||
"来源典籍数": len(source_counter),
|
||
"TOP15来源": source_counter.most_common(15),
|
||
"高频药材数": len(herb_usage),
|
||
"剂型分布": dict(ji_xing.most_common())
|
||
}
|
||
with open(os.path.join(TOP_DIR, "数据摘要_药膳食疗.json"), "w", encoding='utf-8') as f:
|
||
json.dump(summary, f, ensure_ascii=False, indent=2)
|
||
|
||
print("\n=== 索引文件已保存 ===")
|
||
print("\n归档数据保存完毕!")
|