Files
health/大医网/05_脚本工具/药膳食疗归档分析.py
T

199 lines
7.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""药膳食疗数据考古:盘点、分类、索引、交叉关联"""
import json, os, glob, re
from collections import Counter, defaultdict
TOP_DIR = "/home/songyi/Documents/ai_agent_scraper_study/data/大医网/"
DATA_DIR = os.path.join(TOP_DIR, "药膳食疗/")
# ====== Step 1: 加载全部数据 ======
all_items = []
for f in glob.glob(os.path.join(DATA_DIR, "*.json")):
with open(f, encoding='utf-8') as fh:
all_items.append(json.load(fh))
total = len(all_items)
print(f"=== 总条目数: {total} ===\n")
# ====== Step 2: 字段完整率 ======
required_fields = ['url', '名称', '简介', '功效', '配方', '来源', '适宜人群', '相关配伍']
optional_fields = ['做法', '食用方法', '元_名称', '认证者']
has_field = Counter()
for field in required_fields + optional_fields:
for item in all_items:
if item.get(field) and item[field].strip():
has_field[field] += 1
print("=== 字段完整率 ===")
for field in required_fields + optional_fields:
count = has_field[field]
pct = count / total * 100
print(f" {field}: {count} ({pct:.1f}%)")
# ====== Step 3: 功效关键词 TOP 20 ======
gongxiao_keywords = [
'补血', '补气', '滋阴', '温阳', '清热', '解毒', '祛风', '除湿',
'活血', '化瘀', '润肺', '止咳', '化痰', '安神', '止痛', '利水',
'消食', '理气', '止血', '补脾', '益肾', '疏肝', '健脾', '通便', '泻火',
'滋补肝肾', '健脾益气', '养胃', '养阴', '固本', '养心', '养肺',
'解表', '散寒', '润燥', '开窍', '固表', '益精', '温中', '润肠',
'补肾', '壮阳', '明目', '消肿', '降火', '养肤', '美容', '益智'
]
gongxiao_index = defaultdict(set)
gongxiao_counter = Counter()
for item in all_items:
gx = item.get('功效', '') + item.get('简介', '')
for kw in gongxiao_keywords:
if kw in gx:
gongxiao_index[kw].add(item['名称'])
gongxiao_counter[kw] += 1
print(f"\n=== 功效关键词 TOP 20 ===")
for kw, cnt in gongxiao_counter.most_common(20):
print(f" {kw}: {cnt}条 ({cnt/total*100:.1f}%)")
# ====== Step 4: 药膳分类统计 ======
cat_pattern = re.compile(r'为([^,,、.]+?)类药膳配方')
categories = Counter()
for item in all_items:
m = cat_pattern.search(item.get('简介', ''))
if m:
categories[m.group(1)] += 1
unclassified = total - sum(categories.values())
print(f"\n=== 药膳分类统计 ===")
for cat, cnt in categories.most_common():
print(f" {cat}: {cnt}")
print(f" 未标注类别: {unclassified}")
# ====== Step 5: 来源典籍统计 ======
source_counter = Counter()
for item in all_items:
src = item.get('来源', '')
if src:
source_counter[src] += 1
print(f"\n=== 来源典籍 TOP 15 ===")
for src, cnt in source_counter.most_common(15):
print(f" {src}: {cnt}")
print(f" 总来源种类: {len(source_counter)}")
# ====== Step 6: 剂型统计 ======
ji_xing = Counter()
for item in all_items:
name = item['名称']
if name.endswith('粥'): ji_xing['粥'] += 1
elif name.endswith('汤'): ji_xing['汤'] += 1
elif name.endswith('羹'): ji_xing['羹'] += 1
elif name.endswith('膏'): ji_xing['膏'] += 1
elif name.endswith('酒'): ji_xing['酒'] += 1
elif name.endswith('茶'): ji_xing['茶'] += 1
elif name.endswith('饼'): ji_xing['饼'] += 1
elif name.endswith('糕'): ji_xing['糕'] += 1
elif name.endswith('丸'): ji_xing['丸'] += 1
elif name.endswith('饮'): ji_xing['饮'] += 1
elif name.endswith('饭'): ji_xing['饭'] += 1
elif name.endswith('汁'): ji_xing['汁'] += 1
elif name.endswith('包') or name.endswith('包子'): ji_xing['包子'] += 1
else: ji_xing['其他'] += 1
print(f"\n=== 药膳剂型统计 ===")
for p, cnt in ji_xing.most_common():
print(f" {p}: {cnt}")
# ====== Step 7: 高频药材匹配 ======
known_herbs = set()
herbs_dir = os.path.join(TOP_DIR, "中药材/")
if os.path.exists(herbs_dir):
for hf in glob.glob(os.path.join(herbs_dir, "*.json")):
with open(hf, encoding='utf-8') as fh:
h = json.load(fh)
known_herbs.add(h.get('名称', ''))
for alias in str(h.get('别名', '')).split('、'):
if alias.strip():
known_herbs.add(alias.strip())
known_herbs.discard('')
sorted_herbs = sorted(known_herbs, key=lambda x: -len(x))
herb_usage = Counter()
for item in all_items:
pei_fang = item.get('配方', '')
for hname in sorted_herbs:
if len(hname) >= 2 and hname in pei_fang:
herb_usage[hname] += 1
print(f"\n=== 高频中药材 TOP 20 ===")
for herb, cnt in herb_usage.most_common(20):
print(f" {herb}: {cnt}条")
print(f"\n 中药材库已知药材数: {len(known_herbs)}")
print(f" 在药膳中出现的药材数: {len(herb_usage)}")
# ====== Step 8: 交叉关联 — 药膳 vs 方剂 ======
formula_dir = os.path.join(TOP_DIR, "方剂/")
formulas = []
if os.path.exists(formula_dir):
for ff in glob.glob(os.path.join(formula_dir, "*.json")):
with open(ff, encoding='utf-8') as fh:
formulas.append(json.load(fh))
# 提取药膳功效关键词,匹配方剂
diet_gx_words = defaultdict(set)
for item in all_items:
gx_text = item.get('功效', '') + item.get('简介', '')
# 提取 2-4 字功效短语
for kw in gongxiao_keywords:
if kw in gx_text:
for f_item in formulas:
f_text = f_item.get('简介', '') + f_item.get('运用', '')
if kw in f_text:
diet_gx_words[kw].add((item['名称'], f_item.get('名称', '')))
cross_diet_formula = Counter()
for kw, pairs in diet_gx_words.items():
if pairs:
cross_diet_formula[kw] = len(pairs)
print(f"\n=== 药膳-方剂跨库功效关联 TOP 15 ===")
for kw, cnt in cross_diet_formula.most_common(15):
print(f" {kw}: {cnt}对关联")
# ====== Step 9: 保存索引文件 ======
# 功效索引
gx_save = {kw: sorted(names) for kw, names in gongxiao_index.items()}
with open(os.path.join(TOP_DIR, "索引_药膳功效关键词.json"), "w", encoding='utf-8') as f:
json.dump(gx_save, f, ensure_ascii=False, indent=2)
# 分类索引
cat_index = defaultdict(list)
for item in all_items:
m = cat_pattern.search(item.get('简介', ''))
cat = m.group(1) if m else '未标注'
cat_index[cat].append(item['名称'])
with open(os.path.join(TOP_DIR, "索引_药膳分类.json"), "w", encoding='utf-8') as f:
json.dump({k: sorted(v) for k, v in cat_index.items()}, f, ensure_ascii=False, indent=2)
# 高频药材
with open(os.path.join(TOP_DIR, "索引_药膳高频药材.json"), "w", encoding='utf-8') as f:
json.dump(herb_usage.most_common(50), f, ensure_ascii=False, indent=2)
# 数据摘要
summary = {
"总条目数": total,
"字段完整率": {field: round(has_field[field] / total * 100, 1) for field in required_fields + optional_fields},
"药膳分类": dict(categories.most_common()),
"来源典籍数": len(source_counter),
"TOP15来源": source_counter.most_common(15),
"高频药材数": len(herb_usage),
"剂型分布": dict(ji_xing.most_common())
}
with open(os.path.join(TOP_DIR, "数据摘要_药膳食疗.json"), "w", encoding='utf-8') as f:
json.dump(summary, f, ensure_ascii=False, indent=2)
print("\n=== 索引文件已保存 ===")
print("\n归档数据保存完毕!")