初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)

This commit is contained in:
512song committed 2026-09-23 21:59:25 +08:00
commit 80ae3811cf
7712 files changed
+4628547

No files matched your search

@@ -0,0 +1,511 @@
#!/usr/bin/env python3
"""
《国民体质测定标准(2023年修订)》评分表解析脚本
提取所有HTML内联表格为结构化JSON,按年龄段/性别/指标组织
"""
import re
import json
from bs4 import BeautifulSoup
def parse_value(val):
"""解析单元格值,返回 (min, max, operator) 或原始字符串"""
val = val.strip()
# 清理 HTML 实体
val = val.replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
# 处理含BMI/体脂率等文本的值域表达式: "18.5≤BMI<24.0", "BMI≥28.0", "BMI<18.5"
# 去除 BMI 等干扰文本
clean_val = re.sub(r'[A-Za-z\u4e00-\u9fff]+', '', val).strip()
# 处理 LaTeX 公式
if '$' in val:
# 去掉$和反斜杠,统一符号
clean = val.replace('$', ' ').replace('\\', ' ').replace('\\\\', ' ')
clean = clean.replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
# 统一LaTeX命令为符号(长命令在前,避免被短命令截断)
clean = clean.replace('geq', '≥').replace('leq', '≤').replace('ge', '≥').replace('le', '≤')
clean = clean.replace('gt', '>').replace('lt', '<')
# "X分 ≤ a < Y分"
m = re.search(r'(\d+)\s*分?\s*[≤<]\s*a\s*[<]\s*(\d+)', clean)
if m:
return {"min": float(m.group(1)), "max": float(m.group(2))}
# a ≥ X
m = re.search(r'a\s*[≥>]\s*(\d+)', clean)
if m:
return {"min": float(m.group(1))}
# a < X
m = re.search(r'a\s*[<]\s*(\d+)', clean)
if m:
return {"max": float(m.group(1))}
# 尝试用clean_val匹配标准模式(去除文本干扰后)
for test_val in [clean_val, val]:
# X ≤ Y < Z 或 X ≤ Y ≤ Z 或 X < Y < Z
m = re.search(r'(-?[\d.]+)\s*[≤<]\s*[\d.]*\s*[≤<]\s*(-?[\d.]+)', test_val)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
# ≥X
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', test_val)
if m:
return {"min": float(m.group(1))}
# ≤X
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', test_val)
if m:
return {"max": float(m.group(1))}
# X-Y range — 处理正负值(如 -14.9--12.5 或 3.5-7.2)
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', test_val)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
# standalone number
m = re.match(r'^([\d.]+)$', test_val)
if m:
return {"exact": float(m.group(1))}
# <X 或 <-X
m = re.match(r'<\s*(-?[\d.]+)', test_val)
if m:
return {"max": float(m.group(1))}
# >X 或 >-X
m = re.match(r'>\s*(-?[\d.]+)', test_val)
if m:
return {"min": float(m.group(1))}
return {"raw": val.strip()}
def extract_tables(md_content):
"""从Markdown内容中提取所有表格及其前面的表名"""
# 用BeautifulSoup解析HTML表格
# 先找到所有 <div>表名</div> + <table> 对
results = []
# 按行处理,寻找表名div和table
lines = md_content.split('\n')
i = 0
while i < len(lines):
line = lines[i]
# 找表名 <div> 行
m = re.search(r'<div[^>]*>(表\s*[\d.-]+[^<]*)</div>', line)
if m:
table_name = m.group(1).strip()
table_name = re.sub(r'\s+', ' ', table_name)
# 寻找单位行
unit = None
for j in range(i+1, min(i+10, len(lines))):
unit_m = re.match(r'单位[::]\s*(.+)', lines[j].strip())
if unit_m:
unit = unit_m.group(1).strip()
break
# 查找接下来的全部<table>块(直到下一个表名div或文档结束)
all_table_blocks = []
j = i + 1
while j < len(lines):
if '<table' in lines[j]:
# 收集完整的table HTML
table_lines = []
k = j
while k < len(lines) and '</table>' not in lines[k]:
table_lines.append(lines[k])
k += 1
if k < len(lines):
table_lines.append(lines[k]) # 包含</table>的行
all_table_blocks.append('\n'.join(table_lines))
j = k + 1
elif '<div' in lines[j] and '表' in lines[j] and j != i:
# 下一个表名,停止
break
else:
j += 1
if all_table_blocks:
combined_html = '\n'.join(all_table_blocks)
soup = BeautifulSoup(combined_html, 'html.parser')
tables = soup.find_all('table')
if tables:
all_rows = []
for table_tag in tables:
rows = table_tag.find_all('tr')
for tr in rows:
cells = []
for td in tr.find_all(['td', 'th']):
txt = td.get_text(strip=True)
txt = txt.replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
cells.append(txt)
if cells:
all_rows.append(cells)
if all_rows:
results.append({
'name': table_name,
'unit': unit,
'rows': all_rows,
'combined_tables': len(tables)
})
i = j # 跳到已处理的末尾
continue
i += 1
return results
def classify_table(table):
"""分类表格并解析为结构化JSON"""
name = table['name']
rows = table['rows']
if not rows:
return None
# 基本信息
info = {
'name': name,
'unit': table['unit'],
'type': 'unknown',
'parsed': {}
}
# 提取表号
m = re.match(r'表\s*([\d.-]+)\s*(.*)', name)
if m:
info['table_id'] = m.group(1).strip()
info['title'] = m.group(2).strip()
else:
info['table_id'] = ''
info['title'] = name
# 解析性别和指标
title = info['title']
if '男性' in title:
info['gender'] = '男'
info['gender_en'] = 'male'
elif '女性' in title:
info['gender'] = '女'
info['gender_en'] = 'female'
else:
info['gender'] = '通用'
info['gender_en'] = '通用' # 统一用中文"通用"而非"all"
# 判断年龄段 - 用更丰富的关键词
section_keywords = {
'幼儿': ['幼儿', '3岁', '3.5岁', '4岁', '5岁', '6岁', '36月', '37月', '38月', '39月', '40月'],
'成年人': ['成年', '20-24', '25-29', '30-34', '35-39', '40-44', '45-49', '50-54', '55-59'],
'老年人': ['老年', '60-64', '65-69', '70-74', '75-79']
}
age_ranges = {'幼儿': '3-6岁', '成年人': '20-59岁', '老年人': '60-79岁'}
for sec, keywords in section_keywords.items():
if any(k in name or k in title for k in keywords):
info['section'] = sec
info['age_range'] = age_ranges.get(sec, '')
break
else:
# 用表格ID判断:1.x = 幼儿, 2.x = 成年人, 3.x = 老年人
tid = info.get('table_id', '')
if tid.startswith('1'):
info['section'] = '幼儿'
info['age_range'] = '3-6岁'
elif tid.startswith('2'):
info['section'] = '成年人'
info['age_range'] = '20-59岁'
elif tid.startswith('3'):
info['section'] = '老年人'
info['age_range'] = '60-79岁'
# 解析指标名称
indicator_keywords = ['身高', '体重', 'BMI', '体脂率', '肺活量', '功率车', '握力', '纵跳',
'俯卧撑', '跪卧撑', '仰卧起坐', '坐位体前屈', '立定跳远', '双脚连续跳',
'绕障碍跑', '走平衡木', '闭眼单脚站立', '选择反应时', '高抬腿', '坐站']
for kw in indicator_keywords:
if kw in title:
info['indicator'] = kw
break
else:
info['indicator'] = title
# 判断表格类型
header = rows[0] if rows else []
is_bmi_style = ('年龄' in str(header) or '年龄段' in str(header)) and len(header) <= 5
is_weight_table = '权重' in name or ('权重' in str(rows[1] if len(rows) > 1 else ''))
is_rating_table = '评级' in name or '等级' in name
is_indicator_list = ('类别' in str(header) and '测试指标' in str(header)) or \
('一级指标' in str(header) and '二级指标' in str(header) and '权重' not in name) or \
(len(header) == 2 and any(k in str(header[1]) for k in ['测试指标', '指标', '20-49', '50-59', '60-64']))
if is_weight_table:
info['type'] = 'weight_table'
info['parsed'] = parse_weight_table(rows)
elif is_indicator_list:
info['type'] = 'indicator_list'
info['parsed'] = parse_indicator_list(rows)
elif is_rating_table:
info['type'] = 'rating_levels'
info['parsed'] = parse_rating_levels(rows)
elif is_bmi_style:
info['type'] = 'bmi_banded'
info['parsed'] = parse_bmi_banded(rows, info)
else:
info['type'] = 'standard_scoring'
info['parsed'] = parse_standard_scoring(rows)
return info
def parse_indicator_list(rows):
"""解析指标列表(如表1-1)"""
items = []
for row in rows[1:]: # 跳过表头
if len(row) >= 2:
items.append({
'category': row[0],
'indicator': row[1]
})
return {'items': items}
def parse_weight_table(rows):
"""解析权重表(如表1-2)- 需要处理 rowspan"""
weights = {}
# 记录上一行的类别值,用于 rowspan 填充
prev_category = None
for row in rows[1:]: # 跳过表头
if len(row) >= 3:
indicator = row[1].strip()
weight = row[2].strip()
category = row[0].strip() if row[0].strip() else prev_category
prev_category = category
try:
weight_val = float(weight)
weights[indicator] = weight_val
except ValueError:
pass
elif len(row) == 2 and prev_category:
# rowspan 行:只有指标和权重,类别从上一行继承
indicator = row[0].strip()
weight = row[1].strip()
try:
weight_val = float(weight)
weights[indicator] = weight_val
except ValueError:
pass
return {'weights': weights}
def parse_rating_levels(rows):
"""解析评级等级表(如表1-3)"""
levels = {}
for row in rows[1:]:
if len(row) >= 2:
level_name = row[0].strip()
score_text = row[1].strip()
parsed = parse_value(score_text)
levels[level_name] = parsed
return {'levels': levels}
def parse_bmi_banded(rows, info):
"""解析BMI分段评分表(如表1-6)"""
header = rows[0]
# 列名:年龄 | 60分 | 100分 | 60分 | 20分
# 或:年龄段 | 40分 | 100分 | 60分 | 20分
# 解析列对应的分数
score_cols = []
for h in header[1:]:
m = re.search(r'(\d+)\s*分', str(h))
score_cols.append(int(m.group(1)) if m else None)
data = {}
for row in rows[1:]:
age_key = row[0].strip()
ranges = []
for i, val in enumerate(row[1:], 1):
if i <= len(score_cols) and score_cols[i-1] is not None:
parsed_range = parse_value(val)
parsed_range['score'] = score_cols[i-1]
ranges.append(parsed_range)
data[age_key] = ranges
return {
'score_categories': [{'col': h, 'score': s} for h, s in zip(header[1:], score_cols)],
'age_groups': list(data.keys()),
'data': data
}
def parse_standard_scoring(rows):
"""解析标准评分表(行=分值,列=年龄组,如表1-4)"""
header = rows[0]
age_groups = [str(h).strip() for h in header[1:]]
score_bands = []
for row in rows[1:]:
score_text = row[0].strip()
m = re.search(r'(\d+)分?', score_text)
if not m:
continue
score = int(m.group(1))
ranges = {}
for i, age in enumerate(age_groups):
if i + 1 < len(row):
ranges[age] = parse_value(row[i + 1])
score_bands.append({
'score': score,
'ranges': ranges
})
return {
'age_groups': age_groups,
'score_bands': score_bands
}
def generate_scoring_lookup(all_tables):
"""生成按 section → gender → indicator 组织的查询结构"""
lookup = {
'幼儿': {'male': {}, 'female': {}, '通用': {}},
'成年人': {'male': {}, 'female': {}, '通用': {}},
'老年人': {'male': {}, 'female': {}, '通用': {}},
'weights': {'幼儿': {}, '成年人': {}, '老年人': {}},
'rating_levels': {'幼儿': {}, '成年人': {}, '老年人': {}},
'metadata': {'tables_count': len(all_tables)}
}
for t in all_tables:
if not t:
continue
section = t.get('section', '未知')
gender = t.get('gender_en', '通用')
tbl_type = t.get('type', 'unknown')
if section == '未知':
continue
if tbl_type == 'weight_table':
# 按年龄范围分别存储权重表(成年人有20-49和50-59两个表)
title_lower = t.get('title', '')
age_tag = re.search(r'(\d+[-~–]\d+)\s*岁', title_lower)
if age_tag:
weight_key = f"{section}_{age_tag.group(1).replace('~','-').replace('–','-')}岁"
else:
weight_key = section
lookup['weights'][weight_key] = t['parsed']['weights']
elif tbl_type == 'rating_levels':
lookup['rating_levels'][section] = t['parsed']['levels']
elif tbl_type == 'indicator_list':
key = f"indicators_{t['table_id']}"
if section not in lookup:
lookup[section] = {}
lookup[section].setdefault('通用', {})[key] = t['parsed']
else:
# 评分表
indicator = t.get('indicator', t.get('title', '未知'))
if gender not in lookup[section]:
lookup[section][gender] = {}
entry = {
'table_id': t['table_id'],
'title': t['title'],
'unit': t['unit'],
'type': tbl_type,
'data': t['parsed']
}
lookup[section][gender][indicator] = entry
return lookup
def main():
base = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
doc_path = f'{base}/01_来源数据/2023修订版/《国民体质测定标准(2023年修订)》.md'
with open(doc_path, 'r', encoding='utf-8') as f:
content = f.read()
print(f"文件大小: {len(content)} 字符")
tables = extract_tables(content)
print(f"提取到 {len(tables)} 个表格")
# 分类所有表格
classified = []
for t in tables:
result = classify_table(t)
if result:
classified.append(result)
print(f"成功分类 {len(classified)} 个表格")
# 统计
types = {}
for t in classified:
tt = t['type']
types[tt] = types.get(tt, 0) + 1
print(f"表格类型分布: {json.dumps(types, ensure_ascii=False, indent=2)}")
# 生成查询结构
lookup = generate_scoring_lookup(classified)
# 输出全部表格原始解析
all_tables_output = []
for t in classified:
all_tables_output.append({
'table_id': t.get('table_id', ''),
'title': t.get('title', ''),
'section': t.get('section', ''),
'gender': t.get('gender', ''),
'indicator': t.get('indicator', ''),
'unit': t.get('unit'),
'type': t.get('type', ''),
'parsed': t.get('parsed', {})
})
output = {
'metadata': {
'source': '《国民体质测定标准(2023年修订)》',
'publisher': '国家国民体质监测中心',
'year': 2023,
'total_tables': len(classified),
'table_types': types
},
'scoring_lookup': lookup,
'all_tables': all_tables_output,
'weights': lookup['weights'],
'rating_levels': lookup['rating_levels']
}
out_path = f'{base}/02_加工数据/2023修订版/评分标准结构化数据.json'
with open(out_path, 'w', encoding='utf-8') as f:
json.dump(output, f, ensure_ascii=False, indent=2)
print(f"\nJSON 输出: {out_path}")
print(f"JSON 大小: {len(json.dumps(output, ensure_ascii=False))} 字符")
# 打印摘要
print("\n=== 评分表摘要 ===")
seen = set()
for t in classified:
key = (t.get('section',''), t.get('gender',''), t.get('indicator',''))
if key not in seen:
seen.add(key)
print(f" [{t.get('section','')}] {t.get('gender','')} - {t.get('indicator','')} ({t.get('table_id','')}) [{t.get('type','')}]")
print(f"\n总计 {len(seen)} 个评分表/指标组合")
if __name__ == '__main__':
main()