Files

512 lines
18 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
《国民体质测定标准(2023年修订)》评分表解析脚本
提取所有HTML内联表格为结构化JSON,按年龄段/性别/指标组织
"""
import re
import json
from bs4 import BeautifulSoup
def parse_value(val):
"""解析单元格值,返回 (min, max, operator) 或原始字符串"""
val = val.strip()
# 清理 HTML 实体
val = val.replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
# 处理含BMI/体脂率等文本的值域表达式: "18.5≤BMI<24.0", "BMI≥28.0", "BMI<18.5"
# 去除 BMI 等干扰文本
clean_val = re.sub(r'[A-Za-z\u4e00-\u9fff]+', '', val).strip()
# 处理 LaTeX 公式
if '$' in val:
# 去掉$和反斜杠,统一符号
clean = val.replace('$', ' ').replace('\\', ' ').replace('\\\\', ' ')
clean = clean.replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
# 统一LaTeX命令为符号(长命令在前,避免被短命令截断)
clean = clean.replace('geq', '≥').replace('leq', '≤').replace('ge', '≥').replace('le', '≤')
clean = clean.replace('gt', '>').replace('lt', '<')
# "X分 ≤ a < Y分"
m = re.search(r'(\d+)\s*分?\s*[≤<]\s*a\s*[<]\s*(\d+)', clean)
if m:
return {"min": float(m.group(1)), "max": float(m.group(2))}
# a ≥ X
m = re.search(r'a\s*[≥>]\s*(\d+)', clean)
if m:
return {"min": float(m.group(1))}
# a < X
m = re.search(r'a\s*[<]\s*(\d+)', clean)
if m:
return {"max": float(m.group(1))}
# 尝试用clean_val匹配标准模式(去除文本干扰后)
for test_val in [clean_val, val]:
# X ≤ Y < Z 或 X ≤ Y ≤ Z 或 X < Y < Z
m = re.search(r'(-?[\d.]+)\s*[≤<]\s*[\d.]*\s*[≤<]\s*(-?[\d.]+)', test_val)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
# ≥X
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', test_val)
if m:
return {"min": float(m.group(1))}
# ≤X
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', test_val)
if m:
return {"max": float(m.group(1))}
# X-Y range — 处理正负值(如 -14.9--12.5 或 3.5-7.2)
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', test_val)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
# standalone number
m = re.match(r'^([\d.]+)$', test_val)
if m:
return {"exact": float(m.group(1))}
# <X 或 <-X
m = re.match(r'<\s*(-?[\d.]+)', test_val)
if m:
return {"max": float(m.group(1))}
# >X 或 >-X
m = re.match(r'>\s*(-?[\d.]+)', test_val)
if m:
return {"min": float(m.group(1))}
return {"raw": val.strip()}
def extract_tables(md_content):
"""从Markdown内容中提取所有表格及其前面的表名"""
# 用BeautifulSoup解析HTML表格
# 先找到所有 <div>表名</div> + <table> 对
results = []
# 按行处理,寻找表名div和table
lines = md_content.split('\n')
i = 0
while i < len(lines):
line = lines[i]
# 找表名 <div> 行
m = re.search(r'<div[^>]*>(表\s*[\d.-]+[^<]*)</div>', line)
if m:
table_name = m.group(1).strip()
table_name = re.sub(r'\s+', ' ', table_name)
# 寻找单位行
unit = None
for j in range(i+1, min(i+10, len(lines))):
unit_m = re.match(r'单位[::]\s*(.+)', lines[j].strip())
if unit_m:
unit = unit_m.group(1).strip()
break
# 查找接下来的全部<table>块(直到下一个表名div或文档结束)
all_table_blocks = []
j = i + 1
while j < len(lines):
if '<table' in lines[j]:
# 收集完整的table HTML
table_lines = []
k = j
while k < len(lines) and '</table>' not in lines[k]:
table_lines.append(lines[k])
k += 1
if k < len(lines):
table_lines.append(lines[k]) # 包含</table>的行
all_table_blocks.append('\n'.join(table_lines))
j = k + 1
elif '<div' in lines[j] and '表' in lines[j] and j != i:
# 下一个表名,停止
break
else:
j += 1
if all_table_blocks:
combined_html = '\n'.join(all_table_blocks)
soup = BeautifulSoup(combined_html, 'html.parser')
tables = soup.find_all('table')
if tables:
all_rows = []
for table_tag in tables:
rows = table_tag.find_all('tr')
for tr in rows:
cells = []
for td in tr.find_all(['td', 'th']):
txt = td.get_text(strip=True)
txt = txt.replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
cells.append(txt)
if cells:
all_rows.append(cells)
if all_rows:
results.append({
'name': table_name,
'unit': unit,
'rows': all_rows,
'combined_tables': len(tables)
})
i = j # 跳到已处理的末尾
continue
i += 1
return results
def classify_table(table):
"""分类表格并解析为结构化JSON"""
name = table['name']
rows = table['rows']
if not rows:
return None
# 基本信息
info = {
'name': name,
'unit': table['unit'],
'type': 'unknown',
'parsed': {}
}
# 提取表号
m = re.match(r'表\s*([\d.-]+)\s*(.*)', name)
if m:
info['table_id'] = m.group(1).strip()
info['title'] = m.group(2).strip()
else:
info['table_id'] = ''
info['title'] = name
# 解析性别和指标
title = info['title']
if '男性' in title:
info['gender'] = '男'
info['gender_en'] = 'male'
elif '女性' in title:
info['gender'] = '女'
info['gender_en'] = 'female'
else:
info['gender'] = '通用'
info['gender_en'] = '通用' # 统一用中文"通用"而非"all"
# 判断年龄段 - 用更丰富的关键词
section_keywords = {
'幼儿': ['幼儿', '3岁', '3.5岁', '4岁', '5岁', '6岁', '36月', '37月', '38月', '39月', '40月'],
'成年人': ['成年', '20-24', '25-29', '30-34', '35-39', '40-44', '45-49', '50-54', '55-59'],
'老年人': ['老年', '60-64', '65-69', '70-74', '75-79']
}
age_ranges = {'幼儿': '3-6岁', '成年人': '20-59岁', '老年人': '60-79岁'}
for sec, keywords in section_keywords.items():
if any(k in name or k in title for k in keywords):
info['section'] = sec
info['age_range'] = age_ranges.get(sec, '')
break
else:
# 用表格ID判断:1.x = 幼儿, 2.x = 成年人, 3.x = 老年人
tid = info.get('table_id', '')
if tid.startswith('1'):
info['section'] = '幼儿'
info['age_range'] = '3-6岁'
elif tid.startswith('2'):
info['section'] = '成年人'
info['age_range'] = '20-59岁'
elif tid.startswith('3'):
info['section'] = '老年人'
info['age_range'] = '60-79岁'
# 解析指标名称
indicator_keywords = ['身高', '体重', 'BMI', '体脂率', '肺活量', '功率车', '握力', '纵跳',
'俯卧撑', '跪卧撑', '仰卧起坐', '坐位体前屈', '立定跳远', '双脚连续跳',
'绕障碍跑', '走平衡木', '闭眼单脚站立', '选择反应时', '高抬腿', '坐站']
for kw in indicator_keywords:
if kw in title:
info['indicator'] = kw
break
else:
info['indicator'] = title
# 判断表格类型
header = rows[0] if rows else []
is_bmi_style = ('年龄' in str(header) or '年龄段' in str(header)) and len(header) <= 5
is_weight_table = '权重' in name or ('权重' in str(rows[1] if len(rows) > 1 else ''))
is_rating_table = '评级' in name or '等级' in name
is_indicator_list = ('类别' in str(header) and '测试指标' in str(header)) or \
('一级指标' in str(header) and '二级指标' in str(header) and '权重' not in name) or \
(len(header) == 2 and any(k in str(header[1]) for k in ['测试指标', '指标', '20-49', '50-59', '60-64']))
if is_weight_table:
info['type'] = 'weight_table'
info['parsed'] = parse_weight_table(rows)
elif is_indicator_list:
info['type'] = 'indicator_list'
info['parsed'] = parse_indicator_list(rows)
elif is_rating_table:
info['type'] = 'rating_levels'
info['parsed'] = parse_rating_levels(rows)
elif is_bmi_style:
info['type'] = 'bmi_banded'
info['parsed'] = parse_bmi_banded(rows, info)
else:
info['type'] = 'standard_scoring'
info['parsed'] = parse_standard_scoring(rows)
return info
def parse_indicator_list(rows):
"""解析指标列表(如表1-1)"""
items = []
for row in rows[1:]: # 跳过表头
if len(row) >= 2:
items.append({
'category': row[0],
'indicator': row[1]
})
return {'items': items}
def parse_weight_table(rows):
"""解析权重表(如表1-2)- 需要处理 rowspan"""
weights = {}
# 记录上一行的类别值,用于 rowspan 填充
prev_category = None
for row in rows[1:]: # 跳过表头
if len(row) >= 3:
indicator = row[1].strip()
weight = row[2].strip()
category = row[0].strip() if row[0].strip() else prev_category
prev_category = category
try:
weight_val = float(weight)
weights[indicator] = weight_val
except ValueError:
pass
elif len(row) == 2 and prev_category:
# rowspan 行:只有指标和权重,类别从上一行继承
indicator = row[0].strip()
weight = row[1].strip()
try:
weight_val = float(weight)
weights[indicator] = weight_val
except ValueError:
pass
return {'weights': weights}
def parse_rating_levels(rows):
"""解析评级等级表(如表1-3)"""
levels = {}
for row in rows[1:]:
if len(row) >= 2:
level_name = row[0].strip()
score_text = row[1].strip()
parsed = parse_value(score_text)
levels[level_name] = parsed
return {'levels': levels}
def parse_bmi_banded(rows, info):
"""解析BMI分段评分表(如表1-6)"""
header = rows[0]
# 列名:年龄 | 60分 | 100分 | 60分 | 20分
# 或:年龄段 | 40分 | 100分 | 60分 | 20分
# 解析列对应的分数
score_cols = []
for h in header[1:]:
m = re.search(r'(\d+)\s*分', str(h))
score_cols.append(int(m.group(1)) if m else None)
data = {}
for row in rows[1:]:
age_key = row[0].strip()
ranges = []
for i, val in enumerate(row[1:], 1):
if i <= len(score_cols) and score_cols[i-1] is not None:
parsed_range = parse_value(val)
parsed_range['score'] = score_cols[i-1]
ranges.append(parsed_range)
data[age_key] = ranges
return {
'score_categories': [{'col': h, 'score': s} for h, s in zip(header[1:], score_cols)],
'age_groups': list(data.keys()),
'data': data
}
def parse_standard_scoring(rows):
"""解析标准评分表(行=分值,列=年龄组,如表1-4)"""
header = rows[0]
age_groups = [str(h).strip() for h in header[1:]]
score_bands = []
for row in rows[1:]:
score_text = row[0].strip()
m = re.search(r'(\d+)分?', score_text)
if not m:
continue
score = int(m.group(1))
ranges = {}
for i, age in enumerate(age_groups):
if i + 1 < len(row):
ranges[age] = parse_value(row[i + 1])
score_bands.append({
'score': score,
'ranges': ranges
})
return {
'age_groups': age_groups,
'score_bands': score_bands
}
def generate_scoring_lookup(all_tables):
"""生成按 section → gender → indicator 组织的查询结构"""
lookup = {
'幼儿': {'male': {}, 'female': {}, '通用': {}},
'成年人': {'male': {}, 'female': {}, '通用': {}},
'老年人': {'male': {}, 'female': {}, '通用': {}},
'weights': {'幼儿': {}, '成年人': {}, '老年人': {}},
'rating_levels': {'幼儿': {}, '成年人': {}, '老年人': {}},
'metadata': {'tables_count': len(all_tables)}
}
for t in all_tables:
if not t:
continue
section = t.get('section', '未知')
gender = t.get('gender_en', '通用')
tbl_type = t.get('type', 'unknown')
if section == '未知':
continue
if tbl_type == 'weight_table':
# 按年龄范围分别存储权重表(成年人有20-49和50-59两个表)
title_lower = t.get('title', '')
age_tag = re.search(r'(\d+[-~–]\d+)\s*岁', title_lower)
if age_tag:
weight_key = f"{section}_{age_tag.group(1).replace('~','-').replace('–','-')}岁"
else:
weight_key = section
lookup['weights'][weight_key] = t['parsed']['weights']
elif tbl_type == 'rating_levels':
lookup['rating_levels'][section] = t['parsed']['levels']
elif tbl_type == 'indicator_list':
key = f"indicators_{t['table_id']}"
if section not in lookup:
lookup[section] = {}
lookup[section].setdefault('通用', {})[key] = t['parsed']
else:
# 评分表
indicator = t.get('indicator', t.get('title', '未知'))
if gender not in lookup[section]:
lookup[section][gender] = {}
entry = {
'table_id': t['table_id'],
'title': t['title'],
'unit': t['unit'],
'type': tbl_type,
'data': t['parsed']
}
lookup[section][gender][indicator] = entry
return lookup
def main():
base = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
doc_path = f'{base}/01_来源数据/2023修订版/《国民体质测定标准(2023年修订)》.md'
with open(doc_path, 'r', encoding='utf-8') as f:
content = f.read()
print(f"文件大小: {len(content)} 字符")
tables = extract_tables(content)
print(f"提取到 {len(tables)} 个表格")
# 分类所有表格
classified = []
for t in tables:
result = classify_table(t)
if result:
classified.append(result)
print(f"成功分类 {len(classified)} 个表格")
# 统计
types = {}
for t in classified:
tt = t['type']
types[tt] = types.get(tt, 0) + 1
print(f"表格类型分布: {json.dumps(types, ensure_ascii=False, indent=2)}")
# 生成查询结构
lookup = generate_scoring_lookup(classified)
# 输出全部表格原始解析
all_tables_output = []
for t in classified:
all_tables_output.append({
'table_id': t.get('table_id', ''),
'title': t.get('title', ''),
'section': t.get('section', ''),
'gender': t.get('gender', ''),
'indicator': t.get('indicator', ''),
'unit': t.get('unit'),
'type': t.get('type', ''),
'parsed': t.get('parsed', {})
})
output = {
'metadata': {
'source': '《国民体质测定标准(2023年修订)》',
'publisher': '国家国民体质监测中心',
'year': 2023,
'total_tables': len(classified),
'table_types': types
},
'scoring_lookup': lookup,
'all_tables': all_tables_output,
'weights': lookup['weights'],
'rating_levels': lookup['rating_levels']
}
out_path = f'{base}/02_加工数据/2023修订版/评分标准结构化数据.json'
with open(out_path, 'w', encoding='utf-8') as f:
json.dump(output, f, ensure_ascii=False, indent=2)
print(f"\nJSON 输出: {out_path}")
print(f"JSON 大小: {len(json.dumps(output, ensure_ascii=False))} 字符")
# 打印摘要
print("\n=== 评分表摘要 ===")
seen = set()
for t in classified:
key = (t.get('section',''), t.get('gender',''), t.get('indicator',''))
if key not in seen:
seen.add(key)
print(f" [{t.get('section','')}] {t.get('gender','')} - {t.get('indicator','')} ({t.get('table_id','')}) [{t.get('type','')}]")
print(f"\n总计 {len(seen)} 个评分表/指标组合")
if __name__ == '__main__':
main()