#!/usr/bin/env python3 """ 《国民体质测定标准(2023年修订)》评分表解析脚本 提取所有HTML内联表格为结构化JSON,按年龄段/性别/指标组织 """ import re import json from bs4 import BeautifulSoup def parse_value(val): """解析单元格值,返回 (min, max, operator) 或原始字符串""" val = val.strip() # 清理 HTML 实体 val = val.replace('<', '<').replace('>', '>').replace('&', '&') # 处理含BMI/体脂率等文本的值域表达式: "18.5≤BMI<24.0", "BMI≥28.0", "BMI<18.5" # 去除 BMI 等干扰文本 clean_val = re.sub(r'[A-Za-z\u4e00-\u9fff]+', '', val).strip() # 处理 LaTeX 公式 if '$' in val: # 去掉$和反斜杠,统一符号 clean = val.replace('$', ' ').replace('\\', ' ').replace('\\\\', ' ') clean = clean.replace('<', '<').replace('>', '>').replace('&', '&') # 统一LaTeX命令为符号(长命令在前,避免被短命令截断) clean = clean.replace('geq', '≥').replace('leq', '≤').replace('ge', '≥').replace('le', '≤') clean = clean.replace('gt', '>').replace('lt', '<') # "X分 ≤ a < Y分" m = re.search(r'(\d+)\s*分?\s*[≤<]\s*a\s*[<]\s*(\d+)', clean) if m: return {"min": float(m.group(1)), "max": float(m.group(2))} # a ≥ X m = re.search(r'a\s*[≥>]\s*(\d+)', clean) if m: return {"min": float(m.group(1))} # a < X m = re.search(r'a\s*[<]\s*(\d+)', clean) if m: return {"max": float(m.group(1))} # 尝试用clean_val匹配标准模式(去除文本干扰后) for test_val in [clean_val, val]: # X ≤ Y < Z 或 X ≤ Y ≤ Z 或 X < Y < Z m = re.search(r'(-?[\d.]+)\s*[≤<]\s*[\d.]*\s*[≤<]\s*(-?[\d.]+)', test_val) if m: lo, hi = float(m.group(1)), float(m.group(2)) return {"min": min(lo, hi), "max": max(lo, hi)} # ≥X m = re.match(r'[≥>]=?\s*(-?[\d.]+)', test_val) if m: return {"min": float(m.group(1))} # ≤X m = re.match(r'[≤<]=?\s*(-?[\d.]+)', test_val) if m: return {"max": float(m.group(1))} # X-Y range — 处理正负值(如 -14.9--12.5 或 3.5-7.2) m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', test_val) if m: lo, hi = float(m.group(1)), float(m.group(2)) return {"min": min(lo, hi), "max": max(lo, hi)} # standalone number m = re.match(r'^([\d.]+)$', test_val) if m: return {"exact": float(m.group(1))} # X 或 >-X m = re.match(r'>\s*(-?[\d.]+)', test_val) if m: return {"min": float(m.group(1))} return {"raw": val.strip()} def extract_tables(md_content): """从Markdown内容中提取所有表格及其前面的表名""" # 用BeautifulSoup解析HTML表格 # 先找到所有
表名
+ 对 results = [] # 按行处理,寻找表名div和table lines = md_content.split('\n') i = 0 while i < len(lines): line = lines[i] # 找表名
行 m = re.search(r']*>(表\s*[\d.-]+[^<]*)
', line) if m: table_name = m.group(1).strip() table_name = re.sub(r'\s+', ' ', table_name) # 寻找单位行 unit = None for j in range(i+1, min(i+10, len(lines))): unit_m = re.match(r'单位[::]\s*(.+)', lines[j].strip()) if unit_m: unit = unit_m.group(1).strip() break # 查找接下来的全部
块(直到下一个表名div或文档结束) all_table_blocks = [] j = i + 1 while j < len(lines): if '' not in lines[k]: table_lines.append(lines[k]) k += 1 if k < len(lines): table_lines.append(lines[k]) # 包含
的行 all_table_blocks.append('\n'.join(table_lines)) j = k + 1 elif '').replace('&', '&') cells.append(txt) if cells: all_rows.append(cells) if all_rows: results.append({ 'name': table_name, 'unit': unit, 'rows': all_rows, 'combined_tables': len(tables) }) i = j # 跳到已处理的末尾 continue i += 1 return results def classify_table(table): """分类表格并解析为结构化JSON""" name = table['name'] rows = table['rows'] if not rows: return None # 基本信息 info = { 'name': name, 'unit': table['unit'], 'type': 'unknown', 'parsed': {} } # 提取表号 m = re.match(r'表\s*([\d.-]+)\s*(.*)', name) if m: info['table_id'] = m.group(1).strip() info['title'] = m.group(2).strip() else: info['table_id'] = '' info['title'] = name # 解析性别和指标 title = info['title'] if '男性' in title: info['gender'] = '男' info['gender_en'] = 'male' elif '女性' in title: info['gender'] = '女' info['gender_en'] = 'female' else: info['gender'] = '通用' info['gender_en'] = '通用' # 统一用中文"通用"而非"all" # 判断年龄段 - 用更丰富的关键词 section_keywords = { '幼儿': ['幼儿', '3岁', '3.5岁', '4岁', '5岁', '6岁', '36月', '37月', '38月', '39月', '40月'], '成年人': ['成年', '20-24', '25-29', '30-34', '35-39', '40-44', '45-49', '50-54', '55-59'], '老年人': ['老年', '60-64', '65-69', '70-74', '75-79'] } age_ranges = {'幼儿': '3-6岁', '成年人': '20-59岁', '老年人': '60-79岁'} for sec, keywords in section_keywords.items(): if any(k in name or k in title for k in keywords): info['section'] = sec info['age_range'] = age_ranges.get(sec, '') break else: # 用表格ID判断:1.x = 幼儿, 2.x = 成年人, 3.x = 老年人 tid = info.get('table_id', '') if tid.startswith('1'): info['section'] = '幼儿' info['age_range'] = '3-6岁' elif tid.startswith('2'): info['section'] = '成年人' info['age_range'] = '20-59岁' elif tid.startswith('3'): info['section'] = '老年人' info['age_range'] = '60-79岁' # 解析指标名称 indicator_keywords = ['身高', '体重', 'BMI', '体脂率', '肺活量', '功率车', '握力', '纵跳', '俯卧撑', '跪卧撑', '仰卧起坐', '坐位体前屈', '立定跳远', '双脚连续跳', '绕障碍跑', '走平衡木', '闭眼单脚站立', '选择反应时', '高抬腿', '坐站'] for kw in indicator_keywords: if kw in title: info['indicator'] = kw break else: info['indicator'] = title # 判断表格类型 header = rows[0] if rows else [] is_bmi_style = ('年龄' in str(header) or '年龄段' in str(header)) and len(header) <= 5 is_weight_table = '权重' in name or ('权重' in str(rows[1] if len(rows) > 1 else '')) is_rating_table = '评级' in name or '等级' in name is_indicator_list = ('类别' in str(header) and '测试指标' in str(header)) or \ ('一级指标' in str(header) and '二级指标' in str(header) and '权重' not in name) or \ (len(header) == 2 and any(k in str(header[1]) for k in ['测试指标', '指标', '20-49', '50-59', '60-64'])) if is_weight_table: info['type'] = 'weight_table' info['parsed'] = parse_weight_table(rows) elif is_indicator_list: info['type'] = 'indicator_list' info['parsed'] = parse_indicator_list(rows) elif is_rating_table: info['type'] = 'rating_levels' info['parsed'] = parse_rating_levels(rows) elif is_bmi_style: info['type'] = 'bmi_banded' info['parsed'] = parse_bmi_banded(rows, info) else: info['type'] = 'standard_scoring' info['parsed'] = parse_standard_scoring(rows) return info def parse_indicator_list(rows): """解析指标列表(如表1-1)""" items = [] for row in rows[1:]: # 跳过表头 if len(row) >= 2: items.append({ 'category': row[0], 'indicator': row[1] }) return {'items': items} def parse_weight_table(rows): """解析权重表(如表1-2)- 需要处理 rowspan""" weights = {} # 记录上一行的类别值,用于 rowspan 填充 prev_category = None for row in rows[1:]: # 跳过表头 if len(row) >= 3: indicator = row[1].strip() weight = row[2].strip() category = row[0].strip() if row[0].strip() else prev_category prev_category = category try: weight_val = float(weight) weights[indicator] = weight_val except ValueError: pass elif len(row) == 2 and prev_category: # rowspan 行:只有指标和权重,类别从上一行继承 indicator = row[0].strip() weight = row[1].strip() try: weight_val = float(weight) weights[indicator] = weight_val except ValueError: pass return {'weights': weights} def parse_rating_levels(rows): """解析评级等级表(如表1-3)""" levels = {} for row in rows[1:]: if len(row) >= 2: level_name = row[0].strip() score_text = row[1].strip() parsed = parse_value(score_text) levels[level_name] = parsed return {'levels': levels} def parse_bmi_banded(rows, info): """解析BMI分段评分表(如表1-6)""" header = rows[0] # 列名:年龄 | 60分 | 100分 | 60分 | 20分 # 或:年龄段 | 40分 | 100分 | 60分 | 20分 # 解析列对应的分数 score_cols = [] for h in header[1:]: m = re.search(r'(\d+)\s*分', str(h)) score_cols.append(int(m.group(1)) if m else None) data = {} for row in rows[1:]: age_key = row[0].strip() ranges = [] for i, val in enumerate(row[1:], 1): if i <= len(score_cols) and score_cols[i-1] is not None: parsed_range = parse_value(val) parsed_range['score'] = score_cols[i-1] ranges.append(parsed_range) data[age_key] = ranges return { 'score_categories': [{'col': h, 'score': s} for h, s in zip(header[1:], score_cols)], 'age_groups': list(data.keys()), 'data': data } def parse_standard_scoring(rows): """解析标准评分表(行=分值,列=年龄组,如表1-4)""" header = rows[0] age_groups = [str(h).strip() for h in header[1:]] score_bands = [] for row in rows[1:]: score_text = row[0].strip() m = re.search(r'(\d+)分?', score_text) if not m: continue score = int(m.group(1)) ranges = {} for i, age in enumerate(age_groups): if i + 1 < len(row): ranges[age] = parse_value(row[i + 1]) score_bands.append({ 'score': score, 'ranges': ranges }) return { 'age_groups': age_groups, 'score_bands': score_bands } def generate_scoring_lookup(all_tables): """生成按 section → gender → indicator 组织的查询结构""" lookup = { '幼儿': {'male': {}, 'female': {}, '通用': {}}, '成年人': {'male': {}, 'female': {}, '通用': {}}, '老年人': {'male': {}, 'female': {}, '通用': {}}, 'weights': {'幼儿': {}, '成年人': {}, '老年人': {}}, 'rating_levels': {'幼儿': {}, '成年人': {}, '老年人': {}}, 'metadata': {'tables_count': len(all_tables)} } for t in all_tables: if not t: continue section = t.get('section', '未知') gender = t.get('gender_en', '通用') tbl_type = t.get('type', 'unknown') if section == '未知': continue if tbl_type == 'weight_table': # 按年龄范围分别存储权重表(成年人有20-49和50-59两个表) title_lower = t.get('title', '') age_tag = re.search(r'(\d+[-~–]\d+)\s*岁', title_lower) if age_tag: weight_key = f"{section}_{age_tag.group(1).replace('~','-').replace('–','-')}岁" else: weight_key = section lookup['weights'][weight_key] = t['parsed']['weights'] elif tbl_type == 'rating_levels': lookup['rating_levels'][section] = t['parsed']['levels'] elif tbl_type == 'indicator_list': key = f"indicators_{t['table_id']}" if section not in lookup: lookup[section] = {} lookup[section].setdefault('通用', {})[key] = t['parsed'] else: # 评分表 indicator = t.get('indicator', t.get('title', '未知')) if gender not in lookup[section]: lookup[section][gender] = {} entry = { 'table_id': t['table_id'], 'title': t['title'], 'unit': t['unit'], 'type': tbl_type, 'data': t['parsed'] } lookup[section][gender][indicator] = entry return lookup def main(): base = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测' doc_path = f'{base}/01_来源数据/2023修订版/《国民体质测定标准(2023年修订)》.md' with open(doc_path, 'r', encoding='utf-8') as f: content = f.read() print(f"文件大小: {len(content)} 字符") tables = extract_tables(content) print(f"提取到 {len(tables)} 个表格") # 分类所有表格 classified = [] for t in tables: result = classify_table(t) if result: classified.append(result) print(f"成功分类 {len(classified)} 个表格") # 统计 types = {} for t in classified: tt = t['type'] types[tt] = types.get(tt, 0) + 1 print(f"表格类型分布: {json.dumps(types, ensure_ascii=False, indent=2)}") # 生成查询结构 lookup = generate_scoring_lookup(classified) # 输出全部表格原始解析 all_tables_output = [] for t in classified: all_tables_output.append({ 'table_id': t.get('table_id', ''), 'title': t.get('title', ''), 'section': t.get('section', ''), 'gender': t.get('gender', ''), 'indicator': t.get('indicator', ''), 'unit': t.get('unit'), 'type': t.get('type', ''), 'parsed': t.get('parsed', {}) }) output = { 'metadata': { 'source': '《国民体质测定标准(2023年修订)》', 'publisher': '国家国民体质监测中心', 'year': 2023, 'total_tables': len(classified), 'table_types': types }, 'scoring_lookup': lookup, 'all_tables': all_tables_output, 'weights': lookup['weights'], 'rating_levels': lookup['rating_levels'] } out_path = f'{base}/02_加工数据/2023修订版/评分标准结构化数据.json' with open(out_path, 'w', encoding='utf-8') as f: json.dump(output, f, ensure_ascii=False, indent=2) print(f"\nJSON 输出: {out_path}") print(f"JSON 大小: {len(json.dumps(output, ensure_ascii=False))} 字符") # 打印摘要 print("\n=== 评分表摘要 ===") seen = set() for t in classified: key = (t.get('section',''), t.get('gender',''), t.get('indicator','')) if key not in seen: seen.add(key) print(f" [{t.get('section','')}] {t.get('gender','')} - {t.get('indicator','')} ({t.get('table_id','')}) [{t.get('type','')}]") print(f"\n总计 {len(seen)} 个评分表/指标组合") if __name__ == '__main__': main()