#!/usr/bin/env python3 """ 《国民体质测定标准(2003年)》(成年人部分)评分数据提取 + 评测引擎 5分制,含身高体重分档评分(对称1-3-5-3-1)、台阶指数、肺活量等 """ import re import json import math import sys import os from bs4 import BeautifulSoup BASE_DIR = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测' DOC_PATH = os.path.join(BASE_DIR, '01_来源数据/2003版/《国民体质测定标准(2003年)》(成年人部分).md') JSON_PATH = os.path.join(BASE_DIR, '02_加工数据/2003版/评分标准结构化数据.json') # ============================================================ # 第一部分:表格提取与解析 # ============================================================ def parse_range(val): """解析值域:'<47.7', '47.7-50.2', '>70.2', '7-12', '>40', '1-5'""" val = val.strip().replace('<', '<').replace('>', '>') # >=X m = re.match(r'[≥>]=?\s*(-?[\d.]+)', val) if m: return {"min": float(m.group(1))} # <=X m = re.match(r'[≤<]=?\s*(-?[\d.]+)', val) if m: return {"max": float(m.group(1))} # X - Y m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val) if m: lo, hi = float(m.group(1)), float(m.group(2)) return {"min": min(lo, hi), "max": max(lo, hi)} # standalone m = re.match(r'^(-?[\d.]+)$', val) if m: return {"exact": float(m.group(1))} return {"raw": val} def extract_2003_tables(filepath=DOC_PATH): """从2003版OCR文档提取所有HTML表格""" with open(filepath, 'r', encoding='utf-8') as f: content = f.read() tables = [] # 用BeautifulSoup批量提取所有table soup = BeautifulSoup(content, 'html.parser') for table_tag in soup.find_all('table'): rows = [] for tr in table_tag.find_all('tr'): cells = [] for td in tr.find_all(['td', 'th']): txt = td.get_text(strip=True) cells.append(txt) if cells: rows.append(cells) if rows: tables.append(rows) return tables def classify_2003_table(rows): """分类2003版表格并结构化提取""" if not rows or len(rows) < 2: return None header = rows[0] header_str = ' '.join(str(h) for h in header) # 判断是否有标准表头 is_height_weight = ('身高' in header_str and '体重' in header_str) or \ ('身高段' in header_str) # 身高体重续表:首格是"X.X-X.X"身高段,6列 if not is_height_weight and len(header) == 6: first_cell = str(header[0]) if re.match(r'\d+\.\d+[-~–]\d+\.\d+', first_cell): is_height_weight = True is_standard_indicator = '性别' in header_str and '年龄' in header_str and '1分' in header_str # 判断是否为续表(无表头,第一行数据以年龄组+性别开头) is_continuation = False if not is_height_weight and not is_standard_indicator and len(header) >= 7: first_cell = str(header[0]) second_cell = str(header[1]) if len(header) > 1 else '' age_match = re.match(r'\d+[-~–]\d+岁', first_cell) gender_match = second_cell in ['男', '女'] if age_match and gender_match: is_continuation = True if is_height_weight: return parse_height_weight_table(rows) elif is_standard_indicator: return parse_indicator_table(rows) elif is_continuation: # 续表:用默认5分制解析,返回带"continuation"标记 result = parse_indicator_table(rows) result['continuation'] = True return result return None def parse_height_weight_table(rows): """解析身高体重分档评分表(对称1-3-5-3-1或1-2-3-4-5)""" if len(rows) < 2: return None # 检测双层表头:第一行 ["身高段(厘米)", "体重(千克)"], 第二行 ["1分", "3分", "5分", "3分", "1分"] first_row = rows[0] second_row = rows[1] if len(rows) > 1 else [] second_has_scores = any('分' in str(c) for c in second_row) first_is_stub = len(first_row) <= 2 if first_is_stub and second_has_scores: # 双层表头:用第二行作为分数列 score_row = second_row data_start = 2 else: # 单层表头 score_row = first_row data_start = 1 # 提取分数列 score_cols = [] for h in score_row: m = re.search(r'(\d+)分', str(h)) score_cols.append(int(m.group(1)) if m else 0) data = {} for row in rows[data_start:]: if len(row) < 2: continue height_key = row[0].strip() ranges = [] for i in range(1, min(len(row), len(score_cols) + 1)): if i - 1 < len(score_cols): parsed = parse_range(row[i]) parsed['score'] = score_cols[i - 1] ranges.append(parsed) if ranges: data[height_key] = ranges return { 'type': 'height_weight', 'header': first_row, 'score_columns': score_cols, 'data': data } def parse_indicator_table(rows): """解析指标评分表(年龄×性别×5分制)""" header = rows[0] header_str = ' '.join(str(h) for h in header) # 判断是否有标准表头 has_header = '年龄' in header_str and '性别' in header_str and '1分' in header_str if has_header: # 标准表头:['年龄', '性别', '1分', '2分', '3分', '4分', '5分'] score_cols = [] for h in header[2:]: m = re.search(r'(\d+)分', str(h)) score_cols.append(int(m.group(1)) if m else 0) data_start = 1 # 从第1行开始读数据 else: # 续表,无表头:第一行就是数据 ['35-39岁', '男', '31.3-37.2', ...] # 使用默认分数列 [1, 2, 3, 4, 5] score_cols = [1, 2, 3, 4, 5] data_start = 0 result = {'type': 'indicator_5point', 'score_columns': score_cols, 'data': {}} for row in rows[data_start:]: if len(row) < 4: continue age_group = row[0].strip() gender = row[1].strip() if age_group not in result['data']: result['data'][age_group] = {} ranges = [] for i, val in enumerate(row[2:2+len(score_cols)]): if i < len(score_cols): parsed = parse_range(val) parsed['score'] = score_cols[i] ranges.append(parsed) result['data'][age_group][gender] = ranges return result def extract_all_2003_data(filepath=DOC_PATH): """完全提取2003版所有评分标准""" raw_tables = extract_2003_tables(filepath) print(f"2003版文档提取到 {len(raw_tables)} 个HTML表格") classified = [] for rows in raw_tables: result = classify_2003_table(rows) if result: classified.append(result) print(f"成功分类 {len(classified)} 个表格") # 按类型统计 types = {} for t in classified: tt = t['type'] types[tt] = types.get(tt, 0) + 1 print(f"类型分布: {json.dumps(types, ensure_ascii=False)}") return classified def extract_2003_tables_with_context(filepath=DOC_PATH): """提取表格及其前300字符上下文(用于指标名称识别)""" with open(filepath, 'r', encoding='utf-8') as f: content = f.read() # 用正则找到每个前的位置 table_matches = list(re.finditer(r'= r['min']: return score elif 'max' in r and value <= r['max']: return score elif 'exact' in r and abs(value - r['exact']) < 0.01: return score return 0 def score_height_weight(table, height_cm, weight_kg): """在身高体重分档评分表中评分""" data = table.get('data', {}) # 找到对应身高段(精确匹配) height_key = None for hk in data: m = re.match(r'([\d.]+)\s*[-~–]\s*([\d.]+)', hk) if m: lo, hi = float(m.group(1)), float(m.group(2)) if lo <= height_cm <= hi: height_key = hk break if not height_key: return 0 ranges = data[height_key] for r in ranges: score = r['score'] if 'min' in r and 'max' in r: if r['min'] <= weight_kg <= r['max']: return score elif 'min' in r and weight_kg >= r['min']: return score elif 'max' in r and weight_kg <= r['max']: return score return 0 class FitnessEval2003: """2003版国民体质测定标准评测引擎""" def __init__(self, data_path=None): self.tables = None self.parsed_data = {} self.load_data(data_path) def load_data(self, data_path=None): """加载并解析2003版所有评分表(优先从JSON加载)""" # 优先从JSON加载 if os.path.exists(JSON_PATH): return self._load_from_json() # 回退:从文档解析 return self._load_from_doc(data_path) def _load_from_json(self): with open(JSON_PATH, 'r', encoding='utf-8') as f: data = json.load(f) self.height_weight_tables = {} for key, hw in data.get('height_weight_tables', {}).items(): self.height_weight_tables[key] = hw self.indicator_tables = {} for name, ind in data.get('indicator_tables', {}).items(): self.indicator_tables[name] = ind print(f"从JSON加载完成:{len(self.height_weight_tables)}个身高体重表 + {len(self.indicator_tables)}个指标评分表") for name in sorted(self.indicator_tables.keys()): ages = list(self.indicator_tables[name]['data'].keys()) print(f" {name}: {len(ages)}个年龄组") return True def _load_from_doc(self, data_path=None): path = data_path or DOC_PATH raw_tables, context_list = extract_2003_tables_with_context(path) self.height_weight_tables = {} self.indicator_tables = {} last_indicator = None last_hw_info = None # (age_range, gender) 用于身高体重数据表继承 for table_idx, rows in enumerate(raw_tables): context_before = context_list[table_idx] if table_idx < len(context_list) else "" result = classify_2003_table(rows) if not result: continue tbl_type = result['type'] is_continuation = result.get('continuation', False) if tbl_type == 'height_weight': header_first = str(rows[0][0]) if rows[0] else '' is_header = '身高段' in header_first or '身高' in header_first if is_header: # 头表:从上下文获取年龄性别 age_match = re.search(r'(\d+)[-~–—]\s*(\d+)\s*岁', context_before) gender = '男' if '男' in context_before else '女' if age_match: key = f"{age_match.group(1)}-{age_match.group(2)}岁_{gender}" last_hw_info = (f"{age_match.group(1)}-{age_match.group(2)}岁", gender) else: key = f"table_{table_idx}" else: # 数据表:继承上一个头表的年龄性别 if last_hw_info: key = f"{last_hw_info[0]}_{last_hw_info[1]}" else: key = f"table_{table_idx}" # 合并身高体重表数据(多个数据表覆盖不同身高段) if key in self.height_weight_tables: existing_data = self.height_weight_tables[key].get('data', {}) existing_data.update(result.get('data', {})) self.height_weight_tables[key]['data'] = existing_data else: self.height_weight_tables[key] = result elif tbl_type == 'indicator_5point': if is_continuation and last_indicator: # 续表:继承上一个指标名 indicator = last_indicator else: indicator = self._detect_indicator_name(context_before) if indicator: last_indicator = indicator # 更新跟踪 if indicator: if indicator not in self.indicator_tables: self.indicator_tables[indicator] = result else: for age_group, genders in result['data'].items(): if age_group not in self.indicator_tables[indicator]['data']: self.indicator_tables[indicator]['data'][age_group] = {} for g, ranges in genders.items(): self.indicator_tables[indicator]['data'][age_group][g] = ranges self.parsed_data[indicator] = result print(f"加载完成:{len(self.height_weight_tables)}个身高体重表 + {len(self.indicator_tables)}个指标评分表") for name in sorted(self.indicator_tables.keys()): ages = list(self.indicator_tables[name]['data'].keys()) print(f" {name}: {len(ages)}个年龄组") return True def _detect_indicator_name(self, context): """从上下文文本判断指标名称""" indicators = [ ('肺活量', '肺活量'), ('台阶指数', '台阶指数'), ('台阶', '台阶指数'), ('握力', '握力'), ('俯卧撑', '俯卧撑'), ('仰卧起坐', '仰卧起坐'), ('纵跳', '纵跳'), ('坐位体前屈', '坐位体前屈'), ('体前屈', '坐位体前屈'), ('选择反应时', '选择反应时'), ('反应时', '选择反应时'), ('闭眼单脚站立', '闭眼单脚站立'), ('单脚站立', '闭眼单脚站立'), ] for kw, name in indicators: if kw in context: return name return None def _get_height_weight_table(self, age, gender): """获取对应年龄性别的身高体重评分表(2003版按10岁分组)""" # 2003版身高体重按10岁分组 decade_low = (age // 10) * 10 decade_high = decade_low + 9 decade_key = f"{decade_low}-{decade_high}岁" for key, table in self.height_weight_tables.items(): if decade_key in key and gender in key: return table # 退一步:用部分匹配 decade_prefix = str(decade_low) for key, table in self.height_weight_tables.items(): if key.startswith(decade_prefix) and gender in key: return table return None def evaluate(self, age, gender, height_cm, weight_kg, test_values): """ 完整评测 test_values: dict,键名支持: '肺活量(ml)', '台阶指数', '握力(kg)', '俯卧撑(次)', '仰卧起坐(次/分)', '纵跳(cm)', '坐位体前屈(cm)', '选择反应时(秒)', '闭眼单脚站立(秒)' """ results = { 'age': age, 'gender': gender, 'age_group': find_age_group(age), 'standard': '2003版', 'indicators': {}, 'total_score': 0, 'rating': '' } indicator_alias = { '肺活量(ml)': '肺活量', '肺活量': '肺活量', '台阶指数': '台阶指数', '台阶': '台阶指数', '握力(kg)': '握力', '握力': '握力', '俯卧撑(次)': '俯卧撑', '俯卧撑': '俯卧撑', '仰卧起坐(次/分)': '仰卧起坐', '仰卧起坐': '仰卧起坐', '纵跳(cm)': '纵跳', '纵跳': '纵跳', '坐位体前屈(cm)': '坐位体前屈', '坐位体前屈': '坐位体前屈', '选择反应时(秒)': '选择反应时', '选择反应时': '选择反应时', '闭眼单脚站立(秒)': '闭眼单脚站立', '闭眼单脚站立': '闭眼单脚站立', } age_group = find_age_group(age) gender_char = '男' if gender in ['男', 'male', 'M'] else '女' # 1. 身高体重评分 hw_table = self._get_height_weight_table(age, gender_char) if hw_table and height_cm and weight_kg: hw_score = score_height_weight(hw_table, height_cm, weight_kg) results['indicators']['身高标准体重'] = { 'value': f"{height_cm}cm/{weight_kg}kg", 'score': hw_score } else: results['indicators']['身高标准体重'] = {'value': '-', 'score': 0, 'note': '未找到对应年龄身高体重表'} # 2. 其他指标评分 for raw_key, value in test_values.items(): if value is None or value == '': continue try: value = float(value) except (ValueError, TypeError): continue indicator = indicator_alias.get(raw_key, raw_key) # 有些指标有年龄限制 if indicator == '俯卧撑' and gender_char == '女': continue if indicator == '仰卧起坐' and gender_char == '男': continue if indicator in ['俯卧撑', '仰卧起坐', '纵跳'] and age > 39: continue table = self.indicator_tables.get(indicator) if table: score = score_in_indicator_table(table, age_group, gender_char, value) results['indicators'][indicator] = {'value': value, 'score': score} else: results['indicators'][indicator] = {'value': value, 'score': 0, 'note': '评分表未找到'} # 3. 计算总分 scores = [v['score'] for v in results['indicators'].values()] results['total_score'] = sum(scores) # 4. 评级(2003版:按总分,越高越好) max_possible = len(scores) * 5 total = results['total_score'] if total >= max_possible * 0.8: results['rating'] = '优秀' elif total >= max_possible * 0.6: results['rating'] = '良好' elif total >= max_possible * 0.4: results['rating'] = '合格' else: results['rating'] = '不合格' return results def format_report(result): """格式化输出2003版评测报告""" lines = [] lines.append("=" * 60) lines.append(" 国民体质测定标准(2003版)— 评测报告") lines.append("=" * 60) lines.append(f" 年龄:{result['age']}岁 性别:{result['gender']}") lines.append(f" 年龄组:{result['age_group']} 评分制:5分制") lines.append("-" * 60) lines.append(f" {'指标':<18} {'值':<12} {'得分':<6}") lines.append("-" * 60) for name, info in result['indicators'].items(): val = info.get('value', '') if isinstance(val, (int, float)): val_str = f"{val:.1f}" else: val_str = str(val) note = info.get('note', '') score_str = f"{info['score']}" + ("*" if note else "") lines.append(f" {name:<18} {val_str:<12} {score_str:<6}") lines.append("-" * 60) lines.append(f" 总分:{result['total_score']}") lines.append(f" 评级:{result['rating']}") lines.append("=" * 60) return '\n'.join(lines) # ============================================================ # 主程序 # ============================================================ def main(): if len(sys.argv) > 1 and sys.argv[1] in ('--extract', '-e'): # 只提取数据 tables = extract_all_2003_data() print(f"\n共提取 {len(tables)} 个评分表") return # 评测模式 engine = FitnessEval2003() if len(sys.argv) > 1 and sys.argv[1] in ('--demo', '-d'): print("=" * 60) print(" 演示1:成年男性,35岁,身高170cm,体重70kg") print("=" * 60) r = engine.evaluate(35, '男', 170, 70, { '肺活量(ml)': 3500, '台阶指数': 58, '握力(kg)': 42, '俯卧撑(次)': 20, '纵跳(cm)': 32, '坐位体前屈(cm)': 8, '选择反应时(秒)': 0.42, '闭眼单脚站立(秒)': 35 }) print(format_report(r)) print() print("=" * 60) print(" 演示2:成年女性,25岁,身高163cm,体重52kg") print("=" * 60) r2 = engine.evaluate(25, '女', 163, 52, { '肺活量(ml)': 2800, '台阶指数': 60, '握力(kg)': 25, '仰卧起坐(次/分)': 18, '纵跳(cm)': 22, '坐位体前屈(cm)': 12, '选择反应时(秒)': 0.45, '闭眼单脚站立(秒)': 40 }) print(format_report(r2)) print() print("=" * 60) print(" 演示3:成年男性,50岁,身高175cm,体重80kg") print("=" * 60) r3 = engine.evaluate(50, '男', 175, 80, { '肺活量(ml)': 3000, '台阶指数': 55, '握力(kg)': 38, '坐位体前屈(cm)': 5, '选择反应时(秒)': 0.55, '闭眼单脚站立(秒)': 25 }) print(format_report(r3)) else: print("2003版国民体质测定标准评测引擎") print("用法:") print(" python3 fitness_eval_2003.py -d 运行演示") print(" python3 fitness_eval_2003.py -e 仅提取数据") print() print("Python调用:") print(" from fitness_eval_2003 import FitnessEval2003, format_report") print(" engine = FitnessEval2003()") print(' r = engine.evaluate(35, "男", 170, 70, {...})') print(" print(format_report(r))") if __name__ == '__main__': main()