#!/usr/bin/env python3
"""
《国民体质测定标准(2003年)》(成年人部分)评分数据提取 + 评测引擎
5分制,含身高体重分档评分(对称1-3-5-3-1)、台阶指数、肺活量等
"""
import re
import json
import math
import sys
import os
from bs4 import BeautifulSoup
BASE_DIR = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
DOC_PATH = os.path.join(BASE_DIR, '01_来源数据/2003版/《国民体质测定标准(2003年)》(成年人部分).md')
JSON_PATH = os.path.join(BASE_DIR, '02_加工数据/2003版/评分标准结构化数据.json')
# ============================================================
# 第一部分:表格提取与解析
# ============================================================
def parse_range(val):
"""解析值域:'<47.7', '47.7-50.2', '>70.2', '7-12', '>40', '1-5'"""
val = val.strip().replace('<', '<').replace('>', '>')
# >=X
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', val)
if m:
return {"min": float(m.group(1))}
# <=X
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', val)
if m:
return {"max": float(m.group(1))}
# X - Y
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
# standalone
m = re.match(r'^(-?[\d.]+)$', val)
if m:
return {"exact": float(m.group(1))}
return {"raw": val}
def extract_2003_tables(filepath=DOC_PATH):
"""从2003版OCR文档提取所有HTML表格"""
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
tables = []
# 用BeautifulSoup批量提取所有table
soup = BeautifulSoup(content, 'html.parser')
for table_tag in soup.find_all('table'):
rows = []
for tr in table_tag.find_all('tr'):
cells = []
for td in tr.find_all(['td', 'th']):
txt = td.get_text(strip=True)
cells.append(txt)
if cells:
rows.append(cells)
if rows:
tables.append(rows)
return tables
def classify_2003_table(rows):
"""分类2003版表格并结构化提取"""
if not rows or len(rows) < 2:
return None
header = rows[0]
header_str = ' '.join(str(h) for h in header)
# 判断是否有标准表头
is_height_weight = ('身高' in header_str and '体重' in header_str) or \
('身高段' in header_str)
# 身高体重续表:首格是"X.X-X.X"身高段,6列
if not is_height_weight and len(header) == 6:
first_cell = str(header[0])
if re.match(r'\d+\.\d+[-~–]\d+\.\d+', first_cell):
is_height_weight = True
is_standard_indicator = '性别' in header_str and '年龄' in header_str and '1分' in header_str
# 判断是否为续表(无表头,第一行数据以年龄组+性别开头)
is_continuation = False
if not is_height_weight and not is_standard_indicator and len(header) >= 7:
first_cell = str(header[0])
second_cell = str(header[1]) if len(header) > 1 else ''
age_match = re.match(r'\d+[-~–]\d+岁', first_cell)
gender_match = second_cell in ['男', '女']
if age_match and gender_match:
is_continuation = True
if is_height_weight:
return parse_height_weight_table(rows)
elif is_standard_indicator:
return parse_indicator_table(rows)
elif is_continuation:
# 续表:用默认5分制解析,返回带"continuation"标记
result = parse_indicator_table(rows)
result['continuation'] = True
return result
return None
def parse_height_weight_table(rows):
"""解析身高体重分档评分表(对称1-3-5-3-1或1-2-3-4-5)"""
if len(rows) < 2:
return None
# 检测双层表头:第一行 ["身高段(厘米)", "体重(千克)"], 第二行 ["1分", "3分", "5分", "3分", "1分"]
first_row = rows[0]
second_row = rows[1] if len(rows) > 1 else []
second_has_scores = any('分' in str(c) for c in second_row)
first_is_stub = len(first_row) <= 2
if first_is_stub and second_has_scores:
# 双层表头:用第二行作为分数列
score_row = second_row
data_start = 2
else:
# 单层表头
score_row = first_row
data_start = 1
# 提取分数列
score_cols = []
for h in score_row:
m = re.search(r'(\d+)分', str(h))
score_cols.append(int(m.group(1)) if m else 0)
data = {}
for row in rows[data_start:]:
if len(row) < 2:
continue
height_key = row[0].strip()
ranges = []
for i in range(1, min(len(row), len(score_cols) + 1)):
if i - 1 < len(score_cols):
parsed = parse_range(row[i])
parsed['score'] = score_cols[i - 1]
ranges.append(parsed)
if ranges:
data[height_key] = ranges
return {
'type': 'height_weight',
'header': first_row,
'score_columns': score_cols,
'data': data
}
def parse_indicator_table(rows):
"""解析指标评分表(年龄×性别×5分制)"""
header = rows[0]
header_str = ' '.join(str(h) for h in header)
# 判断是否有标准表头
has_header = '年龄' in header_str and '性别' in header_str and '1分' in header_str
if has_header:
# 标准表头:['年龄', '性别', '1分', '2分', '3分', '4分', '5分']
score_cols = []
for h in header[2:]:
m = re.search(r'(\d+)分', str(h))
score_cols.append(int(m.group(1)) if m else 0)
data_start = 1 # 从第1行开始读数据
else:
# 续表,无表头:第一行就是数据 ['35-39岁', '男', '31.3-37.2', ...]
# 使用默认分数列 [1, 2, 3, 4, 5]
score_cols = [1, 2, 3, 4, 5]
data_start = 0
result = {'type': 'indicator_5point', 'score_columns': score_cols, 'data': {}}
for row in rows[data_start:]:
if len(row) < 4:
continue
age_group = row[0].strip()
gender = row[1].strip()
if age_group not in result['data']:
result['data'][age_group] = {}
ranges = []
for i, val in enumerate(row[2:2+len(score_cols)]):
if i < len(score_cols):
parsed = parse_range(val)
parsed['score'] = score_cols[i]
ranges.append(parsed)
result['data'][age_group][gender] = ranges
return result
def extract_all_2003_data(filepath=DOC_PATH):
"""完全提取2003版所有评分标准"""
raw_tables = extract_2003_tables(filepath)
print(f"2003版文档提取到 {len(raw_tables)} 个HTML表格")
classified = []
for rows in raw_tables:
result = classify_2003_table(rows)
if result:
classified.append(result)
print(f"成功分类 {len(classified)} 个表格")
# 按类型统计
types = {}
for t in classified:
tt = t['type']
types[tt] = types.get(tt, 0) + 1
print(f"类型分布: {json.dumps(types, ensure_ascii=False)}")
return classified
def extract_2003_tables_with_context(filepath=DOC_PATH):
"""提取表格及其前300字符上下文(用于指标名称识别)"""
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
# 用正则找到每个
前的位置
table_matches = list(re.finditer(r'= r['min']:
return score
elif 'max' in r and value <= r['max']:
return score
elif 'exact' in r and abs(value - r['exact']) < 0.01:
return score
return 0
def score_height_weight(table, height_cm, weight_kg):
"""在身高体重分档评分表中评分"""
data = table.get('data', {})
# 找到对应身高段(精确匹配)
height_key = None
for hk in data:
m = re.match(r'([\d.]+)\s*[-~–]\s*([\d.]+)', hk)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
if lo <= height_cm <= hi:
height_key = hk
break
if not height_key:
return 0
ranges = data[height_key]
for r in ranges:
score = r['score']
if 'min' in r and 'max' in r:
if r['min'] <= weight_kg <= r['max']:
return score
elif 'min' in r and weight_kg >= r['min']:
return score
elif 'max' in r and weight_kg <= r['max']:
return score
return 0
class FitnessEval2003:
"""2003版国民体质测定标准评测引擎"""
def __init__(self, data_path=None):
self.tables = None
self.parsed_data = {}
self.load_data(data_path)
def load_data(self, data_path=None):
"""加载并解析2003版所有评分表(优先从JSON加载)"""
# 优先从JSON加载
if os.path.exists(JSON_PATH):
return self._load_from_json()
# 回退:从文档解析
return self._load_from_doc(data_path)
def _load_from_json(self):
with open(JSON_PATH, 'r', encoding='utf-8') as f:
data = json.load(f)
self.height_weight_tables = {}
for key, hw in data.get('height_weight_tables', {}).items():
self.height_weight_tables[key] = hw
self.indicator_tables = {}
for name, ind in data.get('indicator_tables', {}).items():
self.indicator_tables[name] = ind
print(f"从JSON加载完成:{len(self.height_weight_tables)}个身高体重表 + {len(self.indicator_tables)}个指标评分表")
for name in sorted(self.indicator_tables.keys()):
ages = list(self.indicator_tables[name]['data'].keys())
print(f" {name}: {len(ages)}个年龄组")
return True
def _load_from_doc(self, data_path=None):
path = data_path or DOC_PATH
raw_tables, context_list = extract_2003_tables_with_context(path)
self.height_weight_tables = {}
self.indicator_tables = {}
last_indicator = None
last_hw_info = None # (age_range, gender) 用于身高体重数据表继承
for table_idx, rows in enumerate(raw_tables):
context_before = context_list[table_idx] if table_idx < len(context_list) else ""
result = classify_2003_table(rows)
if not result:
continue
tbl_type = result['type']
is_continuation = result.get('continuation', False)
if tbl_type == 'height_weight':
header_first = str(rows[0][0]) if rows[0] else ''
is_header = '身高段' in header_first or '身高' in header_first
if is_header:
# 头表:从上下文获取年龄性别
age_match = re.search(r'(\d+)[-~–—]\s*(\d+)\s*岁', context_before)
gender = '男' if '男' in context_before else '女'
if age_match:
key = f"{age_match.group(1)}-{age_match.group(2)}岁_{gender}"
last_hw_info = (f"{age_match.group(1)}-{age_match.group(2)}岁", gender)
else:
key = f"table_{table_idx}"
else:
# 数据表:继承上一个头表的年龄性别
if last_hw_info:
key = f"{last_hw_info[0]}_{last_hw_info[1]}"
else:
key = f"table_{table_idx}"
# 合并身高体重表数据(多个数据表覆盖不同身高段)
if key in self.height_weight_tables:
existing_data = self.height_weight_tables[key].get('data', {})
existing_data.update(result.get('data', {}))
self.height_weight_tables[key]['data'] = existing_data
else:
self.height_weight_tables[key] = result
elif tbl_type == 'indicator_5point':
if is_continuation and last_indicator:
# 续表:继承上一个指标名
indicator = last_indicator
else:
indicator = self._detect_indicator_name(context_before)
if indicator:
last_indicator = indicator # 更新跟踪
if indicator:
if indicator not in self.indicator_tables:
self.indicator_tables[indicator] = result
else:
for age_group, genders in result['data'].items():
if age_group not in self.indicator_tables[indicator]['data']:
self.indicator_tables[indicator]['data'][age_group] = {}
for g, ranges in genders.items():
self.indicator_tables[indicator]['data'][age_group][g] = ranges
self.parsed_data[indicator] = result
print(f"加载完成:{len(self.height_weight_tables)}个身高体重表 + {len(self.indicator_tables)}个指标评分表")
for name in sorted(self.indicator_tables.keys()):
ages = list(self.indicator_tables[name]['data'].keys())
print(f" {name}: {len(ages)}个年龄组")
return True
def _detect_indicator_name(self, context):
"""从上下文文本判断指标名称"""
indicators = [
('肺活量', '肺活量'),
('台阶指数', '台阶指数'),
('台阶', '台阶指数'),
('握力', '握力'),
('俯卧撑', '俯卧撑'),
('仰卧起坐', '仰卧起坐'),
('纵跳', '纵跳'),
('坐位体前屈', '坐位体前屈'),
('体前屈', '坐位体前屈'),
('选择反应时', '选择反应时'),
('反应时', '选择反应时'),
('闭眼单脚站立', '闭眼单脚站立'),
('单脚站立', '闭眼单脚站立'),
]
for kw, name in indicators:
if kw in context:
return name
return None
def _get_height_weight_table(self, age, gender):
"""获取对应年龄性别的身高体重评分表(2003版按10岁分组)"""
# 2003版身高体重按10岁分组
decade_low = (age // 10) * 10
decade_high = decade_low + 9
decade_key = f"{decade_low}-{decade_high}岁"
for key, table in self.height_weight_tables.items():
if decade_key in key and gender in key:
return table
# 退一步:用部分匹配
decade_prefix = str(decade_low)
for key, table in self.height_weight_tables.items():
if key.startswith(decade_prefix) and gender in key:
return table
return None
def evaluate(self, age, gender, height_cm, weight_kg, test_values):
"""
完整评测
test_values: dict,键名支持:
'肺活量(ml)', '台阶指数', '握力(kg)', '俯卧撑(次)',
'仰卧起坐(次/分)', '纵跳(cm)', '坐位体前屈(cm)',
'选择反应时(秒)', '闭眼单脚站立(秒)'
"""
results = {
'age': age,
'gender': gender,
'age_group': find_age_group(age),
'standard': '2003版',
'indicators': {},
'total_score': 0,
'rating': ''
}
indicator_alias = {
'肺活量(ml)': '肺活量', '肺活量': '肺活量',
'台阶指数': '台阶指数', '台阶': '台阶指数',
'握力(kg)': '握力', '握力': '握力',
'俯卧撑(次)': '俯卧撑', '俯卧撑': '俯卧撑',
'仰卧起坐(次/分)': '仰卧起坐', '仰卧起坐': '仰卧起坐',
'纵跳(cm)': '纵跳', '纵跳': '纵跳',
'坐位体前屈(cm)': '坐位体前屈', '坐位体前屈': '坐位体前屈',
'选择反应时(秒)': '选择反应时', '选择反应时': '选择反应时',
'闭眼单脚站立(秒)': '闭眼单脚站立', '闭眼单脚站立': '闭眼单脚站立',
}
age_group = find_age_group(age)
gender_char = '男' if gender in ['男', 'male', 'M'] else '女'
# 1. 身高体重评分
hw_table = self._get_height_weight_table(age, gender_char)
if hw_table and height_cm and weight_kg:
hw_score = score_height_weight(hw_table, height_cm, weight_kg)
results['indicators']['身高标准体重'] = {
'value': f"{height_cm}cm/{weight_kg}kg",
'score': hw_score
}
else:
results['indicators']['身高标准体重'] = {'value': '-', 'score': 0, 'note': '未找到对应年龄身高体重表'}
# 2. 其他指标评分
for raw_key, value in test_values.items():
if value is None or value == '':
continue
try:
value = float(value)
except (ValueError, TypeError):
continue
indicator = indicator_alias.get(raw_key, raw_key)
# 有些指标有年龄限制
if indicator == '俯卧撑' and gender_char == '女':
continue
if indicator == '仰卧起坐' and gender_char == '男':
continue
if indicator in ['俯卧撑', '仰卧起坐', '纵跳'] and age > 39:
continue
table = self.indicator_tables.get(indicator)
if table:
score = score_in_indicator_table(table, age_group, gender_char, value)
results['indicators'][indicator] = {'value': value, 'score': score}
else:
results['indicators'][indicator] = {'value': value, 'score': 0, 'note': '评分表未找到'}
# 3. 计算总分
scores = [v['score'] for v in results['indicators'].values()]
results['total_score'] = sum(scores)
# 4. 评级(2003版:按总分,越高越好)
max_possible = len(scores) * 5
total = results['total_score']
if total >= max_possible * 0.8:
results['rating'] = '优秀'
elif total >= max_possible * 0.6:
results['rating'] = '良好'
elif total >= max_possible * 0.4:
results['rating'] = '合格'
else:
results['rating'] = '不合格'
return results
def format_report(result):
"""格式化输出2003版评测报告"""
lines = []
lines.append("=" * 60)
lines.append(" 国民体质测定标准(2003版)— 评测报告")
lines.append("=" * 60)
lines.append(f" 年龄:{result['age']}岁 性别:{result['gender']}")
lines.append(f" 年龄组:{result['age_group']} 评分制:5分制")
lines.append("-" * 60)
lines.append(f" {'指标':<18} {'值':<12} {'得分':<6}")
lines.append("-" * 60)
for name, info in result['indicators'].items():
val = info.get('value', '')
if isinstance(val, (int, float)):
val_str = f"{val:.1f}"
else:
val_str = str(val)
note = info.get('note', '')
score_str = f"{info['score']}" + ("*" if note else "")
lines.append(f" {name:<18} {val_str:<12} {score_str:<6}")
lines.append("-" * 60)
lines.append(f" 总分:{result['total_score']}")
lines.append(f" 评级:{result['rating']}")
lines.append("=" * 60)
return '\n'.join(lines)
# ============================================================
# 主程序
# ============================================================
def main():
if len(sys.argv) > 1 and sys.argv[1] in ('--extract', '-e'):
# 只提取数据
tables = extract_all_2003_data()
print(f"\n共提取 {len(tables)} 个评分表")
return
# 评测模式
engine = FitnessEval2003()
if len(sys.argv) > 1 and sys.argv[1] in ('--demo', '-d'):
print("=" * 60)
print(" 演示1:成年男性,35岁,身高170cm,体重70kg")
print("=" * 60)
r = engine.evaluate(35, '男', 170, 70, {
'肺活量(ml)': 3500, '台阶指数': 58,
'握力(kg)': 42, '俯卧撑(次)': 20,
'纵跳(cm)': 32, '坐位体前屈(cm)': 8,
'选择反应时(秒)': 0.42, '闭眼单脚站立(秒)': 35
})
print(format_report(r))
print()
print("=" * 60)
print(" 演示2:成年女性,25岁,身高163cm,体重52kg")
print("=" * 60)
r2 = engine.evaluate(25, '女', 163, 52, {
'肺活量(ml)': 2800, '台阶指数': 60,
'握力(kg)': 25, '仰卧起坐(次/分)': 18,
'纵跳(cm)': 22, '坐位体前屈(cm)': 12,
'选择反应时(秒)': 0.45, '闭眼单脚站立(秒)': 40
})
print(format_report(r2))
print()
print("=" * 60)
print(" 演示3:成年男性,50岁,身高175cm,体重80kg")
print("=" * 60)
r3 = engine.evaluate(50, '男', 175, 80, {
'肺活量(ml)': 3000, '台阶指数': 55,
'握力(kg)': 38,
'坐位体前屈(cm)': 5,
'选择反应时(秒)': 0.55, '闭眼单脚站立(秒)': 25
})
print(format_report(r3))
else:
print("2003版国民体质测定标准评测引擎")
print("用法:")
print(" python3 fitness_eval_2003.py -d 运行演示")
print(" python3 fitness_eval_2003.py -e 仅提取数据")
print()
print("Python调用:")
print(" from fitness_eval_2003 import FitnessEval2003, format_report")
print(" engine = FitnessEval2003()")
print(' r = engine.evaluate(35, "男", 170, 70, {...})')
print(" print(format_report(r))")
if __name__ == '__main__':
main()