Files
health/国家国民体质监测/02_加工数据/2003版/fitness_eval_2003.py
T

663 lines
24 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
《国民体质测定标准(2003年)》(成年人部分)评分数据提取 + 评测引擎
5分制,含身高体重分档评分(对称1-3-5-3-1)、台阶指数、肺活量等
"""
import re
import json
import math
import sys
import os
from bs4 import BeautifulSoup
BASE_DIR = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
DOC_PATH = os.path.join(BASE_DIR, '01_来源数据/2003版/《国民体质测定标准(2003年)》(成年人部分).md')
JSON_PATH = os.path.join(BASE_DIR, '02_加工数据/2003版/评分标准结构化数据.json')
# ============================================================
# 第一部分:表格提取与解析
# ============================================================
def parse_range(val):
"""解析值域:'<47.7', '47.7-50.2', '>70.2', '7-12', '>40', '1-5'"""
val = val.strip().replace('&lt;', '<').replace('&gt;', '>')
# >=X
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', val)
if m:
return {"min": float(m.group(1))}
# <=X
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', val)
if m:
return {"max": float(m.group(1))}
# X - Y
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
# standalone
m = re.match(r'^(-?[\d.]+)$', val)
if m:
return {"exact": float(m.group(1))}
return {"raw": val}
def extract_2003_tables(filepath=DOC_PATH):
"""从2003版OCR文档提取所有HTML表格"""
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
tables = []
# 用BeautifulSoup批量提取所有table
soup = BeautifulSoup(content, 'html.parser')
for table_tag in soup.find_all('table'):
rows = []
for tr in table_tag.find_all('tr'):
cells = []
for td in tr.find_all(['td', 'th']):
txt = td.get_text(strip=True)
cells.append(txt)
if cells:
rows.append(cells)
if rows:
tables.append(rows)
return tables
def classify_2003_table(rows):
"""分类2003版表格并结构化提取"""
if not rows or len(rows) < 2:
return None
header = rows[0]
header_str = ' '.join(str(h) for h in header)
# 判断是否有标准表头
is_height_weight = ('身高' in header_str and '体重' in header_str) or \
('身高段' in header_str)
# 身高体重续表:首格是"X.X-X.X"身高段,6列
if not is_height_weight and len(header) == 6:
first_cell = str(header[0])
if re.match(r'\d+\.\d+[-~–]\d+\.\d+', first_cell):
is_height_weight = True
is_standard_indicator = '性别' in header_str and '年龄' in header_str and '1分' in header_str
# 判断是否为续表(无表头,第一行数据以年龄组+性别开头)
is_continuation = False
if not is_height_weight and not is_standard_indicator and len(header) >= 7:
first_cell = str(header[0])
second_cell = str(header[1]) if len(header) > 1 else ''
age_match = re.match(r'\d+[-~–]\d+岁', first_cell)
gender_match = second_cell in ['男', '女']
if age_match and gender_match:
is_continuation = True
if is_height_weight:
return parse_height_weight_table(rows)
elif is_standard_indicator:
return parse_indicator_table(rows)
elif is_continuation:
# 续表:用默认5分制解析,返回带"continuation"标记
result = parse_indicator_table(rows)
result['continuation'] = True
return result
return None
def parse_height_weight_table(rows):
"""解析身高体重分档评分表(对称1-3-5-3-1或1-2-3-4-5)"""
if len(rows) < 2:
return None
# 检测双层表头:第一行 ["身高段(厘米)", "体重(千克)"], 第二行 ["1分", "3分", "5分", "3分", "1分"]
first_row = rows[0]
second_row = rows[1] if len(rows) > 1 else []
second_has_scores = any('分' in str(c) for c in second_row)
first_is_stub = len(first_row) <= 2
if first_is_stub and second_has_scores:
# 双层表头:用第二行作为分数列
score_row = second_row
data_start = 2
else:
# 单层表头
score_row = first_row
data_start = 1
# 提取分数列
score_cols = []
for h in score_row:
m = re.search(r'(\d+)分', str(h))
score_cols.append(int(m.group(1)) if m else 0)
data = {}
for row in rows[data_start:]:
if len(row) < 2:
continue
height_key = row[0].strip()
ranges = []
for i in range(1, min(len(row), len(score_cols) + 1)):
if i - 1 < len(score_cols):
parsed = parse_range(row[i])
parsed['score'] = score_cols[i - 1]
ranges.append(parsed)
if ranges:
data[height_key] = ranges
return {
'type': 'height_weight',
'header': first_row,
'score_columns': score_cols,
'data': data
}
def parse_indicator_table(rows):
"""解析指标评分表(年龄×性别×5分制)"""
header = rows[0]
header_str = ' '.join(str(h) for h in header)
# 判断是否有标准表头
has_header = '年龄' in header_str and '性别' in header_str and '1分' in header_str
if has_header:
# 标准表头:['年龄', '性别', '1分', '2分', '3分', '4分', '5分']
score_cols = []
for h in header[2:]:
m = re.search(r'(\d+)分', str(h))
score_cols.append(int(m.group(1)) if m else 0)
data_start = 1 # 从第1行开始读数据
else:
# 续表,无表头:第一行就是数据 ['35-39岁', '男', '31.3-37.2', ...]
# 使用默认分数列 [1, 2, 3, 4, 5]
score_cols = [1, 2, 3, 4, 5]
data_start = 0
result = {'type': 'indicator_5point', 'score_columns': score_cols, 'data': {}}
for row in rows[data_start:]:
if len(row) < 4:
continue
age_group = row[0].strip()
gender = row[1].strip()
if age_group not in result['data']:
result['data'][age_group] = {}
ranges = []
for i, val in enumerate(row[2:2+len(score_cols)]):
if i < len(score_cols):
parsed = parse_range(val)
parsed['score'] = score_cols[i]
ranges.append(parsed)
result['data'][age_group][gender] = ranges
return result
def extract_all_2003_data(filepath=DOC_PATH):
"""完全提取2003版所有评分标准"""
raw_tables = extract_2003_tables(filepath)
print(f"2003版文档提取到 {len(raw_tables)} 个HTML表格")
classified = []
for rows in raw_tables:
result = classify_2003_table(rows)
if result:
classified.append(result)
print(f"成功分类 {len(classified)} 个表格")
# 按类型统计
types = {}
for t in classified:
tt = t['type']
types[tt] = types.get(tt, 0) + 1
print(f"类型分布: {json.dumps(types, ensure_ascii=False)}")
return classified
def extract_2003_tables_with_context(filepath=DOC_PATH):
"""提取表格及其前300字符上下文(用于指标名称识别)"""
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
# 用正则找到每个<table>前的位置
table_matches = list(re.finditer(r'<table\b', content))
soup = BeautifulSoup(content, 'html.parser')
tables = soup.find_all('table')
result_tables = []
contexts = []
for i, table_tag in enumerate(tables):
rows = []
for tr in table_tag.find_all('tr'):
cells = [td.get_text(strip=True) for td in tr.find_all(['td', 'th'])]
if cells:
rows.append(cells)
if rows:
result_tables.append(rows)
# 获取该表格前的上下文文本
if i < len(table_matches):
pos = table_matches[i].start()
context_start = max(0, pos - 500)
context_text = content[context_start:pos]
contexts.append(context_text)
else:
contexts.append('')
return result_tables, contexts
# ============================================================
# 第二部分:评测引擎
# ============================================================
def find_age_group(age):
"""确定5岁年龄组"""
low = (age // 5) * 5
high = low + 4
return f"{low}-{high}岁"
def score_in_indicator_table(table, age_group, gender, value):
"""在5分制指标表中查分"""
data = table.get('data', {})
if age_group not in data:
return 0
if gender not in data[age_group]:
return 0
ranges = data[age_group][gender]
for r in ranges:
score = r['score']
if 'min' in r and 'max' in r:
if r['min'] <= value <= r['max']:
return score
elif 'min' in r and value >= r['min']:
return score
elif 'max' in r and value <= r['max']:
return score
elif 'exact' in r and abs(value - r['exact']) < 0.01:
return score
return 0
def score_height_weight(table, height_cm, weight_kg):
"""在身高体重分档评分表中评分"""
data = table.get('data', {})
# 找到对应身高段(精确匹配)
height_key = None
for hk in data:
m = re.match(r'([\d.]+)\s*[-~–]\s*([\d.]+)', hk)
if m:
lo, hi = float(m.group(1)), float(m.group(2))
if lo <= height_cm <= hi:
height_key = hk
break
if not height_key:
return 0
ranges = data[height_key]
for r in ranges:
score = r['score']
if 'min' in r and 'max' in r:
if r['min'] <= weight_kg <= r['max']:
return score
elif 'min' in r and weight_kg >= r['min']:
return score
elif 'max' in r and weight_kg <= r['max']:
return score
return 0
class FitnessEval2003:
"""2003版国民体质测定标准评测引擎"""
def __init__(self, data_path=None):
self.tables = None
self.parsed_data = {}
self.load_data(data_path)
def load_data(self, data_path=None):
"""加载并解析2003版所有评分表(优先从JSON加载)"""
# 优先从JSON加载
if os.path.exists(JSON_PATH):
return self._load_from_json()
# 回退:从文档解析
return self._load_from_doc(data_path)
def _load_from_json(self):
with open(JSON_PATH, 'r', encoding='utf-8') as f:
data = json.load(f)
self.height_weight_tables = {}
for key, hw in data.get('height_weight_tables', {}).items():
self.height_weight_tables[key] = hw
self.indicator_tables = {}
for name, ind in data.get('indicator_tables', {}).items():
self.indicator_tables[name] = ind
print(f"从JSON加载完成:{len(self.height_weight_tables)}个身高体重表 + {len(self.indicator_tables)}个指标评分表")
for name in sorted(self.indicator_tables.keys()):
ages = list(self.indicator_tables[name]['data'].keys())
print(f" {name}: {len(ages)}个年龄组")
return True
def _load_from_doc(self, data_path=None):
path = data_path or DOC_PATH
raw_tables, context_list = extract_2003_tables_with_context(path)
self.height_weight_tables = {}
self.indicator_tables = {}
last_indicator = None
last_hw_info = None # (age_range, gender) 用于身高体重数据表继承
for table_idx, rows in enumerate(raw_tables):
context_before = context_list[table_idx] if table_idx < len(context_list) else ""
result = classify_2003_table(rows)
if not result:
continue
tbl_type = result['type']
is_continuation = result.get('continuation', False)
if tbl_type == 'height_weight':
header_first = str(rows[0][0]) if rows[0] else ''
is_header = '身高段' in header_first or '身高' in header_first
if is_header:
# 头表:从上下文获取年龄性别
age_match = re.search(r'(\d+)[-~–—]\s*(\d+)\s*岁', context_before)
gender = '男' if '男' in context_before else '女'
if age_match:
key = f"{age_match.group(1)}-{age_match.group(2)}岁_{gender}"
last_hw_info = (f"{age_match.group(1)}-{age_match.group(2)}岁", gender)
else:
key = f"table_{table_idx}"
else:
# 数据表:继承上一个头表的年龄性别
if last_hw_info:
key = f"{last_hw_info[0]}_{last_hw_info[1]}"
else:
key = f"table_{table_idx}"
# 合并身高体重表数据(多个数据表覆盖不同身高段)
if key in self.height_weight_tables:
existing_data = self.height_weight_tables[key].get('data', {})
existing_data.update(result.get('data', {}))
self.height_weight_tables[key]['data'] = existing_data
else:
self.height_weight_tables[key] = result
elif tbl_type == 'indicator_5point':
if is_continuation and last_indicator:
# 续表:继承上一个指标名
indicator = last_indicator
else:
indicator = self._detect_indicator_name(context_before)
if indicator:
last_indicator = indicator # 更新跟踪
if indicator:
if indicator not in self.indicator_tables:
self.indicator_tables[indicator] = result
else:
for age_group, genders in result['data'].items():
if age_group not in self.indicator_tables[indicator]['data']:
self.indicator_tables[indicator]['data'][age_group] = {}
for g, ranges in genders.items():
self.indicator_tables[indicator]['data'][age_group][g] = ranges
self.parsed_data[indicator] = result
print(f"加载完成:{len(self.height_weight_tables)}个身高体重表 + {len(self.indicator_tables)}个指标评分表")
for name in sorted(self.indicator_tables.keys()):
ages = list(self.indicator_tables[name]['data'].keys())
print(f" {name}: {len(ages)}个年龄组")
return True
def _detect_indicator_name(self, context):
"""从上下文文本判断指标名称"""
indicators = [
('肺活量', '肺活量'),
('台阶指数', '台阶指数'),
('台阶', '台阶指数'),
('握力', '握力'),
('俯卧撑', '俯卧撑'),
('仰卧起坐', '仰卧起坐'),
('纵跳', '纵跳'),
('坐位体前屈', '坐位体前屈'),
('体前屈', '坐位体前屈'),
('选择反应时', '选择反应时'),
('反应时', '选择反应时'),
('闭眼单脚站立', '闭眼单脚站立'),
('单脚站立', '闭眼单脚站立'),
]
for kw, name in indicators:
if kw in context:
return name
return None
def _get_height_weight_table(self, age, gender):
"""获取对应年龄性别的身高体重评分表(2003版按10岁分组)"""
# 2003版身高体重按10岁分组
decade_low = (age // 10) * 10
decade_high = decade_low + 9
decade_key = f"{decade_low}-{decade_high}岁"
for key, table in self.height_weight_tables.items():
if decade_key in key and gender in key:
return table
# 退一步:用部分匹配
decade_prefix = str(decade_low)
for key, table in self.height_weight_tables.items():
if key.startswith(decade_prefix) and gender in key:
return table
return None
def evaluate(self, age, gender, height_cm, weight_kg, test_values):
"""
完整评测
test_values: dict,键名支持:
'肺活量(ml)', '台阶指数', '握力(kg)', '俯卧撑(次)',
'仰卧起坐(次/分)', '纵跳(cm)', '坐位体前屈(cm)',
'选择反应时(秒)', '闭眼单脚站立(秒)'
"""
results = {
'age': age,
'gender': gender,
'age_group': find_age_group(age),
'standard': '2003版',
'indicators': {},
'total_score': 0,
'rating': ''
}
indicator_alias = {
'肺活量(ml)': '肺活量', '肺活量': '肺活量',
'台阶指数': '台阶指数', '台阶': '台阶指数',
'握力(kg)': '握力', '握力': '握力',
'俯卧撑(次)': '俯卧撑', '俯卧撑': '俯卧撑',
'仰卧起坐(次/分)': '仰卧起坐', '仰卧起坐': '仰卧起坐',
'纵跳(cm)': '纵跳', '纵跳': '纵跳',
'坐位体前屈(cm)': '坐位体前屈', '坐位体前屈': '坐位体前屈',
'选择反应时(秒)': '选择反应时', '选择反应时': '选择反应时',
'闭眼单脚站立(秒)': '闭眼单脚站立', '闭眼单脚站立': '闭眼单脚站立',
}
age_group = find_age_group(age)
gender_char = '男' if gender in ['男', 'male', 'M'] else '女'
# 1. 身高体重评分
hw_table = self._get_height_weight_table(age, gender_char)
if hw_table and height_cm and weight_kg:
hw_score = score_height_weight(hw_table, height_cm, weight_kg)
results['indicators']['身高标准体重'] = {
'value': f"{height_cm}cm/{weight_kg}kg",
'score': hw_score
}
else:
results['indicators']['身高标准体重'] = {'value': '-', 'score': 0, 'note': '未找到对应年龄身高体重表'}
# 2. 其他指标评分
for raw_key, value in test_values.items():
if value is None or value == '':
continue
try:
value = float(value)
except (ValueError, TypeError):
continue
indicator = indicator_alias.get(raw_key, raw_key)
# 有些指标有年龄限制
if indicator == '俯卧撑' and gender_char == '女':
continue
if indicator == '仰卧起坐' and gender_char == '男':
continue
if indicator in ['俯卧撑', '仰卧起坐', '纵跳'] and age > 39:
continue
table = self.indicator_tables.get(indicator)
if table:
score = score_in_indicator_table(table, age_group, gender_char, value)
results['indicators'][indicator] = {'value': value, 'score': score}
else:
results['indicators'][indicator] = {'value': value, 'score': 0, 'note': '评分表未找到'}
# 3. 计算总分
scores = [v['score'] for v in results['indicators'].values()]
results['total_score'] = sum(scores)
# 4. 评级(2003版:按总分,越高越好)
max_possible = len(scores) * 5
total = results['total_score']
if total >= max_possible * 0.8:
results['rating'] = '优秀'
elif total >= max_possible * 0.6:
results['rating'] = '良好'
elif total >= max_possible * 0.4:
results['rating'] = '合格'
else:
results['rating'] = '不合格'
return results
def format_report(result):
"""格式化输出2003版评测报告"""
lines = []
lines.append("=" * 60)
lines.append(" 国民体质测定标准(2003版)— 评测报告")
lines.append("=" * 60)
lines.append(f" 年龄:{result['age']}岁 性别:{result['gender']}")
lines.append(f" 年龄组:{result['age_group']} 评分制:5分制")
lines.append("-" * 60)
lines.append(f" {'指标':<18} {'值':<12} {'得分':<6}")
lines.append("-" * 60)
for name, info in result['indicators'].items():
val = info.get('value', '')
if isinstance(val, (int, float)):
val_str = f"{val:.1f}"
else:
val_str = str(val)
note = info.get('note', '')
score_str = f"{info['score']}" + ("*" if note else "")
lines.append(f" {name:<18} {val_str:<12} {score_str:<6}")
lines.append("-" * 60)
lines.append(f" 总分:{result['total_score']}")
lines.append(f" 评级:{result['rating']}")
lines.append("=" * 60)
return '\n'.join(lines)
# ============================================================
# 主程序
# ============================================================
def main():
if len(sys.argv) > 1 and sys.argv[1] in ('--extract', '-e'):
# 只提取数据
tables = extract_all_2003_data()
print(f"\n共提取 {len(tables)} 个评分表")
return
# 评测模式
engine = FitnessEval2003()
if len(sys.argv) > 1 and sys.argv[1] in ('--demo', '-d'):
print("=" * 60)
print(" 演示1:成年男性,35岁,身高170cm,体重70kg")
print("=" * 60)
r = engine.evaluate(35, '男', 170, 70, {
'肺活量(ml)': 3500, '台阶指数': 58,
'握力(kg)': 42, '俯卧撑(次)': 20,
'纵跳(cm)': 32, '坐位体前屈(cm)': 8,
'选择反应时(秒)': 0.42, '闭眼单脚站立(秒)': 35
})
print(format_report(r))
print()
print("=" * 60)
print(" 演示2:成年女性,25岁,身高163cm,体重52kg")
print("=" * 60)
r2 = engine.evaluate(25, '女', 163, 52, {
'肺活量(ml)': 2800, '台阶指数': 60,
'握力(kg)': 25, '仰卧起坐(次/分)': 18,
'纵跳(cm)': 22, '坐位体前屈(cm)': 12,
'选择反应时(秒)': 0.45, '闭眼单脚站立(秒)': 40
})
print(format_report(r2))
print()
print("=" * 60)
print(" 演示3:成年男性,50岁,身高175cm,体重80kg")
print("=" * 60)
r3 = engine.evaluate(50, '男', 175, 80, {
'肺活量(ml)': 3000, '台阶指数': 55,
'握力(kg)': 38,
'坐位体前屈(cm)': 5,
'选择反应时(秒)': 0.55, '闭眼单脚站立(秒)': 25
})
print(format_report(r3))
else:
print("2003版国民体质测定标准评测引擎")
print("用法:")
print(" python3 fitness_eval_2003.py -d 运行演示")
print(" python3 fitness_eval_2003.py -e 仅提取数据")
print()
print("Python调用:")
print(" from fitness_eval_2003 import FitnessEval2003, format_report")
print(" engine = FitnessEval2003()")
print(' r = engine.evaluate(35, "男", 170, 70, {...})')
print(" print(format_report(r))")
if __name__ == '__main__':
main()