242 lines
8.3 KiB
Python
242 lines
8.3 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
2003版国民体质测定标准 — 评分数据提取脚本
|
||
从OCR文档的HTML表格中提取所有评分数据,输出为独立JSON
|
||
"""
|
||
import json, os, re, sys
|
||
from bs4 import BeautifulSoup
|
||
|
||
BASE = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
|
||
DOC_PATH = f'{BASE}/01_来源数据/2003版/《国民体质测定标准(2003年)》(成年人部分).md'
|
||
OUT_PATH = f'{BASE}/02_加工数据/2003版/评分标准结构化数据.json'
|
||
|
||
|
||
def parse_range(val):
|
||
val = val.strip().replace('<', '<').replace('>', '>')
|
||
if re.match(r'[≥>]=?\s*(-?[\d.]+)', val):
|
||
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', val)
|
||
return {"min": float(m.group(1))}
|
||
if re.match(r'[≤<]=?\s*(-?[\d.]+)', val):
|
||
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', val)
|
||
return {"max": float(m.group(1))}
|
||
if re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val):
|
||
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val)
|
||
lo, hi = float(m.group(1)), float(m.group(2))
|
||
return {"min": min(lo, hi), "max": max(lo, hi)}
|
||
if re.match(r'^(-?[\d.]+)$', val):
|
||
m = re.match(r'^(-?[\d.]+)$', val)
|
||
return {"exact": float(m.group(1))}
|
||
return {"raw": val}
|
||
|
||
|
||
def classify_table(rows):
|
||
if not rows or len(rows) < 2:
|
||
return None
|
||
header = rows[0]
|
||
hdr = ' '.join(str(h) for h in header)
|
||
|
||
is_hw = ('身高' in hdr and '体重' in hdr) or ('身高段' in hdr)
|
||
if not is_hw and len(header) == 6:
|
||
if re.match(r'\d+\.\d+[-~–]\d+\.\d+', str(header[0])):
|
||
is_hw = True
|
||
|
||
is_ind = '性别' in hdr and '年龄' in hdr and '1分' in hdr
|
||
|
||
is_cont = False
|
||
if not is_hw and not is_ind and len(header) >= 7:
|
||
if re.match(r'\d+[-~–]\d+岁', str(header[0])) and str(header[1]) in ['男','女']:
|
||
is_cont = True
|
||
|
||
if is_hw:
|
||
return parse_hw(rows)
|
||
elif is_ind:
|
||
return ('indicator', parse_indicator(rows))
|
||
elif is_cont:
|
||
return ('indicator_cont', parse_indicator(rows))
|
||
return None
|
||
|
||
|
||
def parse_hw(rows):
|
||
first, second = rows[0], rows[1] if len(rows) > 1 else []
|
||
has_scores = any('分' in str(c) for c in second)
|
||
if len(first) <= 2 and has_scores:
|
||
score_row, start = second, 2
|
||
else:
|
||
score_row, start = first, 1
|
||
score_cols = []
|
||
for h in score_row:
|
||
m = re.search(r'(\d+)分', str(h))
|
||
score_cols.append(int(m.group(1)) if m else 0)
|
||
data = {}
|
||
for row in rows[start:]:
|
||
if len(row) < 2:
|
||
continue
|
||
hk = row[0].strip()
|
||
ranges = []
|
||
for i in range(1, min(len(row), len(score_cols) + 1)):
|
||
if i - 1 < len(score_cols):
|
||
r = parse_range(row[i])
|
||
r['score'] = score_cols[i - 1]
|
||
ranges.append(r)
|
||
if ranges:
|
||
data[hk] = ranges
|
||
return ('height_weight', {'score_columns': score_cols, 'data': data})
|
||
|
||
|
||
def parse_indicator(rows):
|
||
header = rows[0]
|
||
hdr_str = ' '.join(str(h) for h in header)
|
||
has_hdr = '年龄' in hdr_str and '性别' in hdr_str and '1分' in hdr_str
|
||
if has_hdr:
|
||
score_cols = []
|
||
for h in header[2:]:
|
||
m = re.search(r'(\d+)分', str(h))
|
||
score_cols.append(int(m.group(1)) if m else 0)
|
||
start = 1
|
||
else:
|
||
score_cols = [1, 2, 3, 4, 5]
|
||
start = 0
|
||
data = {}
|
||
for row in rows[start:]:
|
||
if len(row) < 4:
|
||
continue
|
||
age, gender = row[0].strip(), row[1].strip()
|
||
data.setdefault(age, {})
|
||
ranges = []
|
||
for i, val in enumerate(row[2:2+len(score_cols)]):
|
||
if i < len(score_cols):
|
||
r = parse_range(val)
|
||
r['score'] = score_cols[i]
|
||
ranges.append(r)
|
||
data[age][gender] = ranges
|
||
return {'score_columns': score_cols, 'data': data}
|
||
|
||
|
||
def extract_with_context(filepath):
|
||
with open(filepath, 'r', encoding='utf-8') as f:
|
||
content = f.read()
|
||
table_positions = [m.start() for m in re.finditer(r'<table\b', content)]
|
||
soup = BeautifulSoup(content, 'html.parser')
|
||
raw_tables = []
|
||
contexts = []
|
||
for i, table_tag in enumerate(soup.find_all('table')):
|
||
rows = []
|
||
for tr in table_tag.find_all('tr'):
|
||
cells = [td.get_text(strip=True) for td in tr.find_all(['td', 'th'])]
|
||
if cells:
|
||
rows.append(cells)
|
||
if rows:
|
||
raw_tables.append(rows)
|
||
pos = table_positions[i] if i < len(table_positions) else 0
|
||
contexts.append(content[max(0, pos - 400):pos])
|
||
return raw_tables, contexts
|
||
|
||
|
||
DETECT = [
|
||
('肺活量', '肺活量'), ('台阶', '台阶指数'), ('握力', '握力'),
|
||
('俯卧撑', '俯卧撑'), ('仰卧起坐', '仰卧起坐'), ('纵跳', '纵跳'),
|
||
('体前屈', '坐位体前屈'), ('反应时', '选择反应时'), ('单脚站立', '闭眼单脚站立'),
|
||
]
|
||
|
||
|
||
def detect_indicator(ctx):
|
||
for kw, name in DETECT:
|
||
if kw in ctx:
|
||
return name
|
||
return None
|
||
|
||
|
||
def main():
|
||
raw_tables, contexts = extract_with_context(DOC_PATH)
|
||
print(f"提取到 {len(raw_tables)} 个HTML表格")
|
||
|
||
result = {
|
||
'metadata': {
|
||
'title': '国民体质测定标准(2003年)(成年人部分)',
|
||
'publisher': '国家体育总局',
|
||
'year': 2003,
|
||
'scoring_system': '5分制(1-5分)',
|
||
'total_tables': len(raw_tables),
|
||
'rating_levels': None
|
||
},
|
||
'height_weight_tables': {},
|
||
'indicator_tables': {}
|
||
}
|
||
|
||
last_indicator = None
|
||
last_hw = None
|
||
|
||
for idx, rows in enumerate(raw_tables):
|
||
ctx = contexts[idx] if idx < len(contexts) else ''
|
||
cls = classify_table(rows)
|
||
if cls is None:
|
||
continue
|
||
|
||
if cls[0] == 'height_weight':
|
||
hw_data = cls[1]
|
||
first_cell = str(rows[0][0]) if rows[0] else ''
|
||
is_header = '身高段' in first_cell or '身高' in first_cell
|
||
|
||
if is_header:
|
||
age_m = re.search(r'(\d+)[-~–—]\s*(\d+)\s*岁', ctx)
|
||
gender = '男' if '男' in ctx else '女'
|
||
if age_m:
|
||
key = f"{age_m.group(1)}-{age_m.group(2)}岁_{gender}"
|
||
last_hw = key
|
||
else:
|
||
key = f"table_{idx}"
|
||
last_hw = key
|
||
else:
|
||
key = last_hw or f"table_{idx}"
|
||
|
||
if key in result['height_weight_tables']:
|
||
result['height_weight_tables'][key]['data'].update(hw_data['data'])
|
||
else:
|
||
result['height_weight_tables'][key] = hw_data
|
||
|
||
elif cls[0] in ('indicator', 'indicator_cont'):
|
||
ind_data = cls[1]
|
||
is_cont = cls[0] == 'indicator_cont'
|
||
|
||
if is_cont and last_indicator:
|
||
name = last_indicator
|
||
else:
|
||
name = detect_indicator(ctx)
|
||
if name:
|
||
last_indicator = name
|
||
|
||
if not name:
|
||
continue
|
||
|
||
if name not in result['indicator_tables']:
|
||
result['indicator_tables'][name] = ind_data
|
||
else:
|
||
for age, genders in ind_data['data'].items():
|
||
if age not in result['indicator_tables'][name]['data']:
|
||
result['indicator_tables'][name]['data'][age] = {}
|
||
for g, ranges in genders.items():
|
||
result['indicator_tables'][name]['data'][age][g] = ranges
|
||
|
||
# 输出统计
|
||
print(f"身高体重表: {len(result['height_weight_tables'])} 张")
|
||
for k, v in sorted(result['height_weight_tables'].items()):
|
||
d = v.get('data', {})
|
||
keys = sorted(d.keys())
|
||
print(f" {k}: {len(keys)}个身高段 ({keys[0] if keys else '?'} ~ {keys[-1] if keys else '?'})")
|
||
|
||
print(f"\n指标评分表: {len(result['indicator_tables'])} 项")
|
||
for name in sorted(result['indicator_tables'].keys()):
|
||
t = result['indicator_tables'][name]
|
||
ages = sorted(t['data'].keys())
|
||
print(f" {name}: {len(ages)}个年龄组 ({ages[0]}~{ages[-1]})")
|
||
|
||
# 输出JSON
|
||
with open(OUT_PATH, 'w', encoding='utf-8') as f:
|
||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||
size = os.path.getsize(OUT_PATH)
|
||
print(f"\nJSON输出: {OUT_PATH} ({size/1024:.0f}KB)")
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|