Files

242 lines
8.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
2003版国民体质测定标准 — 评分数据提取脚本
从OCR文档的HTML表格中提取所有评分数据,输出为独立JSON
"""
import json, os, re, sys
from bs4 import BeautifulSoup
BASE = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
DOC_PATH = f'{BASE}/01_来源数据/2003版/《国民体质测定标准(2003年)》(成年人部分).md'
OUT_PATH = f'{BASE}/02_加工数据/2003版/评分标准结构化数据.json'
def parse_range(val):
val = val.strip().replace('&lt;', '<').replace('&gt;', '>')
if re.match(r'[≥>]=?\s*(-?[\d.]+)', val):
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', val)
return {"min": float(m.group(1))}
if re.match(r'[≤<]=?\s*(-?[\d.]+)', val):
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', val)
return {"max": float(m.group(1))}
if re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val):
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val)
lo, hi = float(m.group(1)), float(m.group(2))
return {"min": min(lo, hi), "max": max(lo, hi)}
if re.match(r'^(-?[\d.]+)$', val):
m = re.match(r'^(-?[\d.]+)$', val)
return {"exact": float(m.group(1))}
return {"raw": val}
def classify_table(rows):
if not rows or len(rows) < 2:
return None
header = rows[0]
hdr = ' '.join(str(h) for h in header)
is_hw = ('身高' in hdr and '体重' in hdr) or ('身高段' in hdr)
if not is_hw and len(header) == 6:
if re.match(r'\d+\.\d+[-~–]\d+\.\d+', str(header[0])):
is_hw = True
is_ind = '性别' in hdr and '年龄' in hdr and '1分' in hdr
is_cont = False
if not is_hw and not is_ind and len(header) >= 7:
if re.match(r'\d+[-~–]\d+岁', str(header[0])) and str(header[1]) in ['男','女']:
is_cont = True
if is_hw:
return parse_hw(rows)
elif is_ind:
return ('indicator', parse_indicator(rows))
elif is_cont:
return ('indicator_cont', parse_indicator(rows))
return None
def parse_hw(rows):
first, second = rows[0], rows[1] if len(rows) > 1 else []
has_scores = any('分' in str(c) for c in second)
if len(first) <= 2 and has_scores:
score_row, start = second, 2
else:
score_row, start = first, 1
score_cols = []
for h in score_row:
m = re.search(r'(\d+)分', str(h))
score_cols.append(int(m.group(1)) if m else 0)
data = {}
for row in rows[start:]:
if len(row) < 2:
continue
hk = row[0].strip()
ranges = []
for i in range(1, min(len(row), len(score_cols) + 1)):
if i - 1 < len(score_cols):
r = parse_range(row[i])
r['score'] = score_cols[i - 1]
ranges.append(r)
if ranges:
data[hk] = ranges
return ('height_weight', {'score_columns': score_cols, 'data': data})
def parse_indicator(rows):
header = rows[0]
hdr_str = ' '.join(str(h) for h in header)
has_hdr = '年龄' in hdr_str and '性别' in hdr_str and '1分' in hdr_str
if has_hdr:
score_cols = []
for h in header[2:]:
m = re.search(r'(\d+)分', str(h))
score_cols.append(int(m.group(1)) if m else 0)
start = 1
else:
score_cols = [1, 2, 3, 4, 5]
start = 0
data = {}
for row in rows[start:]:
if len(row) < 4:
continue
age, gender = row[0].strip(), row[1].strip()
data.setdefault(age, {})
ranges = []
for i, val in enumerate(row[2:2+len(score_cols)]):
if i < len(score_cols):
r = parse_range(val)
r['score'] = score_cols[i]
ranges.append(r)
data[age][gender] = ranges
return {'score_columns': score_cols, 'data': data}
def extract_with_context(filepath):
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
table_positions = [m.start() for m in re.finditer(r'<table\b', content)]
soup = BeautifulSoup(content, 'html.parser')
raw_tables = []
contexts = []
for i, table_tag in enumerate(soup.find_all('table')):
rows = []
for tr in table_tag.find_all('tr'):
cells = [td.get_text(strip=True) for td in tr.find_all(['td', 'th'])]
if cells:
rows.append(cells)
if rows:
raw_tables.append(rows)
pos = table_positions[i] if i < len(table_positions) else 0
contexts.append(content[max(0, pos - 400):pos])
return raw_tables, contexts
DETECT = [
('肺活量', '肺活量'), ('台阶', '台阶指数'), ('握力', '握力'),
('俯卧撑', '俯卧撑'), ('仰卧起坐', '仰卧起坐'), ('纵跳', '纵跳'),
('体前屈', '坐位体前屈'), ('反应时', '选择反应时'), ('单脚站立', '闭眼单脚站立'),
]
def detect_indicator(ctx):
for kw, name in DETECT:
if kw in ctx:
return name
return None
def main():
raw_tables, contexts = extract_with_context(DOC_PATH)
print(f"提取到 {len(raw_tables)} 个HTML表格")
result = {
'metadata': {
'title': '国民体质测定标准(2003年)(成年人部分)',
'publisher': '国家体育总局',
'year': 2003,
'scoring_system': '5分制(1-5分)',
'total_tables': len(raw_tables),
'rating_levels': None
},
'height_weight_tables': {},
'indicator_tables': {}
}
last_indicator = None
last_hw = None
for idx, rows in enumerate(raw_tables):
ctx = contexts[idx] if idx < len(contexts) else ''
cls = classify_table(rows)
if cls is None:
continue
if cls[0] == 'height_weight':
hw_data = cls[1]
first_cell = str(rows[0][0]) if rows[0] else ''
is_header = '身高段' in first_cell or '身高' in first_cell
if is_header:
age_m = re.search(r'(\d+)[-~–—]\s*(\d+)\s*岁', ctx)
gender = '男' if '男' in ctx else '女'
if age_m:
key = f"{age_m.group(1)}-{age_m.group(2)}岁_{gender}"
last_hw = key
else:
key = f"table_{idx}"
last_hw = key
else:
key = last_hw or f"table_{idx}"
if key in result['height_weight_tables']:
result['height_weight_tables'][key]['data'].update(hw_data['data'])
else:
result['height_weight_tables'][key] = hw_data
elif cls[0] in ('indicator', 'indicator_cont'):
ind_data = cls[1]
is_cont = cls[0] == 'indicator_cont'
if is_cont and last_indicator:
name = last_indicator
else:
name = detect_indicator(ctx)
if name:
last_indicator = name
if not name:
continue
if name not in result['indicator_tables']:
result['indicator_tables'][name] = ind_data
else:
for age, genders in ind_data['data'].items():
if age not in result['indicator_tables'][name]['data']:
result['indicator_tables'][name]['data'][age] = {}
for g, ranges in genders.items():
result['indicator_tables'][name]['data'][age][g] = ranges
# 输出统计
print(f"身高体重表: {len(result['height_weight_tables'])} 张")
for k, v in sorted(result['height_weight_tables'].items()):
d = v.get('data', {})
keys = sorted(d.keys())
print(f" {k}: {len(keys)}个身高段 ({keys[0] if keys else '?'} ~ {keys[-1] if keys else '?'})")
print(f"\n指标评分表: {len(result['indicator_tables'])} 项")
for name in sorted(result['indicator_tables'].keys()):
t = result['indicator_tables'][name]
ages = sorted(t['data'].keys())
print(f" {name}: {len(ages)}个年龄组 ({ages[0]}~{ages[-1]})")
# 输出JSON
with open(OUT_PATH, 'w', encoding='utf-8') as f:
json.dump(result, f, ensure_ascii=False, indent=2)
size = os.path.getsize(OUT_PATH)
print(f"\nJSON输出: {OUT_PATH} ({size/1024:.0f}KB)")
if __name__ == '__main__':
main()