#!/usr/bin/env python3 """ 2003版国民体质测定标准 — 评分数据提取脚本 从OCR文档的HTML表格中提取所有评分数据,输出为独立JSON """ import json, os, re, sys from bs4 import BeautifulSoup BASE = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测' DOC_PATH = f'{BASE}/01_来源数据/2003版/《国民体质测定标准(2003年)》(成年人部分).md' OUT_PATH = f'{BASE}/02_加工数据/2003版/评分标准结构化数据.json' def parse_range(val): val = val.strip().replace('<', '<').replace('>', '>') if re.match(r'[≥>]=?\s*(-?[\d.]+)', val): m = re.match(r'[≥>]=?\s*(-?[\d.]+)', val) return {"min": float(m.group(1))} if re.match(r'[≤<]=?\s*(-?[\d.]+)', val): m = re.match(r'[≤<]=?\s*(-?[\d.]+)', val) return {"max": float(m.group(1))} if re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val): m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', val) lo, hi = float(m.group(1)), float(m.group(2)) return {"min": min(lo, hi), "max": max(lo, hi)} if re.match(r'^(-?[\d.]+)$', val): m = re.match(r'^(-?[\d.]+)$', val) return {"exact": float(m.group(1))} return {"raw": val} def classify_table(rows): if not rows or len(rows) < 2: return None header = rows[0] hdr = ' '.join(str(h) for h in header) is_hw = ('身高' in hdr and '体重' in hdr) or ('身高段' in hdr) if not is_hw and len(header) == 6: if re.match(r'\d+\.\d+[-~–]\d+\.\d+', str(header[0])): is_hw = True is_ind = '性别' in hdr and '年龄' in hdr and '1分' in hdr is_cont = False if not is_hw and not is_ind and len(header) >= 7: if re.match(r'\d+[-~–]\d+岁', str(header[0])) and str(header[1]) in ['男','女']: is_cont = True if is_hw: return parse_hw(rows) elif is_ind: return ('indicator', parse_indicator(rows)) elif is_cont: return ('indicator_cont', parse_indicator(rows)) return None def parse_hw(rows): first, second = rows[0], rows[1] if len(rows) > 1 else [] has_scores = any('分' in str(c) for c in second) if len(first) <= 2 and has_scores: score_row, start = second, 2 else: score_row, start = first, 1 score_cols = [] for h in score_row: m = re.search(r'(\d+)分', str(h)) score_cols.append(int(m.group(1)) if m else 0) data = {} for row in rows[start:]: if len(row) < 2: continue hk = row[0].strip() ranges = [] for i in range(1, min(len(row), len(score_cols) + 1)): if i - 1 < len(score_cols): r = parse_range(row[i]) r['score'] = score_cols[i - 1] ranges.append(r) if ranges: data[hk] = ranges return ('height_weight', {'score_columns': score_cols, 'data': data}) def parse_indicator(rows): header = rows[0] hdr_str = ' '.join(str(h) for h in header) has_hdr = '年龄' in hdr_str and '性别' in hdr_str and '1分' in hdr_str if has_hdr: score_cols = [] for h in header[2:]: m = re.search(r'(\d+)分', str(h)) score_cols.append(int(m.group(1)) if m else 0) start = 1 else: score_cols = [1, 2, 3, 4, 5] start = 0 data = {} for row in rows[start:]: if len(row) < 4: continue age, gender = row[0].strip(), row[1].strip() data.setdefault(age, {}) ranges = [] for i, val in enumerate(row[2:2+len(score_cols)]): if i < len(score_cols): r = parse_range(val) r['score'] = score_cols[i] ranges.append(r) data[age][gender] = ranges return {'score_columns': score_cols, 'data': data} def extract_with_context(filepath): with open(filepath, 'r', encoding='utf-8') as f: content = f.read() table_positions = [m.start() for m in re.finditer(r'