213 lines
7.6 KiB
Python
213 lines
7.6 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
国家学生体质健康标准(2014年修订)— 评分数据提取脚本
|
||
教育部发布,适用于全日制小学至大学学生
|
||
满分120分制(100基础分 + 20附加分)
|
||
"""
|
||
import json, os, re, sys
|
||
from docx import Document
|
||
|
||
BASE = '/home/songyi/Documents/ai_agent_scraper_study/data/国家学生体质健康标准'
|
||
DOCX = os.path.join(BASE, '01_来源数据/国家学生体质健康标准(2014年修订).docx')
|
||
|
||
|
||
def parse_range(val):
|
||
"""解析值域:'13.5~18.1', '≤13.4', '≥17.9'"""
|
||
val = val.strip().replace(' ', '').replace('\n','')
|
||
if re.match(r'[≥>]=?\s*([\d.]+)', val):
|
||
m = re.match(r'[≥>]=?\s*([\d.]+)', val)
|
||
return {"min": float(m.group(1))}
|
||
if re.match(r'[≤<]=?\s*([\d.]+)', val):
|
||
m = re.match(r'[≤<]=?\s*([\d.]+)', val)
|
||
return {"max": float(m.group(1))}
|
||
if re.match(r'([\d.]+)\s*[~-]\s*([\d.]+)', val):
|
||
m = re.match(r'([\d.]+)\s*[~-]\s*([\d.]+)', val)
|
||
lo, hi = float(m.group(1)), float(m.group(2))
|
||
return {"min": min(lo, hi), "max": max(lo, hi)}
|
||
if re.match(r'(\d+)[::·]\s*(\d+[\']?[\d.]*)', val):
|
||
m = re.match(r'(\d+)[::·]\s*(\d+[\']?[\d.]*)', val)
|
||
minutes = int(m.group(1))
|
||
rest = m.group(2).replace("'", ".")
|
||
seconds = float(rest) if '.' in rest else int(rest)
|
||
return {"exact": minutes * 60 + seconds}
|
||
if re.match(r'^([\d.]+)$', val):
|
||
m = re.match(r'^([\d.]+)$', val)
|
||
return {"exact": float(m.group(1))}
|
||
return {"raw": val}
|
||
|
||
|
||
def extract_all():
|
||
doc = Document(DOCX)
|
||
|
||
result = {
|
||
'metadata': {
|
||
'title': '国家学生体质健康标准(2014年修订)',
|
||
'publisher': '教育部',
|
||
'year': 2014,
|
||
'total_tables': len(doc.tables),
|
||
'scoring_system': '100分基础 + 20分附加 = 120分满分',
|
||
'rating_levels': {
|
||
'优秀': {'min': 90.0},
|
||
'良好': {'min': 80.0, 'max': 89.9},
|
||
'及格': {'min': 60.0, 'max': 79.9},
|
||
'不及格': {'max': 59.9}
|
||
}
|
||
},
|
||
'weight_table': {},
|
||
'scoring_tables': [],
|
||
'bonus_tables': [],
|
||
'registration_tables': []
|
||
}
|
||
|
||
for i, table in enumerate(doc.tables):
|
||
rows_data = []
|
||
for row in table.rows:
|
||
cells = [cell.text.strip().replace('\n', ' ') for cell in row.cells]
|
||
rows_data.append(cells)
|
||
|
||
if not rows_data:
|
||
continue
|
||
|
||
header = rows_data[0]
|
||
header_str = ' '.join(str(h) for h in header).strip()
|
||
|
||
# 权重表(表0)
|
||
if i == 0:
|
||
for row in rows_data[1:]:
|
||
if len(row) >= 3:
|
||
obj = row[0].strip()
|
||
indicator = row[1].strip()
|
||
weight = row[2].strip()
|
||
try:
|
||
w = float(weight)
|
||
result['weight_table'].setdefault(obj, {})[indicator] = w
|
||
except ValueError:
|
||
pass
|
||
continue
|
||
|
||
# 登记卡表
|
||
if '学校签章' in header_str or ('姓名' in header_str and '性别' in header_str):
|
||
result['registration_tables'].append({
|
||
'table_index': i, 'rows': len(rows_data), 'cols': len(header)
|
||
})
|
||
continue
|
||
|
||
# 判断加分表
|
||
is_bonus = '加分' in header_str and '成绩' not in header_str
|
||
|
||
# 解析年级列
|
||
grade_cols = []
|
||
start_col = 1 if is_bonus else 2
|
||
for ci, h in enumerate(header):
|
||
h_clean = h.strip().replace('\n', ' ')
|
||
if h_clean and ci >= start_col:
|
||
# 跳过'单项得分'列
|
||
if '单项' in h_clean or '得分' in h_clean:
|
||
continue
|
||
grade_cols.append((ci, h_clean))
|
||
|
||
if not grade_cols:
|
||
continue
|
||
|
||
table_info = {
|
||
'table_index': i,
|
||
'header_sample': [h.strip() for h in header[:5]],
|
||
'grade_names': [g[1] for g in grade_cols],
|
||
'type': 'bonus' if is_bonus else 'standard',
|
||
'rows_count': len(rows_data) - 1,
|
||
'data': []
|
||
}
|
||
|
||
for row in rows_data[1:]:
|
||
if is_bonus:
|
||
if len(row) < 2:
|
||
continue
|
||
bonus_str = row[0].strip()
|
||
try:
|
||
bonus_score = int(bonus_str) if bonus_str.lstrip('-').isdigit() else 0
|
||
except:
|
||
continue
|
||
entry = {'bonus_score': bonus_score, 'grades': {}}
|
||
for ci, gname in grade_cols:
|
||
if ci < len(row):
|
||
val = row[ci].strip()
|
||
if val and val not in ['—', '-', '', '------']:
|
||
entry['grades'][gname] = parse_range(val)
|
||
if entry['grades']:
|
||
table_info['data'].append(entry)
|
||
else:
|
||
if len(row) < 3:
|
||
continue
|
||
level_name = row[0].strip()
|
||
score_str = row[1].strip()
|
||
try:
|
||
score_val = int(score_str) if score_str.isdigit() else 0
|
||
except:
|
||
continue
|
||
entry = {'level': level_name, 'score': score_val, 'grades': {}}
|
||
for ci, gname in grade_cols:
|
||
if ci < len(row):
|
||
val = row[ci].strip()
|
||
if val and val not in ['—', '-', '', '------']:
|
||
entry['grades'][gname] = parse_range(val)
|
||
if entry['grades']:
|
||
table_info['data'].append(entry)
|
||
|
||
if table_info['data']:
|
||
if is_bonus:
|
||
result['bonus_tables'].append(table_info)
|
||
else:
|
||
result['scoring_tables'].append(table_info)
|
||
|
||
return result
|
||
|
||
|
||
def print_summary(data):
|
||
m = data['metadata']
|
||
print(f"总Word表格: {m['total_tables']}")
|
||
print(f"标准评分表: {len(data['scoring_tables'])} 个")
|
||
print(f"加分评分表: {len(data['bonus_tables'])} 个")
|
||
print(f"登记卡/附表: {len(data['registration_tables'])} 个")
|
||
|
||
wt = data['weight_table']
|
||
print(f"\n权重表 ({len(wt)}组):")
|
||
for obj in sorted(wt.keys()):
|
||
items = ', '.join(f"{k}={v}%" for k, v in sorted(wt[obj].items()))
|
||
print(f" {obj}: {items}")
|
||
|
||
print(f"\n标准评分表:")
|
||
for t in data['scoring_tables']:
|
||
gn = t['grade_names']
|
||
gs = ', '.join(gn[:3])
|
||
if len(gn) > 3:
|
||
gs += f"...({len(gn)}个年级)"
|
||
levels = sorted(set(e['level'] for e in t['data']))
|
||
ti = t['table_index']
|
||
rc = len(t['data'])
|
||
print(f" [表{ti}] {rc}行x{len(gn)}年级 等级={levels} 年级={gs}")
|
||
|
||
print(f"\n加分评分表:")
|
||
for t in data['bonus_tables']:
|
||
gn = t['grade_names']
|
||
gs = ', '.join(gn[:3])
|
||
if len(gn) > 3:
|
||
gs += f"...({len(gn)}个年级)"
|
||
ti = t['table_index']
|
||
rc = len(t['data'])
|
||
print(f" [表{ti}] {rc}行 年级={gs}")
|
||
|
||
|
||
def output_json(data):
|
||
path = os.path.join(BASE, '02_加工数据/评分标准结构化数据.json')
|
||
with open(path, 'w', encoding='utf-8') as f:
|
||
json.dump(data, f, ensure_ascii=False, indent=2)
|
||
size = os.path.getsize(path)
|
||
print(f"\nJSON输出: {path} ({size/1024:.0f}KB)")
|
||
return path
|
||
|
||
|
||
if __name__ == '__main__':
|
||
data = extract_all()
|
||
print_summary(data)
|
||
output_json(data)
|