初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)
This commit is contained in:
commit
80ae3811cf
7712 files changed
+4628547
No files matched your search
@@ -0,0 +1,511 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
《国民体质测定标准(2023年修订)》评分表解析脚本
|
||||
提取所有HTML内联表格为结构化JSON,按年龄段/性别/指标组织
|
||||
"""
|
||||
import re
|
||||
import json
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
def parse_value(val):
|
||||
"""解析单元格值,返回 (min, max, operator) 或原始字符串"""
|
||||
val = val.strip()
|
||||
# 清理 HTML 实体
|
||||
val = val.replace('<', '<').replace('>', '>').replace('&', '&')
|
||||
|
||||
# 处理含BMI/体脂率等文本的值域表达式: "18.5≤BMI<24.0", "BMI≥28.0", "BMI<18.5"
|
||||
# 去除 BMI 等干扰文本
|
||||
clean_val = re.sub(r'[A-Za-z\u4e00-\u9fff]+', '', val).strip()
|
||||
|
||||
# 处理 LaTeX 公式
|
||||
if '$' in val:
|
||||
# 去掉$和反斜杠,统一符号
|
||||
clean = val.replace('$', ' ').replace('\\', ' ').replace('\\\\', ' ')
|
||||
clean = clean.replace('<', '<').replace('>', '>').replace('&', '&')
|
||||
# 统一LaTeX命令为符号(长命令在前,避免被短命令截断)
|
||||
clean = clean.replace('geq', '≥').replace('leq', '≤').replace('ge', '≥').replace('le', '≤')
|
||||
clean = clean.replace('gt', '>').replace('lt', '<')
|
||||
# "X分 ≤ a < Y分"
|
||||
m = re.search(r'(\d+)\s*分?\s*[≤<]\s*a\s*[<]\s*(\d+)', clean)
|
||||
if m:
|
||||
return {"min": float(m.group(1)), "max": float(m.group(2))}
|
||||
# a ≥ X
|
||||
m = re.search(r'a\s*[≥>]\s*(\d+)', clean)
|
||||
if m:
|
||||
return {"min": float(m.group(1))}
|
||||
# a < X
|
||||
m = re.search(r'a\s*[<]\s*(\d+)', clean)
|
||||
if m:
|
||||
return {"max": float(m.group(1))}
|
||||
|
||||
# 尝试用clean_val匹配标准模式(去除文本干扰后)
|
||||
for test_val in [clean_val, val]:
|
||||
# X ≤ Y < Z 或 X ≤ Y ≤ Z 或 X < Y < Z
|
||||
m = re.search(r'(-?[\d.]+)\s*[≤<]\s*[\d.]*\s*[≤<]\s*(-?[\d.]+)', test_val)
|
||||
if m:
|
||||
lo, hi = float(m.group(1)), float(m.group(2))
|
||||
return {"min": min(lo, hi), "max": max(lo, hi)}
|
||||
|
||||
# ≥X
|
||||
m = re.match(r'[≥>]=?\s*(-?[\d.]+)', test_val)
|
||||
if m:
|
||||
return {"min": float(m.group(1))}
|
||||
|
||||
# ≤X
|
||||
m = re.match(r'[≤<]=?\s*(-?[\d.]+)', test_val)
|
||||
if m:
|
||||
return {"max": float(m.group(1))}
|
||||
|
||||
# X-Y range — 处理正负值(如 -14.9--12.5 或 3.5-7.2)
|
||||
m = re.match(r'(-?[\d.]+)\s*[-~–]\s*(-?[\d.]+)', test_val)
|
||||
if m:
|
||||
lo, hi = float(m.group(1)), float(m.group(2))
|
||||
return {"min": min(lo, hi), "max": max(lo, hi)}
|
||||
|
||||
# standalone number
|
||||
m = re.match(r'^([\d.]+)$', test_val)
|
||||
if m:
|
||||
return {"exact": float(m.group(1))}
|
||||
|
||||
# <X 或 <-X
|
||||
m = re.match(r'<\s*(-?[\d.]+)', test_val)
|
||||
if m:
|
||||
return {"max": float(m.group(1))}
|
||||
|
||||
# >X 或 >-X
|
||||
m = re.match(r'>\s*(-?[\d.]+)', test_val)
|
||||
if m:
|
||||
return {"min": float(m.group(1))}
|
||||
|
||||
return {"raw": val.strip()}
|
||||
|
||||
|
||||
def extract_tables(md_content):
|
||||
"""从Markdown内容中提取所有表格及其前面的表名"""
|
||||
# 用BeautifulSoup解析HTML表格
|
||||
# 先找到所有 <div>表名</div> + <table> 对
|
||||
|
||||
results = []
|
||||
|
||||
# 按行处理,寻找表名div和table
|
||||
lines = md_content.split('\n')
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
|
||||
# 找表名 <div> 行
|
||||
m = re.search(r'<div[^>]*>(表\s*[\d.-]+[^<]*)</div>', line)
|
||||
if m:
|
||||
table_name = m.group(1).strip()
|
||||
table_name = re.sub(r'\s+', ' ', table_name)
|
||||
|
||||
# 寻找单位行
|
||||
unit = None
|
||||
for j in range(i+1, min(i+10, len(lines))):
|
||||
unit_m = re.match(r'单位[::]\s*(.+)', lines[j].strip())
|
||||
if unit_m:
|
||||
unit = unit_m.group(1).strip()
|
||||
break
|
||||
|
||||
# 查找接下来的全部<table>块(直到下一个表名div或文档结束)
|
||||
all_table_blocks = []
|
||||
j = i + 1
|
||||
while j < len(lines):
|
||||
if '<table' in lines[j]:
|
||||
# 收集完整的table HTML
|
||||
table_lines = []
|
||||
k = j
|
||||
while k < len(lines) and '</table>' not in lines[k]:
|
||||
table_lines.append(lines[k])
|
||||
k += 1
|
||||
if k < len(lines):
|
||||
table_lines.append(lines[k]) # 包含</table>的行
|
||||
all_table_blocks.append('\n'.join(table_lines))
|
||||
j = k + 1
|
||||
elif '<div' in lines[j] and '表' in lines[j] and j != i:
|
||||
# 下一个表名,停止
|
||||
break
|
||||
else:
|
||||
j += 1
|
||||
|
||||
if all_table_blocks:
|
||||
combined_html = '\n'.join(all_table_blocks)
|
||||
soup = BeautifulSoup(combined_html, 'html.parser')
|
||||
tables = soup.find_all('table')
|
||||
|
||||
if tables:
|
||||
all_rows = []
|
||||
for table_tag in tables:
|
||||
rows = table_tag.find_all('tr')
|
||||
for tr in rows:
|
||||
cells = []
|
||||
for td in tr.find_all(['td', 'th']):
|
||||
txt = td.get_text(strip=True)
|
||||
txt = txt.replace('<', '<').replace('>', '>').replace('&', '&')
|
||||
cells.append(txt)
|
||||
if cells:
|
||||
all_rows.append(cells)
|
||||
|
||||
if all_rows:
|
||||
results.append({
|
||||
'name': table_name,
|
||||
'unit': unit,
|
||||
'rows': all_rows,
|
||||
'combined_tables': len(tables)
|
||||
})
|
||||
|
||||
i = j # 跳到已处理的末尾
|
||||
continue
|
||||
|
||||
i += 1
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def classify_table(table):
|
||||
"""分类表格并解析为结构化JSON"""
|
||||
name = table['name']
|
||||
rows = table['rows']
|
||||
|
||||
if not rows:
|
||||
return None
|
||||
|
||||
# 基本信息
|
||||
info = {
|
||||
'name': name,
|
||||
'unit': table['unit'],
|
||||
'type': 'unknown',
|
||||
'parsed': {}
|
||||
}
|
||||
|
||||
# 提取表号
|
||||
m = re.match(r'表\s*([\d.-]+)\s*(.*)', name)
|
||||
if m:
|
||||
info['table_id'] = m.group(1).strip()
|
||||
info['title'] = m.group(2).strip()
|
||||
else:
|
||||
info['table_id'] = ''
|
||||
info['title'] = name
|
||||
|
||||
# 解析性别和指标
|
||||
title = info['title']
|
||||
if '男性' in title:
|
||||
info['gender'] = '男'
|
||||
info['gender_en'] = 'male'
|
||||
elif '女性' in title:
|
||||
info['gender'] = '女'
|
||||
info['gender_en'] = 'female'
|
||||
else:
|
||||
info['gender'] = '通用'
|
||||
info['gender_en'] = '通用' # 统一用中文"通用"而非"all"
|
||||
|
||||
# 判断年龄段 - 用更丰富的关键词
|
||||
section_keywords = {
|
||||
'幼儿': ['幼儿', '3岁', '3.5岁', '4岁', '5岁', '6岁', '36月', '37月', '38月', '39月', '40月'],
|
||||
'成年人': ['成年', '20-24', '25-29', '30-34', '35-39', '40-44', '45-49', '50-54', '55-59'],
|
||||
'老年人': ['老年', '60-64', '65-69', '70-74', '75-79']
|
||||
}
|
||||
age_ranges = {'幼儿': '3-6岁', '成年人': '20-59岁', '老年人': '60-79岁'}
|
||||
for sec, keywords in section_keywords.items():
|
||||
if any(k in name or k in title for k in keywords):
|
||||
info['section'] = sec
|
||||
info['age_range'] = age_ranges.get(sec, '')
|
||||
break
|
||||
else:
|
||||
# 用表格ID判断:1.x = 幼儿, 2.x = 成年人, 3.x = 老年人
|
||||
tid = info.get('table_id', '')
|
||||
if tid.startswith('1'):
|
||||
info['section'] = '幼儿'
|
||||
info['age_range'] = '3-6岁'
|
||||
elif tid.startswith('2'):
|
||||
info['section'] = '成年人'
|
||||
info['age_range'] = '20-59岁'
|
||||
elif tid.startswith('3'):
|
||||
info['section'] = '老年人'
|
||||
info['age_range'] = '60-79岁'
|
||||
|
||||
# 解析指标名称
|
||||
indicator_keywords = ['身高', '体重', 'BMI', '体脂率', '肺活量', '功率车', '握力', '纵跳',
|
||||
'俯卧撑', '跪卧撑', '仰卧起坐', '坐位体前屈', '立定跳远', '双脚连续跳',
|
||||
'绕障碍跑', '走平衡木', '闭眼单脚站立', '选择反应时', '高抬腿', '坐站']
|
||||
for kw in indicator_keywords:
|
||||
if kw in title:
|
||||
info['indicator'] = kw
|
||||
break
|
||||
else:
|
||||
info['indicator'] = title
|
||||
|
||||
# 判断表格类型
|
||||
header = rows[0] if rows else []
|
||||
|
||||
is_bmi_style = ('年龄' in str(header) or '年龄段' in str(header)) and len(header) <= 5
|
||||
is_weight_table = '权重' in name or ('权重' in str(rows[1] if len(rows) > 1 else ''))
|
||||
is_rating_table = '评级' in name or '等级' in name
|
||||
is_indicator_list = ('类别' in str(header) and '测试指标' in str(header)) or \
|
||||
('一级指标' in str(header) and '二级指标' in str(header) and '权重' not in name) or \
|
||||
(len(header) == 2 and any(k in str(header[1]) for k in ['测试指标', '指标', '20-49', '50-59', '60-64']))
|
||||
|
||||
if is_weight_table:
|
||||
info['type'] = 'weight_table'
|
||||
info['parsed'] = parse_weight_table(rows)
|
||||
elif is_indicator_list:
|
||||
info['type'] = 'indicator_list'
|
||||
info['parsed'] = parse_indicator_list(rows)
|
||||
elif is_rating_table:
|
||||
info['type'] = 'rating_levels'
|
||||
info['parsed'] = parse_rating_levels(rows)
|
||||
elif is_bmi_style:
|
||||
info['type'] = 'bmi_banded'
|
||||
info['parsed'] = parse_bmi_banded(rows, info)
|
||||
else:
|
||||
info['type'] = 'standard_scoring'
|
||||
info['parsed'] = parse_standard_scoring(rows)
|
||||
|
||||
return info
|
||||
|
||||
|
||||
def parse_indicator_list(rows):
|
||||
"""解析指标列表(如表1-1)"""
|
||||
items = []
|
||||
for row in rows[1:]: # 跳过表头
|
||||
if len(row) >= 2:
|
||||
items.append({
|
||||
'category': row[0],
|
||||
'indicator': row[1]
|
||||
})
|
||||
return {'items': items}
|
||||
|
||||
|
||||
def parse_weight_table(rows):
|
||||
"""解析权重表(如表1-2)- 需要处理 rowspan"""
|
||||
weights = {}
|
||||
# 记录上一行的类别值,用于 rowspan 填充
|
||||
prev_category = None
|
||||
for row in rows[1:]: # 跳过表头
|
||||
if len(row) >= 3:
|
||||
indicator = row[1].strip()
|
||||
weight = row[2].strip()
|
||||
category = row[0].strip() if row[0].strip() else prev_category
|
||||
prev_category = category
|
||||
try:
|
||||
weight_val = float(weight)
|
||||
weights[indicator] = weight_val
|
||||
except ValueError:
|
||||
pass
|
||||
elif len(row) == 2 and prev_category:
|
||||
# rowspan 行:只有指标和权重,类别从上一行继承
|
||||
indicator = row[0].strip()
|
||||
weight = row[1].strip()
|
||||
try:
|
||||
weight_val = float(weight)
|
||||
weights[indicator] = weight_val
|
||||
except ValueError:
|
||||
pass
|
||||
return {'weights': weights}
|
||||
|
||||
|
||||
def parse_rating_levels(rows):
|
||||
"""解析评级等级表(如表1-3)"""
|
||||
levels = {}
|
||||
for row in rows[1:]:
|
||||
if len(row) >= 2:
|
||||
level_name = row[0].strip()
|
||||
score_text = row[1].strip()
|
||||
parsed = parse_value(score_text)
|
||||
levels[level_name] = parsed
|
||||
return {'levels': levels}
|
||||
|
||||
|
||||
def parse_bmi_banded(rows, info):
|
||||
"""解析BMI分段评分表(如表1-6)"""
|
||||
header = rows[0]
|
||||
# 列名:年龄 | 60分 | 100分 | 60分 | 20分
|
||||
# 或:年龄段 | 40分 | 100分 | 60分 | 20分
|
||||
|
||||
# 解析列对应的分数
|
||||
score_cols = []
|
||||
for h in header[1:]:
|
||||
m = re.search(r'(\d+)\s*分', str(h))
|
||||
score_cols.append(int(m.group(1)) if m else None)
|
||||
|
||||
data = {}
|
||||
for row in rows[1:]:
|
||||
age_key = row[0].strip()
|
||||
ranges = []
|
||||
for i, val in enumerate(row[1:], 1):
|
||||
if i <= len(score_cols) and score_cols[i-1] is not None:
|
||||
parsed_range = parse_value(val)
|
||||
parsed_range['score'] = score_cols[i-1]
|
||||
ranges.append(parsed_range)
|
||||
data[age_key] = ranges
|
||||
|
||||
return {
|
||||
'score_categories': [{'col': h, 'score': s} for h, s in zip(header[1:], score_cols)],
|
||||
'age_groups': list(data.keys()),
|
||||
'data': data
|
||||
}
|
||||
|
||||
|
||||
def parse_standard_scoring(rows):
|
||||
"""解析标准评分表(行=分值,列=年龄组,如表1-4)"""
|
||||
header = rows[0]
|
||||
age_groups = [str(h).strip() for h in header[1:]]
|
||||
|
||||
score_bands = []
|
||||
for row in rows[1:]:
|
||||
score_text = row[0].strip()
|
||||
m = re.search(r'(\d+)分?', score_text)
|
||||
if not m:
|
||||
continue
|
||||
score = int(m.group(1))
|
||||
|
||||
ranges = {}
|
||||
for i, age in enumerate(age_groups):
|
||||
if i + 1 < len(row):
|
||||
ranges[age] = parse_value(row[i + 1])
|
||||
|
||||
score_bands.append({
|
||||
'score': score,
|
||||
'ranges': ranges
|
||||
})
|
||||
|
||||
return {
|
||||
'age_groups': age_groups,
|
||||
'score_bands': score_bands
|
||||
}
|
||||
|
||||
|
||||
def generate_scoring_lookup(all_tables):
|
||||
"""生成按 section → gender → indicator 组织的查询结构"""
|
||||
lookup = {
|
||||
'幼儿': {'male': {}, 'female': {}, '通用': {}},
|
||||
'成年人': {'male': {}, 'female': {}, '通用': {}},
|
||||
'老年人': {'male': {}, 'female': {}, '通用': {}},
|
||||
'weights': {'幼儿': {}, '成年人': {}, '老年人': {}},
|
||||
'rating_levels': {'幼儿': {}, '成年人': {}, '老年人': {}},
|
||||
'metadata': {'tables_count': len(all_tables)}
|
||||
}
|
||||
|
||||
for t in all_tables:
|
||||
if not t:
|
||||
continue
|
||||
section = t.get('section', '未知')
|
||||
gender = t.get('gender_en', '通用')
|
||||
tbl_type = t.get('type', 'unknown')
|
||||
|
||||
if section == '未知':
|
||||
continue
|
||||
|
||||
if tbl_type == 'weight_table':
|
||||
# 按年龄范围分别存储权重表(成年人有20-49和50-59两个表)
|
||||
title_lower = t.get('title', '')
|
||||
age_tag = re.search(r'(\d+[-~–]\d+)\s*岁', title_lower)
|
||||
if age_tag:
|
||||
weight_key = f"{section}_{age_tag.group(1).replace('~','-').replace('–','-')}岁"
|
||||
else:
|
||||
weight_key = section
|
||||
lookup['weights'][weight_key] = t['parsed']['weights']
|
||||
elif tbl_type == 'rating_levels':
|
||||
lookup['rating_levels'][section] = t['parsed']['levels']
|
||||
elif tbl_type == 'indicator_list':
|
||||
key = f"indicators_{t['table_id']}"
|
||||
if section not in lookup:
|
||||
lookup[section] = {}
|
||||
lookup[section].setdefault('通用', {})[key] = t['parsed']
|
||||
else:
|
||||
# 评分表
|
||||
indicator = t.get('indicator', t.get('title', '未知'))
|
||||
if gender not in lookup[section]:
|
||||
lookup[section][gender] = {}
|
||||
|
||||
entry = {
|
||||
'table_id': t['table_id'],
|
||||
'title': t['title'],
|
||||
'unit': t['unit'],
|
||||
'type': tbl_type,
|
||||
'data': t['parsed']
|
||||
}
|
||||
lookup[section][gender][indicator] = entry
|
||||
|
||||
return lookup
|
||||
|
||||
|
||||
def main():
|
||||
base = '/home/songyi/Documents/ai_agent_scraper_study/data/国家国民体质监测'
|
||||
doc_path = f'{base}/01_来源数据/2023修订版/《国民体质测定标准(2023年修订)》.md'
|
||||
with open(doc_path, 'r', encoding='utf-8') as f:
|
||||
content = f.read()
|
||||
|
||||
print(f"文件大小: {len(content)} 字符")
|
||||
|
||||
tables = extract_tables(content)
|
||||
print(f"提取到 {len(tables)} 个表格")
|
||||
|
||||
# 分类所有表格
|
||||
classified = []
|
||||
for t in tables:
|
||||
result = classify_table(t)
|
||||
if result:
|
||||
classified.append(result)
|
||||
|
||||
print(f"成功分类 {len(classified)} 个表格")
|
||||
|
||||
# 统计
|
||||
types = {}
|
||||
for t in classified:
|
||||
tt = t['type']
|
||||
types[tt] = types.get(tt, 0) + 1
|
||||
print(f"表格类型分布: {json.dumps(types, ensure_ascii=False, indent=2)}")
|
||||
|
||||
# 生成查询结构
|
||||
lookup = generate_scoring_lookup(classified)
|
||||
|
||||
# 输出全部表格原始解析
|
||||
all_tables_output = []
|
||||
for t in classified:
|
||||
all_tables_output.append({
|
||||
'table_id': t.get('table_id', ''),
|
||||
'title': t.get('title', ''),
|
||||
'section': t.get('section', ''),
|
||||
'gender': t.get('gender', ''),
|
||||
'indicator': t.get('indicator', ''),
|
||||
'unit': t.get('unit'),
|
||||
'type': t.get('type', ''),
|
||||
'parsed': t.get('parsed', {})
|
||||
})
|
||||
|
||||
output = {
|
||||
'metadata': {
|
||||
'source': '《国民体质测定标准(2023年修订)》',
|
||||
'publisher': '国家国民体质监测中心',
|
||||
'year': 2023,
|
||||
'total_tables': len(classified),
|
||||
'table_types': types
|
||||
},
|
||||
'scoring_lookup': lookup,
|
||||
'all_tables': all_tables_output,
|
||||
'weights': lookup['weights'],
|
||||
'rating_levels': lookup['rating_levels']
|
||||
}
|
||||
|
||||
out_path = f'{base}/02_加工数据/2023修订版/评分标准结构化数据.json'
|
||||
with open(out_path, 'w', encoding='utf-8') as f:
|
||||
json.dump(output, f, ensure_ascii=False, indent=2)
|
||||
|
||||
print(f"\nJSON 输出: {out_path}")
|
||||
print(f"JSON 大小: {len(json.dumps(output, ensure_ascii=False))} 字符")
|
||||
|
||||
# 打印摘要
|
||||
print("\n=== 评分表摘要 ===")
|
||||
seen = set()
|
||||
for t in classified:
|
||||
key = (t.get('section',''), t.get('gender',''), t.get('indicator',''))
|
||||
if key not in seen:
|
||||
seen.add(key)
|
||||
print(f" [{t.get('section','')}] {t.get('gender','')} - {t.get('indicator','')} ({t.get('table_id','')}) [{t.get('type','')}]")
|
||||
|
||||
print(f"\n总计 {len(seen)} 个评分表/指标组合")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in new issue
Block a user