This commit is contained in:
512song committed 2025-12-05 22:47:40 +08:00
1 parent 1036605b50
commit 3c756eda13
3 files changed
+470 -687

No files matched your search

+399 -90
View File
@@ -2793,7 +2793,9 @@
{
"cell_type": "markdown",
"id": "08bcbf58-4356-4f30-850f-0d0ff0e2528b",
"metadata": {},
"metadata": {
"jp-MarkdownHeadingCollapsed": true
},
"source": [
"# 第三次体测(2024年10月)"
]
@@ -3247,27 +3249,12 @@
},
{
"cell_type": "code",
"execution_count": 48,
"execution_count": null,
"id": "9dcbc866-2481-4cf6-a563-df4bb6596dfd",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-03T13:04:55.617841Z",
"iopub.status.busy": "2025-12-03T13:04:55.617249Z",
"iopub.status.idle": "2025-12-03T13:04:56.054475Z",
"shell.execute_reply": "2025-12-03T13:04:56.053478Z",
"shell.execute_reply.started": "2025-12-03T13:04:55.617812Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import json\n",
@@ -3832,34 +3819,10 @@
},
{
"cell_type": "code",
"execution_count": 40,
"execution_count": null,
"id": "227b2b23-1eb4-4da9-896f-5db070228f5b",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-03T11:08:04.377758Z",
"iopub.status.busy": "2025-12-03T11:08:04.377114Z",
"iopub.status.idle": "2025-12-03T11:08:04.744096Z",
"shell.execute_reply": "2025-12-03T11:08:04.743518Z",
"shell.execute_reply.started": "2025-12-03T11:08:04.377699Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"01738573 曹志玲 研究院 2\n",
"01730621 张敬东 电仪部 2\n",
"01733160 王金江 烯烃部 2\n",
"01736847 章洪 热电部 2\n",
"01736378 杨桂强 热电部 1\n",
"01737731 徐欣 运输销售部 1\n",
"01736804 李振江 热电部 2\n",
"01737858 于学宁 运输销售部 2\n",
"01737542 刘呈健 水务部 1\n"
]
}
],
"metadata": {},
"outputs": [],
"source": [
"import json\n",
"\n",
@@ -3978,15 +3941,15 @@
},
{
"cell_type": "code",
"execution_count": 44,
"execution_count": 66,
"id": "d0368e0f-270b-47fd-aec0-cae3cadd68b2",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-03T11:31:43.458591Z",
"iopub.status.busy": "2025-12-03T11:31:43.458312Z",
"iopub.status.idle": "2025-12-03T11:31:44.427029Z",
"shell.execute_reply": "2025-12-03T11:31:44.426528Z",
"shell.execute_reply.started": "2025-12-03T11:31:43.458563Z"
"iopub.execute_input": "2025-12-04T12:11:07.227210Z",
"iopub.status.busy": "2025-12-04T12:11:07.226601Z",
"iopub.status.idle": "2025-12-04T12:11:08.296200Z",
"shell.execute_reply": "2025-12-04T12:11:08.295131Z",
"shell.execute_reply.started": "2025-12-04T12:11:07.227149Z"
}
},
"outputs": [
@@ -4018,7 +3981,7 @@
"list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi','wh']\n",
"i=0\n",
"list2 = []\n",
"person = ['03603316']\n",
"person = ['01738966']\n",
"bumen =['南港乙烯项目管理部']\n",
"for k, v in dict1.items():\n",
" list1 = []\n",
@@ -4074,35 +4037,10 @@
},
{
"cell_type": "code",
"execution_count": 41,
"execution_count": null,
"id": "8467b539-c5ca-4db3-8231-f634b70dff88",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-03T11:13:39.490227Z",
"iopub.status.busy": "2025-12-03T11:13:39.489101Z",
"iopub.status.idle": "2025-12-03T11:13:39.648115Z",
"shell.execute_reply": "2025-12-03T11:13:39.647529Z",
"shell.execute_reply.started": "2025-12-03T11:13:39.490158Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"03603316 马新淼 炼油部\n",
"01738573 曹志玲 研究院\n",
"01730621 张敬东 电仪部\n",
"01733160 王金江 烯烃部\n",
"01736847 章洪 热电部\n",
"01736378 杨桂强 热电部\n",
"01737731 徐欣 运输销售部\n",
"01736804 李振江 热电部\n",
"01737858 于学宁 运输销售部\n",
"01737542 刘呈健 水务部\n"
]
}
],
"metadata": {},
"outputs": [],
"source": [
"import os,sys,shutil\n",
"import json\n",
@@ -4138,17 +4076,9 @@
},
{
"cell_type": "code",
"execution_count": 46,
"execution_count": null,
"id": "b9f88d36-e72a-4f26-8357-62c7b0e2128d",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-03T12:03:30.161578Z",
"iopub.status.busy": "2025-12-03T12:03:30.161039Z",
"iopub.status.idle": "2025-12-03T12:03:31.205217Z",
"shell.execute_reply": "2025-12-03T12:03:31.204754Z",
"shell.execute_reply.started": "2025-12-03T12:03:30.161525Z"
}
},
"metadata": {},
"outputs": [],
"source": [
"import os,sys,shutil\n",
@@ -4175,6 +4105,385 @@
" shutil.copyfile(fn,n_name)"
]
},
{
"cell_type": "markdown",
"id": "d4950051-c196-461a-a862-758ad014923d",
"metadata": {},
"source": [
"## 获得报告数据"
]
},
{
"cell_type": "code",
"execution_count": 52,
"id": "d5ec9835-0f8d-4755-8fa1-b0a1f7f35bfc",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-04T01:42:19.668220Z",
"iopub.status.busy": "2025-12-04T01:42:19.667596Z",
"iopub.status.idle": "2025-12-04T01:46:02.421978Z",
"shell.execute_reply": "2025-12-04T01:46:02.421489Z",
"shell.execute_reply.started": "2025-12-04T01:42:19.668162Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"0\n"
]
}
],
"source": [
"import pdfplumber\n",
"import os,sys,shutil\n",
"import json\n",
"import glob\n",
"from pathlib import Path\n",
"\n",
"dict1 = {}\n",
"\n",
"fi_path = '/home/songyi/pdf-typescript-fit2023/天津石化2025'\n",
"fls = glob.glob(f'{fi_path}/*.pdf')\n",
"for fn in fls:\n",
" code = Path(fn).stem.split('-')[0]\n",
" dict1.setdefault(code,{})\n",
" pdf = pdfplumber.open(fn)\n",
" text = pdf.pages[1].extract_text()#######页码从0开始计数\n",
" #print(text)\n",
" list1 = text.split('\\n')\n",
" for item in list1:\n",
" if '测试标准 国民体质测定标准' in item:\n",
" list_min = list1.index(item)\n",
" if '请注意:以上测试项目' in item or '感谢您完成测试' in item:\n",
" list_max = list1.index(item)\n",
" #print(code,list_min,list_max)\n",
" for i in range(list_min+1,list_max):\n",
" #dict2 = {}\n",
" xm = list1[i].split(' ')[0]\n",
" dict1[code][xm] = ','.join(list1[i].split(' ')[1:])\n",
"filename = 'data/天津石化体测报告提取数据.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False) \n",
"print(len(dict1)) \n",
" \n"
]
},
{
"cell_type": "markdown",
"id": "c3dca4b9-cc70-4e29-8848-97c8e2044219",
"metadata": {},
"source": [
"## 获得报告中得分及等级"
]
},
{
"cell_type": "code",
"execution_count": 69,
"id": "e48afdab-13f5-492a-9c14-72e00a883cdd",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-05T04:28:27.297909Z",
"iopub.status.busy": "2025-12-05T04:28:27.297070Z",
"iopub.status.idle": "2025-12-05T04:32:11.169088Z",
"shell.execute_reply": "2025-12-05T04:32:11.168527Z",
"shell.execute_reply.started": "2025-12-05T04:28:27.297834Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"3740\n"
]
}
],
"source": [
"import pdfplumber\n",
"import os,sys,shutil\n",
"import json\n",
"import glob\n",
"from pathlib import Path\n",
"import re\n",
"\n",
"dict1 = {}\n",
"\n",
"fi_path = '/home/songyi/pdf-typescript-fit2023/天津石化2025'\n",
"fls = glob.glob(f'{fi_path}/*.pdf')\n",
"for fn in fls:\n",
" code = Path(fn).stem.split('-')[0]\n",
" dict1.setdefault(code,{})\n",
" pdf = pdfplumber.open(fn)\n",
" text = pdf.pages[1].extract_text()#######页码从0开始计数\n",
" #print(text)\n",
" list1 = text.split('\\n')\n",
" for item in list1:\n",
" if '感谢您完成测试' in item:\n",
" score_pattern = r\"分为(\\d+)\"\n",
" score_match = re.search(score_pattern, text)\n",
" if score_match:\n",
" average_score = score_match.group(1)\n",
" else:\n",
" average_score = ''\n",
"\n",
"# 提取等级信息(包括括号内的内容)\n",
" grade_pattern = r\"等级为([^,]+)\"\n",
" grade_match = re.search(grade_pattern, text)\n",
" if grade_match:\n",
" grade = grade_match.group(1) \n",
" else:\n",
" grade = ''\n",
" dict1[code]['平均分'] = average_score\n",
" dict1[code]['等级'] = grade.split('(')[0] \n",
" \n",
" \n",
"filename = 'data/天津石化体测报告提取数据(得分及等级).json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False) \n",
"print(len(dict1)) "
]
},
{
"cell_type": "code",
"execution_count": 70,
"id": "c624a091-4851-4f3a-83c9-83520b20603b",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-05T06:39:21.668327Z",
"iopub.status.busy": "2025-12-05T06:39:21.667654Z",
"iopub.status.idle": "2025-12-05T06:39:21.677684Z",
"shell.execute_reply": "2025-12-05T06:39:21.676710Z",
"shell.execute_reply.started": "2025-12-05T06:39:21.668269Z"
}
},
"outputs": [],
"source": [
"import json\n",
"\n",
"filename = 'data/天津石化体测报告提取数据(得分及等级).json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"\n",
"for k, v in dict1.items():\n",
" if not v['平均分'].isdigit:\n",
" print(k)"
]
},
{
"cell_type": "code",
"execution_count": 55,
"id": "e40caa62-9f50-4480-964e-177e88549425",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-04T01:50:28.395443Z",
"iopub.status.busy": "2025-12-04T01:50:28.394915Z",
"iopub.status.idle": "2025-12-04T01:50:28.426527Z",
"shell.execute_reply": "2025-12-04T01:50:28.425998Z",
"shell.execute_reply.started": "2025-12-04T01:50:28.395395Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"{'纵跳', '请注意:以上测试项目及格线为60分。列表中红色项目需重点关注,建议在专家指导下进行科学锻炼,', '选择反应时', '肺活量', '俯卧撑', '闭眼单脚站立', '坐位体前屈', '1分钟仰卧起坐', '腰臀比', '减少潜在的运动风险。', '握力', '身高体重指数', '台阶指数'}\n"
]
}
],
"source": [
"filename = 'data/天津石化体测报告提取数据.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"set1 = set()\n",
"for k, v in dict1.items():\n",
" for item in v.keys():\n",
" set1.add(item)\n",
"print(set1)"
]
},
{
"cell_type": "code",
"execution_count": 67,
"id": "1edd3188-e28f-459c-a7f3-369393e43b47",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-04T12:18:14.866809Z",
"iopub.status.busy": "2025-12-04T12:18:14.866039Z",
"iopub.status.idle": "2025-12-04T12:18:15.086401Z",
"shell.execute_reply": "2025-12-04T12:18:15.085861Z",
"shell.execute_reply.started": "2025-12-04T12:18:14.866737Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"3740\n"
]
}
],
"source": [
"import json\n",
"\n",
"filename = 'data/result_天津石化2025-2.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"filename = 'data/天津石化人员2025.json'\n",
"with open(filename,'r') as fl:\n",
" dict4 = json.load(fl)\n",
" \n",
"filename = 'data/天津石化体测报告提取数据.json'\n",
"with open(filename,'r') as fl:\n",
" dict2 = json.load(fl)\n",
"items = ['纵跳','选择反应时','肺活量','俯卧撑','闭眼单脚站立','坐位体前屈','1分钟仰卧起坐','握力','台阶指数']\n",
"items_1 = ['腰臀比','身高体重指数']\n",
"dict3 = {}\n",
"for k, v in dict2.items():\n",
" dict3.setdefault(k,{})\n",
" dict3[k]['name'] = dict4[k]['name']\n",
" dict3[k]['sex'] = dict4[k]['sex']\n",
" dict3[k]['unit'] = dict4[k]['unit']\n",
" dict3[k]['sub_unit'] = dict4[k]['sub_unit']\n",
" dict3[k]['age'] = dict1[k]['age']\n",
" for item in items:\n",
" if item in v.keys():\n",
" #print(k,item)\n",
" dict3[k].setdefault(item,{})\n",
" dict3[k][item]['mark'] = v[item].split(',')[0]\n",
" dict3[k][item]['score'] = v[item].split(',')[2].replace('分','')\n",
" for item in items_1:\n",
" if item in v.keys():\n",
" dict3[k].setdefault(item,{})\n",
" dict3[k][item]['mark'] = v[item].split(',')[0]\n",
" dict3[k][item]['score'] = v[item].split(',')[1].replace('分','')\n",
" \n",
"filename = 'data/data_天津石化2025.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict3, fl, ensure_ascii=False) \n",
"print(len(dict3)) \n",
" "
]
},
{
"cell_type": "markdown",
"id": "6694af03-76f3-4d7f-8f80-3bf755f94aa8",
"metadata": {},
"source": [
"## 合并整理报告数据"
]
},
{
"cell_type": "code",
"execution_count": 74,
"id": "ed42e2f1-c024-4d29-86c3-b47edb97724f",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-05T06:51:04.517628Z",
"iopub.status.busy": "2025-12-05T06:51:04.516907Z",
"iopub.status.idle": "2025-12-05T06:51:04.714115Z",
"shell.execute_reply": "2025-12-05T06:51:04.713629Z",
"shell.execute_reply.started": "2025-12-05T06:51:04.517561Z"
}
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"3740\n"
]
}
],
"source": [
"filename = 'data/data_天津石化2025.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"filename = 'data/天津石化体测报告提取数据(得分及等级).json'\n",
"with open(filename,'r') as fl:\n",
" dict2 = json.load(fl)\n",
"items = ['纵跳','选择反应时','肺活量','俯卧撑','闭眼单脚站立','坐位体前屈','1分钟仰卧起坐','握力','台阶指数','身高体重指数']\n",
"for k, v in dict1.items():\n",
" for item in items:\n",
" if item in v.keys():\n",
" score = int(v[item]['score'])\n",
" v[item]['score'] = score\n",
" v['平均分'] = dict2[k]['平均分']\n",
" v['等级'] = dict2[k]['等级']\n",
"\n",
"filename = 'data/data_天津石化2025.json'\n",
"with open(filename,'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False) \n",
"print(len(dict1)) "
]
},
{
"cell_type": "markdown",
"id": "aa8a5641-d5dc-4452-8c8a-8d2c161e9b9a",
"metadata": {},
"source": [
"## 导出报告数据"
]
},
{
"cell_type": "code",
"execution_count": 76,
"id": "5f7cbf7c-ad59-4c84-8592-34ae71c10e26",
"metadata": {
"execution": {
"iopub.execute_input": "2025-12-05T07:11:47.233380Z",
"iopub.status.busy": "2025-12-05T07:11:47.232806Z",
"iopub.status.idle": "2025-12-05T07:11:48.772073Z",
"shell.execute_reply": "2025-12-05T07:11:48.771510Z",
"shell.execute_reply.started": "2025-12-05T07:11:47.233324Z"
}
},
"outputs": [],
"source": [
"import json\n",
"import openpyxl\n",
"\n",
"title = ['编号', '姓名', '性别', '部门', '车间','年龄', '身高体重指数', '肺活量', '得分', '握力', '得分', '坐位体前屈', '得分', '纵跳', '得分', '俯卧撑', '得分', '单脚站立', '得分', '选择反应时', '得分', '一分钟仰卧起坐', '得分','台阶指数', '得分','腰臀比','状态','平均分','等级']\n",
"items = ['身高体重指数','肺活量','握力','坐位体前屈','纵跳','俯卧撑','闭眼单脚站立','选择反应时','1分钟仰卧起坐','台阶指数','腰臀比']\n",
"\n",
"list1 = []\n",
"filename = 'data/data_天津石化2025.json'\n",
"with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n",
"\n",
"for k, v in dict1.items():\n",
" list2 = []\n",
" list2.append(k)\n",
" list2.append(v['name'])\n",
" list2.append(v['sex'])\n",
" list2.append(v['unit'])\n",
" list2.append(v['sub_unit'])\n",
" for item in items:\n",
" if item in v.keys():\n",
" list2.append(v[item]['mark'])\n",
" list2.append(v[item]['score'])\n",
" else:\n",
" list2.append('')\n",
" list2.append('') \n",
" list2.append(v['平均分'])\n",
" list2.append(v['等级'])\n",
" list1.append(list2)\n",
" \n",
"\n",
"\n",
"filename = 'data/天津石化体测报告数据(2025年).xlsx'\n",
"wb = openpyxl.Workbook()\n",
"sheet = wb.active\n",
"sheet.append(title)\n",
"for row in list1:\n",
" sheet.append(row)\n",
" \n",
"wb.save(filename) \n"
]
},
{
"cell_type": "markdown",
"id": "85a03b00-9624-4ae9-aa6d-386798d457df",