From fe91eda8ccef31349b17f9329eedf8bd0282a69b Mon Sep 17 00:00:00 2001 From: 512song Date: Fri, 14 Mar 2025 16:29:36 +0800 Subject: [PATCH] 20250314 --- 体测单位/东营老年大学.ipynb | 249 +++++++------------ 体测单位/延庆.ipynb | 54 ++-- 体测单位/海淀老干部大学.ipynb | 455 +++++++++++++++++++++++++++++++++- 数据采集.ipynb | 2 +- 文件操作.ipynb | 133 +++++++++- 5 files changed, 685 insertions(+), 208 deletions(-) diff --git a/体测单位/东营老年大学.ipynb b/体测单位/东营老年大学.ipynb index 3fc4043..fb64955 100644 --- a/体测单位/东营老年大学.ipynb +++ b/体测单位/东营老年大学.ipynb @@ -10,15 +10,15 @@ }, { "cell_type": "code", - "execution_count": 41, + "execution_count": 70, "id": "bbba6efc-73cd-4db6-bae7-014724fee731", "metadata": { "execution": { - "iopub.execute_input": "2025-03-03T06:31:35.177159Z", - "iopub.status.busy": "2025-03-03T06:31:35.176415Z", - "iopub.status.idle": "2025-03-03T06:31:35.214675Z", - "shell.execute_reply": "2025-03-03T06:31:35.214069Z", - "shell.execute_reply.started": "2025-03-03T06:31:35.177089Z" + "iopub.execute_input": "2025-03-14T08:07:40.941181Z", + "iopub.status.busy": "2025-03-14T08:07:40.940504Z", + "iopub.status.idle": "2025-03-14T08:07:40.971889Z", + "shell.execute_reply": "2025-03-14T08:07:40.971330Z", + "shell.execute_reply.started": "2025-03-14T08:07:40.941123Z" }, "tags": [] }, @@ -27,7 +27,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "89\n", + "96\n", "ok\n" ] } @@ -37,24 +37,26 @@ "import json\n", "from datetime import date\n", "\n", - "wb = openpyxl.load_workbook('data/北体体质康健2班.xlsx')\n", + "wb = openpyxl.load_workbook('data/东营市老年大学第三期体质班.xlsx')\n", "sheet = wb.active\n", "# sheets = wb.sheetnames\n", "person = {}\n", "\n", - "for n in range(3, sheet.max_row+1):\n", + "for n in range(2, sheet.max_row+1):\n", " code = int(sheet.cell(n, 1).value)\n", " person.setdefault(code, {})\n", " dict1 = {}\n", - " dict1['name'] = sheet.cell(n, 3).value\n", - " dict1['sex'] = sheet.cell(n, 4).value\n", - " dict1['unit'] = sheet.cell(n, 2).value\n", - " dict1['birth'] = str(sheet.cell(n, 5).value).replace('/','-').split(' ')[0]\n", - " dict1['age'] = sheet.cell(n, 6).value\n", - " if sheet.cell(n,7).value is not None:\n", - " dict1['phone'] = str(sheet.cell(n,7).value) \n", + " dict1['name'] = sheet.cell(n, 2).value\n", + " dict1['sex'] = sheet.cell(n, 3).value\n", + " dict1['unit'] = '第三期体质班'\n", + " dict1['birth'] = str(sheet.cell(n, 4).value).replace('/','-').split(' ')[0]\n", + " #dict1['age'] = sheet.cell(n, 6).value\n", + " if sheet.cell(n,6).value is not None:\n", + " dict1['phone'] = str(sheet.cell(n,6).value) \n", + " if sheet.cell(n,5).value is not None:\n", + " dict1['id'] = str(sheet.cell(n,5).value)\n", " person[code] = dict1\n", - "filename = 'data/北体体质康健班202412.json'\n", + "filename = 'data/东营市老年大学第三期体质班.json'\n", "with open(filename, 'w') as fl:\n", " json.dump(person, fl, ensure_ascii=False)\n", "print(len(person))\n", @@ -128,12 +130,27 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 71, "id": "970e171e-1360-448f-a28c-520ccb8f314a", "metadata": { + "execution": { + "iopub.execute_input": "2025-03-14T08:07:44.886839Z", + "iopub.status.busy": "2025-03-14T08:07:44.886215Z", + "iopub.status.idle": "2025-03-14T08:07:44.904464Z", + "shell.execute_reply": "2025-03-14T08:07:44.903941Z", + "shell.execute_reply.started": "2025-03-14T08:07:44.886779Z" + }, "tags": [] }, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "93\n" + ] + } + ], "source": [ "import json\n", "import datetime\n", @@ -143,11 +160,11 @@ "\n", "re_ta = {}\n", "list1 = []\n", - "filename = 'data/北体体质康健班202412.json'\n", + "filename = 'data/东营市老年大学第三期体质班.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", "\n", - "filename = 'data/marks_20241222.csv'\n", + "filename = 'data/marks_20250314.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -184,7 +201,7 @@ " score = result[4] \n", " re_ta[user][item_name]['成绩'] = score\n", "\n", - "filename = 'data/result_北体体质康健班202412.json'\n", + "filename = 'data/result_东营市老年大学第三期体质班.json'\n", "with open(filename,'w') as fl:\n", " json.dump(re_ta, fl, ensure_ascii=False) \n", "print(len(re_ta))" @@ -192,15 +209,15 @@ }, { "cell_type": "code", - "execution_count": 58, + "execution_count": 72, "id": "e2526429-e779-4ec5-8a0e-bea623f54f29", "metadata": { "execution": { - "iopub.execute_input": "2025-03-04T01:28:20.346704Z", - "iopub.status.busy": "2025-03-04T01:28:20.345188Z", - "iopub.status.idle": "2025-03-04T01:28:20.368327Z", - "shell.execute_reply": "2025-03-04T01:28:20.367809Z", - "shell.execute_reply.started": "2025-03-04T01:28:20.346598Z" + "iopub.execute_input": "2025-03-14T08:07:48.554811Z", + "iopub.status.busy": "2025-03-14T08:07:48.554094Z", + "iopub.status.idle": "2025-03-14T08:07:48.579590Z", + "shell.execute_reply": "2025-03-14T08:07:48.579110Z", + "shell.execute_reply.started": "2025-03-14T08:07:48.554748Z" } }, "outputs": [ @@ -278,15 +295,15 @@ }, { "cell_type": "code", - "execution_count": 59, + "execution_count": 73, "id": "c3b99428-00d5-45ea-9803-261194525fa4", "metadata": { "execution": { - "iopub.execute_input": "2025-03-04T01:28:24.460728Z", - "iopub.status.busy": "2025-03-04T01:28:24.459968Z", - "iopub.status.idle": "2025-03-04T01:28:24.473509Z", - "shell.execute_reply": "2025-03-04T01:28:24.472999Z", - "shell.execute_reply.started": "2025-03-04T01:28:24.460657Z" + "iopub.execute_input": "2025-03-14T08:08:03.162240Z", + "iopub.status.busy": "2025-03-14T08:08:03.161488Z", + "iopub.status.idle": "2025-03-14T08:08:03.178033Z", + "shell.execute_reply": "2025-03-14T08:08:03.177422Z", + "shell.execute_reply.started": "2025-03-14T08:08:03.162173Z" } }, "outputs": [ @@ -305,7 +322,7 @@ "\n", "#list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','height','weight']\n", "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']\n", - "filename = 'data/result_陈庄检测人员.json'\n", + "filename = 'data/result_东营市老年大学第三期体质班.json'\n", "with open(filename,'r') as fl:\n", " dict2 = json.load(fl) \n", "for k, v in dict2.items():\n", @@ -328,7 +345,7 @@ " dict2[k][item_en]['score'] = My.cal_score(data1)\n", " #print(k,v[item_en]['成绩'],cal_score(data1))\n", "\n", - "filename = f'data/result_陈庄检测人员.json'\n", + "filename = f'data/result_东营市老年大学第三期体质班.json'\n", "with open(filename,'w') as fl:\n", " json.dump(dict2,fl , ensure_ascii=False) \n", "print('ok!') " @@ -344,15 +361,15 @@ }, { "cell_type": "code", - "execution_count": 60, + "execution_count": 75, "id": "0c60a509-dfc0-426a-a44e-e9a760139463", "metadata": { "execution": { - "iopub.execute_input": "2025-03-04T01:28:32.981359Z", - "iopub.status.busy": "2025-03-04T01:28:32.980597Z", - "iopub.status.idle": "2025-03-04T01:28:33.024339Z", - "shell.execute_reply": "2025-03-04T01:28:33.023768Z", - "shell.execute_reply.started": "2025-03-04T01:28:32.981288Z" + "iopub.execute_input": "2025-03-14T08:10:25.874988Z", + "iopub.status.busy": "2025-03-14T08:10:25.874253Z", + "iopub.status.idle": "2025-03-14T08:10:25.912497Z", + "shell.execute_reply": "2025-03-14T08:10:25.912021Z", + "shell.execute_reply.started": "2025-03-14T08:10:25.874920Z" }, "tags": [] }, @@ -364,11 +381,11 @@ "items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']\n", "title = ['编号','姓名','性别','单位','身高','体重','肺活量','握力','坐位体前屈','纵跳','俯卧撑','一分钟仰卧起坐','单脚站立','选择反应时','台阶指数']\n", "\n", - "filename = 'data/result_陈庄检测人员.json'\n", + "filename = 'data/result_东营市老年大学第三期体质班.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "\n", - "filename = 'data/陈庄检测人员.json'\n", + "filename = 'data/东营市老年大学第三期体质班.json'\n", "with open(filename,'r') as fl:\n", " dict2 = json.load(fl)\n", " \n", @@ -396,7 +413,7 @@ " list2.append('')\n", " \n", " list1.append(list2)\n", - "filename = 'data/陈庄检测人员20250303.xlsx'\n", + "filename = 'data/东营市老年大学第三期体质班成绩.xlsx'\n", "wb = openpyxl.Workbook()\n", "sheet = wb.active\n", "sheet.append(title)\n", @@ -416,115 +433,15 @@ }, { "cell_type": "code", - "execution_count": null, - "id": "79fb7799-3693-4b7b-ae82-c0d8dee98d36", - "metadata": { - "tags": [] - }, - "outputs": [], - "source": [ - "import json\n", - "import csv\n", - "import openpyxl\n", - "import time\n", - "from datetime import date\n", - "\n", - "dict1 = {}\n", - "\n", - "filename = 'data/result_陈庄检测人员.json'\n", - "with open(filename,'r') as fl:\n", - " dict1 = json.load(fl)\n", - "filename = 'data/陈庄检测人员.json'\n", - "with open(filename,'r') as fl:\n", - " dict3 = json.load(fl)\n", - "\n", - "phone = {}\n", - "for k,v in dict3.items():\n", - " if 'phone' in v.keys():\n", - " phone[v['phone']] = k\n", - "\n", - "list1 = []\n", - "filename = 'data/Survey_20250303.csv'\n", - "with open(filename,'r',newline='') as csv_file:\n", - " fl = csv.reader(csv_file,delimiter=',')\n", - " header = next(fl) \n", - " for line in fl:\n", - " list1.append(line)\n", - "\n", - "\n", - "\n", - "nn = 0\n", - "for item in list1:\n", - " if item[2] in phone.keys():\n", - " psy =[]\n", - " tcm = []\n", - " spine = []\n", - " for i in range(0,45):\n", - " psy.append(0)\n", - " psy[44] = []\n", - " for i in range(0,60):\n", - " tcm.append(0)\n", - " for i in range(0,26):\n", - " spine.append(0)\n", - " \n", - " content = json.loads(item[5])\n", - " if phone[item[2]] not in dict1.keys():\n", - " dict1[phone[item[2]]] = dict3[phone[item[2]]]\n", - " rq = date.fromisoformat(item[6].replace('/','-').split(' ')[0])\n", - " dict1[phone[item[2]]]['rq'] = item[6].replace('/','-').split(' ')[0]\n", - " for k, v in content.items():\n", - " if 'psyOld' in k:\n", - " i = int(k[6:])\n", - " psy[i-1] = int(v)\n", - " if 'tcm' in k:\n", - " i = int(k[3:])\n", - " tcm[i-1] = int(v)\n", - " if 'spine' in k:\n", - " i = int(k[5:])\n", - " spine[i-1] = int(v)\n", - " if 'psyOld' in item[5]:\n", - " #for i in range(5,26):\n", - " # new_valve = 5-psy[i]\n", - " # psy[i] = new_valve\n", - " for i in range(26,40):\n", - " new_valve = 1+psy[i]\n", - " psy[i] = new_valve\n", - " \n", - " dict1[phone[item[2]]]['psy_yangmiao_old'] = psy\n", - " if 'tcm' in item[5]:\n", - " if tcm[40] ==0:\n", - " tcm[40] =1\n", - " dict1[phone[item[2]]]['tcm'] = tcm\n", - " if 'spine' in item[5]:\n", - " for ii in range(25,23,-1):\n", - " spine[ii] = spine[ii-1]\n", - " spine[22] = 0 \n", - " dict1[phone[item[2]]]['spine'] = spine\n", - " birth = date.fromisoformat(dict3[phone[item[2]]]['birth'].replace('/','-'))\n", - " \n", - " days = (rq-birth).days \n", - " dict1[phone[item[2]]]['age'] = int(days/365)\n", - " dict1[phone[item[2]]]['month'] = int(days/365*12)\n", - " #print(phone[item[2]])\n", - " nn+=1\n", - "filename = 'data/result_陈庄检测人员-1.json'\n", - "\n", - "with open(filename,'w') as fl:\n", - " json.dump(dict1, fl, ensure_ascii=False)\n", - "print(nn)" - ] - }, - { - "cell_type": "code", - "execution_count": 61, + "execution_count": 78, "id": "891347c4-cd65-41f8-b5dd-b3ff3d1e249f", "metadata": { "execution": { - "iopub.execute_input": "2025-03-04T01:34:28.818217Z", - "iopub.status.busy": "2025-03-04T01:34:28.817471Z", - "iopub.status.idle": "2025-03-04T01:34:28.854884Z", - "shell.execute_reply": "2025-03-04T01:34:28.854277Z", - "shell.execute_reply.started": "2025-03-04T01:34:28.818153Z" + "iopub.execute_input": "2025-03-14T08:17:25.959454Z", + "iopub.status.busy": "2025-03-14T08:17:25.958763Z", + "iopub.status.idle": "2025-03-14T08:17:25.994196Z", + "shell.execute_reply": "2025-03-14T08:17:25.993587Z", + "shell.execute_reply.started": "2025-03-14T08:17:25.959390Z" } }, "outputs": [ @@ -532,7 +449,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "83\n" + "87\n" ] } ], @@ -545,10 +462,10 @@ "\n", "dict1 = {}\n", "\n", - "filename = 'data/result_陈庄检测人员.json'\n", + "filename = 'data/result_东营市老年大学第三期体质班.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", - "filename = 'data/陈庄检测人员.json'\n", + "filename = 'data/东营市老年大学第三期体质班.json'\n", "with open(filename,'r') as fl:\n", " dict3 = json.load(fl)\n", "\n", @@ -558,7 +475,7 @@ " phone[v['phone']] = k\n", "\n", "list1 = []\n", - "filename = 'data/Survey_20250303.csv'\n", + "filename = 'data/Survey_20250314.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -614,7 +531,7 @@ " dict1[phone[item[2]]]['month'] = int(days/365*12)\n", " #print(phone[item[2]])\n", " nn+=1\n", - "filename = 'data/result_陈庄检测人员-1.json'\n", + "filename = 'data/result_东营市老年大学第三期体质班-1.json'\n", "\n", "with open(filename,'w') as fl:\n", " json.dump(dict1, fl, ensure_ascii=False)\n", @@ -911,15 +828,15 @@ }, { "cell_type": "code", - "execution_count": 62, + "execution_count": 79, "id": "327c33e2-7437-4421-b70c-ec98234a5e88", "metadata": { "execution": { - "iopub.execute_input": "2025-03-04T01:34:56.351921Z", - "iopub.status.busy": "2025-03-04T01:34:56.351159Z", - "iopub.status.idle": "2025-03-04T01:36:01.625317Z", - "shell.execute_reply": "2025-03-04T01:36:01.624114Z", - "shell.execute_reply.started": "2025-03-04T01:34:56.351849Z" + "iopub.execute_input": "2025-03-14T08:19:54.589069Z", + "iopub.status.busy": "2025-03-14T08:19:54.588348Z", + "iopub.status.idle": "2025-03-14T08:20:46.467552Z", + "shell.execute_reply": "2025-03-14T08:20:46.466556Z", + "shell.execute_reply.started": "2025-03-14T08:19:54.589004Z" } }, "outputs": [ @@ -927,7 +844,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "116\n" + "95\n" ] } ], @@ -940,11 +857,11 @@ "headers = {\n", " \"Content-Type\": \"application/json; charset=UTF-8\"\n", " }\n", - "filename = 'data/result_陈庄检测人员-1.json'\n", + "filename = 'data/result_东营市老年大学第三期体质班-1.json'\n", "with open(filename,'r') as fl:\n", " dict1 = json.load(fl)\n", "list1 = []\n", - "file_path ='./东营陈庄/'\n", + "file_path ='./东营市老年大学第三期体质班/'\n", "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']\n", "i=0\n", "list2 = []\n", @@ -954,7 +871,7 @@ " \n", " id = str(k).rjust(4,\"0\")\n", " mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'\n", - " mydata['title'] = '山东东营陈庄镇'\n", + " mydata['title'] = '东营市老年大学'\n", " mydata['subtitle'] = v['unit']\n", " mydata['id'] = id\n", " mydata['name'] = v['name']\n", diff --git a/体测单位/延庆.ipynb b/体测单位/延庆.ipynb index cdf16f4..1480754 100644 --- a/体测单位/延庆.ipynb +++ b/体测单位/延庆.ipynb @@ -10,15 +10,15 @@ }, { "cell_type": "code", - "execution_count": 1, + "execution_count": 29, "id": "3d5ddc7a-a331-438d-aa8d-1ad74c7250d0", "metadata": { "execution": { - "iopub.execute_input": "2025-02-23T10:13:29.334391Z", - "iopub.status.busy": "2025-02-23T10:13:29.333636Z", - "iopub.status.idle": "2025-02-23T10:13:29.512654Z", - "shell.execute_reply": "2025-02-23T10:13:29.511636Z", - "shell.execute_reply.started": "2025-02-23T10:13:29.334322Z" + "iopub.execute_input": "2025-03-11T12:15:27.766643Z", + "iopub.status.busy": "2025-03-11T12:15:27.765962Z", + "iopub.status.idle": "2025-03-11T12:15:27.793362Z", + "shell.execute_reply": "2025-03-11T12:15:27.792793Z", + "shell.execute_reply.started": "2025-03-11T12:15:27.766564Z" } }, "outputs": [ @@ -65,15 +65,15 @@ }, { "cell_type": "code", - "execution_count": 2, + "execution_count": 30, "id": "72ec680f-ec0d-44f9-919c-17c8e614b885", "metadata": { "execution": { - "iopub.execute_input": "2025-02-23T10:13:35.977412Z", - "iopub.status.busy": "2025-02-23T10:13:35.976513Z", - "iopub.status.idle": "2025-02-23T10:13:35.998305Z", - "shell.execute_reply": "2025-02-23T10:13:35.997869Z", - "shell.execute_reply.started": "2025-02-23T10:13:35.977335Z" + "iopub.execute_input": "2025-03-11T12:15:59.585308Z", + "iopub.status.busy": "2025-03-11T12:15:59.584674Z", + "iopub.status.idle": "2025-03-11T12:15:59.596989Z", + "shell.execute_reply": "2025-03-11T12:15:59.595883Z", + "shell.execute_reply.started": "2025-03-11T12:15:59.585249Z" } }, "outputs": [ @@ -81,7 +81,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "113\n" + "3\n" ] } ], @@ -98,7 +98,7 @@ "with open(filename,'r') as fl:\n", " dict1 = json.load(fl) \n", "\n", - "filename = 'data/marks_20250223.csv'\n", + "filename = 'data/marks_20250311.csv'\n", "with open(filename,'r',newline='') as csv_file:\n", " fl = csv.reader(csv_file,delimiter=',')\n", " header = next(fl) \n", @@ -149,15 +149,15 @@ }, { "cell_type": "code", - "execution_count": 3, + "execution_count": 31, "id": "8f459ed3-fc14-404f-998b-5ed2a15958c3", "metadata": { "execution": { - "iopub.execute_input": "2025-02-23T10:13:40.888693Z", - "iopub.status.busy": "2025-02-23T10:13:40.887898Z", - "iopub.status.idle": "2025-02-23T10:13:40.910971Z", - "shell.execute_reply": "2025-02-23T10:13:40.910112Z", - "shell.execute_reply.started": "2025-02-23T10:13:40.888581Z" + "iopub.execute_input": "2025-03-11T12:16:02.677677Z", + "iopub.status.busy": "2025-03-11T12:16:02.677141Z", + "iopub.status.idle": "2025-03-11T12:16:02.688486Z", + "shell.execute_reply": "2025-03-11T12:16:02.687588Z", + "shell.execute_reply.started": "2025-03-11T12:16:02.677626Z" } }, "outputs": [ @@ -215,15 +215,15 @@ }, { "cell_type": "code", - "execution_count": 12, + "execution_count": 32, "id": "9e843585-fef3-4f5c-baae-76e711f2a42f", "metadata": { "execution": { - "iopub.execute_input": "2025-02-16T15:29:54.316042Z", - "iopub.status.busy": "2025-02-16T15:29:54.315217Z", - "iopub.status.idle": "2025-02-16T15:30:07.279126Z", - "shell.execute_reply": "2025-02-16T15:30:07.278134Z", - "shell.execute_reply.started": "2025-02-16T15:29:54.315964Z" + "iopub.execute_input": "2025-03-11T12:16:06.507493Z", + "iopub.status.busy": "2025-03-11T12:16:06.506739Z", + "iopub.status.idle": "2025-03-11T12:16:08.382833Z", + "shell.execute_reply": "2025-03-11T12:16:08.381828Z", + "shell.execute_reply.started": "2025-03-11T12:16:06.507425Z" } }, "outputs": [ @@ -231,7 +231,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "28\n" + "3\n" ] } ], diff --git a/体测单位/海淀老干部大学.ipynb b/体测单位/海淀老干部大学.ipynb index 742f73d..65e6ef0 100644 --- a/体测单位/海淀老干部大学.ipynb +++ b/体测单位/海淀老干部大学.ipynb @@ -10,15 +10,15 @@ }, { "cell_type": "code", - "execution_count": 7, + "execution_count": 15, "id": "eb9aa782-ce2c-4814-bdf6-92c7e892854d", "metadata": { "execution": { - "iopub.execute_input": "2025-03-04T04:48:50.513279Z", - "iopub.status.busy": "2025-03-04T04:48:50.512517Z", - "iopub.status.idle": "2025-03-04T04:48:50.542095Z", - "shell.execute_reply": "2025-03-04T04:48:50.541445Z", - "shell.execute_reply.started": "2025-03-04T04:48:50.513207Z" + "iopub.execute_input": "2025-03-13T02:11:09.976784Z", + "iopub.status.busy": "2025-03-13T02:11:09.976043Z", + "iopub.status.idle": "2025-03-13T02:11:09.999996Z", + "shell.execute_reply": "2025-03-13T02:11:09.999500Z", + "shell.execute_reply.started": "2025-03-13T02:11:09.976712Z" } }, "outputs": [ @@ -112,10 +112,451 @@ "print('ok')" ] }, + { + "cell_type": "markdown", + "id": "188fb9fc-a6e4-4aec-b503-5e96e4981ef5", + "metadata": {}, + "source": [ + "## 获取人员测试成绩" + ] + }, + { + "cell_type": "code", + "execution_count": 8, + "id": "fd005ed3-24ee-4b39-a619-6c83e64c2852", + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-11T06:50:46.618660Z", + "iopub.status.busy": "2025-03-11T06:50:46.617709Z", + "iopub.status.idle": "2025-03-11T06:50:46.633813Z", + "shell.execute_reply": "2025-03-11T06:50:46.633056Z", + "shell.execute_reply.started": "2025-03-11T06:50:46.618581Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "29\n" + ] + } + ], + "source": [ + "import json\n", + "import datetime\n", + "import csv\n", + "from datetime import date\n", + "\n", + "\n", + "re_ta = {}\n", + "list1 = []\n", + "filename = 'data/海淀区老干部大学第一期.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl) \n", + "\n", + "filename = 'data/marks_20250311.csv'\n", + "with open(filename,'r',newline='') as csv_file:\n", + " fl = csv.reader(csv_file,delimiter=',')\n", + " header = next(fl) \n", + " for line in fl:\n", + " #line = re.sub('[\\r\\n\\f ]{1,}', '', line)\n", + " list1.append(line)\n", + "\n", + "for result in list1:\n", + " user = str(result[2])\n", + " rq = date.fromisoformat(result[5].replace('/','-'))\n", + " if user in dict1.keys():\n", + " l_xm = []\n", + " m_item = str(result[3]) \n", + " re_ta.setdefault(user,{}) \n", + " re_ta[user]['name'] = dict1[user]['name']\n", + " re_ta[user]['sex'] = dict1[user]['sex']\n", + " if 'phone' in dict1[user].keys():\n", + " re_ta[user]['phone'] = dict1[user]['phone']\n", + " if dict1[user]['sex'] == '男':\n", + " l_xm = ['bmi','lung','grip','flexion','jump','pushup','balance','reaction','step']\n", + " else:\n", + " l_xm = ['bmi','lung','grip','flexion','jump','balance','reaction','step','situp']\n", + " #re_ta[user]['unit'] = dict1[user]['unit']\n", + " birth = date.fromisoformat(dict1[user]['birth'].replace('/','-'))\n", + " item_name = result[3] \n", + " if item_name in l_xm: \n", + " days = (rq-birth).days \n", + " re_ta[user]['age'] = int(days/365)\n", + " re_ta[user]['month'] = int(days/365*12)\n", + " re_ta[user]['rq'] = result[5]\n", + " re_ta[user].setdefault(item_name,{}) \n", + " score = result[4] \n", + " re_ta[user][item_name]['成绩'] = score\n", + "\n", + "filename = 'data/result_海淀区老干部大学第一期.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(re_ta, fl, ensure_ascii=False) \n", + "print(len(re_ta))" + ] + }, + { + "cell_type": "markdown", + "id": "37c2ccf4-5d95-4f32-a9c2-28669aa90a3d", + "metadata": {}, + "source": [ + "## 生成测试得分" + ] + }, + { + "cell_type": "code", + "execution_count": 9, + "id": "9481765c-92b7-4a58-b192-582bb001f167", + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-11T06:51:42.447230Z", + "iopub.status.busy": "2025-03-11T06:51:42.446485Z", + "iopub.status.idle": "2025-03-11T06:51:42.461166Z", + "shell.execute_reply": "2025-03-11T06:51:42.459770Z", + "shell.execute_reply.started": "2025-03-11T06:51:42.447161Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok!\n" + ] + } + ], + "source": [ + "import json\n", + "import time\n", + "import my_module as My\n", + "\n", + "#list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','height','weight']\n", + "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp']\n", + "filename = 'data/result_海淀区老干部大学第一期.json'\n", + "with open(filename,'r') as fl:\n", + " dict2 = json.load(fl) \n", + "for k, v in dict2.items():\n", + " #print(k)\n", + " if v['sex'] == '男':\n", + " sex = 'M'\n", + " else:\n", + " sex = 'F' \n", + " if 'bmi' in v.keys():\n", + " #bmi_data = v['height']['成绩'].split()[0]+','+ v['weight']['成绩'].split()[0]\n", + " bmi_data = v['bmi']['成绩']\n", + " data1 = {'code':k,'sex':sex,'age':v['age'],'item':'HeightWeight','result':bmi_data}\n", + " dict2[k]['bmi'] = {}\n", + " dict2[k]['bmi']['成绩'] = bmi_data\n", + " dict2[k]['bmi']['score'] = My.cal_bmi(data1)\n", + " for item_en in list_item:\n", + " if item_en in v.keys(): \n", + " data1 = {'code':k,'sex':sex,'age':v['age'],'item':item_en,'result':float(v[item_en]['成绩'].split()[0])}\n", + " #print(k,v['name'])\n", + " dict2[k][item_en]['score'] = My.cal_score(data1)\n", + " #print(k,v[item_en]['成绩'],cal_score(data1))\n", + "\n", + "filename = f'data/result_海淀区老干部大学第一期.json'\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict2,fl , ensure_ascii=False) \n", + "print('ok!') " + ] + }, + { + "cell_type": "markdown", + "id": "bb21f690-115d-4b04-8752-5962a7eafeaf", + "metadata": {}, + "source": [ + "## 导出测试人员信息" + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "id": "63c81b75-9b3d-4001-a5da-32b7cf992843", + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-11T06:54:01.231518Z", + "iopub.status.busy": "2025-03-11T06:54:01.230854Z", + "iopub.status.idle": "2025-03-11T06:54:01.260859Z", + "shell.execute_reply": "2025-03-11T06:54:01.260166Z", + "shell.execute_reply.started": "2025-03-11T06:54:01.231459Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ok\n" + ] + } + ], + "source": [ + "import json\n", + "import openpyxl\n", + "\n", + "items = ['lung','grip','flexion','jump','pushup','situp','balance','reaction','step']\n", + "bmi = ['height','weight']\n", + "title = ['编号','姓名','性别','身高','体重','bmi','肺活量','得分','握力','得分','坐位体前屈','得分','纵跳','得分','俯卧撑','得分','一分钟仰卧起坐','得分','单脚站立','得分','选择反应时','得分','台阶指数','得分']\n", + "\n", + "filename = 'data/result_海淀区老干部大学第一期.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "\n", + "list1 = []\n", + "for k, v in dict1.items(): #print(k,dict2[str(k)]['name'])\n", + " list2 = []\n", + " list2.append(str(k).rjust(5,'0'))\n", + " list2.append(dict1[k]['name']) \n", + " list2.append(dict1[k]['sex'])\n", + " i = 0\n", + " if 'bmi' in dict1[k].keys():\n", + " list2.append(dict1[k]['bmi']['成绩'].split(',')[0])\n", + " list2.append(dict1[k]['bmi']['成绩'].split(',')[1])\n", + " list2.append(dict1[k]['bmi']['score'])\n", + " i+=1\n", + " else:\n", + " list2.append('') \n", + " list2.append('') \n", + " list2.append('') \n", + " \n", + " \n", + " for item in items:\n", + " if item in dict1[k].keys():\n", + " list2.append(dict1[k][item]['成绩'])\n", + " list2.append(dict1[k][item]['score']) \n", + " i+=1\n", + " elif item =='name':\n", + " list2.append(dict1[k][item])\n", + " else:\n", + " list2.append('') \n", + " list2.append('') \n", + " if i>2:\n", + " list1.append(list2)\n", + "filename = 'data/海淀区老干部大学第一期.xlsx'\n", + "wb = openpyxl.Workbook()\n", + "sheet = wb.active\n", + "sheet.append(title)\n", + "for row in list1:\n", + " sheet.append(row)\n", + " \n", + "wb.save(filename)\n", + "print('ok') " + ] + }, + { + "cell_type": "markdown", + "id": "c9bacc5b-6d55-41da-a30e-2862cf544a95", + "metadata": {}, + "source": [ + "## 导入问卷信息" + ] + }, + { + "cell_type": "code", + "execution_count": 18, + "id": "dd8ed4fe-5014-455f-993d-64a41519de6d", + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-13T07:54:13.739008Z", + "iopub.status.busy": "2025-03-13T07:54:13.738365Z", + "iopub.status.idle": "2025-03-13T07:54:13.762948Z", + "shell.execute_reply": "2025-03-13T07:54:13.762206Z", + "shell.execute_reply.started": "2025-03-13T07:54:13.738946Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "27\n" + ] + } + ], + "source": [ + "import json\n", + "import csv\n", + "import openpyxl\n", + "import time\n", + "from datetime import date\n", + "\n", + "dict1 = {}\n", + "\n", + "filename = 'data/result_海淀区老干部大学第一期.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "filename = 'data/海淀区老干部大学第一期.json'\n", + "with open(filename,'r') as fl:\n", + " dict3 = json.load(fl)\n", + "\n", + "phone = {}\n", + "for k,v in dict3.items():\n", + " if 'phone' in v.keys():\n", + " phone[v['phone']] = k\n", + "\n", + "list1 = []\n", + "filename = 'data/Survey_20250313.csv'\n", + "with open(filename,'r',newline='') as csv_file:\n", + " fl = csv.reader(csv_file,delimiter=',')\n", + " header = next(fl) \n", + " for line in fl:\n", + " list1.append(line)\n", + "\n", + "\n", + "\n", + "nn = 0\n", + "for item in list1:\n", + " if item[2] in phone.keys():\n", + " psy =[]\n", + " tcm = []\n", + " spine = []\n", + " for i in range(0,30):\n", + " psy.append(0)\n", + " for i in range(0,60):\n", + " tcm.append(0)\n", + " for i in range(0,26):\n", + " spine.append(0)\n", + " \n", + " content = json.loads(item[5])\n", + " if phone[item[2]] not in dict1.keys():\n", + " dict1[phone[item[2]]] = dict3[phone[item[2]]]\n", + " rq = date.fromisoformat(item[6].replace('/','-').split(' ')[0])\n", + " dict1[phone[item[2]]]['rq'] = item[6].replace('/','-').split(' ')[0]\n", + " for k, v in content.items():\n", + " if 'psyOld' in k:\n", + " i = int(k[6:])\n", + " psy[i-1] = int(v)\n", + " if 'tcm' in k:\n", + " i = int(k[3:])\n", + " tcm[i-1] = int(v)\n", + " if 'spine' in k:\n", + " i = int(k[5:])\n", + " spine[i-1] = int(v)\n", + " if 'psyOld' in item[5]:\n", + " \n", + " \n", + " dict1[phone[item[2]]]['psy_yangmiao_old'] = psy\n", + " if 'tcm' in item[5]:\n", + " \n", + " dict1[phone[item[2]]]['tcm'] = tcm\n", + " if 'spine' in item[5]:\n", + " for ii in range(25,23,-1):\n", + " spine[ii] = spine[ii-1]\n", + " spine[22] = 0 \n", + " dict1[phone[item[2]]]['spine'] = spine\n", + " birth = date.fromisoformat(dict3[phone[item[2]]]['birth'].replace('/','-'))\n", + " \n", + " days = (rq-birth).days \n", + " dict1[phone[item[2]]]['age'] = int(days/365)\n", + " dict1[phone[item[2]]]['month'] = int(days/365*12)\n", + " #print(phone[item[2]])\n", + " nn+=1\n", + "filename = 'data/result_海淀区老干部大学第一期-1.json'\n", + "\n", + "with open(filename,'w') as fl:\n", + " json.dump(dict1, fl, ensure_ascii=False)\n", + "print(nn)" + ] + }, + { + "cell_type": "markdown", + "id": "6b4ffd63-e38b-4a19-9ed7-5cfcc8a1882a", + "metadata": {}, + "source": [ + "## 生成报告" + ] + }, + { + "cell_type": "code", + "execution_count": 19, + "id": "b8b2fc93-9e8b-4bde-bc13-2de179de2b2d", + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-13T07:54:20.836739Z", + "iopub.status.busy": "2025-03-13T07:54:20.836013Z", + "iopub.status.idle": "2025-03-13T07:54:37.198424Z", + "shell.execute_reply": "2025-03-13T07:54:37.197192Z", + "shell.execute_reply.started": "2025-03-13T07:54:20.836671Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "29\n" + ] + } + ], + "source": [ + "import requests\n", + "import json\n", + "import openpyxl\n", + "\n", + "\n", + "headers = {\n", + " \"Content-Type\": \"application/json; charset=UTF-8\"\n", + " }\n", + "filename = 'data/result_海淀区老干部大学第一期-1.json'\n", + "with open(filename,'r') as fl:\n", + " dict1 = json.load(fl)\n", + "list1 = []\n", + "file_path ='./海淀老年大学第一期/'\n", + "list_item = ['lung','grip','flexion','jump','pushup','balance','reaction','step','situp','bmi']\n", + "i=0\n", + "list2 = []\n", + "for k, v in dict1.items():\n", + " list1 = []\n", + " mydata = {}\n", + " \n", + " id = str(k).rjust(4,\"0\")\n", + " mydata['path'] = file_path+id+'-'+ v['name']+'.pdf'\n", + " mydata['title'] = '海淀区老干部大学'\n", + " mydata['subtitle'] = '2025年春季体质提升班'\n", + " mydata['id'] = id\n", + " mydata['name'] = v['name']\n", + " if v['sex'] == '男':\n", + " mydata['gender'] = 'male'\n", + " else:\n", + " mydata['gender'] = 'female'\n", + " \n", + " mydata['month'] = v['month']\n", + " mydata['fits'] = {}\n", + " survey_list = ['tcm','psy_yangmiao_old','spine']\n", + " for item in survey_list:\n", + " if item in v.keys():\n", + " mydata.setdefault('surveys',{})\n", + " mydata['surveys'][item] = v[item]\n", + " \n", + " \n", + " #mydata['fits'] = {}\n", + " for item in list_item:\n", + " if item in v.keys():\n", + " mydata.setdefault('fits',{})\n", + " if item in ['lung','pushup','step','situp']:\n", + " mark = v[item]['成绩'].split()[0].split('.')[0]\n", + " else:\n", + " mark = v[item]['成绩'].split()[0]\n", + " mydata['fits'][item] = {'mark':mark,'score':v[item]['score']}\n", + " #if len(mydata['fits']) >2 or len(mydata['surveys']) >0:\n", + " #if len(mydata['fits']) >2 : \n", + " if len(mydata['fits']) >2 or 'surveys' in mydata.keys():\n", + " list1.append(mydata)\n", + " list2.append([k,v['name']])\n", + " i+=1\n", + " x = requests.post('http://localhost:3003', data = json.dumps(list1), headers=headers)\n", + " #print(id,v['name'],x.text)\n", + " #print(mydata)\n", + " #x.close()\n", + "print(i)" + ] + }, { "cell_type": "code", "execution_count": null, - "id": "26dbb945-2b62-4a46-9c72-aa321be7321d", + "id": "974b19a1-b88e-45a6-9426-9fdb9893101c", "metadata": {}, "outputs": [], "source": [] diff --git a/数据采集.ipynb b/数据采集.ipynb index 46b0faf..8c04d6f 100644 --- a/数据采集.ipynb +++ b/数据采集.ipynb @@ -1595,7 +1595,7 @@ "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", - "version": "3.10.12" + "version": "3.12.3" } }, "nbformat": 4, diff --git a/文件操作.ipynb b/文件操作.ipynb index 101fbba..0b3d51d 100644 --- a/文件操作.ipynb +++ b/文件操作.ipynb @@ -1,12 +1,5 @@ { "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "# 数字文件名转换为文本文件名" - ] - }, { "cell_type": "markdown", "metadata": {}, @@ -412,6 +405,132 @@ " fl1.writelines(list1)" ] }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 读取html文件" + ] + }, + { + "cell_type": "code", + "execution_count": 33, + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-09T02:47:43.965042Z", + "iopub.status.busy": "2025-03-09T02:47:43.964342Z", + "iopub.status.idle": "2025-03-09T02:47:44.398962Z", + "shell.execute_reply": "2025-03-09T02:47:44.398531Z", + "shell.execute_reply.started": "2025-03-09T02:47:43.964980Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n" + ] + } + ], + "source": [ + "from bs4 import BeautifulSoup\n", + "import re\n", + "from pathlib import Path\n", + "\n", + "def get_html_title(html_file_path):\n", + " with open(html_file_path, 'r', encoding='utf-8') as f:\n", + " html_content = f.read()\n", + " \n", + " soup = BeautifulSoup(html_content, 'html.parser')\n", + " title_tag = soup.title\n", + " \n", + " if title_tag and hasattr(title_tag, 'string'):\n", + " return title_tag.string.strip()\n", + " else:\n", + " return None # 标签不存在或内容为空\n", + "\n", + "\n", + "path = Path('./file/guzi')\n", + "markdown_content = []\n", + "files = [file for file in path.iterdir() if file.is_file()]\n", + "sorted_files = sorted(files, key=lambda f: f.name)\n", + "print(type(sorted_files))\n", + "for file in sorted_files:\n", + " if file.suffix=='.html':\n", + " markdown_content.append('## '+get_html_title(file))\n", + " with open(file, 'r', encoding='utf-8') as f:\n", + " html_content = f.read() \n", + " soup = BeautifulSoup(html_content, 'html.parser')\n", + " div = soup.find('div', class_=\"show-content\")\n", + " p_elements = div.find_all('p') \n", + " for div in p_elements:\n", + " markdown_content.append(div.get_text(strip=True))\n", + " #markdown_content.append('\\n')\n", + " if file.suffix=='.md':\n", + " with open(file, 'r', encoding='utf-8') as f:\n", + " txt_content = f.read()\n", + " lines = txt_content.splitlines()\n", + " for line in lines:\n", + " line = line.strip()\n", + " markdown_content.append(line)\n", + " \n", + "markdown_content = '\\n'.join(markdown_content)\n", + "with open('guzi.md', 'w', encoding='utf-8') as file:\n", + " file.write(markdown_content)" + ] + }, + { + "cell_type": "code", + "execution_count": 29, + "metadata": { + "execution": { + "iopub.execute_input": "2025-03-09T02:36:40.207084Z", + "iopub.status.busy": "2025-03-09T02:36:40.206317Z", + "iopub.status.idle": "2025-03-09T02:36:40.223987Z", + "shell.execute_reply": "2025-03-09T02:36:40.223140Z", + "shell.execute_reply.started": "2025-03-09T02:36:40.207017Z" + } + }, + "outputs": [], + "source": [ + "from bs4 import BeautifulSoup\n", + "import re\n", + "from pathlib import Path\n", + "\n", + "\n", + "def get_html_title(html_file_path):\n", + " with open(html_file_path, 'r', encoding='utf-8') as f:\n", + " html_content = f.read()\n", + " \n", + " soup = BeautifulSoup(html_content, 'html.parser')\n", + " title_tag = soup.title\n", + " \n", + " if title_tag and hasattr(title_tag, 'string'):\n", + " return title_tag.string.strip()\n", + " else:\n", + " return None # 标签不存在或内容为空\n", + "\n", + "\n", + "file='file/guzi/1-01.html'\n", + "markdown_content = []\n", + "markdown_content.append('## '+get_html_title(file))\n", + "with open(file, 'r', encoding='utf-8') as f:\n", + " html_content = f.read()\n", + " \n", + " soup = BeautifulSoup(html_content, 'html.parser')\n", + " div = soup.find('div', class_=\"show-content\")\n", + " p_elements = div.find_all('p')\n", + " \n", + " for div in p_elements:\n", + " markdown_content.append(div.get_text(strip=True))\n", + " markdown_content.append('\\n')\n", + "markdown_content = '\\n'.join(markdown_content)\n", + "with open('1-01.md', 'w', encoding='utf-8') as file:\n", + " file.write(markdown_content)\n", + "\n" + ] + }, { "cell_type": "markdown", "metadata": {},