This commit is contained in:
512song committed 2023-12-25 22:26:20 +08:00
1 parent bc7cc4f24d
commit 69fba3ca28
1 file changed
+82 -18
+82 -18
View File
@@ -2251,9 +2251,16 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": 325,
"id": "3fa6a1a9-63af-4f4e-9d30-b94f4357f7dc", "id": "3fa6a1a9-63af-4f4e-9d30-b94f4357f7dc",
"metadata": { "metadata": {
"execution": {
"iopub.execute_input": "2023-12-20T01:58:47.982585Z",
"iopub.status.busy": "2023-12-20T01:58:47.982111Z",
"iopub.status.idle": "2023-12-20T01:58:48.241957Z",
"shell.execute_reply": "2023-12-20T01:58:48.241442Z",
"shell.execute_reply.started": "2023-12-20T01:58:47.982530Z"
},
"tags": [] "tags": []
}, },
"outputs": [], "outputs": [],
@@ -2263,8 +2270,8 @@
"import glob\n", "import glob\n",
"from pathlib import Path\n", "from pathlib import Path\n",
"\n", "\n",
"fi_path = '/home/songyi/pdf-typescript/134'\n", "fi_path = '/home/songyi/pdf-typescript/天津石化问卷'\n",
"new_path = 'file/134'\n", "new_path = 'file/天津石化问卷1'\n",
"old = []\n", "old = []\n",
"dict2 = {}\n", "dict2 = {}\n",
"\n", "\n",
@@ -2275,7 +2282,7 @@
"for fn in fls:\n", "for fn in fls:\n",
" fi_name =Path(fn).stem.split('-')[0]\n", " fi_name =Path(fn).stem.split('-')[0]\n",
" code = int(fi_name)\n", " code = int(fi_name)\n",
" unit_path = Path(new_path,dict1[str(code)]['unit'])\n", " unit_path = Path(new_path,dict1[str(code)]['unit'],dict1[str(code)]['sub_unit'])\n",
" unit_path.mkdir(parents = True, exist_ok = True)\n", " unit_path.mkdir(parents = True, exist_ok = True)\n",
" n_name = Path(unit_path,Path(fn).stem+'.pdf')\n", " n_name = Path(unit_path,Path(fn).stem+'.pdf')\n",
" if not os.path.exists(n_name):\n", " if not os.path.exists(n_name):\n",
@@ -2491,22 +2498,37 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": 324,
"id": "8397633f-fddb-4cfa-b9bd-e38f3fe6772c", "id": "8397633f-fddb-4cfa-b9bd-e38f3fe6772c",
"metadata": { "metadata": {
"execution": {
"iopub.execute_input": "2023-12-19T04:46:55.234792Z",
"iopub.status.busy": "2023-12-19T04:46:55.234584Z",
"iopub.status.idle": "2023-12-19T04:46:55.720872Z",
"shell.execute_reply": "2023-12-19T04:46:55.720417Z",
"shell.execute_reply.started": "2023-12-19T04:46:55.234777Z"
},
"tags": [] "tags": []
}, },
"outputs": [], "outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [ "source": [
"import os,sys,shutil\n", "import os,sys,shutil\n",
"import json\n", "import json\n",
"import glob\n", "import glob\n",
"from pathlib import Path\n", "from pathlib import Path\n",
"\n", "\n",
"fi_path = '/home/songyi/pdf-typescript/134'\n", "fi_path = '/home/songyi/pdf-typescript/天津石化问卷'\n",
"new_path = 'file/134_1'\n", "new_path = 'file/134_2'\n",
"old = []\n", "old = []\n",
"filename = 'data/result_天津231017.json'\n", "filename = 'data/result_天津问卷2.json'\n",
"with open(filename,'r') as fl:\n", "with open(filename,'r') as fl:\n",
" dict1 = json.load(fl)\n", " dict1 = json.load(fl)\n",
"\n", "\n",
@@ -2535,9 +2557,16 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": 323,
"id": "35d1738a-1ec1-4764-bcbd-c1333fac6bfd", "id": "35d1738a-1ec1-4764-bcbd-c1333fac6bfd",
"metadata": { "metadata": {
"execution": {
"iopub.execute_input": "2023-12-19T04:45:06.677002Z",
"iopub.status.busy": "2023-12-19T04:45:06.676614Z",
"iopub.status.idle": "2023-12-19T04:45:06.915694Z",
"shell.execute_reply": "2023-12-19T04:45:06.915217Z",
"shell.execute_reply.started": "2023-12-19T04:45:06.676971Z"
},
"tags": [] "tags": []
}, },
"outputs": [], "outputs": [],
@@ -2612,8 +2641,9 @@
" days = (rq-birth).days \n", " days = (rq-birth).days \n",
" dict1[str(item[4])]['age'] = int(days/365)\n", " dict1[str(item[4])]['age'] = int(days/365)\n",
" dict1[str(item[4])]['month'] = int(days/365*12)\n", " dict1[str(item[4])]['month'] = int(days/365*12)\n",
" dict1[str(item[4])]['rq'] = item[6].replace('/','-').split(' ')[0]\n",
"#print(dict1)\n", "#print(dict1)\n",
"filename = 'data/result_天津问卷1.json'\n", "filename = 'data/result_天津问卷2.json'\n",
"\n", "\n",
"with open(filename,'w') as fl:\n", "with open(filename,'w') as fl:\n",
" json.dump(dict1, fl, ensure_ascii=False) " " json.dump(dict1, fl, ensure_ascii=False) "
@@ -2656,10 +2686,10 @@
"for k,v in dict1.items(): \n", "for k,v in dict1.items(): \n",
" list1 = []\n", " list1 = []\n",
" mydata = {} \n", " mydata = {} \n",
" id = k.rjust(3,\"0\")\n", " id = k.rjust(8,\"0\")\n",
" mydata['path'] = fiie_path+id+'-'+ v['name']+'.pdf'\n", " mydata['path'] = fiie_path+id+'-'+ v['name']+'.pdf'\n",
" mydata['title'] = '中石化(天津)石油化工\\n有限公司'\n", " mydata['title'] = '中石化(天津)石油化工\\n有限公司'\n",
" mydata['subtitle'] = ''\n", " mydata['subtitle'] = v['unit']\n",
" mydata['id'] = id\n", " mydata['id'] = id\n",
" mydata['name'] = v['name']\n", " mydata['name'] = v['name']\n",
" if v['sex'] == '男':\n", " if v['sex'] == '男':\n",
@@ -2703,12 +2733,27 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": 319,
"id": "ca0780fe-5d74-4d4e-ad1d-38813040c778", "id": "ca0780fe-5d74-4d4e-ad1d-38813040c778",
"metadata": { "metadata": {
"execution": {
"iopub.execute_input": "2023-12-18T11:26:08.026183Z",
"iopub.status.busy": "2023-12-18T11:26:08.025794Z",
"iopub.status.idle": "2023-12-18T11:26:08.295041Z",
"shell.execute_reply": "2023-12-18T11:26:08.294596Z",
"shell.execute_reply.started": "2023-12-18T11:26:08.026152Z"
},
"tags": [] "tags": []
}, },
"outputs": [], "outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ok\n"
]
}
],
"source": [ "source": [
"import json\n", "import json\n",
"import csv\n", "import csv\n",
@@ -2750,10 +2795,29 @@
}, },
{ {
"cell_type": "code", "cell_type": "code",
"execution_count": null, "execution_count": 318,
"id": "9b469138-1da9-43ca-98e8-8dc6cb3916ce", "id": "9b469138-1da9-43ca-98e8-8dc6cb3916ce",
"metadata": {}, "metadata": {
"outputs": [], "execution": {
"iopub.execute_input": "2023-12-18T11:18:59.322901Z",
"iopub.status.busy": "2023-12-18T11:18:59.322423Z",
"iopub.status.idle": "2023-12-18T11:18:59.414582Z",
"shell.execute_reply": "2023-12-18T11:18:59.414174Z",
"shell.execute_reply.started": "2023-12-18T11:18:59.322852Z"
},
"tags": []
},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"1730629\n",
"3499711\n",
"3145556\n"
]
}
],
"source": [ "source": [
"import json\n", "import json\n",
"import csv\n", "import csv\n",