Files
jupyter/围棋点校文献/文件处理.ipynb
T
2024-08-28 22:46:30 +08:00

259 lines
7.3 KiB
Plaintext

{
"cells": [
{
"cell_type": "markdown",
"id": "0f939dd1-de12-4b54-bd08-16ebbbfdbfdd",
"metadata": {},
"source": [
"## 选择白子黑子图片"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0569e182-b008-4173-ab0d-7f4a2904f72e",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import glob\n",
"import os,shutil\n",
"from pathlib import Path\n",
"\n",
"filepath = 'img/白/'\n",
"\n",
"files = glob.glob(f'{filepath}*.jpg')\n",
"for fn in files:\n",
" fi_name =Path(fn).stem\n",
" if int(fi_name) % 2 !=0:\n",
" shutil.copy(fn, './img')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "44683ca5-166c-4d3a-9447-6fe46ab7c12d",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import glob\n",
"import os,shutil\n",
"from pathlib import Path\n",
"\n",
"filepath = 'img/黑/'\n",
"\n",
"files = glob.glob(f'{filepath}*.jpg')\n",
"for fn in files:\n",
" fi_name =Path(fn).stem\n",
" if int(fi_name) % 2 ==0:\n",
" shutil.copy(fn, './img')"
]
},
{
"cell_type": "markdown",
"id": "230c5cb3-ee5b-4490-b0a2-6eb62461204b",
"metadata": {},
"source": [
"## 棋子替代"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "3ecfe0a8-3af1-4f48-a2b0-6c3eb25df07d",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import glob\n",
"import os,shutil\n",
"from pathlib import Path\n",
"import re\n",
"\n",
"def find_all_numbers(text):\n",
" return re.findall(r'\\d+', text)\n",
"\n",
"def replace_numbers(text):\n",
" pattern = r'\\d+'\n",
" replacement = lambda x:'{{ '+ f'img_{x.group()}'+' }}'\n",
" modified_text = re.sub(pattern, replacement, text)\n",
" return modified_text\n",
"\n",
"def replace_numbers_with_braces(text):\n",
" # 使用正则表达式查找连续的数字\n",
" pattern = re.compile(r'\\d+')\n",
" \n",
" # 使用re.sub()进行替换\n",
" result = pattern.sub(lambda x: f\"{{{{ img_{x.group(0)} }}}}\", text)\n",
" \n",
" return result\n",
"\n",
"list1 = []\n",
"filepath = '文本文件/'\n",
"files = glob.glob(f'{filepath}*.md')\n",
"for fn in files:\n",
" with open(fn, \"r\") as f:\n",
" data = f.readlines()\n",
" for s in data:\n",
" ss = replace_numbers_with_braces(s)\n",
" list1 = find_all_numbers(ss)\n",
" list1 = list(set(list1))\n",
" list1.sort()\n",
" print(ss,list1)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "14696c7a-d44e-4c99-9d28-a810d991ed20",
"metadata": {},
"outputs": [],
"source": [
"import glob\n",
"import os,shutil\n",
"from pathlib import Path\n",
"import re\n",
"from docx import Document\n",
"import openpyxl\n",
"from docxtpl import DocxTemplate,InlineImage\n",
"from docx.shared import Mm\n",
"\n",
"def find_all_numbers(text):\n",
" return re.findall(r'\\d+', text)\n",
"\n",
"def replace_numbers(text):\n",
" pattern = r'\\d+'\n",
" replacement = lambda x:'{{ '+ f'img_{x.group()}'+' }}'\n",
" modified_text = re.sub(pattern, replacement, text)\n",
" return modified_text\n",
"\n",
"def replace_numbers_with_braces(text):\n",
" # 使用正则表达式查找连续的数字\n",
" pattern = re.compile(r'\\d+')\n",
" \n",
" # 使用re.sub()进行替换\n",
" result = pattern.sub(lambda x: f\"{{{{ img_{x.group(0)} }}}}\", text)\n",
" \n",
" return result\n",
"#doc = Document()\n",
"\n",
"\n",
"filepath = '文本文件/'\n",
"files = glob.glob(f'{filepath}*.txt')\n",
"for fn in files:\n",
" p = Path(fn)\n",
" with open(fn, \"r\") as f:\n",
" list1 = []\n",
" doc = Document()\n",
" paragraph3 = doc.add_paragraph()\n",
" data = f.readlines()\n",
" for s in data:\n",
" list2 = []\n",
" ss = replace_numbers_with_braces(s)\n",
" list2 = find_all_numbers(s)\n",
" if len(list2)>0:\n",
" for item in list2:\n",
" list1.append(item)\n",
" \n",
" list1 = list(set(list1))\n",
" list1.sort()\n",
" #paragraph3 = doc.add_paragraph()\n",
" paragraph3.add_run(ss.replace('\\r', ''))\n",
" print(p.stem)\n",
" print(list1)\n",
" if len(list1) >0:\n",
" #paragraph3 = doc.add_paragraph()\n",
" #paragraph3.add_run(ss.replace('\\r', ''))\n",
" doc.save(f'./templete/{p.stem}.docx')\n",
" dict1 = {}\n",
" tpl = DocxTemplate(f'./templete/{p.stem}.docx')\n",
" for item in list1:\n",
" dict1['img_'+str(item)] = InlineImage(tpl, image_descriptor=f'./img/{str(item)}.jpg',width=Mm(4))\n",
" tpl.render(dict1)\n",
" tpl.save(f'./docx/{p.stem}.docx')\n",
"#print(list1)\n",
"#doc.save('离垢居谈棋.docx')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8d3f1a7a-1322-48b5-9b14-c7ba8f38bacb",
"metadata": {
"tags": []
},
"outputs": [],
"source": [
"import openpyxl\n",
"from docxtpl import DocxTemplate,InlineImage\n",
"from docx.shared import Mm\n",
"\n",
"dict1 = {}\n",
"\n",
"tpl = DocxTemplate(\"离垢居谈棋.docx\")\n",
"for item in list1: \n",
" dict1['img_'+str(item)] = InlineImage(tpl, image_descriptor=f'./img/黑先/{str(item)}.jpg',width=Mm(4))\n",
"#dict1['img_radar'] = InlineImage(tpl, image_descriptor=dict1['radar'])\n",
"\n",
"tpl.render(dict1)\n",
"tpl.save('离垢居谈棋(黑先).docx')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "efc716f8-0ec7-4e0d-9960-676764965be4",
"metadata": {},
"outputs": [],
"source": [
"import openpyxl\n",
"from docxtpl import DocxTemplate,InlineImage\n",
"from docx.shared import Mm\n",
"\n",
"dict1 = {}\n",
"\n",
"tpl = DocxTemplate(\"离垢居谈棋.docx\")\n",
"for item in list1: \n",
" dict1['img_'+str(item)] = InlineImage(tpl, image_descriptor=f'./img/白先/{str(item)}.jpg',width=Mm(4))\n",
"#dict1['img_radar'] = InlineImage(tpl, image_descriptor=dict1['radar'])\n",
"\n",
"tpl.render(dict1)\n",
"tpl.save('离垢居谈棋(白先).docx')"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "fcfd6cab-d83f-4b79-9a0b-05570ca6cecb",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.12.3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}