Files
jupyter/围棋点校文献/文件处理.ipynb
T
2024-08-04 00:13:12 +08:00

7.9 KiB

选择白子黑子图片

In [ ]:
import glob
import os,shutil
from pathlib import Path

filepath = 'img/白/'

files = glob.glob(f'{filepath}*.jpg')
for fn in files:
    fi_name =Path(fn).stem
    if int(fi_name) % 2 !=0:
        shutil.copy(fn, './img')
In [ ]:
import glob
import os,shutil
from pathlib import Path

filepath = 'img/黑/'

files = glob.glob(f'{filepath}*.jpg')
for fn in files:
    fi_name =Path(fn).stem
    if int(fi_name) % 2 ==0:
        shutil.copy(fn, './img')

棋子替代

In [ ]:
import glob
import os,shutil
from pathlib import Path
import re

def find_all_numbers(text):
    return re.findall(r'\d+', text)

def replace_numbers(text):
    pattern = r'\d+'
    replacement = lambda x:'{{ '+ f'img_{x.group()}'+' }}'
    modified_text = re.sub(pattern, replacement, text)
    return modified_text

def replace_numbers_with_braces(text):
    # 使用正则表达式查找连续的数字
    pattern = re.compile(r'\d+')
    
    # 使用re.sub()进行替换
    result = pattern.sub(lambda x: f"{{{{ img_{x.group(0)} }}}}", text)
    
    return result

list1 = []
filepath = '文本文件/'
files = glob.glob(f'{filepath}*.txt')
for fn in files:
    with open(fn, "r") as f:
        data = f.readlines()
        for s in data:
            ss = replace_numbers_with_braces(s)
            list1 = find_all_numbers(ss)
            list1 = list(set(list1))
            list1.sort()
            print(ss,list1)
In [36]:
import glob
import os,shutil
from pathlib import Path
import re
from docx import Document

def find_all_numbers(text):
    return re.findall(r'\d+', text)

def replace_numbers(text):
    pattern = r'\d+'
    replacement = lambda x:'{{ '+ f'img_{x.group()}'+' }}'
    modified_text = re.sub(pattern, replacement, text)
    return modified_text

def replace_numbers_with_braces(text):
    # 使用正则表达式查找连续的数字
    pattern = re.compile(r'\d+')
    
    # 使用re.sub()进行替换
    result = pattern.sub(lambda x: f"{{{{ img_{x.group(0)} }}}}", text)
    
    return result
doc = Document()

list1 = []
filepath = '文本文件/'
files = glob.glob(f'{filepath}*.txt')
for fn in files:
    with open(fn, "r") as f:
        paragraph3 = doc.add_paragraph()
        data = f.readlines()
        for s in data:
            list2 = []
            ss = replace_numbers_with_braces(s)
            list2 = find_all_numbers(s)
            if len(list2)>0:
                for item in list2:
                    list1.append(item)
            
            list1 = list(set(list1))
            list1.sort()
            #paragraph3 = doc.add_paragraph()
            paragraph3.add_run(ss.replace('\r', ''))
            
print(list1)
doc.save('槐荫堂9.docx')
['10', '100', '101', '102', '103', '104', '105', '106', '107', '108', '109', '11', '110', '111', '112', '113', '114', '115', '116', '117', '118', '119', '12', '120', '121', '122', '123', '124', '125', '126', '127', '128', '129', '130', '131', '132', '133', '134', '135', '136', '137', '138', '139', '14', '140', '141', '142', '143', '144', '145', '146', '147', '148', '149', '15', '150', '151', '152', '153', '154', '155', '156', '157', '158', '159', '16', '160', '161', '162', '163', '164', '165', '166', '167', '168', '169', '17', '170', '171', '173', '174', '175', '176', '177', '178', '179', '18', '180', '181', '182', '183', '184', '185', '186', '187', '188', '189', '19', '190', '191', '192', '193', '195', '198', '2', '20', '200', '201', '203', '204', '209', '21', '212', '213', '214', '22', '221', '222', '227', '228', '23', '230', '232', '24', '242', '243', '245', '246', '248', '249', '25', '251', '254', '255', '26', '27', '276', '28', '29', '290', '293', '3', '30', '304', '31', '32', '33', '34', '35', '36', '37', '38', '39', '40', '41', '42', '43', '44', '45', '46', '47', '48', '49', '50', '51', '52', '53', '54', '55', '56', '57', '58', '59', '6', '60', '61', '62', '63', '64', '65', '66', '67', '68', '69', '7', '70', '71', '72', '73', '74', '75', '76', '77', '78', '79', '8', '80', '81', '82', '83', '84', '85', '86', '87', '88', '89', '9', '90', '91', '92', '93', '94', '95', '96', '97', '98', '99']
In [37]:
import openpyxl
from docxtpl import DocxTemplate,InlineImage
from docx.shared import Mm

dict1 = {}

tpl = DocxTemplate("槐荫堂9.docx")
for item in list1:    
    dict1['img_'+str(item)] = InlineImage(tpl, image_descriptor=f'./img/{str(item)}.jpg',width=Mm(4))
#dict1['img_radar'] = InlineImage(tpl, image_descriptor=dict1['radar'])

tpl.render(dict1)
tpl.save('槐荫堂第九卷.docx')
In [ ]: