Files

88 lines
5.0 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""05_build_crossref_report.py — 参校本对照报告(初对照)生成
输入:参校本对照_药目核对.json(03 脚本产物)+ _tmp_实字异文.json(04 脚本产物)
输出:02_加工数据/参校本对照_孙本-维基.md"""
import json, os, datetime
BASE = '/home/songyi/Documents/ai_agent_scraper_study/data/神农本草经'
D = json.load(open(os.path.join(BASE, '02_加工数据/参校本对照_药目核对.json'), encoding='utf-8'))
SUB = json.load(open(os.path.join(BASE, '02_加工数据/_tmp_实字异文.json'), encoding='utf-8'))
same = D['同文']
diff = D['异文条目']
only_sun = D['未对齐_仅孙本']
only_wk = D['未对齐_仅维基']
L = []
L.append('# 参校本对照 — 孙本(第一参照本·殆知阁) ⇄ 维基文库(第二参校本) 初步对照')
L.append('')
L.append('> 性质:**两参校本互校**,双方原文均未改动;差异只登记。本对照供实体书 OCR 底本核对时作为「第三源」线索定位——凡两参校本一致的异文、或两方各有主张的异文,都圈出来供底本定夺。')
L.append('> 规整规则(对照镜像,不是改字):① 繁→简单向转换(opencc t2s,仅对照视图)② 去括注(孙本多《御览》括注)③ 去句读(,。、;:)——均不动任何一方的原始文件。')
L.append('> 药名对齐用显式异体映射表(如 丹沙=丹砂、朴消=朴硝、慈石=磁石、茈胡=柴胡),映射SRﺎ只用于对照键。')
L.append('')
today = datetime.date.today().isoformat()
L.append('| 项 | 数 |')
L.append('|----|----|')
L.append('| 孙本药条(合并整理版口径) | 337 |')
L.append('| 维基药条 | 359 |')
L.append('| 对齐条目(名称可对) | %d |' % (len(same)+len(diff)))
L.append('| 两参校本文同 | %d |' % len(same))
L.append('| 对齐但文异(含体例差) | %d |' % len(diff))
L.append('| 剔除体例差后·实字异文条目 | **%d** |' % len(SUB := json.load(open(os.path.join(BASE, '02_加工数据/_tmp_实字异文.json'), encoding='utf-8'))))
L.append('| 未对齐·仅孙本 | %d |' % len(only_sun))
L.append('| 未对齐·仅维基 | %d |' % len(only_wk))
L.append('')
L.append('## 一、未对齐药名清单(四类成因)')
L.append('')
L.append('### 甲. 繁简异体(同一味药,用字不同——映射表已登记,属**版本体例差**,非内容差)')
L.append('')
L.append('如:丹沙/丹砂、朴消/朴硝、慈石/磁石、茈胡/柴胡、署豫/山芋系统、委萎/女萎……此层维基作繁体、孙本作简体+宋本用字,属繁简/异体书写差。')
L.append('')
L.append('### 未登记进映射表·成对残项(待人工归并):')
L.append('')
sun_left = [n for n in only_sun if n not in ('KT','下经','木','本','石','菌','青','黄','萆','蓬','虫','蓄')]
wk_left = only_wk
L.append('| 孙本侧 | 维基侧(可能对应) |')
L.append('|---------|-------------------|')
for s in sun_left:
L.append('| %s | (待人工归并) |' % s)
for w in wk_left[:20]:
L.append('| (待人工归并) | %s |' % w)
L.append('')
L.append('## 二、同文条目(两层规整后完全一致)[%d 条]' % len(same))
L.append('')
L.append('、'.join(same))
L.append('')
L.append('## 三、实字异文[粗筛 %d 条](样本)' % len(SUB))
L.append('')
L.append('> 注意:初筛自动剔除「治/有毒/无毒/小/微」等体例差;**繁简残留与形近字仍混入**,须人工定夺。每条列出「孙[原文]→维[原文]」。')
L.append('')
L.append('| 药名 | 差异(孙→维基) |')
L.append('|------|-----------------|')
for n, frags in list(SUB.items()):
cells = ';'.join('孙[%s]→维[%s]' % (f['孙'], f['维基']) for f in frags[:5])
L.append('| %s | %s |' % (n, cells.replace('|','\\|')))
L.append('')
L.append('## 四、全书结构差异(不在药条层面的差异)')
L.append('')
L.append('1. 孙本有《吴普》曰/《名医》曰/案:三重考据——维基文库全无')
L.append('2. 孙本有《御览》括注援引(《太平御览》引……)——维基无')
L.append('3. 维基有完整序录13行——孙本序录仅3段+目录行')
L.append('4. 维基三品分部按「玉石/草/木/果菜/米穀/蟲獸」——孙本按「卷一/二/三」')
L.append('5. 孙本卷三末附七情制使231种+《药对》五条——维基无')
L.append('6. 孙本药目数371>365、识别口径338——<strong>总计待底本核对后方能定</strong>')
L.append('')
L.append('## 五、下一步')
L.append('')
L.append('1. 未对齐药名人工归并清单→等底本OCR后定字')
L.append('2. 实字异文 228 条→等底本 OCR 后逐条定夺')
L.append('3. 序录 3 段 vs 13 行文本的取舍——以底本为准')
L.append('4. 孙本「KT」占位药条→底本核名')
L.append('')
md = '\n'.join(L)
md = md.replace('>>', '>')
open(os.path.join(BASE, '02_加工数据/参校本对照_孙本-维基.md'), 'w', encoding='utf-8').write(md)
print('已写 02_加工数据/参校本对照_孙本-维基.md,行数', md.count('\n')+1, 'md5:', __import__('hashlib').md5(md.encode()).hexdigest())