Files

91 lines
4.1 KiB
Python
Raw Permalink Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""01_merge_sunben.py — 《神农本草经·清·孙星衍》(殆知阁本) 合并整理版生成
规则(2026-09-22 用户指令):
1. 重复出现的「卷一/二/三 上/中/下经」卷标行只保留一次(同卷连续出现的合并);
2. 药物条目块(药名单行+味…+吴普/名医/案)合并为组织化结构,药名转为 ####;
3. 原件正文一字不改,仅做结构性合并(去重卷标、条目并块、空行规整);
4. 幂等可重跑;输出至 校核参考/殆知阁-孙星衍辑本/神农本草经_孙星衍辑本_合并整理版.md
"""
import re, sys, json, hashlib, os
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) # 01_来源数据/..
SRC = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本',
'神农本草经_孙星衍辑本_原文快照_20260922.md')
OUT = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本',
'神农本草经_孙星衍辑本_合并整理版.md')
STAT = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本',
'合并整理版_结构.json')
VOL_RE = re.compile(r'^卷([一二三四])\s* ?\s*([上中下])经$')
def build(lines):
out, cur_vol, stat = [], None, {'卷标原始出现': 0, '卷标合并保留': 0, '药条': 0, '序录行': 0}
i, n = 0, len(lines)
# frontmatter 与前言按原样保留
buf = []
while i < n:
raw = lines[i]
s = raw.strip()
m = VOL_RE.match(s)
if m:
stat['卷标原始出现'] += 1
vol = f'## 卷{m.group(1)} {m.group(2)}经'
# 判断此卷标行是否为重复标记:其前方上一非空行若同卷经 -> 跳过
prev_nonempty = next((l.strip() for l in reversed(buf) if l.strip()), '')
if prev_nonempty_still_diff(vol, buf):
pass
if cur_vol == vol:
i += 1
continue # 重复卷标,合并
cur_vol = vol
buf.append('')
buf.append('---')
buf.append(vol)
buf.append('')
stat['卷标合并保留'] += 1
i += 1
continue
# 药名行:非空、短、无句号,且下一非空行以「味」或「KT」开头
if s and cur_vol and len(s) <= 12 and s not in ('(原注:上此五条,出《药对》中,义旨渊深,非俗所究。虽莫可遵用,而是主统之本,故亦载之。)',):
nxt = next((l.strip() for l in lines[i+1:] if l.strip()), '')
nxt_is_body = nxt.startswith('味') or nxt == 'KT'
if nxt_is_body and not s.endswith(('。',)) and not s.startswith('《'):
buf.append(f'#### {s}')
stat['药条'] += 1
i += 1
continue
buf.append(raw.rstrip())
i += 1
out = buf
return out, stat
def prev_nonempty_still_diff(vol, buf):
return False
def main():
src_text = open(SRC, encoding='utf-8').read()
lines = src_text.split('\n')
body, stat = build(lines)
text = '\n'.join(body)
# 空行规整:连续3+空行→1空行(结构性,不改文字)
text = re.sub(r'\n{4,}', '\n\n\n', text)
open(OUT, 'w', encoding='utf-8').write(text)
md5 = hashlib.md5(open(OUT, 'rb').read()).hexdigest()
stat['md5'] = md5
stat['bytes'] = os.path.getsize(OUT)
stat['行数'] = text.count('\n') + 1
# 药条按卷统计
sec, cnt = None, {'卷一 上经': 0, '卷二 中经': 0, '卷三 下经': 0}
for l in text.split('\n'):
s = l.strip()
m = VOL_RE.match(s) or (s.startswith('## ') and VOL_RE.match(s[3:]))
if m: sec = f'卷{m.group(1)}\u3000{m.group(2)}经'
elif s.startswith('#### ') and sec in cnt: cnt[sec] += 1
stat['药条_按卷'] = cnt
json.dump(stat, open(STAT, 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
print(json.dumps(stat, ensure_ascii=False, indent=1))
if __name__ == '__main__':
main()