#!/usr/bin/env python3 # -*- coding: utf-8 -*- """01_merge_sunben.py — 《神农本草经·清·孙星衍》(殆知阁本) 合并整理版生成 规则(2026-09-22 用户指令): 1. 重复出现的「卷一/二/三 上/中/下经」卷标行只保留一次(同卷连续出现的合并); 2. 药物条目块(药名单行+味…+吴普/名医/案)合并为组织化结构,药名转为 ####; 3. 原件正文一字不改,仅做结构性合并(去重卷标、条目并块、空行规整); 4. 幂等可重跑;输出至 校核参考/殆知阁-孙星衍辑本/神农本草经_孙星衍辑本_合并整理版.md """ import re, sys, json, hashlib, os BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) # 01_来源数据/.. SRC = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本', '神农本草经_孙星衍辑本_原文快照_20260922.md') OUT = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本', '神农本草经_孙星衍辑本_合并整理版.md') STAT = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本', '合并整理版_结构.json') VOL_RE = re.compile(r'^卷([一二三四])\s* ?\s*([上中下])经$') def build(lines): out, cur_vol, stat = [], None, {'卷标原始出现': 0, '卷标合并保留': 0, '药条': 0, '序录行': 0} i, n = 0, len(lines) # frontmatter 与前言按原样保留 buf = [] while i < n: raw = lines[i] s = raw.strip() m = VOL_RE.match(s) if m: stat['卷标原始出现'] += 1 vol = f'## 卷{m.group(1)} {m.group(2)}经' # 判断此卷标行是否为重复标记:其前方上一非空行若同卷经 -> 跳过 prev_nonempty = next((l.strip() for l in reversed(buf) if l.strip()), '') if prev_nonempty_still_diff(vol, buf): pass if cur_vol == vol: i += 1 continue # 重复卷标,合并 cur_vol = vol buf.append('') buf.append('---') buf.append(vol) buf.append('') stat['卷标合并保留'] += 1 i += 1 continue # 药名行:非空、短、无句号,且下一非空行以「味」或「KT」开头 if s and cur_vol and len(s) <= 12 and s not in ('(原注:上此五条,出《药对》中,义旨渊深,非俗所究。虽莫可遵用,而是主统之本,故亦载之。)',): nxt = next((l.strip() for l in lines[i+1:] if l.strip()), '') nxt_is_body = nxt.startswith('味') or nxt == 'KT' if nxt_is_body and not s.endswith(('。',)) and not s.startswith('《'): buf.append(f'#### {s}') stat['药条'] += 1 i += 1 continue buf.append(raw.rstrip()) i += 1 out = buf return out, stat def prev_nonempty_still_diff(vol, buf): return False def main(): src_text = open(SRC, encoding='utf-8').read() lines = src_text.split('\n') body, stat = build(lines) text = '\n'.join(body) # 空行规整:连续3+空行→1空行(结构性,不改文字) text = re.sub(r'\n{4,}', '\n\n\n', text) open(OUT, 'w', encoding='utf-8').write(text) md5 = hashlib.md5(open(OUT, 'rb').read()).hexdigest() stat['md5'] = md5 stat['bytes'] = os.path.getsize(OUT) stat['行数'] = text.count('\n') + 1 # 药条按卷统计 sec, cnt = None, {'卷一 上经': 0, '卷二 中经': 0, '卷三 下经': 0} for l in text.split('\n'): s = l.strip() m = VOL_RE.match(s) or (s.startswith('## ') and VOL_RE.match(s[3:])) if m: sec = f'卷{m.group(1)}\u3000{m.group(2)}经' elif s.startswith('#### ') and sec in cnt: cnt[sec] += 1 stat['药条_按卷'] = cnt json.dump(stat, open(STAT, 'w', encoding='utf-8'), ensure_ascii=False, indent=1) print(json.dumps(stat, ensure_ascii=False, indent=1)) if __name__ == '__main__': main()