初始归档:健康管理资料库(大医网/药膳/体质监测/中医理论等12个资料集,7712个文件)

This commit is contained in:
512song committed 2026-09-23 21:59:25 +08:00
commit 80ae3811cf
7712 files changed
+4628547

No files matched your search

@@ -0,0 +1,90 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""01_merge_sunben.py — 《神农本草经·清·孙星衍》(殆知阁本) 合并整理版生成
规则(2026-09-22 用户指令):
1. 重复出现的「卷一/二/三 上/中/下经」卷标行只保留一次(同卷连续出现的合并);
2. 药物条目块(药名单行+味…+吴普/名医/案)合并为组织化结构,药名转为 ####;
3. 原件正文一字不改,仅做结构性合并(去重卷标、条目并块、空行规整);
4. 幂等可重跑;输出至 校核参考/殆知阁-孙星衍辑本/神农本草经_孙星衍辑本_合并整理版.md
"""
import re, sys, json, hashlib, os
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) # 01_来源数据/..
SRC = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本',
'神农本草经_孙星衍辑本_原文快照_20260922.md')
OUT = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本',
'神农本草经_孙星衍辑本_合并整理版.md')
STAT = os.path.join(BASE, '01_来源数据', '校核参考', '殆知阁-孙星衍辑本',
'合并整理版_结构.json')
VOL_RE = re.compile(r'^卷([一二三四])\s* ?\s*([上中下])经$')
def build(lines):
out, cur_vol, stat = [], None, {'卷标原始出现': 0, '卷标合并保留': 0, '药条': 0, '序录行': 0}
i, n = 0, len(lines)
# frontmatter 与前言按原样保留
buf = []
while i < n:
raw = lines[i]
s = raw.strip()
m = VOL_RE.match(s)
if m:
stat['卷标原始出现'] += 1
vol = f'## 卷{m.group(1)} {m.group(2)}经'
# 判断此卷标行是否为重复标记:其前方上一非空行若同卷经 -> 跳过
prev_nonempty = next((l.strip() for l in reversed(buf) if l.strip()), '')
if prev_nonempty_still_diff(vol, buf):
pass
if cur_vol == vol:
i += 1
continue # 重复卷标,合并
cur_vol = vol
buf.append('')
buf.append('---')
buf.append(vol)
buf.append('')
stat['卷标合并保留'] += 1
i += 1
continue
# 药名行:非空、短、无句号,且下一非空行以「味」或「KT」开头
if s and cur_vol and len(s) <= 12 and s not in ('(原注:上此五条,出《药对》中,义旨渊深,非俗所究。虽莫可遵用,而是主统之本,故亦载之。)',):
nxt = next((l.strip() for l in lines[i+1:] if l.strip()), '')
nxt_is_body = nxt.startswith('味') or nxt == 'KT'
if nxt_is_body and not s.endswith(('。',)) and not s.startswith('《'):
buf.append(f'#### {s}')
stat['药条'] += 1
i += 1
continue
buf.append(raw.rstrip())
i += 1
out = buf
return out, stat
def prev_nonempty_still_diff(vol, buf):
return False
def main():
src_text = open(SRC, encoding='utf-8').read()
lines = src_text.split('\n')
body, stat = build(lines)
text = '\n'.join(body)
# 空行规整:连续3+空行→1空行(结构性,不改文字)
text = re.sub(r'\n{4,}', '\n\n\n', text)
open(OUT, 'w', encoding='utf-8').write(text)
md5 = hashlib.md5(open(OUT, 'rb').read()).hexdigest()
stat['md5'] = md5
stat['bytes'] = os.path.getsize(OUT)
stat['行数'] = text.count('\n') + 1
# 药条按卷统计
sec, cnt = None, {'卷一 上经': 0, '卷二 中经': 0, '卷三 下经': 0}
for l in text.split('\n'):
s = l.strip()
m = VOL_RE.match(s) or (s.startswith('## ') and VOL_RE.match(s[3:]))
if m: sec = f'卷{m.group(1)}\u3000{m.group(2)}经'
elif s.startswith('#### ') and sec in cnt: cnt[sec] += 1
stat['药条_按卷'] = cnt
json.dump(stat, open(STAT, 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
print(json.dumps(stat, ensure_ascii=False, indent=1))
if __name__ == '__main__':
main()