Files
health/辅行诀五脏用药法要-药性探真/05_脚本工具/01_clean_and_parse.py
T

385 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辅行诀五脏用药法要-药性探真 资料库 ETL 第 1 步(v3)
01_来源数据 -> 02_加工数据
只做"数据清理",不做释义抽取 / 跨库关联 / 主题分析(用户 2026-09-15 指令)。
清理项:
A 页面残留物:页眉(含缺书名号的嵌入形式)、独立页码、行尾空格、全角空格
B 断行错位接合:段落被页眉/页码打断处自动接合
C 形近错字 / 异体字 / 书名误字修正(逐条附核校依据)
E 半角标点规范化(邻接汉字时)
F 存疑登记:无法确证处不改动原文,逐条附疑正与核校来源
输出:
02_加工数据/药性探真_第一章_清洗版.md
02_加工数据/文本清洗报告.md (全部核对数据**按章/节列明**)
02_加工数据/人工核对清单.md (按章/节列明的待人工确认项)
02_加工数据/章节结构_第一章.json (清理配套定位索引)
幂等:可重复运行,输出覆盖生成。
"""
import os, re, json, hashlib
BASE = os.path.expanduser("~/Documents/ai_agent_scraper_study/data/辅行诀五脏用药法要-药性探真")
SRC = os.path.join(BASE, "01_来源数据/药性探真_第一章_原始_20260915.md")
OUT_MD = os.path.join(BASE, "02_加工数据/药性探真_第一章_清洗版.md")
OUT_REPORT = os.path.join(BASE, "02_加工数据/文本清洗报告.md")
OUT_CHECK = os.path.join(BASE, "02_加工数据/人工核对清单.md")
OUT_STRUCT = os.path.join(BASE, "02_加工数据/章节结构_第一章.json")
PAGE_HEADER_RE = re.compile(r'^\s*《?辅行诀五脏用药法要》?\s*药性探真\s*》?\s*$')
PAGE_NUM_RE = re.compile(r'^\s*\d{1,3}\s*$')
CJK = r'[\u4e00-\u9fff]'
# C 类:形近错字 / 异体字 / 书名误字(逐条核校)
FIXES = [
("先骋通使", "先聘通使", "《本经》牡桂条作「为诸药先聘通使」;下文自称「先聘」,本书自证"),
("越洲山阴", "越州山阴", "陆佃籍贯,《宋史·陆佃传》作越州山阴"),
("桜", "梫", "《纲目》桂条引《尔雅》「梫者,能侵害他木也」(梫,音寝)"),
("国老之誊", "国老之誉", "文义:众药之王、和国老之誉;「誊」形近误"),
("酒皰皷鼻", "酒疱皶鼻", "《本经》栀子条作「面赤,酒疱皶鼻」"),
("淚出", "泪出", "异体字正字,全书他处均用「泪」"),
("微字为蒼之讹字", "微字为苍之讹字", "《纲目》人参释名,简体正字"),
("天元幻大论", "天元纪大论", "《素问》篇名作「天元纪大论」"),
("白戸浆", "白酨浆", "酨即酢浆;《辅行诀》原书作「白酨浆」,本书小标题亦作白酨浆"),
]
# F 类:存疑不改(原文保留)
SUSPECT_RULES = [
(r'菌,蕗。从竹,雨声', "《说文》「箘,箘簬。从竹,囷声。一曰博棋也」;疑「菌蕗」「雨声」为形近致误"),
(r'大则性环详而缓', "《本草衍义》枳实条:「大则其性详而缓」,疑衍「环」字"),
(r'决泄愤,地方但', "《本草衍义》:「皆取其疏通决泄、破结实之义。他方但导败风壅之气……」;"
"此处「愤,地方但」疑为OCR丢字串行,待补正"),
(r'由于积分布甚广', "据上下文(论枳之品种)疑当作「枳分布甚广」"),
(r'传动产隻者为雄', "《竹谱》竹根雌雄辨:疑当作「其节单者为雄」(「隻」为「单」之误)"),
(r'[((]?[))]', "原文含存疑符号(?),系作者或OCR存疑处,保留原文"),
(r'[①②③④⑤⑥⑦⑧⑨⑩]', "脚注标记,对应脚注正文在本次来源中缺失,待补"),
]
def clean_text(raw: str):
"""返回 (清洗文本, 事件列表, 删除统计)。事件均锚定到【清洗版行号】。"""
recs = [] # [{"text":..., "fixes":[...], "punct":[...]}]
removed = {"页眉": 0, "页码": 0}
for line in raw.split("\n"):
if PAGE_HEADER_RE.match(line):
removed["页眉"] += 1
continue
if PAGE_NUM_RE.match(line):
removed["页码"] += 1
continue
new = line.replace("\u3000", "").rstrip()
rec = {"text": new, "fixes": [], "punct": []}
for bad, good, why in FIXES:
if bad in new:
rec["fixes"].append({"原": bad, "改": good, "依据": why})
new = new.replace(bad, good)
# 保护「附录N」「第N章/节」后的分隔空格,避免与 OCR 断行空格一并折叠
new = re.sub(r'(附录[一二三四五六七八九十]|第[一二三四五六七八九十]+[章节])[ \t]+',
lambda m: m.group(1) + "\x01", new)
prev = None
while prev != new: # 中文词间半角空格(OCR断行残留)
prev = new
new = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])', '', new)
new = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[,。、;:?!“”()《》])', '', new)
new = re.sub(r'(?<=[,。、;:?!“”()《》])\s+(?=[\u4e00-\u9fff])', '', new)
# E 类:半角标点 ↔ 全角(邻接汉字或全角标点)
if re.search(CJK, line):
for _pass in range(2): # 二次通过:括号先转全角后再判后随标点
for asc, full in ((",", ","), (";", ";"), (":", ":"), ("?", "?"), ("!", "!")):
new2 = re.sub(r'(?<=[\u4e00-\u9fff),。、;:?!”“》)])\s*' + re.escape(asc), full, new)
if new2 != new:
rec["punct"].append({"原": asc, "改": full}); new = new2
for asc, full, side in (("(", "(", "后"), (")", ")", "前")):
pat = (r'\s*' + re.escape(asc) + r'(?=' + CJK + r')') if side == "后" else \
(r'(?<=' + CJK + r')\s*' + re.escape(asc))
new2 = re.sub(pat, full, new)
if new2 != new:
rec["punct"].append({"原": asc, "改": full}); new = new2
# 转换后再次折叠全角标点两侧空格
new = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[,。、;:?!“”()《》])', '', new)
new = re.sub(r'(?<=[,。、;:?!“”()《》])\s+(?=[\u4e00-\u9fff])', '', new)
new = re.sub(r'(?=[,。、;:?!])\s+', '', new)
new = new.replace("\x01", " ") # 还原受保护的分隔空格
rec["text"] = new.rstrip()
recs.append(rec)
# 合并连续空行
tmp = []
prev_blank = False
for r in recs:
if r["text"] == "":
if prev_blank:
continue
prev_blank = True
else:
prev_blank = False
tmp.append(r)
# B 类:断行错位接合
out, joins = [], []
i = 0
while i < len(tmp):
cur = tmp[i]
if (cur["text"] and not cur["text"].startswith("#") and len(cur["text"]) > 8
and re.search(CJK + r'$', cur["text"])):
j = i + 1
while j < len(tmp) and tmp[j]["text"] == "":
j += 1
if j < len(tmp) and re.match(CJK, tmp[j]["text"]) and not tmp[j]["text"].startswith("#"):
merged = {"text": cur["text"] + tmp[j]["text"],
"fixes": cur["fixes"] + tmp[j]["fixes"],
"punct": cur["punct"] + tmp[j]["punct"]}
joins.append({"接合前": cur["text"][-20:], "接合后": tmp[j]["text"][:20],
"_before_line": i + 1, "_cont_line": j + 1})
out.append(merged)
i = j + 1
continue
out.append(cur)
i += 1
# 落定清洗版行号 + 事件汇总
lines, events = [], []
for n, rec in enumerate(out, start=1):
lines.append(rec["text"])
for f in rec["fixes"]:
events.append(dict(类型="C形近错字", 行=n, **f))
for p in rec["punct"]:
events.append(dict(类型="E标点规范化", 行=n, 原=p["原"], 改=p["改"], 依据="半角标点邻接汉字"))
for s in SUSPECT_RULES:
for m in re.finditer(s[0], rec["text"]):
events.append(dict(类型="F存疑", 行=n,
原=f"…{rec['text'][max(0,m.start()-15):m.end()+15]}…", 改="(保留原文)", 依据=s[1]))
for jn in joins:
events.append(dict(类型="B断行接合", 行=jn["_before_line"],
原=f"…{jn['接合前']} ‖ {jn['接合后']}…",
改="接合为一段", 依据=f"段落被页眉/页码打断,接续于原第{jn['_cont_line']}行"))
text = re.sub(r'\n{3,}', '\n\n', "\n".join(lines)).strip() + "\n"
# 去重:同行同类型同说明
seen, uniq = set(), []
for e in events:
k = (e["类型"], e["行"], e["依据"])
if k not in seen:
seen.add(k); uniq.append(e)
uniq.sort(key=lambda x: (x["行"], {"B断行接合": 0, "C形近错字": 1, "E标点规范化": 2, "F存疑": 3}[x["类型"]]))
return text, uniq, removed
def parse_structure(text: str):
lines = text.split("\n")
nodes = []
for i, l in enumerate(lines, start=1):
if not l.startswith("#"):
continue
title = l.lstrip("#").strip()
if re.match(r'^第[一二三四五六七八九十]+章', title):
lvl, typ = 1, "章"
elif re.match(r'^第[一二三四五六七八九十]+节', title):
lvl, typ = 2, "节"
elif re.match(r'^(附[::]|\d+\.\s)', title):
lvl, typ = 4, "附注"
else:
lvl, typ = 3, "条目"
nodes.append({"标题": title, "起始行": i, "类型": typ, "_lvl": lvl})
for n, node in enumerate(nodes):
node["结束行"] = len(lines)
for m in range(n + 1, len(nodes)):
if nodes[m]["_lvl"] <= node["_lvl"]:
node["结束行"] = nodes[m]["起始行"] - 1
break
GATE = {"辛": ("肝", "木", "春"), "咸": ("心", "火", "夏"), "甘": ("脾", "土", "长夏"),
"酸": ("肺", "金", "秋"), "苦": ("肾", "水", "冬")}
chapters = []
for node in nodes:
if node["类型"] == "章":
chapters.append(dict(node, 节=[]))
elif node["类型"] == "节":
m = re.search(r'([辛咸甘酸苦])味?门', node["标题"])
node["味门"] = m.group(1) if m else None
if m:
node["本脏"], node["本行"], node["脏象法象"] = GATE[m.group(1)]
if chapters:
chapters[-1]["节"].append(dict(node, 条目=[]))
else:
m = re.match(r'^[一二三四五六七八九十]+、\s*(.+?)\s*[((](.+?)[))]\s*$', node["标题"])
if m:
node["药名"], node["五行互含名位"] = m.group(1), m.group(2)
else:
node["药名"] = re.sub(r'^[一二三四五六七八九十]+、\s*', '', node["标题"])
node["五行互含名位"] = None
if chapters and chapters[-1].get("节"):
chapters[-1]["节"][-1]["条目"].append(node)
for c in chapters:
for s in c["节"]:
s["药物条目"] = [e for e in s["条目"] if e["类型"] == "条目"]
s["附属标题"] = [e["标题"] for e in s["条目"] if e["类型"] == "附注"]
return chapters
def scan_missing_assets(text: str):
"""登记悬空图片引用(OCR 流水线遗留,图片本体未随源文件提供)。"""
lines = text.split("\n")
out = []
for i, l in enumerate(lines, start=1):
for m in re.finditer(r'!\[\]\(([^)\n]+)\)', l):
out.append({"行": i, "文件名": os.path.basename(m.group(1)), "原始引用": m.group(0)[:80],
"上文": lines[i-3][:60] if i >= 3 else ""})
return out
def main():
raw = open(SRC, encoding="utf-8").read()
text, events, removed = clean_text(raw)
chapters = parse_structure(text)
missing_assets = scan_missing_assets(text)
def wf(path, content):
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "w", encoding="utf-8") as f:
f.write(content); f.flush(); os.fsync(f.fileno())
wf(OUT_MD, text)
struct = {
"来源文件": os.path.basename(SRC), "来源文件md5": hashlib.md5(raw.encode()).hexdigest(),
"清洗版文件": os.path.basename(OUT_MD), "清洗版md5": hashlib.md5(text.encode()).hexdigest(),
"原始字符数": len(raw), "清洗后字符数": len(text),
"说明": "行号均指清洗版;本文件仅作清理配套定位索引,未做释义抽取与跨库关联",
"章": chapters,
}
wf(OUT_STRUCT, json.dumps(struct, ensure_ascii=False, indent=2) + "\n")
ch = chapters[0]
secs = ch["节"]
n_drug = sum(len(s["药物条目"]) for s in secs)
cnt = lambda t: sum(1 for e in events if e["类型"] == t)
# 每节归属
def owner(line):
for s in secs:
if s["起始行"] <= line <= s["结束行"]:
return s
return None
per_sec = {s["标题"]: [] for s in secs}
for e in events:
o = owner(e["行"])
if o is not None:
per_sec[o["标题"]].append(e)
def ev_table(evs):
rows = ["| 行 | 类型 | 原文 | 处理后 | 依据 |", "|----|------|------|--------|------|"]
for e in evs:
rows.append(f"| {e['行']} | {e['类型']} | {e['原'].replace('|','|')} | "
f"{e['改'].replace('|','|')} | {e['依据']} |")
return rows
rep = ["# 《辅行诀五脏用药法要》药性探真 第一章 — 文本清洗报告", "",
"> 本库现阶段只做数据清理。用户 2026-09-15 指令:以第一章为源,待全文内容完整后再做加工/跨库关联/分析。", "",
"## 0 概览", "",
f"- 来源:`01_来源数据/{os.path.basename(SRC)}`(md5 `{struct['来源文件md5']}`)",
f"- 清洗版:`02_加工数据/{os.path.basename(OUT_MD)}`(md5 `{struct['清洗版md5']}`)",
f"- 字符数:{len(raw)} → {len(text)};行数:{raw.count(chr(10))} → {text.count(chr(10))}",
f"- 结构:{ch['标题']} ── {len(secs)} 节 / 药物条目 {n_drug} 味", "",
"| 清理类别 | 处数 | 说明 |", "|---------|------|------|",
f"| A 页面残留物 | {removed['页眉'] + removed['页码']} | 页眉 {removed['页眉']} 行、独立页码 {removed['页码']} 行;另有行尾/全角/词间空格全量清理 |",
f"| B 断行错位接合 | {cnt('B断行接合')} | 段落被页眉打断处复原成整段 |",
f"| C 形近错字/异体字 | {cnt('C形近错字')} | 每条附核校依据,见下逐节明细 |",
f"| E 半角标点规范化 | {cnt('E标点规范化')} | ASCII , ; : ( ) 邻接汉字时改全角 |",
f"| F 存疑(未改原文) | {cnt('F存疑')} | 逐条列出疑正与核校来源,见 `人工核对清单.md` |", "",
"### 各节清理量一览", "",
"| 节 | 行范围 | 药物条目 | B 接合 | C 错字 | E 标点 | F 存疑 |",
"|----|--------|---------|--------|--------|--------|--------|"]
for s in secs:
evs = per_sec[s["标题"]]
c = lambda t: sum(1 for e in evs if e["类型"] == t)
rep.append(f"| {s['标题']} | {s['起始行']}–{s['结束行']} | {len(s['药物条目'])} | "
f"{c('B断行接合')} | {c('C形近错字')} | {c('E标点规范化')} | {c('F存疑')} |")
rep += ["", "---", "", "## 1 逐节核对明细(按章 / 节列明)", "",
f"### {ch['标题']}(清洗版行 {ch['起始行']}–{ch['结束行']})", ""]
for s in secs:
evs = per_sec[s["标题"]]
gate = f"味门{s['味门']}·{s['本脏']}{s['本行']}·法{s['脏象法象']}" if s.get("味门") else "从火土同治论心病(草木方例用药简释)"
rep += [f"#### {s['标题']}", "",
f"- 定位:清洗版行 {s['起始行']}–{s['结束行']};{gate};药物条目 {len(s['药物条目'])} 味",
f"- 本味药:{'、'.join(e['药名'] for e in s['药物条目'])}",
f"- 清理量:B 接合 {sum(1 for e in evs if e['类型']=='B断行接合')} / "
f"C 错字 {sum(1 for e in evs if e['类型']=='C形近错字')} / "
f"E 标点 {sum(1 for e in evs if e['类型']=='E标点规范化')} / "
f"F 存疑 {sum(1 for e in evs if e['类型']=='F存疑')}", ""]
if evs:
rep += ev_table(evs) + [""]
else:
rep += ["- 本节无需修正条目。", ""]
rep += ["---", "", "## 2 章节结构(清洗版行号)", "",
"| 章 | 节 | 味门 | 本脏/行/时 | 起始行 | 结束行 | 药物条目 | 附属标题 |",
"|----|----|------|-----------|--------|--------|---------|---------|"]
for s in secs:
gate = f"{s.get('本脏','—')}/{s.get('本行','—')}/{s.get('脏象法象','—')}" if s.get("味门") else "—"
rep.append(f"| {ch['标题']} | {s['标题']} | {s.get('味门') or '—'} | {gate} | {s['起始行']} | {s['结束行']} | "
f"{len(s['药物条目'])} | {';'.join(s['附属标题']) or '—'} |")
rep += ["", "## 3 缺失内容登记(来源不完整,非错字问题)", "",
"| 项 | 情况 | 位置 | 处理 |", "|----|------|------|------|",
f"| 悬空图片引用 | {len(missing_assets)} 处(OCR 流水线遗留,图片本体未随源文件提供) | "
+ ";".join(f"行 {m['行']}({m['文件名'][:12]}…)" for m in missing_assets) + " | 原文保留引用,待补图 |",
"| 脚注正文 | 正文存 ① ② 等标记 5 处(桂枝、人参、升麻、白醪法 2 处),脚注正文未随文录入 | 见上方逐节明细 F 类 | 待补后回填 |",
"| 第二章及以后各章 | 本次来源仅第一章「五脏补泻草木方例 用药释义」 | — | 待补 |",
"| 附录一至四 | 目录列有:整订稿/略论《汤液经法》与张仲景论著的关系/张大昌先生《处方正范》/张大昌先生《三十六脉名义略述》 | 来源文件首 3 行 | 待补 |", "",
"## 4 药物条目清单(章节 → 药名 → 五行互含名位,原样保留)", "",
"| 节 | 序 | 药名 | 五行互含名位 | 清洗版行范围 |", "|----|----|------|-------------|-------------|"]
for s in secs:
for k, e in enumerate(s["药物条目"], start=1):
rep.append(f"| {s['标题']} | {k} | {e['药名']} | {e['五行互含名位'] or '—'} | {e['起始行']}–{e['结束行']} |")
rep += ["", "> 说明:标题中的「五行互含名位」为原书表述(如「生姜木中火、干姜木中水」),清洗阶段原样保留,不作拆分或解释。"]
wf(OUT_REPORT, "\n".join(rep) + "\n")
chk = ["# 人工核对清单 — 《辅行诀五脏用药法要》药性探真 第一章", "",
"> 用途:需对照纸质原书/影印本确认的条目。核对清单**按章/节列明**,本阶段不改动清洗版正文。",
"> 清洗流程与全部修正记录见 `文本清洗报告.md`。", "",
f"## {ch['标题']}", ""]
for s in secs:
evs = [e for e in per_sec[s["标题"]] if e["类型"] == "F存疑"]
chk += [f"### {s['标题']}(行 {s['起始行']}–{s['结束行']})", ""]
if evs:
chk += ["| 行 | 片段 | 疑正 / 核校来源 |", "|----|------|----------------|"]
for e in evs:
chk.append(f"| {e['行']} | {e['原'].replace('|','|')} | {e['依据']} |")
else:
chk.append("- 无存疑项。")
chk.append("")
chk += ["---", "", "## 缺失内容登记(本次来源不完整,非错字问题)", "",
"| 项 | 情况 | 处理 |", "|----|------|------|",
f"| 悬空图片引用 | {len(missing_assets)} 处:"
+ ";".join(f"行 {m['行']} `{m['文件名']}`" for m in missing_assets)
+ "(OCR 流水线遗留,图片本体未随源文件提供) | 待补图,补后替换为真实路径 |",
"| 脚注正文 | 正文存 ① ② 等标记 5 处(桂枝、人参、升麻、白醪法 2 处),脚注正文未随文录入 | 待补后回填 |",
"| 第二章及以后各章 | 本次来源仅第一章「五脏补泻草木方例 用药释义」 | 待补 |",
"| 附录一至四 | 目录列有:整订稿/略论《汤液经法》与张仲景论著的关系/张大昌先生《处方正范》/张大昌先生《三十六脉名义略述》 | 待补 |",
"| 表格 | 第一章正文不含表格 | — |",
"| 图版 | 正文有 2 处图片引用,图片本体缺失(见上「悬空图片引用」) | 如有影印本可另补 |", "",
"## 书目信息(供核对)", "",
"- 书名:《辅行诀五脏用药法要》药性探真",
"- 著者:衣之镖 撰著;丛书:张大昌先生弟子个人专著",
"- 出版:学苑出版社,2014 年 1 月第 1 版,ISBN 978-7-5077-4414-9",
"- 本次来源:`01_来源数据/药性探真_第一章_原始_20260915.md`(仅第一章)"]
wf(OUT_CHECK, "\n".join(chk) + "\n")
print("清洗版:", OUT_MD, os.path.getsize(OUT_MD), "bytes")
print("结构 :", OUT_STRUCT, os.path.getsize(OUT_STRUCT), "bytes")
print("报告 :", OUT_REPORT, os.path.getsize(OUT_REPORT), "bytes")
print("核对单:", OUT_CHECK, os.path.getsize(OUT_CHECK), "bytes")
print(f"页眉/页码删除 {removed} | B接合 {cnt('B断行接合')} | C错字 {cnt('C形近错字')} | "
f"E标点 {cnt('E标点规范化')} | F存疑 {cnt('F存疑')}")
print(f"节 {len(secs)} 个 / 药物条目 {n_drug} 味")
for s in secs:
evs = per_sec[s["标题"]]
print(f" {s['标题']} 行{s['起始行']}–{s['结束行']} 药物{len(s['药物条目'])} "
f"| 清理项 {len(evs)} 条:" + "、".join(f"{e['类型'][:1]}@{e['行']}" for e in evs))
if __name__ == "__main__":
main()