"""标点规范化:把模型爱用的半角标点改回中文排版。 模型(尤其是 `"`)经常直接吐 ASCII 引号。这既不符合中文排版, 也会让「到底有没有对白」这种统计失真——我第一次统计扩写稿对白时就被骗了 (算出 0 处对白,实际有 32 组,只是引号是英文的)。 思路借自 oh-story-claudecode 的 normalize-punctuation.js(MIT),只保留本书需要的两条: 引号配对、CJK 之间的半角标点转全角。**破折号与省略号不转换**——本书正文禁用它们, 交给诊断器报警比偷偷替换更好。 """ from __future__ import annotations import re from dataclasses import dataclass, field _CJK = r"[\u4e00-\u9fff]" _HALF2FULL = {",": ",", ".": "。", "!": "!", "?": "?", ":": ":", ";": ";"} _CJK_PUNCT = re.compile(f"({_CJK})([,.\u0021?:;])({_CJK})") @dataclass class Normalized: text: str notes: list[str] = field(default_factory=list) @property def changed(self) -> bool: return bool(self.notes) def fix_quotes(text: str) -> tuple[str, int]: """ASCII 直引号 → 中文弯引号“”。 按出现顺序交替开合。史书正文里不会出现嵌套引号(对白里再说引语时 我们用「」),所以交替法够用;真出现奇数个也会被 deslop 诊断出来。 """ if '"' not in text: return text, 0 out: list[str] = [] opening = True for ch in text: if ch == '"': out.append("“" if opening else "”") opening = not opening else: out.append(ch) return "".join(out), text.count('"') def fix_cjk_punct(text: str) -> tuple[str, int]: """夹在两个汉字之间的半角标点 → 全角。不动英文专名里的标点。""" total = 0 while True: text, n = _CJK_PUNCT.subn( lambda m: m.group(1) + _HALF2FULL[m.group(2)] + m.group(3), text ) total += n if not n: return text, total def _drop_separator_lines(text: str) -> tuple[str, int]: """删掉模型自作的分隔线(`---`、`***`)。原稿不用它们,空行已经够。""" kept, dropped = [], 0 for line in text.split("\n"): stripped = line.strip() if stripped and set(stripped) <= {"-", "*", "=", "—", "·"}: dropped += 1 continue kept.append(line) return "\n".join(kept), dropped _HEADING_RE = re.compile(r"^\s*#{1,6}\s*\S*\s*$") _NUMERAL_RE = re.compile(r"^\s*(?:[一二三四五六七八九十]{1,3}|\d{1,4})\s*$") def _drop_stray_markings(text: str) -> tuple[str, int]: """删掉模型自作的小节标记:`## 一`、孤立的「七」。 分块扩写实测:模型每写一块就自制一个编号小节(## 一 … ## 六), 它们不是故事内容,留在正文里就是脏标记。 正文第一行标题(形如 `# 守门人`)要保留——那是原稿的标题。 """ lines = text.split("\n") first_content = next((i for i, line in enumerate(lines) if line.strip()), None) kept, dropped = [], 0 for index, line in enumerate(lines): if index != first_content and line.strip() and ( _HEADING_RE.match(line) or _NUMERAL_RE.match(line) ): dropped += 1 continue kept.append(line) return "\n".join(kept), dropped def _collapse_duplicate_paragraphs(text: str) -> tuple[str, int]: """相邻的完全重复段落只留一段(模型跨块写作时会重写上一块的结尾)。""" kept, dropped, prev = [], 0, None for line in text.split("\n"): stripped = line.strip() if stripped and stripped == prev: dropped += 1 continue kept.append(line) if stripped: prev = stripped return "\n".join(kept), dropped def _collapse_duplicate_sentences(text: str) -> tuple[str, int]: """段内连续重复的句子只留一句(模型爱把同一句复制两遍再往下写)。""" total = 0 out: list[str] = [] for para in text.split("\n"): if not para.strip(): out.append(para) continue pieces = re.split(r"(?<=[。!?])", para) kept: list[str] = [] prev = None for piece in pieces: key = piece.strip() if key and key == prev: total += 1 continue kept.append(piece) if key: prev = key out.append("".join(kept)) return "\n".join(out), total def _tidy_blank_lines(text: str) -> str: """压缩多余空行、去掉行尾空白。""" text = re.sub(r"\n{3,}", "\n\n", text) lines = [line.rstrip() for line in text.split("\n")] return "\n".join(lines).strip() def normalize(text: str) -> Normalized: """返回规范化后的文本与改动说明(不改字词,只改标点与结构性冗余)。""" notes: list[str] = [] text, separators = _drop_separator_lines(text) if separators: notes.append(f"删除模型自作的分隔线 {separators} 行") text, markings = _drop_stray_markings(text) if markings: notes.append(f"删除模型自作的小节标记 {markings} 处") text, dup_paras = _collapse_duplicate_paragraphs(text) if dup_paras: notes.append(f"合并重复段落 {dup_paras} 处") text, dup_sents = _collapse_duplicate_sentences(text) if dup_sents: notes.append(f"合并段内重复句子 {dup_sents} 处") text, quotes = fix_quotes(text) if quotes: notes.append(f"英文引号 {quotes} 个 → 中文引号") if quotes % 2: notes.append("警告:引号个数为奇数,可能有一处未闭合") text, punct = fix_cjk_punct(text) if punct: notes.append(f"汉字间半角标点 {punct} 处 → 全角") return Normalized(text=_tidy_blank_lines(text), notes=notes)