Files
dwarf-fortress-annals/dfannals/normalize.py
T
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

170 lines
5.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""标点规范化:把模型爱用的半角标点改回中文排版。
模型(尤其是 `"`)经常直接吐 ASCII 引号。这既不符合中文排版,
也会让「到底有没有对白」这种统计失真——我第一次统计扩写稿对白时就被骗了
(算出 0 处对白,实际有 32 组,只是引号是英文的)。
思路借自 oh-story-claudecode 的 normalize-punctuation.js(MIT),只保留本书需要的两条:
引号配对、CJK 之间的半角标点转全角。**破折号与省略号不转换**——本书正文禁用它们,
交给诊断器报警比偷偷替换更好。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
_CJK = r"[\u4e00-\u9fff]"
_HALF2FULL = {",": ",", ".": "。", "!": "!", "?": "?", ":": ":", ";": ";"}
_CJK_PUNCT = re.compile(f"({_CJK})([,.\u0021?:;])({_CJK})")
@dataclass
class Normalized:
text: str
notes: list[str] = field(default_factory=list)
@property
def changed(self) -> bool:
return bool(self.notes)
def fix_quotes(text: str) -> tuple[str, int]:
"""ASCII 直引号 → 中文弯引号“”。
按出现顺序交替开合。史书正文里不会出现嵌套引号(对白里再说引语时
我们用「」),所以交替法够用;真出现奇数个也会被 deslop 诊断出来。
"""
if '"' not in text:
return text, 0
out: list[str] = []
opening = True
for ch in text:
if ch == '"':
out.append("“" if opening else "”")
opening = not opening
else:
out.append(ch)
return "".join(out), text.count('"')
def fix_cjk_punct(text: str) -> tuple[str, int]:
"""夹在两个汉字之间的半角标点 → 全角。不动英文专名里的标点。"""
total = 0
while True:
text, n = _CJK_PUNCT.subn(
lambda m: m.group(1) + _HALF2FULL[m.group(2)] + m.group(3), text
)
total += n
if not n:
return text, total
def _drop_separator_lines(text: str) -> tuple[str, int]:
"""删掉模型自作的分隔线(`---`、`***`)。原稿不用它们,空行已经够。"""
kept, dropped = [], 0
for line in text.split("\n"):
stripped = line.strip()
if stripped and set(stripped) <= {"-", "*", "=", "—", "·"}:
dropped += 1
continue
kept.append(line)
return "\n".join(kept), dropped
_HEADING_RE = re.compile(r"^\s*#{1,6}\s*\S*\s*$")
_NUMERAL_RE = re.compile(r"^\s*(?:[一二三四五六七八九十]{1,3}|\d{1,4})\s*$")
def _drop_stray_markings(text: str) -> tuple[str, int]:
"""删掉模型自作的小节标记:`## 一`、孤立的「七」。
分块扩写实测:模型每写一块就自制一个编号小节(## 一 … ## 六),
它们不是故事内容,留在正文里就是脏标记。
正文第一行标题(形如 `# 守门人`)要保留——那是原稿的标题。
"""
lines = text.split("\n")
first_content = next((i for i, line in enumerate(lines) if line.strip()), None)
kept, dropped = [], 0
for index, line in enumerate(lines):
if index != first_content and line.strip() and (
_HEADING_RE.match(line) or _NUMERAL_RE.match(line)
):
dropped += 1
continue
kept.append(line)
return "\n".join(kept), dropped
def _collapse_duplicate_paragraphs(text: str) -> tuple[str, int]:
"""相邻的完全重复段落只留一段(模型跨块写作时会重写上一块的结尾)。"""
kept, dropped, prev = [], 0, None
for line in text.split("\n"):
stripped = line.strip()
if stripped and stripped == prev:
dropped += 1
continue
kept.append(line)
if stripped:
prev = stripped
return "\n".join(kept), dropped
def _collapse_duplicate_sentences(text: str) -> tuple[str, int]:
"""段内连续重复的句子只留一句(模型爱把同一句复制两遍再往下写)。"""
total = 0
out: list[str] = []
for para in text.split("\n"):
if not para.strip():
out.append(para)
continue
pieces = re.split(r"(?<=[。!?])", para)
kept: list[str] = []
prev = None
for piece in pieces:
key = piece.strip()
if key and key == prev:
total += 1
continue
kept.append(piece)
if key:
prev = key
out.append("".join(kept))
return "\n".join(out), total
def _tidy_blank_lines(text: str) -> str:
"""压缩多余空行、去掉行尾空白。"""
text = re.sub(r"\n{3,}", "\n\n", text)
lines = [line.rstrip() for line in text.split("\n")]
return "\n".join(lines).strip()
def normalize(text: str) -> Normalized:
"""返回规范化后的文本与改动说明(不改字词,只改标点与结构性冗余)。"""
notes: list[str] = []
text, separators = _drop_separator_lines(text)
if separators:
notes.append(f"删除模型自作的分隔线 {separators} 行")
text, markings = _drop_stray_markings(text)
if markings:
notes.append(f"删除模型自作的小节标记 {markings} 处")
text, dup_paras = _collapse_duplicate_paragraphs(text)
if dup_paras:
notes.append(f"合并重复段落 {dup_paras} 处")
text, dup_sents = _collapse_duplicate_sentences(text)
if dup_sents:
notes.append(f"合并段内重复句子 {dup_sents} 处")
text, quotes = fix_quotes(text)
if quotes:
notes.append(f"英文引号 {quotes} 个 → 中文引号")
if quotes % 2:
notes.append("警告:引号个数为奇数,可能有一处未闭合")
text, punct = fix_cjk_punct(text)
if punct:
notes.append(f"汉字间半角标点 {punct} 处 → 全角")
return Normalized(text=_tidy_blank_lines(text), notes=notes)