文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)

- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
Chen Yi
2026-10-05 22:27:02 +08:00
parent 08a032120a
commit 8630dcad55
23 changed files with 1971 additions and 56 deletions
+169
View File
@@ -0,0 +1,169 @@
"""标点规范化:把模型爱用的半角标点改回中文排版。
模型(尤其是 `"`)经常直接吐 ASCII 引号。这既不符合中文排版,
也会让「到底有没有对白」这种统计失真——我第一次统计扩写稿对白时就被骗了
(算出 0 处对白,实际有 32 组,只是引号是英文的)。
思路借自 oh-story-claudecode 的 normalize-punctuation.js(MIT),只保留本书需要的两条:
引号配对、CJK 之间的半角标点转全角。**破折号与省略号不转换**——本书正文禁用它们,
交给诊断器报警比偷偷替换更好。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
_CJK = r"[\u4e00-\u9fff]"
_HALF2FULL = {",": ",", ".": "。", "!": "!", "?": "?", ":": ":", ";": ";"}
_CJK_PUNCT = re.compile(f"({_CJK})([,.\u0021?:;])({_CJK})")
@dataclass
class Normalized:
text: str
notes: list[str] = field(default_factory=list)
@property
def changed(self) -> bool:
return bool(self.notes)
def fix_quotes(text: str) -> tuple[str, int]:
"""ASCII 直引号 → 中文弯引号“”。
按出现顺序交替开合。史书正文里不会出现嵌套引号(对白里再说引语时
我们用「」),所以交替法够用;真出现奇数个也会被 deslop 诊断出来。
"""
if '"' not in text:
return text, 0
out: list[str] = []
opening = True
for ch in text:
if ch == '"':
out.append("“" if opening else "”")
opening = not opening
else:
out.append(ch)
return "".join(out), text.count('"')
def fix_cjk_punct(text: str) -> tuple[str, int]:
"""夹在两个汉字之间的半角标点 → 全角。不动英文专名里的标点。"""
total = 0
while True:
text, n = _CJK_PUNCT.subn(
lambda m: m.group(1) + _HALF2FULL[m.group(2)] + m.group(3), text
)
total += n
if not n:
return text, total
def _drop_separator_lines(text: str) -> tuple[str, int]:
"""删掉模型自作的分隔线(`---`、`***`)。原稿不用它们,空行已经够。"""
kept, dropped = [], 0
for line in text.split("\n"):
stripped = line.strip()
if stripped and set(stripped) <= {"-", "*", "=", "—", "·"}:
dropped += 1
continue
kept.append(line)
return "\n".join(kept), dropped
_HEADING_RE = re.compile(r"^\s*#{1,6}\s*\S*\s*$")
_NUMERAL_RE = re.compile(r"^\s*(?:[一二三四五六七八九十]{1,3}|\d{1,4})\s*$")
def _drop_stray_markings(text: str) -> tuple[str, int]:
"""删掉模型自作的小节标记:`## 一`、孤立的「七」。
分块扩写实测:模型每写一块就自制一个编号小节(## 一 … ## 六),
它们不是故事内容,留在正文里就是脏标记。
正文第一行标题(形如 `# 守门人`)要保留——那是原稿的标题。
"""
lines = text.split("\n")
first_content = next((i for i, line in enumerate(lines) if line.strip()), None)
kept, dropped = [], 0
for index, line in enumerate(lines):
if index != first_content and line.strip() and (
_HEADING_RE.match(line) or _NUMERAL_RE.match(line)
):
dropped += 1
continue
kept.append(line)
return "\n".join(kept), dropped
def _collapse_duplicate_paragraphs(text: str) -> tuple[str, int]:
"""相邻的完全重复段落只留一段(模型跨块写作时会重写上一块的结尾)。"""
kept, dropped, prev = [], 0, None
for line in text.split("\n"):
stripped = line.strip()
if stripped and stripped == prev:
dropped += 1
continue
kept.append(line)
if stripped:
prev = stripped
return "\n".join(kept), dropped
def _collapse_duplicate_sentences(text: str) -> tuple[str, int]:
"""段内连续重复的句子只留一句(模型爱把同一句复制两遍再往下写)。"""
total = 0
out: list[str] = []
for para in text.split("\n"):
if not para.strip():
out.append(para)
continue
pieces = re.split(r"(?<=[。!?])", para)
kept: list[str] = []
prev = None
for piece in pieces:
key = piece.strip()
if key and key == prev:
total += 1
continue
kept.append(piece)
if key:
prev = key
out.append("".join(kept))
return "\n".join(out), total
def _tidy_blank_lines(text: str) -> str:
"""压缩多余空行、去掉行尾空白。"""
text = re.sub(r"\n{3,}", "\n\n", text)
lines = [line.rstrip() for line in text.split("\n")]
return "\n".join(lines).strip()
def normalize(text: str) -> Normalized:
"""返回规范化后的文本与改动说明(不改字词,只改标点与结构性冗余)。"""
notes: list[str] = []
text, separators = _drop_separator_lines(text)
if separators:
notes.append(f"删除模型自作的分隔线 {separators} 行")
text, markings = _drop_stray_markings(text)
if markings:
notes.append(f"删除模型自作的小节标记 {markings} 处")
text, dup_paras = _collapse_duplicate_paragraphs(text)
if dup_paras:
notes.append(f"合并重复段落 {dup_paras} 处")
text, dup_sents = _collapse_duplicate_sentences(text)
if dup_sents:
notes.append(f"合并段内重复句子 {dup_sents} 处")
text, quotes = fix_quotes(text)
if quotes:
notes.append(f"英文引号 {quotes} 个 → 中文引号")
if quotes % 2:
notes.append("警告:引号个数为奇数,可能有一处未闭合")
text, punct = fix_cjk_punct(text)
if punct:
notes.append(f"汉字间半角标点 {punct} 处 → 全角")
return Normalized(text=_tidy_blank_lines(text), notes=notes)