文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
@@ -0,0 +1,169 @@
|
||||
"""标点规范化:把模型爱用的半角标点改回中文排版。
|
||||
|
||||
模型(尤其是 `"`)经常直接吐 ASCII 引号。这既不符合中文排版,
|
||||
也会让「到底有没有对白」这种统计失真——我第一次统计扩写稿对白时就被骗了
|
||||
(算出 0 处对白,实际有 32 组,只是引号是英文的)。
|
||||
|
||||
思路借自 oh-story-claudecode 的 normalize-punctuation.js(MIT),只保留本书需要的两条:
|
||||
引号配对、CJK 之间的半角标点转全角。**破折号与省略号不转换**——本书正文禁用它们,
|
||||
交给诊断器报警比偷偷替换更好。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
_CJK = r"[\u4e00-\u9fff]"
|
||||
_HALF2FULL = {",": ",", ".": "。", "!": "!", "?": "?", ":": ":", ";": ";"}
|
||||
_CJK_PUNCT = re.compile(f"({_CJK})([,.\u0021?:;])({_CJK})")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Normalized:
|
||||
text: str
|
||||
notes: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def changed(self) -> bool:
|
||||
return bool(self.notes)
|
||||
|
||||
|
||||
def fix_quotes(text: str) -> tuple[str, int]:
|
||||
"""ASCII 直引号 → 中文弯引号“”。
|
||||
|
||||
按出现顺序交替开合。史书正文里不会出现嵌套引号(对白里再说引语时
|
||||
我们用「」),所以交替法够用;真出现奇数个也会被 deslop 诊断出来。
|
||||
"""
|
||||
if '"' not in text:
|
||||
return text, 0
|
||||
out: list[str] = []
|
||||
opening = True
|
||||
for ch in text:
|
||||
if ch == '"':
|
||||
out.append("“" if opening else "”")
|
||||
opening = not opening
|
||||
else:
|
||||
out.append(ch)
|
||||
return "".join(out), text.count('"')
|
||||
|
||||
|
||||
def fix_cjk_punct(text: str) -> tuple[str, int]:
|
||||
"""夹在两个汉字之间的半角标点 → 全角。不动英文专名里的标点。"""
|
||||
total = 0
|
||||
while True:
|
||||
text, n = _CJK_PUNCT.subn(
|
||||
lambda m: m.group(1) + _HALF2FULL[m.group(2)] + m.group(3), text
|
||||
)
|
||||
total += n
|
||||
if not n:
|
||||
return text, total
|
||||
|
||||
|
||||
def _drop_separator_lines(text: str) -> tuple[str, int]:
|
||||
"""删掉模型自作的分隔线(`---`、`***`)。原稿不用它们,空行已经够。"""
|
||||
kept, dropped = [], 0
|
||||
for line in text.split("\n"):
|
||||
stripped = line.strip()
|
||||
if stripped and set(stripped) <= {"-", "*", "=", "—", "·"}:
|
||||
dropped += 1
|
||||
continue
|
||||
kept.append(line)
|
||||
return "\n".join(kept), dropped
|
||||
|
||||
|
||||
_HEADING_RE = re.compile(r"^\s*#{1,6}\s*\S*\s*$")
|
||||
_NUMERAL_RE = re.compile(r"^\s*(?:[一二三四五六七八九十]{1,3}|\d{1,4})\s*$")
|
||||
|
||||
|
||||
def _drop_stray_markings(text: str) -> tuple[str, int]:
|
||||
"""删掉模型自作的小节标记:`## 一`、孤立的「七」。
|
||||
|
||||
分块扩写实测:模型每写一块就自制一个编号小节(## 一 … ## 六),
|
||||
它们不是故事内容,留在正文里就是脏标记。
|
||||
正文第一行标题(形如 `# 守门人`)要保留——那是原稿的标题。
|
||||
"""
|
||||
lines = text.split("\n")
|
||||
first_content = next((i for i, line in enumerate(lines) if line.strip()), None)
|
||||
kept, dropped = [], 0
|
||||
for index, line in enumerate(lines):
|
||||
if index != first_content and line.strip() and (
|
||||
_HEADING_RE.match(line) or _NUMERAL_RE.match(line)
|
||||
):
|
||||
dropped += 1
|
||||
continue
|
||||
kept.append(line)
|
||||
return "\n".join(kept), dropped
|
||||
|
||||
|
||||
def _collapse_duplicate_paragraphs(text: str) -> tuple[str, int]:
|
||||
"""相邻的完全重复段落只留一段(模型跨块写作时会重写上一块的结尾)。"""
|
||||
kept, dropped, prev = [], 0, None
|
||||
for line in text.split("\n"):
|
||||
stripped = line.strip()
|
||||
if stripped and stripped == prev:
|
||||
dropped += 1
|
||||
continue
|
||||
kept.append(line)
|
||||
if stripped:
|
||||
prev = stripped
|
||||
return "\n".join(kept), dropped
|
||||
|
||||
|
||||
def _collapse_duplicate_sentences(text: str) -> tuple[str, int]:
|
||||
"""段内连续重复的句子只留一句(模型爱把同一句复制两遍再往下写)。"""
|
||||
total = 0
|
||||
out: list[str] = []
|
||||
for para in text.split("\n"):
|
||||
if not para.strip():
|
||||
out.append(para)
|
||||
continue
|
||||
pieces = re.split(r"(?<=[。!?])", para)
|
||||
kept: list[str] = []
|
||||
prev = None
|
||||
for piece in pieces:
|
||||
key = piece.strip()
|
||||
if key and key == prev:
|
||||
total += 1
|
||||
continue
|
||||
kept.append(piece)
|
||||
if key:
|
||||
prev = key
|
||||
out.append("".join(kept))
|
||||
return "\n".join(out), total
|
||||
|
||||
|
||||
def _tidy_blank_lines(text: str) -> str:
|
||||
"""压缩多余空行、去掉行尾空白。"""
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
lines = [line.rstrip() for line in text.split("\n")]
|
||||
return "\n".join(lines).strip()
|
||||
|
||||
|
||||
def normalize(text: str) -> Normalized:
|
||||
"""返回规范化后的文本与改动说明(不改字词,只改标点与结构性冗余)。"""
|
||||
notes: list[str] = []
|
||||
|
||||
text, separators = _drop_separator_lines(text)
|
||||
if separators:
|
||||
notes.append(f"删除模型自作的分隔线 {separators} 行")
|
||||
text, markings = _drop_stray_markings(text)
|
||||
if markings:
|
||||
notes.append(f"删除模型自作的小节标记 {markings} 处")
|
||||
text, dup_paras = _collapse_duplicate_paragraphs(text)
|
||||
if dup_paras:
|
||||
notes.append(f"合并重复段落 {dup_paras} 处")
|
||||
text, dup_sents = _collapse_duplicate_sentences(text)
|
||||
if dup_sents:
|
||||
notes.append(f"合并段内重复句子 {dup_sents} 处")
|
||||
|
||||
text, quotes = fix_quotes(text)
|
||||
if quotes:
|
||||
notes.append(f"英文引号 {quotes} 个 → 中文引号")
|
||||
if quotes % 2:
|
||||
notes.append("警告:引号个数为奇数,可能有一处未闭合")
|
||||
|
||||
text, punct = fix_cjk_punct(text)
|
||||
if punct:
|
||||
notes.append(f"汉字间半角标点 {punct} 处 → 全角")
|
||||
|
||||
return Normalized(text=_tidy_blank_lines(text), notes=notes)
|
||||
Reference in New Issue
Block a user