文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)

- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
Chen Yi
2026-10-05 22:27:02 +08:00
parent 08a032120a
commit 8630dcad55
23 changed files with 1971 additions and 56 deletions
+162
View File
@@ -0,0 +1,162 @@
"""AI 味机械诊断:把「文笔是否自然」变成可度量的指标。
规则来自 oh-story-claudecode 的 story-deslop 与 banned-words(MIT),做了两处适配:
1. **分层**。原文把「仿佛、一丝、坚定、冰冷」放同一张一级表,但「坚定」「冰冷」在
史书叙事里是正常词(矮人要塞的"冰冷"可能是字面温度),一律判罪会刷满假阳性。
因此分成硬伤词(几乎只在 AI 文本里成串出现)与密度词(真人也用,只看密度)。
2. **按密度而非布尔判定**。AI 味是密度问题:出现一次是风格,出现二十次是模板。
输出「每千字命中数」,供闸门与扩写模型参考。
语义类判断(比喻是否套话、是否把一件事掰成三段写)机械检查假阳性太高,
交给扩写模型和自评闸门,这里不猜。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from dfannals.chronicle import cjk_length
# 硬伤词:出现在正常中文里的概率很低,命中一处就该改
HARD_WORDS: dict[str, tuple[str, ...]] = {
"情态": ("仿佛", "犹如", "宛若", "如同", "一丝", "一抹", "些许", "几分", "隐约",
"毫无征兆", "几不可闻", "微不可察"),
"动作": ("深吸一口气", "不禁"),
"表情": ("眼中闪过", "嘴角勾起", "眉头微皱", "眉眼低垂", "瞳孔微缩", "瞳孔收缩",
"瞳孔一缩", "指节泛白", "眼神锐利", "目光锐利"),
"心理": ("心中一动", "心头一震", "心下了然", "心中暗道", "心底泛起", "不由得", "心中一凛"),
"判断": ("不容置疑", "不容置喙", "不易察觉", "显而易见", "毫无疑问", "不可否认", "前所未有"),
"过渡": ("不由自主", "情不自禁", "自然而然", "话锋一转", "闪烁着光芒"),
}
# 密度词:真人写作也用,只有成串出现才算模板腔
SOFT_WORDS: dict[str, tuple[str, ...]] = {
"形容": ("坚定", "狡黠", "深邃", "凛冽", "冰冷"),
"语境敏感": ("突然", "陡然", "骤然", "猛然", "好像", "似乎", "瞬间", "猛地", "死死地"),
"弱化副词": ("缓缓", "微微", "轻轻", "淡淡"),
}
# 句式与标点:每条是(规则名,说明,正则)
PATTERNS: tuple[tuple[str, str, str], ...] = (
("不是A而是B", "最毒的 AI 句式,直接写 B", r"不是[^,。;!?\n]{1,24}[,,]\s*(?:而)?是"),
("万能动宾", "「,带着……」万能状语", r"[,,]\s*带着[^,。;!?\n]{0,20}"),
("声音描写", "「声音不大,却带着……」", r"声音(?:不大|很轻|很平|平静)[^。\n]{0,14}(?:却|但)?带着"),
("认知直述", "告诉而非展示,改为动作或台词", r"(?:他|她|他们|她们)(?:这才|终于)?(?:明白|意识到|感到|知道)"),
("章末预告", "「他不知道的是……」空泛预告", r"(?:他|她)不知道的是"),
("收束腔", "「这一刻/才刚刚开始」式总结", r"这一刻[,,]|从这一刻开始|才刚刚开始|原来[,,]"),
("抽象命运", "命运+齿轮/棋局/獠牙", r"命运[^。\n]{0,8}(?:齿轮|棋局|獠牙|改写|安排)"),
("心理模板", "「心中一震/心头一凛」式心理惊吓", r"心(?:中|头|底)(?:一|猛|顿)(?:震|凛|颤)"),
("过渡模板", "「取而代之的是」", r"取而代之"),
("告诉而非展示", "「显得有些X」「散发着一股X气息」", r"显得(?:有些|十分|非常)?[\u4e00-\u9fff]{1,4}|散发(?:着|出)?[^。\n]{0,8}(?:气息|气场)"),
("公式化对话标签", "「好的,他说道」", r"[,,](?:他|她)(?:说道|开口道|沉声道|低声道)"),
("英文引号", "模型常直接吐 ASCII 直角引号,中文排版用“”", r'"'),
("破折号", "正文禁用,改用句号、逗号或动作断句", r"——|──|(?<!-)--(?!-)"),
("省略号", "正文禁用,用动作或短句表达停顿", r"……|\.\.\."),
)
SAMPLE_WIDTH = 14
SAMPLES_PER_RULE = 2
@dataclass
class Hit:
"""一条诊断:命中哪个规则、几次、原文长什么样。"""
rule: str
category: str
level: str # hard = 硬伤词+句式+标点;soft = 只看密度的词
count: int
samples: list[str] = field(default_factory=list)
def line(self) -> str:
ex = ";".join(self.samples)
return f"{self.category:<6} {self.rule:<8} ×{self.count:<3} {ex}"
def _context(text: str, start: int, end: int) -> str:
left = max(0, start - SAMPLE_WIDTH)
right = min(len(text), end + SAMPLE_WIDTH)
return text[left:right].replace("\n", " ").strip()
def _scan_word(text: str, word: str) -> tuple[int, list[str]]:
count, samples, pos = 0, [], text.find(word)
while pos != -1:
count += 1
if len(samples) < SAMPLES_PER_RULE:
samples.append(_context(text, pos, pos + len(word)))
pos = text.find(word, pos + len(word))
return count, samples
def check(text: str) -> list[Hit]:
"""扫一遍文本,返回全部诊断(硬伤在前,同类按命中数降序)。"""
hits: list[Hit] = []
for category, words in HARD_WORDS.items():
for word in words:
count, samples = _scan_word(text, word)
if count:
hits.append(Hit(word, category, "hard", count, samples))
for rule, _note, pattern in PATTERNS:
rx = re.compile(pattern)
found = list(rx.finditer(text))
if found:
samples = [_context(text, m.start(), m.end()) for m in found[:SAMPLES_PER_RULE]]
hits.append(Hit(rule, "句式" if not rule.startswith(("破折号", "省略号", "英文引号")) else "标点",
"hard", len(found), samples))
for category, words in SOFT_WORDS.items():
for word in words:
count, samples = _scan_word(text, word)
if count:
hits.append(Hit(word, category, "soft", count, samples))
hits.sort(key=lambda h: (h.level != "hard", -h.count, h.rule))
return hits
def hard_hits(hits: list[Hit]) -> list[Hit]:
return [h for h in hits if h.level == "hard"]
def per_k(text: str) -> float:
"""每千字硬伤命中数。空文本返回 0,避免除零。"""
length = cjk_length(text) or 1
return sum(h.count for h in hard_hits(check(text))) * 1000 / length
def report(text: str, hits: list[Hit] | None = None, top: int = 14) -> str:
hits = check(text) if hits is None else hits
hard = hard_hits(hits)
soft = [h for h in hits if h.level == "soft"]
total_hard = sum(h.count for h in hard)
total_soft = sum(h.count for h in soft)
length = cjk_length(text)
density = total_hard * 1000 / (length or 1)
lines = [
f"AI 味诊断:{length} 字|硬伤 {total_hard} 处({density:.1f}/千字)|密度词 {total_soft} 处",
]
if hard:
lines.append("")
lines.append("硬伤(命中即改):")
lines += [" " + h.line() for h in hard[:top]]
if soft:
lines.append("")
lines.append("密度词(成串才处理):")
lines += [" " + h.line() for h in soft[:top]]
if not hard and not soft:
lines.append(" 没有命中:用词干净。")
return "\n".join(lines)
def digest(text: str, top: int = 10) -> str:
"""给扩写模型看的简短诊断:要求它逐条消灭这些。"""
hits = [h for h in check(text) if h.level == "hard"][:top]
if not hits:
return "(机械检查没有发现 AI 味硬伤;重点放在把概述换成场景。)"
return "\n".join(f"- {h.category}『{h.rule}』×{h.count}|例:{h.samples[0] if h.samples else ''}"
for h in hits)