Files
dwarf-fortress-annals/dfannals/deslop.py
T
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

163 lines
7.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""AI 味机械诊断:把「文笔是否自然」变成可度量的指标。
规则来自 oh-story-claudecode 的 story-deslop 与 banned-words(MIT),做了两处适配:
1. **分层**。原文把「仿佛、一丝、坚定、冰冷」放同一张一级表,但「坚定」「冰冷」在
史书叙事里是正常词(矮人要塞的"冰冷"可能是字面温度),一律判罪会刷满假阳性。
因此分成硬伤词(几乎只在 AI 文本里成串出现)与密度词(真人也用,只看密度)。
2. **按密度而非布尔判定**。AI 味是密度问题:出现一次是风格,出现二十次是模板。
输出「每千字命中数」,供闸门与扩写模型参考。
语义类判断(比喻是否套话、是否把一件事掰成三段写)机械检查假阳性太高,
交给扩写模型和自评闸门,这里不猜。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from dfannals.chronicle import cjk_length
# 硬伤词:出现在正常中文里的概率很低,命中一处就该改
HARD_WORDS: dict[str, tuple[str, ...]] = {
"情态": ("仿佛", "犹如", "宛若", "如同", "一丝", "一抹", "些许", "几分", "隐约",
"毫无征兆", "几不可闻", "微不可察"),
"动作": ("深吸一口气", "不禁"),
"表情": ("眼中闪过", "嘴角勾起", "眉头微皱", "眉眼低垂", "瞳孔微缩", "瞳孔收缩",
"瞳孔一缩", "指节泛白", "眼神锐利", "目光锐利"),
"心理": ("心中一动", "心头一震", "心下了然", "心中暗道", "心底泛起", "不由得", "心中一凛"),
"判断": ("不容置疑", "不容置喙", "不易察觉", "显而易见", "毫无疑问", "不可否认", "前所未有"),
"过渡": ("不由自主", "情不自禁", "自然而然", "话锋一转", "闪烁着光芒"),
}
# 密度词:真人写作也用,只有成串出现才算模板腔
SOFT_WORDS: dict[str, tuple[str, ...]] = {
"形容": ("坚定", "狡黠", "深邃", "凛冽", "冰冷"),
"语境敏感": ("突然", "陡然", "骤然", "猛然", "好像", "似乎", "瞬间", "猛地", "死死地"),
"弱化副词": ("缓缓", "微微", "轻轻", "淡淡"),
}
# 句式与标点:每条是(规则名,说明,正则)
PATTERNS: tuple[tuple[str, str, str], ...] = (
("不是A而是B", "最毒的 AI 句式,直接写 B", r"不是[^,。;!?\n]{1,24}[,,]\s*(?:而)?是"),
("万能动宾", "「,带着……」万能状语", r"[,,]\s*带着[^,。;!?\n]{0,20}"),
("声音描写", "「声音不大,却带着……」", r"声音(?:不大|很轻|很平|平静)[^。\n]{0,14}(?:却|但)?带着"),
("认知直述", "告诉而非展示,改为动作或台词", r"(?:他|她|他们|她们)(?:这才|终于)?(?:明白|意识到|感到|知道)"),
("章末预告", "「他不知道的是……」空泛预告", r"(?:他|她)不知道的是"),
("收束腔", "「这一刻/才刚刚开始」式总结", r"这一刻[,,]|从这一刻开始|才刚刚开始|原来[,,]"),
("抽象命运", "命运+齿轮/棋局/獠牙", r"命运[^。\n]{0,8}(?:齿轮|棋局|獠牙|改写|安排)"),
("心理模板", "「心中一震/心头一凛」式心理惊吓", r"心(?:中|头|底)(?:一|猛|顿)(?:震|凛|颤)"),
("过渡模板", "「取而代之的是」", r"取而代之"),
("告诉而非展示", "「显得有些X」「散发着一股X气息」", r"显得(?:有些|十分|非常)?[\u4e00-\u9fff]{1,4}|散发(?:着|出)?[^。\n]{0,8}(?:气息|气场)"),
("公式化对话标签", "「好的,他说道」", r"[,,](?:他|她)(?:说道|开口道|沉声道|低声道)"),
("英文引号", "模型常直接吐 ASCII 直角引号,中文排版用“”", r'"'),
("破折号", "正文禁用,改用句号、逗号或动作断句", r"——|──|(?<!-)--(?!-)"),
("省略号", "正文禁用,用动作或短句表达停顿", r"……|\.\.\."),
)
SAMPLE_WIDTH = 14
SAMPLES_PER_RULE = 2
@dataclass
class Hit:
"""一条诊断:命中哪个规则、几次、原文长什么样。"""
rule: str
category: str
level: str # hard = 硬伤词+句式+标点;soft = 只看密度的词
count: int
samples: list[str] = field(default_factory=list)
def line(self) -> str:
ex = ";".join(self.samples)
return f"{self.category:<6} {self.rule:<8} ×{self.count:<3} {ex}"
def _context(text: str, start: int, end: int) -> str:
left = max(0, start - SAMPLE_WIDTH)
right = min(len(text), end + SAMPLE_WIDTH)
return text[left:right].replace("\n", " ").strip()
def _scan_word(text: str, word: str) -> tuple[int, list[str]]:
count, samples, pos = 0, [], text.find(word)
while pos != -1:
count += 1
if len(samples) < SAMPLES_PER_RULE:
samples.append(_context(text, pos, pos + len(word)))
pos = text.find(word, pos + len(word))
return count, samples
def check(text: str) -> list[Hit]:
"""扫一遍文本,返回全部诊断(硬伤在前,同类按命中数降序)。"""
hits: list[Hit] = []
for category, words in HARD_WORDS.items():
for word in words:
count, samples = _scan_word(text, word)
if count:
hits.append(Hit(word, category, "hard", count, samples))
for rule, _note, pattern in PATTERNS:
rx = re.compile(pattern)
found = list(rx.finditer(text))
if found:
samples = [_context(text, m.start(), m.end()) for m in found[:SAMPLES_PER_RULE]]
hits.append(Hit(rule, "句式" if not rule.startswith(("破折号", "省略号", "英文引号")) else "标点",
"hard", len(found), samples))
for category, words in SOFT_WORDS.items():
for word in words:
count, samples = _scan_word(text, word)
if count:
hits.append(Hit(word, category, "soft", count, samples))
hits.sort(key=lambda h: (h.level != "hard", -h.count, h.rule))
return hits
def hard_hits(hits: list[Hit]) -> list[Hit]:
return [h for h in hits if h.level == "hard"]
def per_k(text: str) -> float:
"""每千字硬伤命中数。空文本返回 0,避免除零。"""
length = cjk_length(text) or 1
return sum(h.count for h in hard_hits(check(text))) * 1000 / length
def report(text: str, hits: list[Hit] | None = None, top: int = 14) -> str:
hits = check(text) if hits is None else hits
hard = hard_hits(hits)
soft = [h for h in hits if h.level == "soft"]
total_hard = sum(h.count for h in hard)
total_soft = sum(h.count for h in soft)
length = cjk_length(text)
density = total_hard * 1000 / (length or 1)
lines = [
f"AI 味诊断:{length} 字|硬伤 {total_hard} 处({density:.1f}/千字)|密度词 {total_soft} 处",
]
if hard:
lines.append("")
lines.append("硬伤(命中即改):")
lines += [" " + h.line() for h in hard[:top]]
if soft:
lines.append("")
lines.append("密度词(成串才处理):")
lines += [" " + h.line() for h in soft[:top]]
if not hard and not soft:
lines.append(" 没有命中:用词干净。")
return "\n".join(lines)
def digest(text: str, top: int = 10) -> str:
"""给扩写模型看的简短诊断:要求它逐条消灭这些。"""
hits = [h for h in check(text) if h.level == "hard"][:top]
if not hits:
return "(机械检查没有发现 AI 味硬伤;重点放在把概述换成场景。)"
return "\n".join(f"- {h.category}『{h.rule}』×{h.count}|例:{h.samples[0] if h.samples else ''}"
for h in hits)