"""AI 味机械诊断:把「文笔是否自然」变成可度量的指标。 规则来自 oh-story-claudecode 的 story-deslop 与 banned-words(MIT),做了两处适配: 1. **分层**。原文把「仿佛、一丝、坚定、冰冷」放同一张一级表,但「坚定」「冰冷」在 史书叙事里是正常词(矮人要塞的"冰冷"可能是字面温度),一律判罪会刷满假阳性。 因此分成硬伤词(几乎只在 AI 文本里成串出现)与密度词(真人也用,只看密度)。 2. **按密度而非布尔判定**。AI 味是密度问题:出现一次是风格,出现二十次是模板。 输出「每千字命中数」,供闸门与扩写模型参考。 语义类判断(比喻是否套话、是否把一件事掰成三段写)机械检查假阳性太高, 交给扩写模型和自评闸门,这里不猜。 """ from __future__ import annotations import re from dataclasses import dataclass, field from dfannals.chronicle import cjk_length # 硬伤词:出现在正常中文里的概率很低,命中一处就该改 HARD_WORDS: dict[str, tuple[str, ...]] = { "情态": ("仿佛", "犹如", "宛若", "如同", "一丝", "一抹", "些许", "几分", "隐约", "毫无征兆", "几不可闻", "微不可察"), "动作": ("深吸一口气", "不禁"), "表情": ("眼中闪过", "嘴角勾起", "眉头微皱", "眉眼低垂", "瞳孔微缩", "瞳孔收缩", "瞳孔一缩", "指节泛白", "眼神锐利", "目光锐利"), "心理": ("心中一动", "心头一震", "心下了然", "心中暗道", "心底泛起", "不由得", "心中一凛"), "判断": ("不容置疑", "不容置喙", "不易察觉", "显而易见", "毫无疑问", "不可否认", "前所未有"), "过渡": ("不由自主", "情不自禁", "自然而然", "话锋一转", "闪烁着光芒"), } # 密度词:真人写作也用,只有成串出现才算模板腔 SOFT_WORDS: dict[str, tuple[str, ...]] = { "形容": ("坚定", "狡黠", "深邃", "凛冽", "冰冷"), "语境敏感": ("突然", "陡然", "骤然", "猛然", "好像", "似乎", "瞬间", "猛地", "死死地"), "弱化副词": ("缓缓", "微微", "轻轻", "淡淡"), } # 句式与标点:每条是(规则名,说明,正则) PATTERNS: tuple[tuple[str, str, str], ...] = ( ("不是A而是B", "最毒的 AI 句式,直接写 B", r"不是[^,。;!?\n]{1,24}[,,]\s*(?:而)?是"), ("万能动宾", "「,带着……」万能状语", r"[,,]\s*带着[^,。;!?\n]{0,20}"), ("声音描写", "「声音不大,却带着……」", r"声音(?:不大|很轻|很平|平静)[^。\n]{0,14}(?:却|但)?带着"), ("认知直述", "告诉而非展示,改为动作或台词", r"(?:他|她|他们|她们)(?:这才|终于)?(?:明白|意识到|感到|知道)"), ("章末预告", "「他不知道的是……」空泛预告", r"(?:他|她)不知道的是"), ("收束腔", "「这一刻/才刚刚开始」式总结", r"这一刻[,,]|从这一刻开始|才刚刚开始|原来[,,]"), ("抽象命运", "命运+齿轮/棋局/獠牙", r"命运[^。\n]{0,8}(?:齿轮|棋局|獠牙|改写|安排)"), ("心理模板", "「心中一震/心头一凛」式心理惊吓", r"心(?:中|头|底)(?:一|猛|顿)(?:震|凛|颤)"), ("过渡模板", "「取而代之的是」", r"取而代之"), ("告诉而非展示", "「显得有些X」「散发着一股X气息」", r"显得(?:有些|十分|非常)?[\u4e00-\u9fff]{1,4}|散发(?:着|出)?[^。\n]{0,8}(?:气息|气场)"), ("公式化对话标签", "「好的,他说道」", r"[,,](?:他|她)(?:说道|开口道|沉声道|低声道)"), ("英文引号", "模型常直接吐 ASCII 直角引号,中文排版用“”", r'"'), ("破折号", "正文禁用,改用句号、逗号或动作断句", r"——|──|(? str: ex = ";".join(self.samples) return f"{self.category:<6} {self.rule:<8} ×{self.count:<3} {ex}" def _context(text: str, start: int, end: int) -> str: left = max(0, start - SAMPLE_WIDTH) right = min(len(text), end + SAMPLE_WIDTH) return text[left:right].replace("\n", " ").strip() def _scan_word(text: str, word: str) -> tuple[int, list[str]]: count, samples, pos = 0, [], text.find(word) while pos != -1: count += 1 if len(samples) < SAMPLES_PER_RULE: samples.append(_context(text, pos, pos + len(word))) pos = text.find(word, pos + len(word)) return count, samples def check(text: str) -> list[Hit]: """扫一遍文本,返回全部诊断(硬伤在前,同类按命中数降序)。""" hits: list[Hit] = [] for category, words in HARD_WORDS.items(): for word in words: count, samples = _scan_word(text, word) if count: hits.append(Hit(word, category, "hard", count, samples)) for rule, _note, pattern in PATTERNS: rx = re.compile(pattern) found = list(rx.finditer(text)) if found: samples = [_context(text, m.start(), m.end()) for m in found[:SAMPLES_PER_RULE]] hits.append(Hit(rule, "句式" if not rule.startswith(("破折号", "省略号", "英文引号")) else "标点", "hard", len(found), samples)) for category, words in SOFT_WORDS.items(): for word in words: count, samples = _scan_word(text, word) if count: hits.append(Hit(word, category, "soft", count, samples)) hits.sort(key=lambda h: (h.level != "hard", -h.count, h.rule)) return hits def hard_hits(hits: list[Hit]) -> list[Hit]: return [h for h in hits if h.level == "hard"] def per_k(text: str) -> float: """每千字硬伤命中数。空文本返回 0,避免除零。""" length = cjk_length(text) or 1 return sum(h.count for h in hard_hits(check(text))) * 1000 / length def report(text: str, hits: list[Hit] | None = None, top: int = 14) -> str: hits = check(text) if hits is None else hits hard = hard_hits(hits) soft = [h for h in hits if h.level == "soft"] total_hard = sum(h.count for h in hard) total_soft = sum(h.count for h in soft) length = cjk_length(text) density = total_hard * 1000 / (length or 1) lines = [ f"AI 味诊断:{length} 字|硬伤 {total_hard} 处({density:.1f}/千字)|密度词 {total_soft} 处", ] if hard: lines.append("") lines.append("硬伤(命中即改):") lines += [" " + h.line() for h in hard[:top]] if soft: lines.append("") lines.append("密度词(成串才处理):") lines += [" " + h.line() for h in soft[:top]] if not hard and not soft: lines.append(" 没有命中:用词干净。") return "\n".join(lines) def digest(text: str, top: int = 10) -> str: """给扩写模型看的简短诊断:要求它逐条消灭这些。""" hits = [h for h in check(text) if h.level == "hard"][:top] if not hits: return "(机械检查没有发现 AI 味硬伤;重点放在把概述换成场景。)" return "\n".join(f"- {h.category}『{h.rule}』×{h.count}|例:{h.samples[0] if h.samples else ''}" for h in hits)