文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)

- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
Chen Yi
2026-10-05 22:27:02 +08:00
parent 08a032120a
commit 8630dcad55
23 changed files with 1971 additions and 56 deletions
+90
View File
@@ -0,0 +1,90 @@
"""AI 味诊断器的回归测试:能抓住植入的 AI 腔,也不冤枉正常句子。"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import deslop
# 故意堆满 AI 腔:禁用词、万能动宾、收束腔、破折号、省略号、认知直述
AI_TEXT = (
"他心中一震,眼中闪过一丝惊讶,仿佛看见了命运的獠牙——这一刻,他终于明白,"
"一切都已经太晚了。他深吸一口气,嘴角勾起一抹冷笑,带着不容置疑的语气说:"
"“走吧。”他知道,属于他的反击才刚刚开始……"
)
CLEAN_TEXT = (
"Ral把绳子绕了两圈,打了个结。她没说话,把账本推过去,指着第三行。"
"Guspu看了很久,把账本合上,还给她。窗外的风把灯吹得晃,两人谁都没去扶。"
)
def test_hard_hits_are_caught() -> None:
hits = deslop.hard_hits(deslop.check(AI_TEXT))
rules = {h.rule for h in hits}
for expected in ("心理模板", "眼中闪过", "一丝", "仿佛", "深吸一口气", "嘴角勾起",
"不容置疑", "破折号", "省略号"):
assert expected in rules, f"漏检: {expected}(实际命中 {sorted(rules)})"
def test_cognitive_and_closure_patterns() -> None:
hits = deslop.hard_hits(deslop.check(AI_TEXT))
rules = {h.rule for h in hits}
assert "认知直述" in rules, "「他知道」应被抓住"
assert "收束腔" in rules, "「这一刻」「才刚刚开始」应被抓住"
assert "万能动宾" in rules, "「,带着……」应被抓住"
assert "抽象命运" in rules, "「命运的獠牙」应被抓住"
def test_clean_text_has_no_hard_hits() -> None:
hits = deslop.hard_hits(deslop.check(CLEAN_TEXT))
assert not hits, f"正常句子被误判: {[h.line() for h in hits]}"
def test_not_a_but_b_pattern() -> None:
hits = deslop.check("他不是冷漠,而是绝望。")
assert any(h.rule == "不是A而是B" for h in deslop.hard_hits(hits))
def test_density_separates_ai_from_clean() -> None:
ai, clean = deslop.per_k(AI_TEXT), deslop.per_k(CLEAN_TEXT)
assert ai > clean, f"AI 稿密度应高于正常稿:{ai:.1f} vs {clean:.1f}"
assert clean == 0.0
def test_report_and_digest_render() -> None:
rep = deslop.report(AI_TEXT)
assert "硬伤" in rep and "字" in rep
dig = deslop.digest(AI_TEXT)
assert dig.startswith("- "), dig[:80]
assert "机械检查没有发现" in deslop.digest(CLEAN_TEXT)
def test_digest_lists_each_rule_once() -> None:
dig = deslop.digest(AI_TEXT)
rules = [line.split("『")[1].split("』")[0] for line in dig.splitlines()]
assert len(rules) == len(set(rules)), f"同一规则重复列出: {rules}"
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
except Exception as exc: # noqa: BLE001
failed += 1
print(f" ✗ {fn.__name__}: {type(exc).__name__}: {exc}")
else:
print(f" ✓ {fn.__name__}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())