Files
dwarf-fortress-annals/tests/test_deslop.py
T
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

91 lines
3.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""AI 味诊断器的回归测试:能抓住植入的 AI 腔,也不冤枉正常句子。"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import deslop
# 故意堆满 AI 腔:禁用词、万能动宾、收束腔、破折号、省略号、认知直述
AI_TEXT = (
"他心中一震,眼中闪过一丝惊讶,仿佛看见了命运的獠牙——这一刻,他终于明白,"
"一切都已经太晚了。他深吸一口气,嘴角勾起一抹冷笑,带着不容置疑的语气说:"
"“走吧。”他知道,属于他的反击才刚刚开始……"
)
CLEAN_TEXT = (
"Ral把绳子绕了两圈,打了个结。她没说话,把账本推过去,指着第三行。"
"Guspu看了很久,把账本合上,还给她。窗外的风把灯吹得晃,两人谁都没去扶。"
)
def test_hard_hits_are_caught() -> None:
hits = deslop.hard_hits(deslop.check(AI_TEXT))
rules = {h.rule for h in hits}
for expected in ("心理模板", "眼中闪过", "一丝", "仿佛", "深吸一口气", "嘴角勾起",
"不容置疑", "破折号", "省略号"):
assert expected in rules, f"漏检: {expected}(实际命中 {sorted(rules)})"
def test_cognitive_and_closure_patterns() -> None:
hits = deslop.hard_hits(deslop.check(AI_TEXT))
rules = {h.rule for h in hits}
assert "认知直述" in rules, "「他知道」应被抓住"
assert "收束腔" in rules, "「这一刻」「才刚刚开始」应被抓住"
assert "万能动宾" in rules, "「,带着……」应被抓住"
assert "抽象命运" in rules, "「命运的獠牙」应被抓住"
def test_clean_text_has_no_hard_hits() -> None:
hits = deslop.hard_hits(deslop.check(CLEAN_TEXT))
assert not hits, f"正常句子被误判: {[h.line() for h in hits]}"
def test_not_a_but_b_pattern() -> None:
hits = deslop.check("他不是冷漠,而是绝望。")
assert any(h.rule == "不是A而是B" for h in deslop.hard_hits(hits))
def test_density_separates_ai_from_clean() -> None:
ai, clean = deslop.per_k(AI_TEXT), deslop.per_k(CLEAN_TEXT)
assert ai > clean, f"AI 稿密度应高于正常稿:{ai:.1f} vs {clean:.1f}"
assert clean == 0.0
def test_report_and_digest_render() -> None:
rep = deslop.report(AI_TEXT)
assert "硬伤" in rep and "字" in rep
dig = deslop.digest(AI_TEXT)
assert dig.startswith("- "), dig[:80]
assert "机械检查没有发现" in deslop.digest(CLEAN_TEXT)
def test_digest_lists_each_rule_once() -> None:
dig = deslop.digest(AI_TEXT)
rules = [line.split("『")[1].split("』")[0] for line in dig.splitlines()]
assert len(rules) == len(set(rules)), f"同一规则重复列出: {rules}"
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
except Exception as exc: # noqa: BLE001
failed += 1
print(f" ✗ {fn.__name__}: {type(exc).__name__}: {exc}")
else:
print(f" ✓ {fn.__name__}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())