文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
@@ -0,0 +1,90 @@
|
||||
"""AI 味诊断器的回归测试:能抓住植入的 AI 腔,也不冤枉正常句子。"""
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from dfannals import deslop
|
||||
|
||||
# 故意堆满 AI 腔:禁用词、万能动宾、收束腔、破折号、省略号、认知直述
|
||||
AI_TEXT = (
|
||||
"他心中一震,眼中闪过一丝惊讶,仿佛看见了命运的獠牙——这一刻,他终于明白,"
|
||||
"一切都已经太晚了。他深吸一口气,嘴角勾起一抹冷笑,带着不容置疑的语气说:"
|
||||
"“走吧。”他知道,属于他的反击才刚刚开始……"
|
||||
)
|
||||
|
||||
CLEAN_TEXT = (
|
||||
"Ral把绳子绕了两圈,打了个结。她没说话,把账本推过去,指着第三行。"
|
||||
"Guspu看了很久,把账本合上,还给她。窗外的风把灯吹得晃,两人谁都没去扶。"
|
||||
)
|
||||
|
||||
|
||||
def test_hard_hits_are_caught() -> None:
|
||||
hits = deslop.hard_hits(deslop.check(AI_TEXT))
|
||||
rules = {h.rule for h in hits}
|
||||
for expected in ("心理模板", "眼中闪过", "一丝", "仿佛", "深吸一口气", "嘴角勾起",
|
||||
"不容置疑", "破折号", "省略号"):
|
||||
assert expected in rules, f"漏检: {expected}(实际命中 {sorted(rules)})"
|
||||
|
||||
|
||||
def test_cognitive_and_closure_patterns() -> None:
|
||||
hits = deslop.hard_hits(deslop.check(AI_TEXT))
|
||||
rules = {h.rule for h in hits}
|
||||
assert "认知直述" in rules, "「他知道」应被抓住"
|
||||
assert "收束腔" in rules, "「这一刻」「才刚刚开始」应被抓住"
|
||||
assert "万能动宾" in rules, "「,带着……」应被抓住"
|
||||
assert "抽象命运" in rules, "「命运的獠牙」应被抓住"
|
||||
|
||||
|
||||
def test_clean_text_has_no_hard_hits() -> None:
|
||||
hits = deslop.hard_hits(deslop.check(CLEAN_TEXT))
|
||||
assert not hits, f"正常句子被误判: {[h.line() for h in hits]}"
|
||||
|
||||
|
||||
def test_not_a_but_b_pattern() -> None:
|
||||
hits = deslop.check("他不是冷漠,而是绝望。")
|
||||
assert any(h.rule == "不是A而是B" for h in deslop.hard_hits(hits))
|
||||
|
||||
|
||||
def test_density_separates_ai_from_clean() -> None:
|
||||
ai, clean = deslop.per_k(AI_TEXT), deslop.per_k(CLEAN_TEXT)
|
||||
assert ai > clean, f"AI 稿密度应高于正常稿:{ai:.1f} vs {clean:.1f}"
|
||||
assert clean == 0.0
|
||||
|
||||
|
||||
def test_report_and_digest_render() -> None:
|
||||
rep = deslop.report(AI_TEXT)
|
||||
assert "硬伤" in rep and "字" in rep
|
||||
dig = deslop.digest(AI_TEXT)
|
||||
assert dig.startswith("- "), dig[:80]
|
||||
assert "机械检查没有发现" in deslop.digest(CLEAN_TEXT)
|
||||
|
||||
|
||||
def test_digest_lists_each_rule_once() -> None:
|
||||
dig = deslop.digest(AI_TEXT)
|
||||
rules = [line.split("『")[1].split("』")[0] for line in dig.splitlines()]
|
||||
assert len(rules) == len(set(rules)), f"同一规则重复列出: {rules}"
|
||||
|
||||
|
||||
def _main() -> int:
|
||||
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
|
||||
failed = 0
|
||||
for fn in tests:
|
||||
try:
|
||||
fn()
|
||||
except AssertionError as exc:
|
||||
failed += 1
|
||||
print(f" ✗ {fn.__name__}: {exc}")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
failed += 1
|
||||
print(f" ✗ {fn.__name__}: {type(exc).__name__}: {exc}")
|
||||
else:
|
||||
print(f" ✓ {fn.__name__}")
|
||||
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
|
||||
return 1 if failed else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(_main())
|
||||
Reference in New Issue
Block a user