Files
dwarf-fortress-annals/tests/test_normalize.py
T
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

87 lines
2.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""标点与结构清理的回归测试。"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import normalize
def test_ascii_quotes_become_chinese() -> None:
res = normalize.normalize('他说:"好。" 她说:"不。"')
assert res.text == "他说:“好。” 她说:“不。”", res.text
assert any("英文引号" in n for n in res.notes)
def test_halfwidth_punct_between_cjk() -> None:
res = normalize.normalize("他说,好。她走,停。")
assert "," in res.text and "," not in res.text, res.text
def test_english_punctuation_untouched() -> None:
src = "Mon Sagus, The Plane of Dawn"
assert normalize.normalize(src).text == src
def test_separator_lines_dropped() -> None:
res = normalize.normalize("甲段。\n\n---\n\n乙段。")
assert "---" not in res.text, res.text
assert any("分隔线" in n for n in res.notes)
def test_stray_section_markings_dropped_but_title_kept() -> None:
res = normalize.normalize("# 守门人\n\n甲段。\n\n## 一\n\n乙段。\n\n七\n\n丙段。")
assert res.text.startswith("# 守门人"), res.text
assert "## 一" not in res.text, res.text
assert "\n七\n" not in res.text, res.text
assert any("小节标记" in n for n in res.notes)
def test_duplicate_paragraph_merged() -> None:
res = normalize.normalize("同一段。\n同一段。\n不同段。")
assert res.text.count("同一段。") == 1, res.text
assert any("重复段落" in n for n in res.notes)
def test_duplicate_sentences_merged() -> None:
res = normalize.normalize("他走了。他走了。她留下。")
assert res.text.count("他走了。") == 1, res.text
assert any("重复句子" in n for n in res.notes)
def test_blank_lines_tidied() -> None:
res = normalize.normalize("甲。\n\n\n\n乙。")
assert "\n\n\n" not in res.text
assert res.text == "甲。\n\n乙。"
def test_clean_text_untouched() -> None:
src = "Ral把绳子绕了两圈,打了个结。她没说话。"
res = normalize.normalize(src)
assert res.text == src
assert not res.changed, res.notes
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
except Exception as exc: # noqa: BLE001
failed += 1
print(f" ✗ {fn.__name__}: {type(exc).__name__}: {exc}")
else:
print(f" ✓ {fn.__name__}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())