Files
dwarf-fortress-annals/tests/test_factcheck.py
T
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

202 lines
7.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。
"""事实校验的回归测试。
重点是按 t6 验收契约做**反例测试**:故意植入史料里不存在的人名/地名,
校验器必须把它揪出来;同时史料里真实存在(或属于游戏枚举值)的词不能被误报。
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import factcheck, legends
FIXTURES = Path(__file__).resolve().parent / "fixtures"
def load_world() -> legends.World:
return legends.load(
[FIXTURES / "sample-legends.xml", FIXTURES / "sample-legends_plus.xml"]
)
def test_world_parsed() -> None:
world = load_world()
assert world.name == "The World of Prophecy"
assert world.altname == "Usthar Adil"
assert world.figure_name(10) == "Urist McMiner"
assert world.site_name(1) == "Boatmurdered"
assert world.entity_name(100) == "The Steel Confederacy"
assert len(world.events) == 7
assert [e.year for e in world.events] == [30, 31, 40, 41, 42, 43, 88]
def test_field_classifier_covers_relationship_fields() -> None:
"""回归:旧版只认 4 个人物字段,导致双方关系在素材里被丢掉。"""
from dfannals.legends import kind_of
for tag in ("hfid", "hfid_target", "group_1_hfid", "group_2_hfid", "snatcher_hfid",
"seeker_hfid", "hfid1", "hfid2", "wounder_hfid", "teacher_hfid",
"conspirator_hfid", "slayer_hfid"):
assert kind_of(tag) == "figure", tag
for tag in ("target_enid", "entity_id", "attacker_civ_id"):
assert kind_of(tag) == "entity", tag
for tag in ("identity_id", "master_wcid", "slayer_item_id",
"slayer_race", "reason", "relationship"):
assert kind_of(tag) is None, tag
def test_relationship_events_render_both_sides() -> None:
"""核心契约:四类关系事件必须渲染出双方,而不是只剩单方面 hfid。"""
world = load_world()
rendered = {e.type: world.render_refs(e) for e in world.events}
link = rendered["add hf hf link"]
assert "hfid=Urist McMiner" in link and "hfid_target=Liri Fogbalded" in link
battle = rendered["hf simple battle event"]
assert "group_1_hfid=Urist McMiner" in battle and "group_2_hfid=Kogan Deathspear" in battle
abduct = rendered["hf abducted"]
assert "snatcher_hfid=Kogan Deathspear" in abduct and "target_hfid=Liri Fogbalded" in abduct
denied = rendered["hf relationship denied"]
assert "seeker_hfid=Liri Fogbalded" in denied and "target_hfid=Urist McMiner" in denied
def test_figure_ids_excludes_placeholder_and_dedups() -> None:
world = load_world()
battle = next(e for e in world.events if e.type == "hf simple battle event" and e.year == 41)
assert sorted(battle.figure_ids()) == [10, 11]
# hfid 与 slayer_hfid 同指一人时不应重复计数
assert len(battle.figure_ids()) == len(set(battle.figure_ids()))
def test_caste_parsed_into_gender() -> None:
"""回归:v53 导出用 <caste> 而非 <sex>,漏读会把女性主角写成"他"。"""
world = load_world()
assert world.figures[12].caste == "FEMALE"
assert world.figures[12].gender == "女"
assert world.figures[10].gender == "", "未载性别的应返回空串,不能猜"
def test_cast_table_shows_gender() -> None:
"""人物表必须带性别列,否则写作者只能猜。"""
from collections import Counter
from dfannals import cast
from dfannals import episodes as ep_mod
from dfannals.threads import Thread
world = load_world()
thread = Thread(
members=[10, 11, 12], interactions=5, turns=1, span=58, first_year=30, last_year=88,
types=Counter(), races=Counter(), non_person_races=[], events=world.events,
)
planned = ep_mod.plan_episodes(thread)
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=False)
assert "| 人物 | 种族 | 性别 |" in text, text.splitlines()[4] if len(text.splitlines()) > 4 else text
assert not suspicions
elf = next(r for r in rows if r.fid == 12)
assert elf.gender == "女"
def test_fake_names_are_caught() -> None:
"""反例:凭空编造的人名地名必须被抓出来。"""
world = load_world()
text = (
"Zoltan the Unmaker 从 Mount Doom 出发,袭击了 Urist McMiner 驻守的 Boatmurdered。"
"Kogan Deathspear 事后表示,The World of Prophecy 从未见过如此荒唐的进军。"
)
found = {s.name for s in factcheck.check(text, world)}
assert "Zoltan" in found or "Zoltan the Unmaker" in found, f"漏掉了假人名:{found}"
assert "Mount Doom" in found or "Doom" in found, f"漏掉了假地名:{found}"
def test_real_names_not_flagged() -> None:
"""真名与枚举值不能被误报(这是上一轮修掉的假阳性)。"""
world = load_world()
text = (
"The World of Prophecy 的开局并不壮丽。他是个 GOBLIN 斧手,而对方是个 MINER。"
"Urist McMiner 死在 Boatmurdered,The Steel Confederacy 随后在 Glazedbolts 立城。"
"Usthar Adil 这个别名也见于档案。"
)
found = {s.name for s in factcheck.check(text, world)}
assert not found, f"真名被误报:{found}"
def test_lowercase_source_names_with_pretty_display() -> None:
"""回归:DF 程序名全小写,正文做首字母大写后不能被误报。
这个 bug 是在真实世界数据上踩出来的:史料里是 'yemi deermoths',
正文渲染成 'Yemi Deermoths',校验器当时把 6 个真名全报成了可疑。
"""
from dfannals.legends import Entity, Figure, Site, World
w = World(name="mon sagus", altname="the plane of dawn")
w.figures[1] = Figure(id=1, name="yemi deermoths", race="ELF")
w.sites[1] = Site(id=1, name="nutssound")
w.entities[1] = Entity(id=1, name="the washed terrors")
text = "Yemi Deermoths 死在 Nutssound,The Washed Terrors 动的手;Mon Sagus 记住了这件事。"
found = {s.name for s in factcheck.check(text, w)}
assert not found, f"展示形式的真名被误报:{found}"
def test_negative_id_placeholders_not_in_known_names() -> None:
"""回归:id 为负表示“无此项”,不能被渲染成专名。"""
from dfannals.legends import Figure, World
w = World(name="w")
w.figures[1] = Figure(id=1, name="someone")
assert w.figure_name(-1) == ""
assert w.site_name(-1) == ""
assert w.entity_name(-1) == ""
def test_annotation_only_when_suspicious() -> None:
world = load_world()
clean = "# 第一章\n\nUrist McMiner 站在 Boatmurdered 的门口。\n"
assert factcheck.annotate(clean, []) == clean
suspicions = factcheck.check("Zoltan 来了。", world)
assert suspicions, "应判定 Zoltan 可疑"
marked = factcheck.annotate(clean, suspicions)
assert "史料核验提示" in marked and "Zoltan" in marked
def test_unsafe_xml_rejected() -> None:
import tempfile
evil = Path(tempfile.mkdtemp()) / "evil.xml"
evil.write_text(
'<?xml version="1.0"?><!DOCTYPE foo [<!ENTITY x "y">]><df_world><name>&x;</name></df_world>'
)
try:
legends.load([evil])
except ValueError as exc:
assert "DTD" in str(exc) or "实体" in str(exc)
else:
raise AssertionError("含 DTD 的文件未被拦截")
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
print(f" ✓ {fn.__name__}")
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())