Files
dwarf-fortress-annals/tests/test_factcheck.py
T

132 lines
4.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。
"""事实校验的回归测试。
重点是按 t6 验收契约做**反例测试**:故意植入史料里不存在的人名/地名,
校验器必须把它揪出来;同时史料里真实存在(或属于游戏枚举值)的词不能被误报。
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import factcheck, legends # noqa: E402
FIXTURES = Path(__file__).resolve().parent / "fixtures"
def load_world() -> legends.World:
return legends.load(
[FIXTURES / "sample-legends.xml", FIXTURES / "sample-legends_plus.xml"]
)
def test_world_parsed() -> None:
world = load_world()
assert world.name == "The World of Prophecy"
assert world.altname == "Usthar Adil"
assert world.figure_name(10) == "Urist McMiner"
assert world.site_name(1) == "Boatmurdered"
assert world.entity_name(100) == "The Steel Confederacy"
assert len(world.events) == 3
assert [e.year for e in world.events] == [30, 31, 88]
def test_fake_names_are_caught() -> None:
"""反例:凭空编造的人名地名必须被抓出来。"""
world = load_world()
text = (
"Zoltan the Unmaker 从 Mount Doom 出发,袭击了 Urist McMiner 驻守的 Boatmurdered。"
"Kogan Deathspear 事后表示,The World of Prophecy 从未见过如此荒唐的进军。"
)
found = {s.name for s in factcheck.check(text, world)}
assert "Zoltan" in found or "Zoltan the Unmaker" in found, f"漏掉了假人名:{found}"
assert "Mount Doom" in found or "Doom" in found, f"漏掉了假地名:{found}"
def test_real_names_not_flagged() -> None:
"""真名与枚举值不能被误报(这是上一轮修掉的假阳性)。"""
world = load_world()
text = (
"The World of Prophecy 的开局并不壮丽。他是个 GOBLIN 斧手,而对方是个 MINER。"
"Urist McMiner 死在 Boatmurdered,The Steel Confederacy 随后在 Glazedbolts 立城。"
"Usthar Adil 这个别名也见于档案。"
)
found = {s.name for s in factcheck.check(text, world)}
assert not found, f"真名被误报:{found}"
def test_lowercase_source_names_with_pretty_display() -> None:
"""回归:DF 程序名全小写,正文做首字母大写后不能被误报。
这个 bug 是在真实世界数据上踩出来的:史料里是 'yemi deermoths',
正文渲染成 'Yemi Deermoths',校验器当时把 6 个真名全报成了可疑。
"""
from dfannals.legends import Entity, Figure, Site, World
w = World(name="mon sagus", altname="the plane of dawn")
w.figures[1] = Figure(id=1, name="yemi deermoths", race="ELF")
w.sites[1] = Site(id=1, name="nutssound")
w.entities[1] = Entity(id=1, name="the washed terrors")
text = "Yemi Deermoths 死在 Nutssound,The Washed Terrors 动的手;Mon Sagus 记住了这件事。"
found = {s.name for s in factcheck.check(text, w)}
assert not found, f"展示形式的真名被误报:{found}"
def test_negative_id_placeholders_not_in_known_names() -> None:
"""回归:id 为负表示“无此项”,不能被渲染成专名。"""
from dfannals.legends import Figure, World
w = World(name="w")
w.figures[1] = Figure(id=1, name="someone")
assert w.figure_name(-1) == ""
assert w.site_name(-1) == ""
assert w.entity_name(-1) == ""
def test_annotation_only_when_suspicious() -> None:
world = load_world()
clean = "# 第一章\n\nUrist McMiner 站在 Boatmurdered 的门口。\n"
assert factcheck.annotate(clean, []) == clean
suspicions = factcheck.check("Zoltan 来了。", world)
assert suspicions, "应判定 Zoltan 可疑"
marked = factcheck.annotate(clean, suspicions)
assert "史料核验提示" in marked and "Zoltan" in marked
def test_unsafe_xml_rejected() -> None:
import tempfile
evil = Path(tempfile.mkdtemp()) / "evil.xml"
evil.write_text(
'<?xml version="1.0"?><!DOCTYPE foo [<!ENTITY x "y">]><df_world><name>&x;</name></df_world>'
)
try:
legends.load([evil])
except ValueError as exc:
assert "DTD" in str(exc) or "实体" in str(exc)
else:
raise AssertionError("含 DTD 的文件未被拦截")
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
print(f" ✓ {fn.__name__}")
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())