- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
202 lines
7.9 KiB
Python
202 lines
7.9 KiB
Python
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
|
||
# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。
|
||
"""事实校验的回归测试。
|
||
|
||
重点是按 t6 验收契约做**反例测试**:故意植入史料里不存在的人名/地名,
|
||
校验器必须把它揪出来;同时史料里真实存在(或属于游戏枚举值)的词不能被误报。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||
|
||
from dfannals import factcheck, legends
|
||
|
||
FIXTURES = Path(__file__).resolve().parent / "fixtures"
|
||
|
||
|
||
def load_world() -> legends.World:
|
||
return legends.load(
|
||
[FIXTURES / "sample-legends.xml", FIXTURES / "sample-legends_plus.xml"]
|
||
)
|
||
|
||
|
||
def test_world_parsed() -> None:
|
||
world = load_world()
|
||
assert world.name == "The World of Prophecy"
|
||
assert world.altname == "Usthar Adil"
|
||
assert world.figure_name(10) == "Urist McMiner"
|
||
assert world.site_name(1) == "Boatmurdered"
|
||
assert world.entity_name(100) == "The Steel Confederacy"
|
||
assert len(world.events) == 7
|
||
assert [e.year for e in world.events] == [30, 31, 40, 41, 42, 43, 88]
|
||
|
||
|
||
def test_field_classifier_covers_relationship_fields() -> None:
|
||
"""回归:旧版只认 4 个人物字段,导致双方关系在素材里被丢掉。"""
|
||
from dfannals.legends import kind_of
|
||
|
||
for tag in ("hfid", "hfid_target", "group_1_hfid", "group_2_hfid", "snatcher_hfid",
|
||
"seeker_hfid", "hfid1", "hfid2", "wounder_hfid", "teacher_hfid",
|
||
"conspirator_hfid", "slayer_hfid"):
|
||
assert kind_of(tag) == "figure", tag
|
||
for tag in ("target_enid", "entity_id", "attacker_civ_id"):
|
||
assert kind_of(tag) == "entity", tag
|
||
for tag in ("identity_id", "master_wcid", "slayer_item_id",
|
||
"slayer_race", "reason", "relationship"):
|
||
assert kind_of(tag) is None, tag
|
||
|
||
|
||
def test_relationship_events_render_both_sides() -> None:
|
||
"""核心契约:四类关系事件必须渲染出双方,而不是只剩单方面 hfid。"""
|
||
world = load_world()
|
||
rendered = {e.type: world.render_refs(e) for e in world.events}
|
||
|
||
link = rendered["add hf hf link"]
|
||
assert "hfid=Urist McMiner" in link and "hfid_target=Liri Fogbalded" in link
|
||
|
||
battle = rendered["hf simple battle event"]
|
||
assert "group_1_hfid=Urist McMiner" in battle and "group_2_hfid=Kogan Deathspear" in battle
|
||
|
||
abduct = rendered["hf abducted"]
|
||
assert "snatcher_hfid=Kogan Deathspear" in abduct and "target_hfid=Liri Fogbalded" in abduct
|
||
|
||
denied = rendered["hf relationship denied"]
|
||
assert "seeker_hfid=Liri Fogbalded" in denied and "target_hfid=Urist McMiner" in denied
|
||
|
||
|
||
def test_figure_ids_excludes_placeholder_and_dedups() -> None:
|
||
world = load_world()
|
||
battle = next(e for e in world.events if e.type == "hf simple battle event" and e.year == 41)
|
||
assert sorted(battle.figure_ids()) == [10, 11]
|
||
# hfid 与 slayer_hfid 同指一人时不应重复计数
|
||
assert len(battle.figure_ids()) == len(set(battle.figure_ids()))
|
||
|
||
|
||
def test_caste_parsed_into_gender() -> None:
|
||
"""回归:v53 导出用 <caste> 而非 <sex>,漏读会把女性主角写成"他"。"""
|
||
world = load_world()
|
||
assert world.figures[12].caste == "FEMALE"
|
||
assert world.figures[12].gender == "女"
|
||
assert world.figures[10].gender == "", "未载性别的应返回空串,不能猜"
|
||
|
||
|
||
def test_cast_table_shows_gender() -> None:
|
||
"""人物表必须带性别列,否则写作者只能猜。"""
|
||
from collections import Counter
|
||
|
||
from dfannals import cast
|
||
from dfannals import episodes as ep_mod
|
||
from dfannals.threads import Thread
|
||
|
||
world = load_world()
|
||
thread = Thread(
|
||
members=[10, 11, 12], interactions=5, turns=1, span=58, first_year=30, last_year=88,
|
||
types=Counter(), races=Counter(), non_person_races=[], events=world.events,
|
||
)
|
||
planned = ep_mod.plan_episodes(thread)
|
||
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=False)
|
||
assert "| 人物 | 种族 | 性别 |" in text, text.splitlines()[4] if len(text.splitlines()) > 4 else text
|
||
assert not suspicions
|
||
elf = next(r for r in rows if r.fid == 12)
|
||
assert elf.gender == "女"
|
||
|
||
|
||
def test_fake_names_are_caught() -> None:
|
||
"""反例:凭空编造的人名地名必须被抓出来。"""
|
||
world = load_world()
|
||
text = (
|
||
"Zoltan the Unmaker 从 Mount Doom 出发,袭击了 Urist McMiner 驻守的 Boatmurdered。"
|
||
"Kogan Deathspear 事后表示,The World of Prophecy 从未见过如此荒唐的进军。"
|
||
)
|
||
found = {s.name for s in factcheck.check(text, world)}
|
||
assert "Zoltan" in found or "Zoltan the Unmaker" in found, f"漏掉了假人名:{found}"
|
||
assert "Mount Doom" in found or "Doom" in found, f"漏掉了假地名:{found}"
|
||
|
||
|
||
def test_real_names_not_flagged() -> None:
|
||
"""真名与枚举值不能被误报(这是上一轮修掉的假阳性)。"""
|
||
world = load_world()
|
||
text = (
|
||
"The World of Prophecy 的开局并不壮丽。他是个 GOBLIN 斧手,而对方是个 MINER。"
|
||
"Urist McMiner 死在 Boatmurdered,The Steel Confederacy 随后在 Glazedbolts 立城。"
|
||
"Usthar Adil 这个别名也见于档案。"
|
||
)
|
||
found = {s.name for s in factcheck.check(text, world)}
|
||
assert not found, f"真名被误报:{found}"
|
||
|
||
|
||
def test_lowercase_source_names_with_pretty_display() -> None:
|
||
"""回归:DF 程序名全小写,正文做首字母大写后不能被误报。
|
||
|
||
这个 bug 是在真实世界数据上踩出来的:史料里是 'yemi deermoths',
|
||
正文渲染成 'Yemi Deermoths',校验器当时把 6 个真名全报成了可疑。
|
||
"""
|
||
from dfannals.legends import Entity, Figure, Site, World
|
||
|
||
w = World(name="mon sagus", altname="the plane of dawn")
|
||
w.figures[1] = Figure(id=1, name="yemi deermoths", race="ELF")
|
||
w.sites[1] = Site(id=1, name="nutssound")
|
||
w.entities[1] = Entity(id=1, name="the washed terrors")
|
||
|
||
text = "Yemi Deermoths 死在 Nutssound,The Washed Terrors 动的手;Mon Sagus 记住了这件事。"
|
||
found = {s.name for s in factcheck.check(text, w)}
|
||
assert not found, f"展示形式的真名被误报:{found}"
|
||
|
||
|
||
def test_negative_id_placeholders_not_in_known_names() -> None:
|
||
"""回归:id 为负表示“无此项”,不能被渲染成专名。"""
|
||
from dfannals.legends import Figure, World
|
||
|
||
w = World(name="w")
|
||
w.figures[1] = Figure(id=1, name="someone")
|
||
assert w.figure_name(-1) == ""
|
||
assert w.site_name(-1) == ""
|
||
assert w.entity_name(-1) == ""
|
||
|
||
|
||
def test_annotation_only_when_suspicious() -> None:
|
||
world = load_world()
|
||
clean = "# 第一章\n\nUrist McMiner 站在 Boatmurdered 的门口。\n"
|
||
assert factcheck.annotate(clean, []) == clean
|
||
|
||
suspicions = factcheck.check("Zoltan 来了。", world)
|
||
assert suspicions, "应判定 Zoltan 可疑"
|
||
marked = factcheck.annotate(clean, suspicions)
|
||
assert "史料核验提示" in marked and "Zoltan" in marked
|
||
|
||
|
||
def test_unsafe_xml_rejected() -> None:
|
||
import tempfile
|
||
|
||
evil = Path(tempfile.mkdtemp()) / "evil.xml"
|
||
evil.write_text(
|
||
'<?xml version="1.0"?><!DOCTYPE foo [<!ENTITY x "y">]><df_world><name>&x;</name></df_world>'
|
||
)
|
||
try:
|
||
legends.load([evil])
|
||
except ValueError as exc:
|
||
assert "DTD" in str(exc) or "实体" in str(exc)
|
||
else:
|
||
raise AssertionError("含 DTD 的文件未被拦截")
|
||
|
||
|
||
def _main() -> int:
|
||
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
|
||
failed = 0
|
||
for fn in tests:
|
||
try:
|
||
fn()
|
||
print(f" ✓ {fn.__name__}")
|
||
except AssertionError as exc:
|
||
failed += 1
|
||
print(f" ✗ {fn.__name__}: {exc}")
|
||
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
|
||
return 1 if failed else 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(_main())
|