Files
dwarf-fortress-annals/dfannals/legends.py
T
Chen Yi e0073e90bd 第 1 卷(人物主线):第 1–3 集 + 人物表
主角线:Guspu Frillyknots / Ral Fastenhatchets / Alath Blottedmine
由线索挖掘器自动选定,按因果链切集,经 5 维度自评闸门(≥12 分)后入库。
2026-10-05 19:28:49 +08:00

409 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""解析矮人要塞的 legends 导出。
数据来源有两份,按 id 合并:
1. ``legends.xml`` —— DF 原生 "Export XML" 产物:名字、年份、事件主体
2. ``legends_plus.xml`` —— DFHack ``exportlegends`` 产物:补充字段(sex/race 等)
因为大世界的导出可达数十 MB,这里用 iterparse 流式解析。
"""
from __future__ import annotations
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
from defusedxml.ElementTree import iterparse
# 字段分类规则来自对真实史料全部 225 种字段的穷举:
# 含 "hfid" 的字段全部指向历史人物(共 40+ 个:hfid / hfid_target / group_1_hfid /
# slayer_hfid / snatcher_hfid / seeker_hfid / wounder_hfid / teacher_hfid /
# conspirator_hfid ...);而 target_enid、identity_id、master_wcid、slayer_item_id 不是。
# 旧版手工维护的字典只列了 4 个人物字段,导致关系信息在素材里被丢掉。
_ENTITY_SUFFIXES = ("_entity_id", "_civ_id", "_enid")
_REGION_FIELDS = {"region_id", "feature_layer_id", "subregion_id"}
def kind_of(tag: str) -> str | None:
"""事件字段名 → 它指向的对象类别(非引用类字段返回 None)。"""
t = tag.lower()
if "hfid" in t or t == "deity" or t.endswith("_deity"):
return "figure"
if "artifact" in t:
return "artifact"
if t == "site_id" or t.endswith("_site_id"):
return "site"
if t in ("entity_id", "target_enid") or t.endswith(_ENTITY_SUFFIXES):
return "entity"
if t in _REGION_FIELDS:
return "region"
return None
@dataclass
class Figure:
id: int
name: str = ""
race: str = ""
sex: int | None = None
birth_year: int | None = None
death_year: int | None = None
profession: str = ""
entities: list[int] = field(default_factory=list)
sites: list[int] = field(default_factory=list)
@property
def alive_span(self) -> str:
# DF 用负数占位表示“无此项”(未记录/未死),不要当成真实年份显示
birth = self.birth_year if (self.birth_year or 0) >= 0 else None
death = self.death_year if (self.death_year or 0) >= 0 else None
if birth is None and death is None:
return "生卒不详"
b = "?" if birth is None else str(birth)
d = "在世/不详" if death is None else str(death)
return f"{b}–{d}"
@dataclass
class Event:
id: int
year: int
seconds72: int = 0
type: str = ""
fields: dict[str, list[str]] = field(default_factory=dict)
def ids(self, tag: str) -> list[int]:
out: list[int] = []
for raw in self.fields.get(tag, []):
try:
out.append(int(raw))
except ValueError:
continue
return out
def text(self, tag: str) -> str:
vals = self.fields.get(tag)
return vals[0] if vals else ""
def ids_of_kind(self, kind: str) -> list[int]:
"""本事件中指向该类对象的全部有效 id(去重、剔除 -1 占位)。"""
out: list[int] = []
for tag, vals in self.fields.items():
if kind_of(tag) != kind:
continue
for raw in vals:
i = _int_or_none(raw)
if i is not None and i >= 0 and i not in out:
out.append(i)
return out
def figure_ids(self) -> list[int]:
return self.ids_of_kind("figure")
def site_ids(self) -> list[int]:
return self.ids_of_kind("site")
def entity_ids(self) -> list[int]:
return self.ids_of_kind("entity")
def field_kinds(self) -> dict[str, list[int]]:
"""{类别: [id, ...]},供渲染与建图使用。"""
out: dict[str, list[int]] = {}
for kind in ("figure", "site", "entity", "artifact", "region"):
ids = self.ids_of_kind(kind)
if ids:
out[kind] = ids
return out
@dataclass
class Entity:
id: int
name: str = ""
type: str = ""
race: str = ""
@dataclass
class Site:
id: int
name: str = ""
type: str = ""
# ------------------------------------------------------------------ 展示用工具
def pretty(name: str) -> str:
"""展示用:DF 的程序化名字全小写,正文里按首字母大写会更像专名。
只影响渲染,不改变史料本身。
"""
if not name:
return name
return " ".join(part[:1].upper() + part[1:] for part in name.split())
@dataclass
class World:
name: str = ""
altname: str = ""
figures: dict[int, Figure] = field(default_factory=dict)
events: list[Event] = field(default_factory=list)
entities: dict[int, Entity] = field(default_factory=dict)
sites: dict[int, Site] = field(default_factory=dict)
sources: list[str] = field(default_factory=list)
# ---------------------------------------------------------- 查询
# 约定:id 为负表示“无此项”(DF 用 -1 占位),返回空串让调用方跳过。
def figure_name(self, fid: int) -> str:
if fid < 0:
return ""
fig = self.figures.get(fid)
return pretty(fig.name) if fig and fig.name else f"HF#{fid}"
def entity_name(self, eid: int) -> str:
if eid < 0:
return ""
ent = self.entities.get(eid)
return pretty(ent.name) if ent and ent.name else f"Entity#{eid}"
def site_name(self, sid: int) -> str:
if sid < 0:
return ""
st = self.sites.get(sid)
return pretty(st.name) if st and st.name else f"Site#{sid}"
def render_refs(self, event: Event) -> list[str]:
"""把一条事件里的引用字段渲染成 ``tag=名字``。
保留字段名是有意的:slayer_hfid=谁 与 hfid=谁 语义完全不同,
丢掉字段名就等于丢掉了“谁对谁做了什么”。
负 id(DF 的“无此项”占位)与查不到名字的字段直接跳过。
"""
out: list[str] = []
for tag, vals in event.fields.items():
kind = kind_of(tag)
if kind is None:
continue
for raw in vals:
i = _int_or_none(raw)
if i is None or i < 0:
continue
if kind == "figure":
name = self.figure_name(i)
elif kind == "site":
name = self.site_name(i)
elif kind == "entity":
name = self.entity_name(i)
else:
name = ""
if name:
out.append(f"{tag}={name}")
return out
def event_years(self) -> tuple[int, int]:
years = [e.year for e in self.events if e.year]
return (min(years), max(years)) if years else (0, 0)
# ------------------------------------------------------------------ 解析工具
def _int_or_none(text: str | None) -> int | None:
if text is None:
return None
try:
return int(text)
except ValueError:
return None
def _children(elem: Any) -> dict[str, list[str]]:
"""把元素的子标签收成 {tag: [文本, ...]}。"""
out: dict[str, list[str]] = {}
for child in elem:
val = (child.text or "").strip()
out.setdefault(child.tag, []).append(val)
return out
def _first(d: dict[str, list[str]], *keys: str) -> str:
for k in keys:
if d.get(k):
return d[k][0]
return ""
# XML 1.0 不允许的控制字符(保留 \t \n \r)
ILLEGAL_BYTES = bytes(range(0x09)) + b"\x0b\x0c" + bytes(range(0x0e, 0x20))
CLEAN_CACHE = Path("data/exports/clean")
def sanitize(src: Path, dst: Path, chunk_size: int = 8 << 20) -> Path:
"""把 DF 导出的 legends XML 转成合规的 UTF-8 文件。
DF 会在名字里嵌入 \\x10/\\x11 等 CP437 控制字节做专名标记,
XML 1.0 禁止这些字节,严格解析器会直接报错。
这里做两件事:剔除非法控制字节;把 CP437 声明与实际字节统一成 UTF-8。
cp437 是单字节编码,因此可以按块转换而不会切断字符。
"""
dst.parent.mkdir(parents=True, exist_ok=True)
with src.open("rb") as fin, dst.open("wb") as fout:
first = True
while True:
buf = fin.read(chunk_size)
if not buf:
break
buf = buf.translate(None, ILLEGAL_BYTES)
if first:
buf = buf.replace(b"encoding='CP437'", b"encoding='UTF-8'", 1)
buf = buf.replace(b'encoding="CP437"', b'encoding="UTF-8"', 1)
first = False
fout.write(buf.decode("cp437").encode("utf-8"))
return dst
def clean_copy(src: Path) -> Path:
"""返回可解析的干净副本,带缓存(源文件更新则重建)。
缓存写到 data/exports/clean/ 子目录,避开 find_exports 的 *.xml 扫描,
否则干净副本会被当作第二份史料,导致事件数翻倍。
"""
dst = CLEAN_CACHE / src.name
if dst.is_file() and dst.stat().st_mtime >= src.stat().st_mtime and dst.stat().st_size > 0:
return dst
return sanitize(src, dst)
def _reject_unsafe_xml(path: Path) -> None:
"""拒绝带 DTD / 实体声明的文件。
DOCTYPE 必须出现在根元素之前,因此只检查文件头部即可拦住
XXE 与 billion-laughs 这两类实体展开攻击。
"""
with path.open("rb") as fh:
head = fh.read(65536)
for marker in (b"<!DOCTYPE", b"<!ENTITY"):
if marker in head:
raise ValueError(f"拒绝解析含 DTD/实体声明的文件({marker.decode()}): {path}")
WRAPPERS = ("historical_figures", "historical_events", "entities", "sites")
def parse_file(path: Path, world: World) -> None:
"""流式解析一个 legends XML,合并进 world。
用元素深度而不是包裹标签的进出状态来控制语义:
只有 depth == 2(即 <df_world> 的直接子元素)才会被当作世界级字段,
这样历史人物、地区里的 <name> 永远不会污染世界名。
"""
_reject_unsafe_xml(path)
world.sources.append(str(path))
depth = 0
parser = iterparse(str(path), events=("start", "end"))
for event, elem in parser:
if event == "start":
depth += 1
continue
tag = elem.tag
if depth == 2:
if tag == "name":
world.name = world.name or (elem.text or "").strip()
elif tag == "altname":
world.altname = world.altname or (elem.text or "").strip()
elif tag == "historical_figure":
_add_figure(world, elem)
elem.clear()
elif tag == "historical_event":
_add_event(world, elem)
elem.clear()
elif tag == "entity":
_add_entity(world, elem)
elem.clear()
elif tag == "site":
_add_site(world, elem)
elem.clear()
depth -= 1
def _add_figure(world: World, elem) -> None:
d = _children(elem)
fid = _int_or_none(_first(d, "id"))
if fid is None:
return
fig = world.figures.setdefault(fid, Figure(id=fid))
fig.name = fig.name or _first(d, "name")
fig.race = fig.race or _first(d, "race")
fig.profession = fig.profession or _first(d, "profession")
if fig.sex is None:
fig.sex = _int_or_none(_first(d, "sex"))
if fig.birth_year is None:
fig.birth_year = _int_or_none(_first(d, "birth_year"))
if fig.death_year is None:
fig.death_year = _int_or_none(_first(d, "death_year"))
def _add_event(world: World, elem) -> None:
d = _children(elem)
eid = _int_or_none(_first(d, "id"))
year = _int_or_none(_first(d, "year"))
if eid is None or year is None:
return
world.events.append(
Event(
id=eid,
year=year,
seconds72=_int_or_none(_first(d, "seconds72")) or 0,
type=_first(d, "type"),
fields=d,
)
)
def _add_entity(world: World, elem) -> None:
d = _children(elem)
eid = _int_or_none(_first(d, "id"))
if eid is None:
return
ent = world.entities.setdefault(eid, Entity(id=eid))
ent.name = ent.name or _first(d, "name")
ent.type = ent.type or _first(d, "type")
ent.race = ent.race or _first(d, "race")
def _add_site(world: World, elem) -> None:
d = _children(elem)
sid = _int_or_none(_first(d, "id"))
if sid is None:
return
st = world.sites.setdefault(sid, Site(id=sid))
st.name = st.name or _first(d, "name")
st.type = st.type or _first(d, "type")
def load(paths: list[Path]) -> World:
world = World()
for p in paths:
if not p.is_file():
raise FileNotFoundError(f"缺少 legends 导出文件: {p}")
if p.stat().st_size == 0:
raise ValueError(f"legends 导出文件为空: {p}")
parse_file(clean_copy(p), world)
world.events.sort(key=lambda e: (e.year, e.seconds72, e.id))
return world
def describe(world: World, top: int = 8) -> str:
lo, hi = world.event_years()
by_type: dict[str, int] = {}
for e in world.events:
by_type[e.type] = by_type.get(e.type, 0) + 1
hot = sorted(by_type.items(), key=lambda kv: -kv[1])[:top]
lines = [
f"世界: {world.name}" + (f"({world.altname})" if world.altname else ""),
f"历史跨度: {lo}–{hi}({hi - lo} 年)| 事件 {len(world.events)} 条",
f"历史人物 {len(world.figures)} | 文明/组织 {len(world.entities)} | 地点 {len(world.sites)}",
"高频事件类型: " + ", ".join(f"{t}×{n}" for t, n in hot),
f"来源: {', '.join(Path(s).name for s in world.sources)}",
]
return "\n".join(lines)