- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
421 lines
14 KiB
Python
421 lines
14 KiB
Python
"""解析矮人要塞的 legends 导出。
|
||
|
||
数据来源有两份,按 id 合并:
|
||
1. ``legends.xml`` —— DF 原生 "Export XML" 产物:名字、年份、事件主体
|
||
2. ``legends_plus.xml`` —— DFHack ``exportlegends`` 产物:补充字段(sex/race 等)
|
||
|
||
因为大世界的导出可达数十 MB,这里用 iterparse 流式解析。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
from dataclasses import dataclass, field
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
from defusedxml.ElementTree import iterparse
|
||
|
||
# 字段分类规则来自对真实史料全部 225 种字段的穷举:
|
||
# 含 "hfid" 的字段全部指向历史人物(共 40+ 个:hfid / hfid_target / group_1_hfid /
|
||
# slayer_hfid / snatcher_hfid / seeker_hfid / wounder_hfid / teacher_hfid /
|
||
# conspirator_hfid ...);而 target_enid、identity_id、master_wcid、slayer_item_id 不是。
|
||
# 旧版手工维护的字典只列了 4 个人物字段,导致关系信息在素材里被丢掉。
|
||
_ENTITY_SUFFIXES = ("_entity_id", "_civ_id", "_enid")
|
||
_REGION_FIELDS = {"region_id", "feature_layer_id", "subregion_id"}
|
||
|
||
|
||
def kind_of(tag: str) -> str | None:
|
||
"""事件字段名 → 它指向的对象类别(非引用类字段返回 None)。"""
|
||
t = tag.lower()
|
||
if "hfid" in t or t == "deity" or t.endswith("_deity"):
|
||
return "figure"
|
||
if "artifact" in t:
|
||
return "artifact"
|
||
if t == "site_id" or t.endswith("_site_id"):
|
||
return "site"
|
||
if t in ("entity_id", "target_enid") or t.endswith(_ENTITY_SUFFIXES):
|
||
return "entity"
|
||
if t in _REGION_FIELDS:
|
||
return "region"
|
||
return None
|
||
|
||
|
||
@dataclass
|
||
class Figure:
|
||
id: int
|
||
name: str = ""
|
||
race: str = ""
|
||
sex: int | None = None
|
||
caste: str = "" # 史料里的性别字段(v53 导出用 <caste>MALE/FEMALE</caste>)
|
||
birth_year: int | None = None
|
||
death_year: int | None = None
|
||
profession: str = ""
|
||
entities: list[int] = field(default_factory=list)
|
||
sites: list[int] = field(default_factory=list)
|
||
|
||
@property
|
||
def gender(self) -> str:
|
||
"""史料记载的性别,用于避免正文里默认称呼"他"。"""
|
||
c = (self.caste or "").upper()
|
||
if c == "FEMALE":
|
||
return "女"
|
||
if c == "MALE":
|
||
return "男"
|
||
return ""
|
||
|
||
@property
|
||
def alive_span(self) -> str:
|
||
# DF 用负数占位表示“无此项”(未记录/未死),不要当成真实年份显示
|
||
birth = self.birth_year if (self.birth_year or 0) >= 0 else None
|
||
death = self.death_year if (self.death_year or 0) >= 0 else None
|
||
if birth is None and death is None:
|
||
return "生卒不详"
|
||
b = "?" if birth is None else str(birth)
|
||
d = "在世/不详" if death is None else str(death)
|
||
return f"{b}–{d}"
|
||
|
||
|
||
@dataclass
|
||
class Event:
|
||
id: int
|
||
year: int
|
||
seconds72: int = 0
|
||
type: str = ""
|
||
fields: dict[str, list[str]] = field(default_factory=dict)
|
||
|
||
def ids(self, tag: str) -> list[int]:
|
||
out: list[int] = []
|
||
for raw in self.fields.get(tag, []):
|
||
try:
|
||
out.append(int(raw))
|
||
except ValueError:
|
||
continue
|
||
return out
|
||
|
||
def text(self, tag: str) -> str:
|
||
vals = self.fields.get(tag)
|
||
return vals[0] if vals else ""
|
||
|
||
def ids_of_kind(self, kind: str) -> list[int]:
|
||
"""本事件中指向该类对象的全部有效 id(去重、剔除 -1 占位)。"""
|
||
out: list[int] = []
|
||
for tag, vals in self.fields.items():
|
||
if kind_of(tag) != kind:
|
||
continue
|
||
for raw in vals:
|
||
i = _int_or_none(raw)
|
||
if i is not None and i >= 0 and i not in out:
|
||
out.append(i)
|
||
return out
|
||
|
||
def figure_ids(self) -> list[int]:
|
||
return self.ids_of_kind("figure")
|
||
|
||
def site_ids(self) -> list[int]:
|
||
return self.ids_of_kind("site")
|
||
|
||
def entity_ids(self) -> list[int]:
|
||
return self.ids_of_kind("entity")
|
||
|
||
def field_kinds(self) -> dict[str, list[int]]:
|
||
"""{类别: [id, ...]},供渲染与建图使用。"""
|
||
out: dict[str, list[int]] = {}
|
||
for kind in ("figure", "site", "entity", "artifact", "region"):
|
||
ids = self.ids_of_kind(kind)
|
||
if ids:
|
||
out[kind] = ids
|
||
return out
|
||
|
||
|
||
@dataclass
|
||
class Entity:
|
||
id: int
|
||
name: str = ""
|
||
type: str = ""
|
||
race: str = ""
|
||
|
||
|
||
@dataclass
|
||
class Site:
|
||
id: int
|
||
name: str = ""
|
||
type: str = ""
|
||
|
||
|
||
# ------------------------------------------------------------------ 展示用工具
|
||
def pretty(name: str) -> str:
|
||
"""展示用:DF 的程序化名字全小写,正文里按首字母大写会更像专名。
|
||
|
||
只影响渲染,不改变史料本身。
|
||
"""
|
||
if not name:
|
||
return name
|
||
return " ".join(part[:1].upper() + part[1:] for part in name.split())
|
||
|
||
|
||
@dataclass
|
||
class World:
|
||
name: str = ""
|
||
altname: str = ""
|
||
figures: dict[int, Figure] = field(default_factory=dict)
|
||
events: list[Event] = field(default_factory=list)
|
||
entities: dict[int, Entity] = field(default_factory=dict)
|
||
sites: dict[int, Site] = field(default_factory=dict)
|
||
sources: list[str] = field(default_factory=list)
|
||
|
||
# ---------------------------------------------------------- 查询
|
||
# 约定:id 为负表示“无此项”(DF 用 -1 占位),返回空串让调用方跳过。
|
||
def figure_name(self, fid: int) -> str:
|
||
if fid < 0:
|
||
return ""
|
||
fig = self.figures.get(fid)
|
||
return pretty(fig.name) if fig and fig.name else f"HF#{fid}"
|
||
|
||
def entity_name(self, eid: int) -> str:
|
||
if eid < 0:
|
||
return ""
|
||
ent = self.entities.get(eid)
|
||
return pretty(ent.name) if ent and ent.name else f"Entity#{eid}"
|
||
|
||
def site_name(self, sid: int) -> str:
|
||
if sid < 0:
|
||
return ""
|
||
st = self.sites.get(sid)
|
||
return pretty(st.name) if st and st.name else f"Site#{sid}"
|
||
|
||
def render_refs(self, event: Event) -> list[str]:
|
||
"""把一条事件里的引用字段渲染成 ``tag=名字``。
|
||
|
||
保留字段名是有意的:slayer_hfid=谁 与 hfid=谁 语义完全不同,
|
||
丢掉字段名就等于丢掉了“谁对谁做了什么”。
|
||
负 id(DF 的“无此项”占位)与查不到名字的字段直接跳过。
|
||
"""
|
||
out: list[str] = []
|
||
for tag, vals in event.fields.items():
|
||
kind = kind_of(tag)
|
||
if kind is None:
|
||
continue
|
||
for raw in vals:
|
||
i = _int_or_none(raw)
|
||
if i is None or i < 0:
|
||
continue
|
||
if kind == "figure":
|
||
name = self.figure_name(i)
|
||
elif kind == "site":
|
||
name = self.site_name(i)
|
||
elif kind == "entity":
|
||
name = self.entity_name(i)
|
||
else:
|
||
name = ""
|
||
if name:
|
||
out.append(f"{tag}={name}")
|
||
return out
|
||
|
||
def event_years(self) -> tuple[int, int]:
|
||
years = [e.year for e in self.events if e.year]
|
||
return (min(years), max(years)) if years else (0, 0)
|
||
|
||
|
||
# ------------------------------------------------------------------ 解析工具
|
||
def _int_or_none(text: str | None) -> int | None:
|
||
if text is None:
|
||
return None
|
||
try:
|
||
return int(text)
|
||
except ValueError:
|
||
return None
|
||
|
||
|
||
def _children(elem: Any) -> dict[str, list[str]]:
|
||
"""把元素的子标签收成 {tag: [文本, ...]}。"""
|
||
out: dict[str, list[str]] = {}
|
||
for child in elem:
|
||
val = (child.text or "").strip()
|
||
out.setdefault(child.tag, []).append(val)
|
||
return out
|
||
|
||
|
||
def _first(d: dict[str, list[str]], *keys: str) -> str:
|
||
for k in keys:
|
||
if d.get(k):
|
||
return d[k][0]
|
||
return ""
|
||
|
||
|
||
# XML 1.0 不允许的控制字符(保留 \t \n \r)
|
||
ILLEGAL_BYTES = bytes(range(0x09)) + b"\x0b\x0c" + bytes(range(0x0e, 0x20))
|
||
CLEAN_CACHE = Path("data/exports/clean")
|
||
|
||
|
||
def sanitize(src: Path, dst: Path, chunk_size: int = 8 << 20) -> Path:
|
||
"""把 DF 导出的 legends XML 转成合规的 UTF-8 文件。
|
||
|
||
DF 会在名字里嵌入 \\x10/\\x11 等 CP437 控制字节做专名标记,
|
||
XML 1.0 禁止这些字节,严格解析器会直接报错。
|
||
这里做两件事:剔除非法控制字节;把 CP437 声明与实际字节统一成 UTF-8。
|
||
cp437 是单字节编码,因此可以按块转换而不会切断字符。
|
||
"""
|
||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||
with src.open("rb") as fin, dst.open("wb") as fout:
|
||
first = True
|
||
while True:
|
||
buf = fin.read(chunk_size)
|
||
if not buf:
|
||
break
|
||
buf = buf.translate(None, ILLEGAL_BYTES)
|
||
if first:
|
||
buf = buf.replace(b"encoding='CP437'", b"encoding='UTF-8'", 1)
|
||
buf = buf.replace(b'encoding="CP437"', b'encoding="UTF-8"', 1)
|
||
first = False
|
||
fout.write(buf.decode("cp437").encode("utf-8"))
|
||
return dst
|
||
|
||
|
||
def clean_copy(src: Path) -> Path:
|
||
"""返回可解析的干净副本,带缓存(源文件更新则重建)。
|
||
|
||
缓存写到 data/exports/clean/ 子目录,避开 find_exports 的 *.xml 扫描,
|
||
否则干净副本会被当作第二份史料,导致事件数翻倍。
|
||
"""
|
||
dst = CLEAN_CACHE / src.name
|
||
if dst.is_file() and dst.stat().st_mtime >= src.stat().st_mtime and dst.stat().st_size > 0:
|
||
return dst
|
||
return sanitize(src, dst)
|
||
|
||
|
||
def _reject_unsafe_xml(path: Path) -> None:
|
||
"""拒绝带 DTD / 实体声明的文件。
|
||
|
||
DOCTYPE 必须出现在根元素之前,因此只检查文件头部即可拦住
|
||
XXE 与 billion-laughs 这两类实体展开攻击。
|
||
"""
|
||
with path.open("rb") as fh:
|
||
head = fh.read(65536)
|
||
for marker in (b"<!DOCTYPE", b"<!ENTITY"):
|
||
if marker in head:
|
||
raise ValueError(f"拒绝解析含 DTD/实体声明的文件({marker.decode()}): {path}")
|
||
|
||
|
||
WRAPPERS = ("historical_figures", "historical_events", "entities", "sites")
|
||
|
||
|
||
def parse_file(path: Path, world: World) -> None:
|
||
"""流式解析一个 legends XML,合并进 world。
|
||
|
||
用元素深度而不是包裹标签的进出状态来控制语义:
|
||
只有 depth == 2(即 <df_world> 的直接子元素)才会被当作世界级字段,
|
||
这样历史人物、地区里的 <name> 永远不会污染世界名。
|
||
"""
|
||
_reject_unsafe_xml(path)
|
||
world.sources.append(str(path))
|
||
depth = 0
|
||
parser = iterparse(str(path), events=("start", "end"))
|
||
for event, elem in parser:
|
||
if event == "start":
|
||
depth += 1
|
||
continue
|
||
|
||
tag = elem.tag
|
||
if depth == 2:
|
||
if tag == "name":
|
||
world.name = world.name or (elem.text or "").strip()
|
||
elif tag == "altname":
|
||
world.altname = world.altname or (elem.text or "").strip()
|
||
elif tag == "historical_figure":
|
||
_add_figure(world, elem)
|
||
elem.clear()
|
||
elif tag == "historical_event":
|
||
_add_event(world, elem)
|
||
elem.clear()
|
||
elif tag == "entity":
|
||
_add_entity(world, elem)
|
||
elem.clear()
|
||
elif tag == "site":
|
||
_add_site(world, elem)
|
||
elem.clear()
|
||
|
||
depth -= 1
|
||
|
||
|
||
def _add_figure(world: World, elem) -> None:
|
||
d = _children(elem)
|
||
fid = _int_or_none(_first(d, "id"))
|
||
if fid is None:
|
||
return
|
||
fig = world.figures.setdefault(fid, Figure(id=fid))
|
||
fig.name = fig.name or _first(d, "name")
|
||
fig.race = fig.race or _first(d, "race")
|
||
fig.caste = fig.caste or _first(d, "caste")
|
||
fig.profession = fig.profession or _first(d, "profession")
|
||
if fig.sex is None:
|
||
fig.sex = _int_or_none(_first(d, "sex"))
|
||
if fig.birth_year is None:
|
||
fig.birth_year = _int_or_none(_first(d, "birth_year"))
|
||
if fig.death_year is None:
|
||
fig.death_year = _int_or_none(_first(d, "death_year"))
|
||
|
||
|
||
def _add_event(world: World, elem) -> None:
|
||
d = _children(elem)
|
||
eid = _int_or_none(_first(d, "id"))
|
||
year = _int_or_none(_first(d, "year"))
|
||
if eid is None or year is None:
|
||
return
|
||
world.events.append(
|
||
Event(
|
||
id=eid,
|
||
year=year,
|
||
seconds72=_int_or_none(_first(d, "seconds72")) or 0,
|
||
type=_first(d, "type"),
|
||
fields=d,
|
||
)
|
||
)
|
||
|
||
|
||
def _add_entity(world: World, elem) -> None:
|
||
d = _children(elem)
|
||
eid = _int_or_none(_first(d, "id"))
|
||
if eid is None:
|
||
return
|
||
ent = world.entities.setdefault(eid, Entity(id=eid))
|
||
ent.name = ent.name or _first(d, "name")
|
||
ent.type = ent.type or _first(d, "type")
|
||
ent.race = ent.race or _first(d, "race")
|
||
|
||
|
||
def _add_site(world: World, elem) -> None:
|
||
d = _children(elem)
|
||
sid = _int_or_none(_first(d, "id"))
|
||
if sid is None:
|
||
return
|
||
st = world.sites.setdefault(sid, Site(id=sid))
|
||
st.name = st.name or _first(d, "name")
|
||
st.type = st.type or _first(d, "type")
|
||
|
||
|
||
def load(paths: list[Path]) -> World:
|
||
world = World()
|
||
for p in paths:
|
||
if not p.is_file():
|
||
raise FileNotFoundError(f"缺少 legends 导出文件: {p}")
|
||
if p.stat().st_size == 0:
|
||
raise ValueError(f"legends 导出文件为空: {p}")
|
||
parse_file(clean_copy(p), world)
|
||
world.events.sort(key=lambda e: (e.year, e.seconds72, e.id))
|
||
return world
|
||
|
||
|
||
def describe(world: World, top: int = 8) -> str:
|
||
lo, hi = world.event_years()
|
||
by_type: dict[str, int] = {}
|
||
for e in world.events:
|
||
by_type[e.type] = by_type.get(e.type, 0) + 1
|
||
hot = sorted(by_type.items(), key=lambda kv: -kv[1])[:top]
|
||
lines = [
|
||
f"世界: {world.name}" + (f"({world.altname})" if world.altname else ""),
|
||
f"历史跨度: {lo}–{hi}({hi - lo} 年)| 事件 {len(world.events)} 条",
|
||
f"历史人物 {len(world.figures)} | 文明/组织 {len(world.entities)} | 地点 {len(world.sites)}",
|
||
"高频事件类型: " + ", ".join(f"{t}×{n}" for t, n in hot),
|
||
f"来源: {', '.join(Path(s).name for s in world.sources)}",
|
||
]
|
||
return "\n".join(lines)
|