- 新增 dfannals/glossary.py(译名表)与 dfannals/localize.py(渲染层) 人名=音译+绰号意译(Ral Fastenhatchets→拉尔·缚斧、朗古德·商熊「剖胎之寒」), 地名整体意译(Fordedwinds→涉风渡、The Volcano Of Ravens→渡鸦火山) - 生成链路本土化:素材渲染、人物表、扩写白名单全部走译名;已有产物做确定性替换 - 修复 factcheck 的 \b 盲点:Unicode 下汉字也算词字符,紧贴汉字的英文名不被检查 (曾因此漏掉编造专名)。修好后抓出并定点修掉长版第 1 集的 2 处编造 - 修正 title_form 大小写(YETI→Yeti),此前 Yeti/Goblin 一直漏译 - 新增 tests/test_localize.py(13 项回归)
232 lines
8.1 KiB
Python
232 lines
8.1 KiB
Python
"""每卷人物表:Markdown 表格 + 每人一两句小传。
|
||
|
||
访谈决策:
|
||
- 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库
|
||
- 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册
|
||
|
||
表格部分完全由史料推导(种族、生卒、出场次数、身份);
|
||
"身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上,
|
||
就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import re
|
||
from dataclasses import dataclass
|
||
from pathlib import Path
|
||
|
||
from dfannals import factcheck, localize
|
||
from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway
|
||
from dfannals.episodes import Episode
|
||
from dfannals.legends import Event, World
|
||
from dfannals.llm import chat
|
||
from dfannals.threads import Thread
|
||
|
||
# 事件字段 → 中文角色名(用于反推"身份")
|
||
ROLE_LABELS: dict[str, str] = {
|
||
"corruptor_hfid": "构陷发起者",
|
||
"target_hfid": "被针对者",
|
||
"wounder_hfid": "行凶者",
|
||
"woundee_hfid": "受伤者",
|
||
"snatcher_hfid": "绑走者",
|
||
"seeker_hfid": "求关系者",
|
||
"winner_hfid": "胜者",
|
||
"competitor_hfid": "参赛者",
|
||
"slayer_hfid": "凶手",
|
||
"hfid": "当事者",
|
||
"hfid_target": "关系对象",
|
||
"teacher_hfid": "师长",
|
||
"student_hfid": "学徒",
|
||
"persecutor_hfid": "迫害者",
|
||
"expelled_hfid": "被逐者",
|
||
"convicted_hfid": "被定罪者",
|
||
"interrogator_hfid": "审讯者",
|
||
"framer_hfid": "构陷者",
|
||
"fooled_hfid": "受骗者",
|
||
}
|
||
|
||
BIO_LIMIT = 10 # 小传最多覆盖多少人
|
||
EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
|
||
|
||
|
||
@dataclass
|
||
class CastRow:
|
||
fid: int
|
||
name: str
|
||
race: str
|
||
gender: str # 男/女/空(史料未载)
|
||
span: str
|
||
role: str # 主角 / 对手 / 配角
|
||
appearances: int
|
||
identity: str
|
||
|
||
def as_row(self) -> str:
|
||
return (f"| {self.name} | {self.race or '—'} | {self.gender or '—'} | {self.span} | "
|
||
f"{self.identity or '—'} | {self.appearances} | {self.role} |")
|
||
|
||
|
||
def _identity_of(world: World, fid: int, events: list[Event]) -> str:
|
||
"""从事件字段反推这个人在这段历史里扮演什么。"""
|
||
counts: dict[str, int] = {}
|
||
for e in events:
|
||
for tag, vals in e.fields.items():
|
||
if tag not in ROLE_LABELS:
|
||
continue
|
||
for raw in vals:
|
||
try:
|
||
if int(raw) == fid:
|
||
counts[tag] = counts.get(tag, 0) + 1
|
||
except ValueError:
|
||
continue
|
||
if not counts:
|
||
return ""
|
||
ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2]
|
||
return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked)
|
||
|
||
|
||
def _span_of(thread: Thread, episodes: list[Episode]) -> str:
|
||
"""表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。"""
|
||
if not episodes:
|
||
return thread.label
|
||
return f"{episodes[0].start_year}–{episodes[-1].end_year}"
|
||
|
||
|
||
def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]:
|
||
"""主角优先,其次是出场最多的对手/配角。"""
|
||
events = [e for ep in episodes for e in ep.events] or thread.events
|
||
appearances: dict[int, int] = {}
|
||
for e in events:
|
||
for fid in e.figure_ids():
|
||
appearances[fid] = appearances.get(fid, 0) + 1
|
||
|
||
members = set(thread.members)
|
||
rows: list[CastRow] = []
|
||
|
||
def make(fid: int, role: str) -> CastRow:
|
||
fig = world.figures.get(fid)
|
||
return CastRow(
|
||
fid=fid,
|
||
name=world.figure_name(fid),
|
||
race=fig.race if fig else "",
|
||
gender=fig.gender if fig else "",
|
||
span=fig.alive_span if fig else "生卒不详",
|
||
role=role,
|
||
appearances=appearances.get(fid, 0),
|
||
identity=_identity_of(world, fid, events),
|
||
)
|
||
|
||
for fid in sorted(members, key=lambda f: -appearances.get(f, 0)):
|
||
rows.append(make(fid, "主角"))
|
||
others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f])
|
||
for fid in others[:max_others]:
|
||
rows.append(make(fid, "对手/配角"))
|
||
return rows
|
||
|
||
|
||
BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。
|
||
|
||
规则:
|
||
- 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。**
|
||
- 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。
|
||
- 姓名一律保留英文原文。
|
||
- 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...}
|
||
"""
|
||
|
||
|
||
def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
|
||
lines = [BIO_INSTRUCTIONS, "", "材料:"]
|
||
for row in rows[:BIO_LIMIT]:
|
||
facts = [
|
||
(
|
||
f"{row.name}({row.race or '种族不详'},{row.gender or '性别史料未载'},{row.span},"
|
||
f"本卷出场 {row.appearances} 次,身份:{row.identity or '不详'},"
|
||
f"在故事里是{row.role})"
|
||
)
|
||
]
|
||
for e in events:
|
||
if row.fid not in e.figure_ids():
|
||
continue
|
||
refs = " · ".join(world.render_refs(e))
|
||
facts.append(f" {e.year}年 {e.type} {refs}")
|
||
if len(facts) > 5:
|
||
break
|
||
lines.extend(facts)
|
||
return "\n".join(lines)
|
||
|
||
|
||
def _parse_bios(raw: str) -> dict[str, str]:
|
||
text = EDGE_RE.sub("", raw).strip()
|
||
start, end = text.find("{"), text.rfind("}")
|
||
if start < 0 or end <= start:
|
||
raise ValueError(f"小传没有返回 JSON:{raw[:200]}")
|
||
try:
|
||
data = json.loads(text[start:end + 1])
|
||
except json.JSONDecodeError as exc:
|
||
raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
|
||
if not isinstance(data, dict):
|
||
raise TypeError("小传返回的不是对象")
|
||
return {str(k): str(v) for k, v in data.items()}
|
||
|
||
|
||
def build_cast(
|
||
world: World,
|
||
thread: Thread,
|
||
episodes: list[Episode],
|
||
*,
|
||
gateway: Gateway | None = None,
|
||
with_bios: bool = True,
|
||
) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]:
|
||
"""返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。"""
|
||
rows = collect_rows(world, thread, episodes)
|
||
events = [e for ep in episodes for e in ep.events] or thread.events
|
||
|
||
bios: dict[str, str] = {}
|
||
if with_bios and rows:
|
||
reply = chat(
|
||
[{"role": "user", "content": _bio_prompt(world, rows, events)}],
|
||
gateway=gateway,
|
||
max_tokens=MAX_TOKENS_STRUCTURED,
|
||
)
|
||
bios = _parse_bios(reply.content)
|
||
|
||
suspicions: list[factcheck.Suspicion] = []
|
||
bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items())
|
||
if bios_text:
|
||
suspicions = factcheck.check(bios_text, world)
|
||
|
||
lines = [
|
||
f"# {world.name} · 人物表",
|
||
"",
|
||
f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
|
||
f"线索跨度:{_span_of(thread, episodes)}",
|
||
"",
|
||
"| 人物 | 种族 | 性别 | 生卒 | 身份 | 本卷出场 | 角色 |",
|
||
"|---|---|---|---|---|---|---|",
|
||
]
|
||
lines.extend(row.as_row() for row in rows)
|
||
|
||
if bios:
|
||
lines += ["", "## 小传", ""]
|
||
for row in rows:
|
||
bio = bios.get(row.name) or bios.get(row.name.lower())
|
||
if bio:
|
||
lines.append(f"**{row.name}** —— {bio}")
|
||
lines.append("")
|
||
|
||
if suspicions:
|
||
lines += [
|
||
"",
|
||
"> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:"
|
||
+ "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]),
|
||
"",
|
||
]
|
||
|
||
# 专名本土化:表格与小传都输出中文译名
|
||
return localize.for_world(world).render("\n".join(lines)) + "\n", rows, suspicions
|
||
|
||
|
||
def write_cast(path: Path, text: str) -> Path:
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(text, encoding="utf-8")
|
||
return path
|