第 1 卷(人物主线):第 1–3 集 + 人物表

主角线:Guspu Frillyknots / Ral Fastenhatchets / Alath Blottedmine
由线索挖掘器自动选定,按因果链切集,经 5 维度自评闸门(≥12 分)后入库。
This commit is contained in:
Chen Yi
2026-10-05 19:28:49 +08:00
parent 594cf48a1a
commit e0073e90bd
24 changed files with 2070 additions and 79 deletions
+225
View File
@@ -0,0 +1,225 @@
"""每卷人物表:Markdown 表格 + 每人一两句小传。
访谈决策:
- 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库
- 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册
表格部分完全由史料推导(种族、生卒、出场次数、身份);
"身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上,
就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。
"""
from __future__ import annotations
import json
import re
from dataclasses import dataclass
from pathlib import Path
from dfannals import factcheck
from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway
from dfannals.episodes import Episode
from dfannals.legends import Event, World
from dfannals.llm import chat
from dfannals.threads import Thread
# 事件字段 → 中文角色名(用于反推"身份")
ROLE_LABELS: dict[str, str] = {
"corruptor_hfid": "构陷发起者",
"target_hfid": "被针对者",
"wounder_hfid": "行凶者",
"woundee_hfid": "受伤者",
"snatcher_hfid": "绑走者",
"seeker_hfid": "求关系者",
"winner_hfid": "胜者",
"competitor_hfid": "参赛者",
"slayer_hfid": "凶手",
"hfid": "当事者",
"hfid_target": "关系对象",
"teacher_hfid": "师长",
"student_hfid": "学徒",
"persecutor_hfid": "迫害者",
"expelled_hfid": "被逐者",
"convicted_hfid": "被定罪者",
"interrogator_hfid": "审讯者",
"framer_hfid": "构陷者",
"fooled_hfid": "受骗者",
}
BIO_LIMIT = 10 # 小传最多覆盖多少人
EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
@dataclass
class CastRow:
fid: int
name: str
race: str
span: str
role: str # 主角 / 对手 / 配角
appearances: int
identity: str
def as_row(self) -> str:
return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | "
f"{self.appearances} | {self.role} |")
def _identity_of(world: World, fid: int, events: list[Event]) -> str:
"""从事件字段反推这个人在这段历史里扮演什么。"""
counts: dict[str, int] = {}
for e in events:
for tag, vals in e.fields.items():
if tag not in ROLE_LABELS:
continue
for raw in vals:
try:
if int(raw) == fid:
counts[tag] = counts.get(tag, 0) + 1
except ValueError:
continue
if not counts:
return ""
ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2]
return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked)
def _span_of(thread: Thread, episodes: list[Episode]) -> str:
"""表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。"""
if not episodes:
return thread.label
return f"{episodes[0].start_year}–{episodes[-1].end_year}"
def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]:
"""主角优先,其次是出场最多的对手/配角。"""
events = [e for ep in episodes for e in ep.events] or thread.events
appearances: dict[int, int] = {}
for e in events:
for fid in e.figure_ids():
appearances[fid] = appearances.get(fid, 0) + 1
members = set(thread.members)
rows: list[CastRow] = []
def make(fid: int, role: str) -> CastRow:
fig = world.figures.get(fid)
return CastRow(
fid=fid,
name=world.figure_name(fid),
race=fig.race if fig else "",
span=fig.alive_span if fig else "生卒不详",
role=role,
appearances=appearances.get(fid, 0),
identity=_identity_of(world, fid, events),
)
for fid in sorted(members, key=lambda f: -appearances.get(f, 0)):
rows.append(make(fid, "主角"))
others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f])
for fid in others[:max_others]:
rows.append(make(fid, "对手/配角"))
return rows
BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。
规则:
- 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。**
- 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。
- 姓名一律保留英文原文。
- 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...}
"""
def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
lines = [BIO_INSTRUCTIONS, "", "材料:"]
for row in rows[:BIO_LIMIT]:
facts = [
f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次,"
f"身份:{row.identity or '不详'},在故事里是{row.role})"
]
for e in events:
if row.fid not in e.figure_ids():
continue
refs = " · ".join(world.render_refs(e))
facts.append(f" {e.year}年 {e.type} {refs}")
if len(facts) > 5:
break
lines.extend(facts)
return "\n".join(lines)
def _parse_bios(raw: str) -> dict[str, str]:
text = EDGE_RE.sub("", raw).strip()
start, end = text.find("{"), text.rfind("}")
if start < 0 or end <= start:
raise ValueError(f"小传没有返回 JSON:{raw[:200]}")
try:
data = json.loads(text[start:end + 1])
except json.JSONDecodeError as exc:
raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
if not isinstance(data, dict):
raise ValueError("小传返回的不是对象")
return {str(k): str(v) for k, v in data.items()}
def build_cast(
world: World,
thread: Thread,
episodes: list[Episode],
*,
gateway: Gateway | None = None,
with_bios: bool = True,
) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]:
"""返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。"""
rows = collect_rows(world, thread, episodes)
events = [e for ep in episodes for e in ep.events] or thread.events
bios: dict[str, str] = {}
if with_bios and rows:
reply = chat(
[{"role": "user", "content": _bio_prompt(world, rows, events)}],
gateway=gateway,
max_tokens=MAX_TOKENS_STRUCTURED,
)
bios = _parse_bios(reply.content)
suspicions: list[factcheck.Suspicion] = []
bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items())
if bios_text:
suspicions = factcheck.check(bios_text, world)
lines = [
f"# {world.name} · 人物表",
"",
f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
f"线索跨度:{_span_of(thread, episodes)}",
"",
"| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |",
"|---|---|---|---|---|---|",
]
lines.extend(row.as_row() for row in rows)
if bios:
lines += ["", "## 小传", ""]
for row in rows:
bio = bios.get(row.name) or bios.get(row.name.lower())
if bio:
lines.append(f"**{row.name}** —— {bio}")
lines.append("")
if suspicions:
lines += [
"",
"> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:"
+ "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]),
"",
]
return "\n".join(lines) + "\n", rows, suspicions
def write_cast(path: Path, text: str) -> Path:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text, encoding="utf-8")
return path