"""每卷人物表:Markdown 表格 + 每人一两句小传。 访谈决策: - 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库 - 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册 表格部分完全由史料推导(种族、生卒、出场次数、身份); "身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上, 就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。 """ from __future__ import annotations import json import re from dataclasses import dataclass from pathlib import Path from dfannals import factcheck from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway from dfannals.episodes import Episode from dfannals.legends import Event, World from dfannals.llm import chat from dfannals.threads import Thread # 事件字段 → 中文角色名(用于反推"身份") ROLE_LABELS: dict[str, str] = { "corruptor_hfid": "构陷发起者", "target_hfid": "被针对者", "wounder_hfid": "行凶者", "woundee_hfid": "受伤者", "snatcher_hfid": "绑走者", "seeker_hfid": "求关系者", "winner_hfid": "胜者", "competitor_hfid": "参赛者", "slayer_hfid": "凶手", "hfid": "当事者", "hfid_target": "关系对象", "teacher_hfid": "师长", "student_hfid": "学徒", "persecutor_hfid": "迫害者", "expelled_hfid": "被逐者", "convicted_hfid": "被定罪者", "interrogator_hfid": "审讯者", "framer_hfid": "构陷者", "fooled_hfid": "受骗者", } BIO_LIMIT = 10 # 小传最多覆盖多少人 EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE) @dataclass class CastRow: fid: int name: str race: str gender: str # 男/女/空(史料未载) span: str role: str # 主角 / 对手 / 配角 appearances: int identity: str def as_row(self) -> str: return (f"| {self.name} | {self.race or '—'} | {self.gender or '—'} | {self.span} | " f"{self.identity or '—'} | {self.appearances} | {self.role} |") def _identity_of(world: World, fid: int, events: list[Event]) -> str: """从事件字段反推这个人在这段历史里扮演什么。""" counts: dict[str, int] = {} for e in events: for tag, vals in e.fields.items(): if tag not in ROLE_LABELS: continue for raw in vals: try: if int(raw) == fid: counts[tag] = counts.get(tag, 0) + 1 except ValueError: continue if not counts: return "" ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2] return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked) def _span_of(thread: Thread, episodes: list[Episode]) -> str: """表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。""" if not episodes: return thread.label return f"{episodes[0].start_year}–{episodes[-1].end_year}" def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]: """主角优先,其次是出场最多的对手/配角。""" events = [e for ep in episodes for e in ep.events] or thread.events appearances: dict[int, int] = {} for e in events: for fid in e.figure_ids(): appearances[fid] = appearances.get(fid, 0) + 1 members = set(thread.members) rows: list[CastRow] = [] def make(fid: int, role: str) -> CastRow: fig = world.figures.get(fid) return CastRow( fid=fid, name=world.figure_name(fid), race=fig.race if fig else "", gender=fig.gender if fig else "", span=fig.alive_span if fig else "生卒不详", role=role, appearances=appearances.get(fid, 0), identity=_identity_of(world, fid, events), ) for fid in sorted(members, key=lambda f: -appearances.get(f, 0)): rows.append(make(fid, "主角")) others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f]) for fid in others[:max_others]: rows.append(make(fid, "对手/配角")) return rows BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。 规则: - 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。** - 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。 - 姓名一律保留英文原文。 - 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...} """ def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str: lines = [BIO_INSTRUCTIONS, "", "材料:"] for row in rows[:BIO_LIMIT]: facts = [ ( f"{row.name}({row.race or '种族不详'},{row.gender or '性别史料未载'},{row.span}," f"本卷出场 {row.appearances} 次,身份:{row.identity or '不详'}," f"在故事里是{row.role})" ) ] for e in events: if row.fid not in e.figure_ids(): continue refs = " · ".join(world.render_refs(e)) facts.append(f" {e.year}年 {e.type} {refs}") if len(facts) > 5: break lines.extend(facts) return "\n".join(lines) def _parse_bios(raw: str) -> dict[str, str]: text = EDGE_RE.sub("", raw).strip() start, end = text.find("{"), text.rfind("}") if start < 0 or end <= start: raise ValueError(f"小传没有返回 JSON:{raw[:200]}") try: data = json.loads(text[start:end + 1]) except json.JSONDecodeError as exc: raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None if not isinstance(data, dict): raise TypeError("小传返回的不是对象") return {str(k): str(v) for k, v in data.items()} def build_cast( world: World, thread: Thread, episodes: list[Episode], *, gateway: Gateway | None = None, with_bios: bool = True, ) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]: """返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。""" rows = collect_rows(world, thread, episodes) events = [e for ep in episodes for e in ep.events] or thread.events bios: dict[str, str] = {} if with_bios and rows: reply = chat( [{"role": "user", "content": _bio_prompt(world, rows, events)}], gateway=gateway, max_tokens=MAX_TOKENS_STRUCTURED, ) bios = _parse_bios(reply.content) suspicions: list[factcheck.Suspicion] = [] bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items()) if bios_text: suspicions = factcheck.check(bios_text, world) lines = [ f"# {world.name} · 人物表", "", f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}", f"线索跨度:{_span_of(thread, episodes)}", "", "| 人物 | 种族 | 性别 | 生卒 | 身份 | 本卷出场 | 角色 |", "|---|---|---|---|---|---|---|", ] lines.extend(row.as_row() for row in rows) if bios: lines += ["", "## 小传", ""] for row in rows: bio = bios.get(row.name) or bios.get(row.name.lower()) if bio: lines.append(f"**{row.name}** —— {bio}") lines.append("") if suspicions: lines += [ "", "> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:" + "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]), "", ] return "\n".join(lines) + "\n", rows, suspicions def write_cast(path: Path, text: str) -> Path: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(text, encoding="utf-8") return path