From e0073e90bd420844bcdc8058f4598937c3ea68c2 Mon Sep 17 00:00:00 2001
From: Chen Yi <466354947@qq.com>
Date: Mon, 5 Oct 2026 19:28:49 +0800
Subject: [PATCH] =?UTF-8?q?=E7=AC=AC=201=20=E5=8D=B7=EF=BC=88=E4=BA=BA?=
=?UTF-8?q?=E7=89=A9=E4=B8=BB=E7=BA=BF=EF=BC=89=EF=BC=9A=E7=AC=AC=201?=
=?UTF-8?q?=E2=80=933=20=E9=9B=86=20+=20=E4=BA=BA=E7=89=A9=E8=A1=A8?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
主角线:Guspu Frillyknots / Ral Fastenhatchets / Alath Blottedmine
由线索挖掘器自动选定,按因果链切集,经 5 维度自评闸门(≥12 分)后入库。
---
dfannals/cast.py | 225 +++++++++++++
dfannals/chronicle.py | 4 +-
dfannals/cli.py | 156 ++++++++-
dfannals/config.py | 5 +-
dfannals/episodes.py | 192 +++++++++++
dfannals/gate.py | 223 ++++++++++++
dfannals/legends.py | 101 ++++--
dfannals/publish.py | 1 +
dfannals/score.py | 21 +-
dfannals/slice.py | 27 +-
dfannals/threads.py | 287 ++++++++++++++++
dfannals/timeline.py | 28 +-
dfannals/volume.py | 336 +++++++++++++++++++
output/volumes/v1/cast.md | 37 ++
output/volumes/v1/chapters/001-181-198.md | 13 +
output/volumes/v1/chapters/002-202-216.md | 21 ++
output/volumes/v1/chapters/003-219-228-p1.md | 19 ++
prompts/biographer.md | 45 +++
scripts/dfannals | 36 ++
scripts/gate-selfcheck.py | 76 +++++
tests/fixtures/sample-legends.xml | 5 +
tests/test_episodes.py | 119 +++++++
tests/test_factcheck.py | 47 ++-
tests/test_gate.py | 125 +++++++
24 files changed, 2070 insertions(+), 79 deletions(-)
create mode 100644 dfannals/cast.py
create mode 100644 dfannals/episodes.py
create mode 100644 dfannals/gate.py
create mode 100644 dfannals/threads.py
create mode 100644 dfannals/volume.py
create mode 100644 output/volumes/v1/cast.md
create mode 100644 output/volumes/v1/chapters/001-181-198.md
create mode 100644 output/volumes/v1/chapters/002-202-216.md
create mode 100644 output/volumes/v1/chapters/003-219-228-p1.md
create mode 100644 prompts/biographer.md
create mode 100755 scripts/dfannals
create mode 100644 scripts/gate-selfcheck.py
create mode 100644 tests/test_episodes.py
create mode 100644 tests/test_gate.py
diff --git a/dfannals/cast.py b/dfannals/cast.py
new file mode 100644
index 0000000..8e406d6
--- /dev/null
+++ b/dfannals/cast.py
@@ -0,0 +1,225 @@
+"""每卷人物表:Markdown 表格 + 每人一两句小传。
+
+访谈决策:
+- 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库
+- 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册
+
+表格部分完全由史料推导(种族、生卒、出场次数、身份);
+"身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上,
+就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。
+"""
+from __future__ import annotations
+
+import json
+import re
+from dataclasses import dataclass
+from pathlib import Path
+
+from dfannals import factcheck
+from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway
+from dfannals.episodes import Episode
+from dfannals.legends import Event, World
+from dfannals.llm import chat
+from dfannals.threads import Thread
+
+# 事件字段 → 中文角色名(用于反推"身份")
+ROLE_LABELS: dict[str, str] = {
+ "corruptor_hfid": "构陷发起者",
+ "target_hfid": "被针对者",
+ "wounder_hfid": "行凶者",
+ "woundee_hfid": "受伤者",
+ "snatcher_hfid": "绑走者",
+ "seeker_hfid": "求关系者",
+ "winner_hfid": "胜者",
+ "competitor_hfid": "参赛者",
+ "slayer_hfid": "凶手",
+ "hfid": "当事者",
+ "hfid_target": "关系对象",
+ "teacher_hfid": "师长",
+ "student_hfid": "学徒",
+ "persecutor_hfid": "迫害者",
+ "expelled_hfid": "被逐者",
+ "convicted_hfid": "被定罪者",
+ "interrogator_hfid": "审讯者",
+ "framer_hfid": "构陷者",
+ "fooled_hfid": "受骗者",
+}
+
+BIO_LIMIT = 10 # 小传最多覆盖多少人
+EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
+
+
+@dataclass
+class CastRow:
+ fid: int
+ name: str
+ race: str
+ span: str
+ role: str # 主角 / 对手 / 配角
+ appearances: int
+ identity: str
+
+ def as_row(self) -> str:
+ return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | "
+ f"{self.appearances} | {self.role} |")
+
+
+def _identity_of(world: World, fid: int, events: list[Event]) -> str:
+ """从事件字段反推这个人在这段历史里扮演什么。"""
+ counts: dict[str, int] = {}
+ for e in events:
+ for tag, vals in e.fields.items():
+ if tag not in ROLE_LABELS:
+ continue
+ for raw in vals:
+ try:
+ if int(raw) == fid:
+ counts[tag] = counts.get(tag, 0) + 1
+ except ValueError:
+ continue
+ if not counts:
+ return ""
+ ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2]
+ return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked)
+
+
+def _span_of(thread: Thread, episodes: list[Episode]) -> str:
+ """表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。"""
+ if not episodes:
+ return thread.label
+ return f"{episodes[0].start_year}–{episodes[-1].end_year}"
+
+
+def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]:
+ """主角优先,其次是出场最多的对手/配角。"""
+ events = [e for ep in episodes for e in ep.events] or thread.events
+ appearances: dict[int, int] = {}
+ for e in events:
+ for fid in e.figure_ids():
+ appearances[fid] = appearances.get(fid, 0) + 1
+
+ members = set(thread.members)
+ rows: list[CastRow] = []
+
+ def make(fid: int, role: str) -> CastRow:
+ fig = world.figures.get(fid)
+ return CastRow(
+ fid=fid,
+ name=world.figure_name(fid),
+ race=fig.race if fig else "",
+ span=fig.alive_span if fig else "生卒不详",
+ role=role,
+ appearances=appearances.get(fid, 0),
+ identity=_identity_of(world, fid, events),
+ )
+
+ for fid in sorted(members, key=lambda f: -appearances.get(f, 0)):
+ rows.append(make(fid, "主角"))
+ others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f])
+ for fid in others[:max_others]:
+ rows.append(make(fid, "对手/配角"))
+ return rows
+
+
+BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。
+
+规则:
+- 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。**
+- 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。
+- 姓名一律保留英文原文。
+- 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...}
+"""
+
+
+def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
+ lines = [BIO_INSTRUCTIONS, "", "材料:"]
+ for row in rows[:BIO_LIMIT]:
+ facts = [
+ f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次,"
+ f"身份:{row.identity or '不详'},在故事里是{row.role})"
+ ]
+ for e in events:
+ if row.fid not in e.figure_ids():
+ continue
+ refs = " · ".join(world.render_refs(e))
+ facts.append(f" {e.year}年 {e.type} {refs}")
+ if len(facts) > 5:
+ break
+ lines.extend(facts)
+ return "\n".join(lines)
+
+
+def _parse_bios(raw: str) -> dict[str, str]:
+ text = EDGE_RE.sub("", raw).strip()
+ start, end = text.find("{"), text.rfind("}")
+ if start < 0 or end <= start:
+ raise ValueError(f"小传没有返回 JSON:{raw[:200]}")
+ try:
+ data = json.loads(text[start:end + 1])
+ except json.JSONDecodeError as exc:
+ raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
+ if not isinstance(data, dict):
+ raise ValueError("小传返回的不是对象")
+ return {str(k): str(v) for k, v in data.items()}
+
+
+def build_cast(
+ world: World,
+ thread: Thread,
+ episodes: list[Episode],
+ *,
+ gateway: Gateway | None = None,
+ with_bios: bool = True,
+) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]:
+ """返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。"""
+ rows = collect_rows(world, thread, episodes)
+ events = [e for ep in episodes for e in ep.events] or thread.events
+
+ bios: dict[str, str] = {}
+ if with_bios and rows:
+ reply = chat(
+ [{"role": "user", "content": _bio_prompt(world, rows, events)}],
+ gateway=gateway,
+ max_tokens=MAX_TOKENS_STRUCTURED,
+ )
+ bios = _parse_bios(reply.content)
+
+ suspicions: list[factcheck.Suspicion] = []
+ bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items())
+ if bios_text:
+ suspicions = factcheck.check(bios_text, world)
+
+ lines = [
+ f"# {world.name} · 人物表",
+ "",
+ f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
+ f"线索跨度:{_span_of(thread, episodes)}",
+ "",
+ "| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |",
+ "|---|---|---|---|---|---|",
+ ]
+ lines.extend(row.as_row() for row in rows)
+
+ if bios:
+ lines += ["", "## 小传", ""]
+ for row in rows:
+ bio = bios.get(row.name) or bios.get(row.name.lower())
+ if bio:
+ lines.append(f"**{row.name}** —— {bio}")
+ lines.append("")
+
+ if suspicions:
+ lines += [
+ "",
+ "> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:"
+ + "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]),
+ "",
+ ]
+
+ return "\n".join(lines) + "\n", rows, suspicions
+
+
+def write_cast(path: Path, text: str) -> Path:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text(text, encoding="utf-8")
+ return path
diff --git a/dfannals/chronicle.py b/dfannals/chronicle.py
index db9181d..08844a0 100644
--- a/dfannals/chronicle.py
+++ b/dfannals/chronicle.py
@@ -62,8 +62,8 @@ class State:
def cjk_length(text: str) -> int:
"""按中文习惯计字数:CJK 字符与拉丁单词都算 1。"""
- without_code = re.sub(r"```.*?```", "", text, flags=re.S)
- without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.M)
+ without_code = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
+ without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.MULTILINE)
n = 0
for ch in without_meta:
if (
diff --git a/dfannals/cli.py b/dfannals/cli.py
index 56b0104..9556ec6 100644
--- a/dfannals/cli.py
+++ b/dfannals/cli.py
@@ -16,9 +16,26 @@ import argparse
import sys
from pathlib import Path
-from dfannals import chronicle, factcheck, legends, publish, timeline
+from dfannals import (
+ cast,
+ chronicle,
+ episodes,
+ factcheck,
+ legends,
+ publish,
+ threads,
+ timeline,
+)
from dfannals import slice as chapter_slice
-from dfannals.config import EXPORT_DIR, GITEA_WEB, chapter_dir, ensure_dirs, volume_dir
+from dfannals import volume as volume_mod
+from dfannals.config import (
+ DATA_DIR,
+ EXPORT_DIR,
+ GITEA_WEB,
+ chapter_dir,
+ ensure_dirs,
+ volume_dir,
+)
def find_exports(export_dir: Path) -> list[Path]:
@@ -142,6 +159,109 @@ def cmd_next(args: argparse.Namespace) -> int:
return 0
+def cmd_threads(args: argparse.Namespace) -> int:
+ ensure_dirs()
+ world = load_world(EXPORT_DIR)
+ found = threads.find_threads(
+ world,
+ min_interactions=args.min_interactions,
+ min_span=args.min_span,
+ limit=args.top,
+ )
+ print(threads.render_report(world, found, per_thread=args.samples))
+ path = threads.save_json(world, found, DATA_DIR / "threads.json")
+ print(f"共 {len(found)} 条候选;明细已写入 {path}")
+ return 0
+
+
+def cmd_episodes(args: argparse.Namespace) -> int:
+ ensure_dirs()
+ world = load_world(EXPORT_DIR)
+ found = threads.find_threads(
+ world, min_interactions=args.min_interactions, min_span=args.min_span
+ )
+ if not found:
+ print("没有候选线索")
+ return 1
+ if not 1 <= args.thread <= len(found):
+ print(f"候选只有 {len(found)} 条,--thread 超出范围")
+ return 1
+ thread = found[args.thread - 1]
+ _, planned = volume_mod.extended_plan(world, thread)
+ print(episodes.render_plan(world, thread, planned))
+ print()
+ print(episodes.budget_report(planned))
+ if args.samples:
+ print("\n=== 前两集事件样例 ===")
+ for ep in planned[:2]:
+ print(f" 第 {ep.index} 集({ep.span})")
+ for e in ep.events[: args.samples]:
+ print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
+ return 0
+
+
+def cmd_cast(args: argparse.Namespace) -> int:
+ ensure_dirs()
+ world = load_world(EXPORT_DIR)
+ found = threads.find_threads(
+ world, min_interactions=args.min_interactions, min_span=args.min_span
+ )
+ if not found:
+ print("没有候选线索")
+ return 1
+ if not 1 <= args.thread <= len(found):
+ print(f"候选只有 {len(found)} 条,--thread 超出范围")
+ return 1
+
+ thread = found[args.thread - 1]
+ _, planned = volume_mod.extended_plan(world, thread)
+ text, rows, suspicions = cast.build_cast(
+ world, thread, planned, with_bios=not args.no_bios
+ )
+ if args.dry_run:
+ print(text[:2000])
+ return 0
+
+ state = chronicle.State.load()
+ volume = state.volume if state.world == world.name else 1
+ path = cast.write_cast(volume_dir(volume) / "cast.md", text)
+ print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
+ print(factcheck.report(suspicions) if suspicions else "专名校验:小传里的专名全部能在史料中找到。")
+ return 0
+
+
+def cmd_episode(args: argparse.Namespace) -> int:
+ ensure_dirs()
+ world = load_world(EXPORT_DIR)
+ found = threads.find_threads(
+ world, min_interactions=args.min_interactions, min_span=args.min_span
+ )
+ if not found:
+ print("没有候选线索")
+ return 1
+
+ out_dir = None if args.dry_run else chapter_dir(1)
+ produced = volume_mod.produce_episode(
+ world, found, episode_index=args.index, out_dir=out_dir
+ )
+
+ print("选择依据:" + produced.selection.reason)
+ print(f"本集字数:{chronicle.cjk_length(produced.text)}")
+ if args.dry_run:
+ print("\n--- 试运行,不落盘 ---\n")
+ print(produced.text[:1600])
+ return 0
+
+ print(f"已写入 {produced.path}")
+ if not args.no_push:
+ publish.ensure_repo()
+ head = publish.commit_and_push(
+ f"第 1 卷第 {produced.episode.index} 集(人物主线):{produced.episode.span}"
+ )
+ print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
+ return 0
+
+
def cmd_run(args: argparse.Namespace) -> int:
rc = cmd_index(args)
if rc:
@@ -163,6 +283,38 @@ def main(argv: list[str] | None = None) -> int:
p = sub.add_parser("plan", help="显示切章方案")
p.set_defaults(func=cmd_plan)
+ p = sub.add_parser("threads", help="挖掘人物线索候选")
+ p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
+ help="强边阈值:一对人物至少反复互动多少次(默认 5)")
+ p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
+ help="线索最小跨度年数(默认 30)")
+ p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
+ p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
+ p.set_defaults(func=cmd_threads)
+
+ p = sub.add_parser("episodes", help="按人物线索切出剧集")
+ p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
+ p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
+ p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
+ p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
+ p.set_defaults(func=cmd_episodes)
+
+ p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
+ p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
+ p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
+ p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
+ p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
+ p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
+ p.set_defaults(func=cmd_cast)
+
+ p = sub.add_parser("episode", help="按人物主线产出下一集")
+ p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
+ p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
+ p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
+ p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
+ p.add_argument("--no-push", action="store_true", help="落盘但不推送")
+ p.set_defaults(func=cmd_episode)
+
p = sub.add_parser("next", help="生成下一章")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.set_defaults(func=cmd_next)
diff --git a/dfannals/config.py b/dfannals/config.py
index 25ae117..34e179a 100644
--- a/dfannals/config.py
+++ b/dfannals/config.py
@@ -47,6 +47,9 @@ MODEL = "cn:deepseek-v4-pro"
# 该模型是推理模型:思考与正文共用 max_tokens,必须留足
MAX_TOKENS_CHAPTER = 16000
MAX_TOKENS_UTIL = 4000
+# 凡是「喂长材料 + 要求结构化 JSON 输出」的调用(闸门自评、人物小传)都容易把预算
+# 烧在思考上而返回空正文,实测 4000 不够。
+MAX_TOKENS_STRUCTURED = 12000
# ---------------------------------------------------------------- 发布
GITEA_SSH_HOST = "124.222.29.26"
@@ -61,7 +64,7 @@ SSH_KEY = HOME / ".ssh/id_ed25519"
WORLD_SIZE = "Medium"
WORLD_HISTORY_YEARS = 250
CHAPTER_MIN_CHARS = 800
-CHAPTER_MAX_CHARS = 1600 # 契约上限 1500,留出标点/换行余量
+CHAPTER_MAX_CHARS = 1500 # 严格按访谈定的 800–1500,不留“余量”
@dataclass
diff --git a/dfannals/episodes.py b/dfannals/episodes.py
new file mode 100644
index 0000000..a2443b4
--- /dev/null
+++ b/dfannals/episodes.py
@@ -0,0 +1,192 @@
+"""因果链切集:把一条人物线索切成"一集一个完整回合"。
+
+参数来自访谈决策:
+
+- 相邻事件间隔 >2 年即断开(比推荐值更紧,节奏更快)
+- 短链只合并,不设线索总量下限
+- 链路过长按字数预算拆上下集
+- 每集正文预算落在 800–1500 字
+
+字数预算用「事件条数 × 每条展开字数」估算。系数 110 是按此前实测校准的:
+12 条史料事件生成的中文正文约 1100–1400 字。
+"""
+from __future__ import annotations
+
+import math
+from dataclasses import dataclass, field
+
+from dfannals.legends import Event, World
+from dfannals.threads import Thread
+
+# 相邻事件间隔超过它即断开
+GAP_YEARS = 2
+# 每条史料在正文里大致展开的字数。实测区间约 85–115(同样 12 条事件,
+# 两次生成分别得到 1026 字与 1388 字),取中值偏保守,避免把集拆得过碎。
+CHARS_PER_EVENT = 95
+MIN_CHARS = 800 # 每集预算下限
+MAX_CHARS = 1500 # 每集预算上限
+
+
+@dataclass
+class Episode:
+ index: int
+ start_year: int
+ end_year: int
+ events: list[Event] = field(default_factory=list)
+ part: int = 1
+ parts: int = 1
+
+ @property
+ def key(self) -> str:
+ base = f"{self.index:03d}-{self.start_year}-{self.end_year}"
+ return base if self.parts == 1 else f"{base}-p{self.part}"
+
+ @property
+ def span(self) -> str:
+ if self.start_year == self.end_year:
+ return f"{self.start_year} 年"
+ return f"{self.start_year}–{self.end_year} 年"
+
+ @property
+ def estimated_chars(self) -> int:
+ return len(self.events) * CHARS_PER_EVENT
+
+ def in_budget(self) -> bool:
+ return MIN_CHARS <= self.estimated_chars <= MAX_CHARS or self.parts > 1
+
+
+def _min_events() -> int:
+ return max(1, math.ceil(MIN_CHARS / CHARS_PER_EVENT))
+
+
+def _max_events() -> int:
+ return max(1, MAX_CHARS // CHARS_PER_EVENT)
+
+
+def _split_chains(events: list[Event], gap_years: int) -> list[list[Event]]:
+ """按时间间隔把事件串切成因果链。"""
+ chains: list[list[Event]] = []
+ current: list[Event] = []
+ prev_year: int | None = None
+ for e in events:
+ if current and prev_year is not None and e.year - prev_year > gap_years:
+ chains.append(current)
+ current = []
+ current.append(e)
+ prev_year = e.year
+ if current:
+ chains.append(current)
+ return chains
+
+
+def _merge_short(chains: list[list[Event]], min_events: int) -> list[list[Event]]:
+ """短链只合并:攒够预算就收一集,尾部不足的并入上一集。"""
+ merged: list[list[Event]] = []
+ buffer: list[Event] = []
+ for chain in chains:
+ buffer.extend(chain)
+ if len(buffer) >= min_events:
+ merged.append(buffer)
+ buffer = []
+ if buffer:
+ if merged:
+ merged[-1].extend(buffer)
+ else:
+ merged = [buffer]
+ return merged
+
+
+def _split_long(chain: list[Event]) -> list[list[Event]]:
+ """过长链路按预算拆成若干段(对应正文的上/下集)。
+
+ 注意不能均分:14 条事件均分成 7+7 会得到两个 770 字的碎片,
+ 两端都跌破 800 字下限。所以先取最小段数,再确保每段不低于下限。
+ """
+ n = len(chain)
+ max_events = _max_events()
+ min_events = _min_events()
+ if n <= max_events:
+ return [chain]
+ parts = math.ceil(n / max_events)
+ while parts > 1 and math.ceil(n / parts) < min_events:
+ parts -= 1
+ size = math.ceil(n / parts)
+ return [chain[i:i + size] for i in range(0, n, size)]
+
+
+def plan_episodes(
+ thread: Thread,
+ gap_years: int = GAP_YEARS,
+ min_chars: int = MIN_CHARS,
+ max_chars: int = MAX_CHARS,
+) -> list[Episode]:
+ """给定主角组,切出剧集序列。"""
+ if not thread.events:
+ return []
+
+ chains = _split_chains(thread.events, gap_years)
+ merged = _merge_short(chains, _min_events())
+
+ episodes: list[Episode] = []
+ for chain in merged:
+ segments = _split_long(chain)
+ for i, segment in enumerate(segments, 1):
+ episodes.append(
+ Episode(
+ index=len(episodes) + 1,
+ start_year=segment[0].year,
+ end_year=segment[-1].year,
+ events=segment,
+ part=i,
+ parts=len(segments),
+ )
+ )
+ return episodes
+
+
+def cast_of(world: World, thread: Thread, episode: Episode) -> tuple[list[int], list[int]]:
+ """返回 (本集出场的主角, 本集出现的对手/外人)。"""
+ counter: dict[int, int] = {}
+ for e in episode.events:
+ for fid in e.figure_ids():
+ counter[fid] = counter.get(fid, 0) + 1
+ members = set(thread.members)
+ heros = sorted((f for f in counter if f in members), key=lambda f: -counter[f])
+ others = sorted((f for f in counter if f not in members), key=lambda f: -counter[f])
+ return heros, others
+
+
+def render_plan(world: World, thread: Thread, episodes: list[Episode], top_others: int = 3) -> str:
+ """剧集清单:集号、年份范围、事件条数、主角与对手。"""
+ lines = [
+ f"线索主角:{'、'.join(world.figure_name(m) for m in thread.members)}",
+ f"线索跨度:{thread.label} | 事件 {len(thread.events)} 条 | 规划 {len(episodes)} 集",
+ "",
+ ]
+ for ep in episodes:
+ heros, others = cast_of(world, thread, ep)
+ hero_names = "、".join(world.figure_name(h) for h in heros[:4]) or "(本集无主角出场)"
+ other_names = "、".join(world.figure_name(o) for o in others[:top_others])
+ budget = f"{ep.estimated_chars} 字估算"
+ mark = "" if MIN_CHARS <= ep.estimated_chars <= MAX_CHARS else " ⚠ 超出预算"
+ lines.append(
+ f" 第 {ep.index:2d} 集({ep.span})| {len(ep.events):2d} 条事件 | {budget}{mark}"
+ + (f" | 上/下:{ep.part}/{ep.parts}" if ep.parts > 1 else "")
+ )
+ lines.append(f" 主角:{hero_names}")
+ if other_names:
+ lines.append(f" 对手/外人:{other_names}")
+ return "\n".join(lines)
+
+
+def budget_report(episodes: list[Episode]) -> str:
+ if not episodes:
+ return "无剧集"
+ inside = sum(1 for e in episodes if MIN_CHARS <= e.estimated_chars <= MAX_CHARS)
+ sizes = [len(e.events) for e in episodes]
+ return (
+ f"共 {len(episodes)} 集;事件条数 {min(sizes)}–{max(sizes)};"
+ f"估算字数 {min(e.estimated_chars for e in episodes)}–{max(e.estimated_chars for e in episodes)};"
+ f"落在 {MIN_CHARS}–{MAX_CHARS} 预算内的 {inside}/{len(episodes)} 集"
+ "(估算按每条约 95 字,真实字数在生成时会再校正)"
+ )
diff --git a/dfannals/gate.py b/dfannals/gate.py
new file mode 100644
index 0000000..15dcf77
--- /dev/null
+++ b/dfannals/gate.py
@@ -0,0 +1,223 @@
+"""5 维度自评闸门:判断一稿是不是"流水账",并决定重写还是换线索。
+
+访谈决策:
+- 5 个维度各 0–5 分,总分 <12 判平淡
+- 不合格先重写本集一次,仍不合格才换线索
+- 放弃原因写进仓库 notes/ 附录
+- 设失败上限,防止「重写→换线索」空转
+
+为什么要这道闸门:上一版章节读起来像流水账,而"平淡"是模型最容易自我感觉良好的东西。
+因此评分必须结构化、维度必须问得具体,且要有一次反例自检证明它真的能判不合格。
+"""
+from __future__ import annotations
+
+import json
+import re
+from collections.abc import Callable
+from dataclasses import dataclass, field
+from pathlib import Path
+
+from dfannals.config import MAX_TOKENS_STRUCTURED, MODEL, Gateway
+from dfannals.llm import chat
+
+# 维度:键、标签、问什么
+DIMENSIONS: tuple[tuple[str, str, str], ...] = (
+ ("goal", "主角目标", "主角在本集里有没有明确的目标,并为此做出主动选择?全程被动得分低"),
+ ("escalation", "冲突升级", "冲突是否逐级升级,而不是同一件事反复发生?"),
+ ("turning", "真正转折", "本集有没有真正的转折(立场改变、关系破裂、意外后果)?"),
+ ("rival", "对手具体", "对手是不是一个具体的人,有自己的动机?只是数字或人群得分低"),
+ ("beyond_bodycount", "超越战斗计数", "抛开对战次数,本集还有没有可讲的内容(算计、情感、抉择)?"),
+)
+
+THRESHOLD = 12 # 总分低于它判平淡
+MAX_SCORE = 5 # 每个维度满分
+EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
+
+# 失败上限:防止反复重写/换线空转
+MAX_REWRITES = 1 # 每集最多重写次数(访谈决定)
+MAX_CANDIDATES = 3 # 最多尝试几条线索
+
+
+@dataclass
+class Verdict:
+ total: int
+ scores: dict[str, int]
+ reason: str
+
+ @property
+ def passed(self) -> bool:
+ return self.total >= THRESHOLD
+
+ def weak(self, limit: int = 2) -> list[str]:
+ labels = {k: label for k, label, _ in DIMENSIONS}
+ ranked = sorted(self.scores.items(), key=lambda kv: kv[1])
+ return [f"{labels.get(k, k)}({v})" for k, v in ranked[:limit]]
+
+ def render(self) -> str:
+ parts = "、".join(f"{label}={self.scores.get(key, '?')}" for key, label, _ in DIMENSIONS)
+ verdict = "通过" if self.passed else "判平淡"
+ return f"{verdict} 总分 {self.total}/{len(DIMENSIONS) * MAX_SCORE}({parts})|{self.reason}"
+
+
+DIMENSION_BLOCK = "\n".join(
+ f'- "{key}": {hint}(0–5 分)' for key, _label, hint in DIMENSIONS
+)
+
+EVAL_INSTRUCTIONS = f"""你是严格的文学编辑,只做一件事:判断下面这集稿子是不是"流水账"。
+
+按 5 个维度各打 0–5 分:
+{DIMENSION_BLOCK}
+
+打分要求:
+- 不要客气,不要因为文笔通顺就给分;只看叙事本身有没有戏。
+- 主角全程被动、只是"发生了很多事",goal 与 escalation 必须给低分。
+- 通篇只是同一类事件重复(例如反复对战、反复被杀),turning 与 beyond_bodycount 必须给低分。
+
+只输出 JSON,不要解释、不要 Markdown 代码块:
+{{"goal": 0-5, "escalation": 0-5, "turning": 0-5, "rival": 0-5, "beyond_bodycount": 0-5, "reason": "一句话说明最弱的地方"}}
+"""
+
+
+def build_eval_prompt(chapter_text: str) -> str:
+ """拼接自评提示词。
+
+ 刻意不用 str.format:提示词里的 JSON 例子含大括号,用 format 会被当成占位符
+ (曾直接报 KeyError: '"goal"')。
+ """
+ return EVAL_INSTRUCTIONS + "\n待评稿件:\n---\n" + chapter_text + "\n---\n"
+
+
+def _parse(raw: str) -> dict:
+ text = EDGE_RE.sub("", raw).strip()
+ start, end = text.find("{"), text.rfind("}")
+ if start < 0 or end <= start:
+ raise ValueError(f"自评没有返回 JSON:{raw[:200]}")
+ try:
+ data = json.loads(text[start:end + 1])
+ except json.JSONDecodeError as exc:
+ raise ValueError(f"自评返回的 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
+ if not isinstance(data, dict):
+ raise ValueError(f"自评返回的不是对象:{type(data).__name__}")
+ return data
+
+
+def evaluate(
+ chapter_text: str,
+ *,
+ gateway: Gateway | None = None,
+ model: str = MODEL,
+) -> Verdict:
+ """让模型按 5 个维度打分。返回结构化结果。"""
+ reply = chat(
+ [{"role": "user", "content": build_eval_prompt(chapter_text)}],
+ gateway=gateway,
+ model=model,
+ max_tokens=MAX_TOKENS_STRUCTURED,
+ temperature=0.0,
+ )
+ data = _parse(reply.content)
+
+ scores: dict[str, int] = {}
+ for key, _label, _hint in DIMENSIONS:
+ value = data.get(key)
+ if not isinstance(value, (int, float)):
+ raise ValueError(f"自评维度 {key} 缺失或非数字:{data.get(key)!r}")
+ scores[key] = max(0, min(MAX_SCORE, int(value)))
+
+ reason = str(data.get("reason", "")).strip() or "(未给出理由)"
+ return Verdict(total=sum(scores.values()), scores=scores, reason=reason)
+
+
+@dataclass
+class GateRun:
+ """一次闸门流程的完整记录。"""
+ episode_key: str
+ verdicts: list[Verdict] = field(default_factory=list)
+ accepted: bool = False
+ rewrites: int = 0
+
+ @property
+ def final(self) -> Verdict | None:
+ return self.verdicts[-1] if self.verdicts else None
+
+ def render(self) -> str:
+ lines = [f"闸门:{self.episode_key},重写 {self.rewrites} 次,"
+ + ("通过" if self.accepted else "未通过")]
+ for i, v in enumerate(self.verdicts, 1):
+ lines.append(f" 第 {i} 稿:{v.render()}")
+ return "\n".join(lines)
+
+
+def gate_episode(
+ episode_key: str,
+ write: Callable[[int, Verdict | None], str],
+ *,
+ max_rewrites: int = MAX_REWRITES,
+ evaluate_fn: Callable[[str], Verdict] = evaluate,
+ on_event: Callable[[str], None] | None = None,
+) -> tuple[str, GateRun]:
+ """写 → 自评 → 不合格则重写一次 → 仍不合格则交给调用方换线索。
+
+ ``write(attempt, previous_verdict)``:attempt 为 0 表示初稿,
+ 大于 0 表示重写,previous_verdict 是上一稿的评分(供重写时指出问题)。
+ """
+ run = GateRun(episode_key=episode_key)
+ text = write(0, None)
+ verdict = evaluate_fn(text)
+ run.verdicts.append(verdict)
+
+ while not verdict.passed and run.rewrites < max_rewrites:
+ run.rewrites += 1
+ if on_event:
+ on_event(f"{episode_key} 第 {run.rewrites} 次重写:{verdict.weak()}")
+ text = write(run.rewrites, verdict)
+ verdict = evaluate_fn(text)
+ run.verdicts.append(verdict)
+
+ run.accepted = verdict.passed
+ return text, run
+
+
+NOTES_DIR_NAME = "notes"
+SWITCH_LOG = "switched-threads.md"
+
+
+def append_switch_note(
+ project_dir: Path,
+ *,
+ world_name: str,
+ members: list[str],
+ span: str,
+ verdict: Verdict | None,
+ reason: str,
+) -> Path:
+ """把"为什么放弃这条线索"写进仓库 notes/ 附录(访谈决定:公开可查)。"""
+ notes = project_dir / NOTES_DIR_NAME
+ notes.mkdir(parents=True, exist_ok=True)
+ path = notes / SWITCH_LOG
+ header = "# 被放弃的线索\n\n记录每一组挑出来、但写不成故事的线索,以及放弃原因。\n"
+ if not path.is_file():
+ path.write_text(header, encoding="utf-8")
+
+ score = verdict.render() if verdict else "(未评分)"
+ entry = (
+ f"\n## {world_name}|{'、'.join(members)}\n\n"
+ f"- 线索跨度:{span}\n"
+ f"- 放弃原因:{reason}\n"
+ f"- 自评:{score}\n"
+ )
+ with path.open("a", encoding="utf-8") as fh:
+ fh.write(entry)
+ return path
+
+
+def append_attempt_note(project_dir: Path, text: str) -> Path:
+ """一般性记录(例如失败上限触发)。"""
+ notes = project_dir / NOTES_DIR_NAME
+ notes.mkdir(parents=True, exist_ok=True)
+ path = notes / SWITCH_LOG
+ if not path.is_file():
+ path.write_text("# 被放弃的线索\n\n", encoding="utf-8")
+ with path.open("a", encoding="utf-8") as fh:
+ fh.write(text if text.startswith("\n") else "\n" + text)
+ return path
diff --git a/dfannals/legends.py b/dfannals/legends.py
index 117ce0c..bde8b25 100644
--- a/dfannals/legends.py
+++ b/dfannals/legends.py
@@ -14,26 +14,29 @@ from typing import Any
from defusedxml.ElementTree import iterparse
-# 事件里指向其他实体的字段 → 指向哪类对象
-ID_FIELDS: dict[str, str] = {
- "hfid": "figure",
- "hist_figure_id": "figure",
- "slayer_hfid": "figure",
- "slayer_item_id": "artifact",
- "target_hfid": "figure",
- "source_hfid": "figure",
- "site_id": "site",
- "site_civ_id": "entity",
- "entity_id": "entity",
- "attacker_civ_id": "entity",
- "defender_civ_id": "entity",
- "artifact_id": "artifact",
- "region_id": "region",
- "feature_layer_id": "region",
- "deity": "figure",
- "worshipper_hfid": "figure",
- "creature_id": "creature",
-}
+# 字段分类规则来自对真实史料全部 225 种字段的穷举:
+# 含 "hfid" 的字段全部指向历史人物(共 40+ 个:hfid / hfid_target / group_1_hfid /
+# slayer_hfid / snatcher_hfid / seeker_hfid / wounder_hfid / teacher_hfid /
+# conspirator_hfid ...);而 target_enid、identity_id、master_wcid、slayer_item_id 不是。
+# 旧版手工维护的字典只列了 4 个人物字段,导致关系信息在素材里被丢掉。
+_ENTITY_SUFFIXES = ("_entity_id", "_civ_id", "_enid")
+_REGION_FIELDS = {"region_id", "feature_layer_id", "subregion_id"}
+
+
+def kind_of(tag: str) -> str | None:
+ """事件字段名 → 它指向的对象类别(非引用类字段返回 None)。"""
+ t = tag.lower()
+ if "hfid" in t or t == "deity" or t.endswith("_deity"):
+ return "figure"
+ if "artifact" in t:
+ return "artifact"
+ if t == "site_id" or t.endswith("_site_id"):
+ return "site"
+ if t in ("entity_id", "target_enid") or t.endswith(_ENTITY_SUFFIXES):
+ return "entity"
+ if t in _REGION_FIELDS:
+ return "region"
+ return None
@dataclass
@@ -81,6 +84,36 @@ class Event:
vals = self.fields.get(tag)
return vals[0] if vals else ""
+ def ids_of_kind(self, kind: str) -> list[int]:
+ """本事件中指向该类对象的全部有效 id(去重、剔除 -1 占位)。"""
+ out: list[int] = []
+ for tag, vals in self.fields.items():
+ if kind_of(tag) != kind:
+ continue
+ for raw in vals:
+ i = _int_or_none(raw)
+ if i is not None and i >= 0 and i not in out:
+ out.append(i)
+ return out
+
+ def figure_ids(self) -> list[int]:
+ return self.ids_of_kind("figure")
+
+ def site_ids(self) -> list[int]:
+ return self.ids_of_kind("site")
+
+ def entity_ids(self) -> list[int]:
+ return self.ids_of_kind("entity")
+
+ def field_kinds(self) -> dict[str, list[int]]:
+ """{类别: [id, ...]},供渲染与建图使用。"""
+ out: dict[str, list[int]] = {}
+ for kind in ("figure", "site", "entity", "artifact", "region"):
+ ids = self.ids_of_kind(kind)
+ if ids:
+ out[kind] = ids
+ return out
+
@dataclass
class Entity:
@@ -138,6 +171,34 @@ class World:
st = self.sites.get(sid)
return pretty(st.name) if st and st.name else f"Site#{sid}"
+ def render_refs(self, event: Event) -> list[str]:
+ """把一条事件里的引用字段渲染成 ``tag=名字``。
+
+ 保留字段名是有意的:slayer_hfid=谁 与 hfid=谁 语义完全不同,
+ 丢掉字段名就等于丢掉了“谁对谁做了什么”。
+ 负 id(DF 的“无此项”占位)与查不到名字的字段直接跳过。
+ """
+ out: list[str] = []
+ for tag, vals in event.fields.items():
+ kind = kind_of(tag)
+ if kind is None:
+ continue
+ for raw in vals:
+ i = _int_or_none(raw)
+ if i is None or i < 0:
+ continue
+ if kind == "figure":
+ name = self.figure_name(i)
+ elif kind == "site":
+ name = self.site_name(i)
+ elif kind == "entity":
+ name = self.entity_name(i)
+ else:
+ name = ""
+ if name:
+ out.append(f"{tag}={name}")
+ return out
+
def event_years(self) -> tuple[int, int]:
years = [e.year for e in self.events if e.year]
return (min(years), max(years)) if years else (0, 0)
diff --git a/dfannals/publish.py b/dfannals/publish.py
index 3383c64..10e568e 100644
--- a/dfannals/publish.py
+++ b/dfannals/publish.py
@@ -32,6 +32,7 @@ def _git(*args: str, check: bool = True) -> subprocess.CompletedProcess:
env={**_base_env(), **env},
capture_output=True,
text=True,
+ check=False, # 下面是自定义错误处理,故意不用 check=True
)
if check and proc.returncode != 0:
raise RuntimeError(f"git {' '.join(args)} 失败:{proc.stderr.strip() or proc.stdout.strip()}")
diff --git a/dfannals/score.py b/dfannals/score.py
index e335dfa..128b0ae 100644
--- a/dfannals/score.py
+++ b/dfannals/score.py
@@ -93,22 +93,23 @@ def category(event: Event) -> str:
def prominence(world: World) -> Counter[int]:
- """每个人物被卷入的事件次数——用来衡量他在史料里有多重要。"""
+ """每个人物被卷入的事件次数——用来衡量他在史料里有多重要。
+
+ 改用字段分类器后覆盖了 40+ 个人物字段(旧版只数 6 个),数值尺度整体变大,
+ 因此 PROMINENT 阈值必须按新尺度重校。结果缓存在 World 上,避免反复重算。
+ """
+ cached = getattr(world, "_prominence", None)
+ if cached is not None:
+ return cached
counts: Counter[int] = Counter()
for e in world.events:
- seen: set[int] = set()
- for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity", "worshipper_hfid"):
- for fid in e.ids(tag):
- seen.add(fid)
- counts.update(seen)
+ counts.update(e.figure_ids())
+ world._prominence = counts # type: ignore[attr-defined]
return counts
def _participants(event: Event) -> list[int]:
- out: list[int] = []
- for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity"):
- out.extend(event.ids(tag))
- return out
+ return event.figure_ids()
# 人物重要度阈值:参与事件数达到这个量级才算"要角"
diff --git a/dfannals/slice.py b/dfannals/slice.py
index 1824ca9..8c11ee0 100644
--- a/dfannals/slice.py
+++ b/dfannals/slice.py
@@ -130,23 +130,10 @@ def material(world: World, chapter: Chapter, max_figures: int = 40) -> str:
participation: dict[int, int] = {}
for e in chapter.events:
- chunks = [f"{e.year}年", e.type]
- for tag, kind in lg.ID_FIELDS.items():
- for i in e.ids(tag):
- # id 为负是“无此项”,不要渲染成 HF#-1 这类噪音
- if kind == "figure":
- name = world.figure_name(i)
- elif kind == "site":
- name = world.site_name(i)
- elif kind == "entity":
- name = world.entity_name(i)
- else:
- name = ""
- if not name:
- continue
- chunks.append(f"{tag}={name}")
- if kind == "figure":
- participation[i] = participation.get(i, 0) + 1
+ # render_refs 会带出双方(比如 slayer_hfid=谁),而不是只剩单方面 hfid
+ chunks = [f"{e.year}年", e.type, *world.render_refs(e)]
+ for i in e.figure_ids():
+ participation[i] = participation.get(i, 0) + 1
extra = []
for field_name in ("state", "reason", "circumstance"):
v = e.text(field_name)
@@ -184,8 +171,10 @@ def overview(world: World, chapters: list[Chapter], limit: int = 10) -> str:
lo, hi = world.event_years()
lines = [
f"世界:{world.name}" + (f"({world.altname})" if world.altname else ""),
- f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
- f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}",
+ (
+ f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
+ f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}"
+ ),
f"规划章节:{len(chapters)} 章(每章约 {YEARS_PER_CHAPTER} 年 / 最多 {EVENTS_PER_CHAPTER} 条史料)",
]
for c in chapters[:limit]:
diff --git a/dfannals/threads.py b/dfannals/threads.py
new file mode 100644
index 0000000..b996f74
--- /dev/null
+++ b/dfannals/threads.py
@@ -0,0 +1,287 @@
+"""人物线索挖掘:从史料里找出「关系密切、有戏」的 3–6 人小圈子。
+
+设计依据全部来自对 Mon Sagus 真实数据(403853 条事件)的实测:
+
+1. **不能用裸互动次数排序。** 实测排名第一的簇 49 次互动里有 35 次是
+ ``hf relationship denied``——反复请求建立关系、反复被拒的循环,是统计噪声。
+ 因此整类剔除(访谈已定)。
+2. **同一对同一类型反复出现也不是故事。** 修掉上一条之后,候选又变成
+ ``对战×27`` 这种「同一对人反复互殴」的机械循环。因此在计数时对
+ (人物对, 事件类型) 做封顶,避免刷量取胜——这也正是访谈定的
+ 「转折/冲突多样性为主、互动次数为次」。
+3. **主线必须是"人"。** 种族分布实测:ELF/GOBLIN/DWARF/HUMAN/KOBOLD 五族加
+ ``*_MAN`` 人形族占全部历史人物的 96.5%,其余 400 多种是夜行怪、野兽、
+ 泰坦、实验体。所以这里用白名单,而不是越列越长的黑名单。
+4. 强边阈值 ≥5 次反复互动;簇规模 3–6 人;跨度 ≥30 年。
+ 实测:阈值提到 8 会一条候选都不剩,5 是正确档位。
+"""
+from __future__ import annotations
+
+import json
+from collections import Counter, defaultdict
+from dataclasses import dataclass, field
+from pathlib import Path
+
+from dfannals.legends import Event, World
+
+# 事件类型 → (权重, 中文标签, 是否算转折点)
+SIGNALS: dict[str, tuple[int, str, bool]] = {
+ "hf simple battle event": (1, "对战", False),
+ "add hf hf link": (1, "结缘", False),
+ "hfs formed reputation relationship": (1, "结缘", False),
+ "competition": (1, "竞争", False),
+ "hf wounded": (2, "搏杀", True),
+ "hf confronted": (2, "对峙", True),
+ "hf interrogated": (2, "审讯", True),
+ "failed intrigue corruption": (2, "阴谋", True),
+ "remove hf hf link": (3, "反目", True),
+ "hf abducted": (3, "绑架", True),
+ "hf convicted": (3, "定罪", True),
+ "entity persecuted": (3, "迫害", True),
+ "hf enslaved": (3, "奴役", True),
+ "hf ransomed": (3, "贖金", True),
+ "entity overthrown": (3, "推翻", True),
+ "failed frame attempt": (3, "构陷", True),
+}
+
+# 明确剔除:重复性请求,实测会刷满排序(访谈已定)
+EXCLUDED_SIGNALS = ("hf relationship denied",)
+
+# 文明种族白名单:实测覆盖 96.5% 的历史人物
+CIVILIZED_RACES = frozenset({"DWARF", "ELF", "HUMAN", "GOBLIN", "KOBOLD"})
+
+DEFAULT_MIN_INTERACTIONS = 5 # 强边阈值:≥5 次反复互动
+DEFAULT_SIZE_RANGE = (3, 6)
+DEFAULT_MIN_SPAN = 30 # 跨度 ≥30 年才撑得起连载
+PAIR_TYPE_CAP = 4 # 同一对、同一类型最多计 4 次
+MIN_SIGNAL_KINDS = 2 # 至少两种不同信号,否则只是单一类型的重复
+INTERACTION_CAP = 40 # 互动总量封顶,防止刷量取胜
+
+
+@dataclass
+class Edge:
+ a: int
+ b: int
+ raw: int = 0
+ types: Counter = field(default_factory=Counter)
+ first_year: int = 0
+ last_year: int = 0
+
+ @property
+ def effective(self) -> int:
+ """封顶后的有效互动次数。"""
+ return sum(min(n, PAIR_TYPE_CAP) for n in self.types.values())
+
+ @property
+ def turns(self) -> int:
+ """封顶后的转折点次数。"""
+ return sum(min(n, PAIR_TYPE_CAP) for t, n in self.types.items() if SIGNALS[t][2])
+
+
+@dataclass
+class Thread:
+ members: list[int]
+ interactions: int
+ turns: int
+ span: int
+ first_year: int
+ last_year: int
+ types: Counter
+ races: Counter
+ non_person_races: list[str]
+ events: list[Event]
+
+ @property
+ def kinds(self) -> int:
+ return len(self.types)
+
+ @property
+ def score(self) -> int:
+ """转折为主、多样性次之、互动次数封顶计入(访谈定的排序原则)。"""
+ return self.turns * 10 + self.kinds * 6 + min(self.interactions, INTERACTION_CAP)
+
+ @property
+ def label(self) -> str:
+ return f"{self.first_year}–{self.last_year}({self.span} 年)"
+
+ def signal_line(self) -> str:
+ return "、".join(f"{SIGNALS[t][1]}×{n}" for t, n in self.types.most_common() if t in SIGNALS)
+
+
+def is_person(race: str) -> bool:
+ """是否属于"可作为主角的人":文明种族或人形族(*_MAN)。"""
+ r = (race or "").upper()
+ return r in CIVILIZED_RACES or r.endswith("_MAN")
+
+
+def build_edges(world: World) -> dict[tuple[int, int], Edge]:
+ """按有叙事含义的信号建立人物之间的边。"""
+ edges: dict[tuple[int, int], Edge] = {}
+ for e in world.events:
+ if e.type not in SIGNALS:
+ continue
+ people = e.figure_ids()
+ if len(people) < 2:
+ continue
+ for i in range(len(people)):
+ for j in range(i + 1, len(people)):
+ key = (min(people[i], people[j]), max(people[i], people[j]))
+ edge = edges.get(key)
+ if edge is None:
+ edge = Edge(a=key[0], b=key[1], first_year=e.year, last_year=e.year)
+ edges[key] = edge
+ edge.raw += 1
+ edge.types[e.type] += 1
+ edge.first_year = min(edge.first_year, e.year)
+ edge.last_year = max(edge.last_year, e.year)
+ return edges
+
+
+def find_threads(
+ world: World,
+ min_interactions: int = DEFAULT_MIN_INTERACTIONS,
+ size_range: tuple[int, int] = DEFAULT_SIZE_RANGE,
+ min_span: int = DEFAULT_MIN_SPAN,
+ require_people: bool = True,
+ min_kinds: int = MIN_SIGNAL_KINDS,
+ limit: int = 0,
+) -> list[Thread]:
+ """返回按剧情张力排序的候选线索。"""
+ edges = build_edges(world)
+ strong = {k: v for k, v in edges.items() if v.effective >= min_interactions}
+
+ parent: dict[int, int] = {}
+
+ def find(x: int) -> int:
+ parent.setdefault(x, x)
+ while parent[x] != x:
+ parent[x] = parent[parent[x]]
+ x = parent[x]
+ return x
+
+ for (a, b) in strong:
+ ra, rb = find(a), find(b)
+ if ra != rb:
+ parent[ra] = rb
+
+ members_of: dict[int, set[int]] = defaultdict(set)
+ for x in parent:
+ members_of[find(x)].add(x)
+
+ events_by_pair: dict[tuple[int, int], list[Event]] = defaultdict(list)
+ for e in world.events:
+ if e.type not in SIGNALS:
+ continue
+ people = e.figure_ids()
+ if len(people) < 2:
+ continue
+ for i in range(len(people)):
+ for j in range(i + 1, len(people)):
+ events_by_pair[(min(people[i], people[j]), max(people[i], people[j]))].append(e)
+
+ lo_size, hi_size = size_range
+ threads: list[Thread] = []
+ for members in members_of.values():
+ if not (lo_size <= len(members) <= hi_size):
+ continue
+
+ races = Counter(world.figures[m].race for m in members if m in world.figures)
+ if require_people and not all(is_person(r) for r in races):
+ continue
+
+ inner = {k: v for k, v in strong.items() if k[0] in members and k[1] in members}
+ if not inner:
+ continue
+
+ types: Counter = Counter()
+ for v in inner.values():
+ types.update({t: min(n, PAIR_TYPE_CAP) for t, n in v.types.items()})
+ if len(types) < min_kinds:
+ continue
+
+ first = min(v.first_year for v in inner.values())
+ last = max(v.last_year for v in inner.values())
+ if last - first < min_span:
+ continue
+
+ evs: list[Event] = []
+ seen_ids: set[int] = set()
+ for k in inner:
+ for e in events_by_pair.get(k, []):
+ if e.id not in seen_ids:
+ seen_ids.add(e.id)
+ evs.append(e)
+ evs.sort(key=lambda e: (e.year, e.seconds72, e.id))
+
+ threads.append(
+ Thread(
+ members=sorted(members),
+ interactions=sum(v.effective for v in inner.values()),
+ turns=sum(v.turns for v in inner.values()),
+ span=last - first,
+ first_year=first,
+ last_year=last,
+ types=types,
+ races=races,
+ non_person_races=sorted(r for r in races if not is_person(r)),
+ events=evs,
+ )
+ )
+
+ threads.sort(key=lambda t: (-t.score, -t.span))
+ return threads[:limit] if limit else threads
+
+
+def render_report(world: World, threads: list[Thread], per_thread: int = 3) -> str:
+ """给人看的候选明细。"""
+ lines = [
+ f"候选线索 {len(threads)} 条(转折为主、多样性次之;同一对同类事件已封顶)",
+ "筛选:强边 ≥5 次互动、3–6 人、跨度 ≥30 年、含巨兽的簇已剔除",
+ "",
+ ]
+ for i, t in enumerate(threads, 1):
+ races = "、".join(f"{r}×{n}" for r, n in t.races.most_common())
+ monsters = "、".join(t.non_person_races) if t.non_person_races else "无"
+ lines.append(
+ f"{i:2d}. 得分 {t.score:4d} | {len(t.members)} 人 | 有效互动 {t.interactions:3d} | "
+ f"转折 {t.turns:2d} | 信号 {t.kinds} 种 | 跨度 {t.label} | 巨兽:{monsters}"
+ )
+ lines.append(f" 种族:{races}")
+ lines.append(f" 信号:{t.signal_line()}")
+ lines.append(f" 成员:{'、'.join(world.figure_name(m) for m in t.members)}")
+ for e in t.events[:per_thread]:
+ lines.append(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
+ lines.append("")
+ return "\n".join(lines)
+
+
+def to_json(world: World, threads: list[Thread]) -> dict:
+ return {
+ "world": world.name,
+ "candidates": [
+ {
+ "rank": i,
+ "score": t.score,
+ "members": [
+ {"id": m, "name": world.figure_name(m),
+ "race": world.figures[m].race if m in world.figures else ""}
+ for m in t.members
+ ],
+ "interactions": t.interactions,
+ "turns": t.turns,
+ "kinds": t.kinds,
+ "span": t.span,
+ "first_year": t.first_year,
+ "last_year": t.last_year,
+ "signals": {SIGNALS[k][1]: v for k, v in t.types.items() if k in SIGNALS},
+ "event_ids": [e.id for e in t.events],
+ }
+ for i, t in enumerate(threads, 1)
+ ],
+ }
+
+
+def save_json(world: World, threads: list[Thread], path: Path) -> Path:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text(json.dumps(to_json(world, threads), ensure_ascii=False, indent=2), encoding="utf-8")
+ return path
diff --git a/dfannals/timeline.py b/dfannals/timeline.py
index d0e4c1c..e2134fc 100644
--- a/dfannals/timeline.py
+++ b/dfannals/timeline.py
@@ -19,19 +19,7 @@ category = score.category
def event_line(world: World, e: Event) -> str:
"""把一条事件压成一行可读文字。"""
- chunks: list[str] = [e.type]
- for tag, kind in lg.ID_FIELDS.items():
- for i in e.ids(tag):
- if kind == "figure":
- name = world.figure_name(i)
- elif kind == "site":
- name = world.site_name(i)
- elif kind == "entity":
- name = world.entity_name(i)
- else:
- name = ""
- if name:
- chunks.append(f"{tag}={name}")
+ chunks: list[str] = [e.type, *world.render_refs(e)]
return f"- **{e.year}** [{category(e)}] " + " · ".join(chunks)
@@ -51,8 +39,10 @@ def build_timeline(world: World, min_score: int = 70, per_decade: int = 14) -> s
f"- 事件总数:{len(world.events)},其中重要事件 {len(important)}",
f"- 历史人物 {len(world.figures)} 位 · 文明与组织 {len(world.entities)} 个 · 地点 {len(world.sites)} 处",
"",
- "> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
- "完整数据留在本地,不入库。",
+ (
+ "> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
+ "完整数据留在本地,不入库。"
+ ),
"",
]
for decade in sorted(buckets):
@@ -72,13 +62,7 @@ def build_figures(world: World, top: int = 120) -> str:
"""人物索引:按参与事件数排序,给出身份与生卒。"""
involvement: Counter[int] = Counter()
for e in world.events:
- seen: set[int] = set()
- for tag, kind in lg.ID_FIELDS.items():
- if kind != "figure":
- continue
- for i in e.ids(tag):
- seen.add(i)
- involvement.update(seen)
+ involvement.update(e.figure_ids())
lines = [
f"# {world.name} · 人物索引",
diff --git a/dfannals/volume.py b/dfannals/volume.py
new file mode 100644
index 0000000..471ea2f
--- /dev/null
+++ b/dfannals/volume.py
@@ -0,0 +1,336 @@
+"""人物主线写作与编排:选主角组 → 切集 → 贴着人物写 → 过闸门 → 落盘。
+
+与旧的 chronicle.py 的区别:那里的主语是"年代",一章按 10 年窗口罗列事件;
+这里的主语是"人",一集是主角线上一段完整回合,世界大事只在影响到主角时提及。
+"""
+from __future__ import annotations
+
+import json
+from collections import Counter
+from collections.abc import Callable
+from dataclasses import dataclass, replace
+from pathlib import Path
+
+from dfannals import episodes as ep_mod
+from dfannals import gate
+from dfannals.cast import collect_rows
+from dfannals.chronicle import cjk_length
+from dfannals.config import (
+ CHAPTER_MAX_CHARS,
+ CHAPTER_MIN_CHARS,
+ MAX_TOKENS_CHAPTER,
+ PROJECT_DIR,
+ PROMPT_DIR,
+)
+from dfannals.legends import Event, World
+from dfannals.llm import chat
+from dfannals.threads import SIGNALS, Thread
+
+SPEC_FILE = PROMPT_DIR / "biographer.md"
+TAIL_CHARS = 600
+SELECTION_FILE = PROJECT_DIR / "data" / "selection.json"
+
+
+@dataclass
+class Selection:
+ rank: int
+ thread: Thread
+ reason: str
+
+
+@dataclass
+class Produced:
+ selection: Selection
+ episode: ep_mod.Episode
+ text: str
+ run: gate.GateRun
+ path: Path | None = None
+
+
+@dataclass
+class Extended:
+ """主角参与的扩展素材。"""
+ events: list[Event]
+ elided: Counter
+
+
+PER_KEY_CAP = 3 # 同一组人 + 同一类型,每集素材最多保留几次
+
+
+def expand_thread_events(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP) -> Extended:
+ """把素材从“成员之间”扩到“主角参与的全部有意义事件”。
+
+ 实测:某 3 人线索成员之间只有 14 条事件(只能切 1 集),但把他们的对外活动
+ (构陷、对战、结义、定罪……)算进来有 136 条,足够支撑一卷。
+
+ 代价是重复:这 136 条里有 64 条是同一类“构陷失败”。所以同一组人 + 同一类型
+ 最多保留 per_key_cap 条,其余计入 elided 供写作时一笔带过——否则又会变流水账。
+ """
+ members = set(thread.members)
+ seen: Counter = Counter()
+ keep: list[Event] = []
+ elided: Counter = Counter()
+
+ relevant = [
+ e for e in world.events
+ if e.type in SIGNALS and (members & set(e.figure_ids()))
+ ]
+ relevant.sort(key=lambda e: (e.year, e.seconds72, e.id))
+
+ for e in relevant:
+ key = (e.type, tuple(sorted(e.figure_ids())))
+ seen[key] += 1
+ if seen[key] <= per_key_cap:
+ keep.append(e)
+ else:
+ elided[e.type] += 1
+ return Extended(events=keep, elided=elided)
+
+
+def extended_plan(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP
+ ) -> tuple[Extended, list[ep_mod.Episode]]:
+ """用扩展素材切集(保持切集规则不变)。"""
+ ext = expand_thread_events(world, thread, per_key_cap=per_key_cap)
+ planned = ep_mod.plan_episodes(replace(thread, events=ext.events))
+ return ext, planned
+
+
+def select_thread(world: World, candidates: list[Thread], rank: int = 1) -> Selection:
+ """按挖掘器排序取第 rank 条,并把选择依据写成可读理由。"""
+ thread = candidates[rank - 1]
+ names = "、".join(world.figure_name(m) for m in thread.members)
+ reason = (
+ f"挖掘器排序第 {rank} 名(转折为主、互动封顶):"
+ f"{len(thread.members)} 人、有效互动 {thread.interactions}、转折 {thread.turns}、"
+ f"信号 {thread.kinds} 种({thread.signal_line()})、跨度 {thread.label};成员:{names}"
+ )
+ return Selection(rank=rank, thread=thread, reason=reason)
+
+
+def save_selection(selection: Selection, world: World, path: Path = SELECTION_FILE) -> Path:
+ """把主角组的选定依据落到本地(不入库,供追溯)。"""
+ path.parent.mkdir(parents=True, exist_ok=True)
+ thread = selection.thread
+ path.write_text(
+ json.dumps(
+ {
+ "world": world.name,
+ "rank": selection.rank,
+ "reason": selection.reason,
+ "score": thread.score,
+ "members": [
+ {"id": m, "name": world.figure_name(m),
+ "race": world.figures[m].race if m in world.figures else ""}
+ for m in thread.members
+ ],
+ "signals": dict(thread.types),
+ "span": thread.label,
+ },
+ ensure_ascii=False,
+ indent=2,
+ ),
+ encoding="utf-8",
+ )
+ return path
+
+
+def render_material(
+ world: World,
+ thread: Thread,
+ episode: ep_mod.Episode,
+ total_episodes: int,
+ prev_tail: str = "",
+ elided: Counter | None = None,
+) -> str:
+ """把一集素材渲染给传记作者:主角是谁、发生了什么、对手是谁。"""
+ rows = {r.fid: r for r in collect_rows(world, thread, [episode], max_others=6)}
+ heros, others = ep_mod.cast_of(world, thread, episode)
+
+ lines = [
+ f"## 本集:第 {episode.index}/{total_episodes} 集,{episode.span}",
+ "",
+ "### 主角(本集要写的就是他们)",
+ "",
+ ]
+ for fid in heros:
+ row = rows.get(fid)
+ if row:
+ lines.append(f"- {row.name}|{row.race}|{row.span}|身份:{row.identity or '不详'}"
+ f"|本集出场 {row.appearances} 次")
+ else:
+ lines.append(f"- {world.figure_name(fid)}")
+
+ lines += ["", "### 本集史料(按时间排序;专名保留游戏原文)", ""]
+ for e in episode.events:
+ refs = " · ".join(world.render_refs(e))
+ extra = [f"{k}={e.text(k)}" for k in ("state", "reason", "circumstance")
+ if e.text(k) and e.text(k) not in ("-1", "")]
+ tail = ("|" + ";".join(extra)) if extra else ""
+ lines.append(f"- {e.year}年 {e.type} · {refs}{tail}")
+
+ if others:
+ lines += ["", "### 对手/相关者(史料里出现,但不是本卷主角)", ""]
+ for fid in others[:6]:
+ row = rows.get(fid)
+ if row:
+ lines.append(f"- {row.name}|{row.race}|身份:{row.identity or '不详'}"
+ f"|本集出场 {row.appearances} 次")
+ else:
+ lines.append(f"- {world.figure_name(fid)}")
+
+ if elided:
+ hot = "、".join(f"{t}×{n}" for t, n in elided.most_common(6))
+ lines += [
+ "",
+ "### 被省略的同类重复事件",
+ "",
+ f"- 本卷还有这些同类事件已被过滤,不要逐条罗列,需要时用一句话带过:{hot}",
+ ]
+
+ if prev_tail:
+ lines += ["", "### 上一集结尾(只用于承接状态,不要复述)", "", prev_tail]
+
+ lines += [
+ "",
+ "### 写作要求",
+ "",
+ f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,专名保留英文。",
+ "- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。",
+ "- 史料之外的世界大事不要写;只有影响到主角时才提一句。",
+ ]
+ return "\n".join(lines)
+
+
+def build_messages(
+ world: World,
+ thread: Thread,
+ episode: ep_mod.Episode,
+ total_episodes: int,
+ prev_tail: str,
+ feedback: str | None = None,
+ elided: Counter | None = None,
+) -> list[dict]:
+ spec = SPEC_FILE.read_text(encoding="utf-8")
+ material = render_material(world, thread, episode, total_episodes, prev_tail, elided)
+ user = [material]
+ if feedback:
+ user += ["", "### 上一稿的问题(重写时必须针对这些改)", "", feedback]
+ return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
+
+
+def _repair(messages: list[dict], draft: str, length: int, gateway) -> str:
+ target = (
+ f"内容太短({length} 字),请扩写到 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,"
+ "补充场景与对白,不要注水。"
+ if length < CHAPTER_MIN_CHARS
+ else f"内容太长({length} 字),请压缩到 {CHAPTER_MAX_CHARS} 字以内,删次要枝节。"
+ )
+ reply = chat(messages + [{"role": "assistant", "content": draft},
+ {"role": "user", "content": target}], gateway=gateway)
+ return reply.content
+
+
+def write_episode(
+ world: World,
+ thread: Thread,
+ episode: ep_mod.Episode,
+ total_episodes: int,
+ prev_tail: str = "",
+ feedback: str | None = None,
+ *,
+ gateway=None,
+ max_repairs: int = 2,
+ elided: Counter | None = None,
+) -> str:
+ """生成一集正文,并把长度校正到 800–1500 字。"""
+ messages = build_messages(world, thread, episode, total_episodes, prev_tail, feedback, elided)
+ text = chat(messages, gateway=gateway, max_tokens=MAX_TOKENS_CHAPTER).content
+ for _ in range(max_repairs):
+ length = cjk_length(text)
+ if CHAPTER_MIN_CHARS <= length <= CHAPTER_MAX_CHARS:
+ break
+ text = _repair(messages, text, length, gateway)
+ return text.strip()
+
+
+def write_file(directory: Path, episode: ep_mod.Episode, text: str) -> Path:
+ directory.mkdir(parents=True, exist_ok=True)
+ path = directory / f"{episode.key}.md"
+ path.write_text(text.strip() + "\n", encoding="utf-8")
+ return path
+
+
+def prev_tail_of(directory: Path, planned: list[ep_mod.Episode], index: int) -> str:
+ """取上一集结尾作为承接线索。"""
+ if index <= 1:
+ return ""
+ path = directory / f"{planned[index - 2].key}.md"
+ if not path.is_file():
+ return ""
+ return path.read_text(encoding="utf-8").strip()[-TAIL_CHARS:]
+
+
+def produce_episode(
+ world: World,
+ candidates: list[Thread],
+ *,
+ episode_index: int = 1,
+ out_dir: Path | None = None,
+ gateway=None,
+ max_candidates: int = gate.MAX_CANDIDATES,
+ on_event: Callable[[str], None] = print,
+) -> Produced:
+ """选线索 → 写这一集 → 过闸门;不过就换下一线索(有上限)。"""
+ if not candidates:
+ raise RuntimeError("没有候选线索可用")
+
+ for rank in range(1, min(max_candidates, len(candidates)) + 1):
+ selection = select_thread(world, candidates, rank=rank)
+ thread = selection.thread
+ ext, planned = extended_plan(world, thread)
+ if episode_index > len(planned):
+ on_event(f"第 {rank} 条线索只有 {len(planned)} 集,跳过")
+ continue
+
+ episode = planned[episode_index - 1]
+ directory = out_dir or (PROJECT_DIR / "output" / "volumes" / "v1" / "chapters")
+ tail = prev_tail_of(directory, planned, episode_index)
+ on_event(f"选定线索 {rank}:{'、'.join(world.figure_name(m) for m in thread.members)}")
+
+ # 把循环变量绑定成默认参数:否则闭包会绑定到循环变量本身(后面还会变)
+ def write(
+ attempt: int,
+ prev: gate.Verdict | None,
+ _thread: Thread = thread,
+ _episode: ep_mod.Episode = episode,
+ _total: int = len(planned),
+ _tail: str = tail,
+ _elided: Counter = ext.elided,
+ ) -> str:
+ feedback = None
+ if prev is not None:
+ feedback = (f"上一稿被判平淡:{prev.render()}。"
+ f"最弱环节是 {'、'.join(prev.weak())},请围绕这些重写,"
+ "让人物真正做选择。")
+ return write_episode(world, _thread, _episode, _total, _tail, feedback,
+ gateway=gateway, elided=_elided)
+
+ text, run = gate.gate_episode(episode.key, write, on_event=on_event)
+ on_event(run.render())
+
+ if run.accepted:
+ save_selection(selection, world)
+ path = write_file(directory, episode, text) if out_dir is not None else None
+ return Produced(selection=selection, episode=episode, text=text, run=run, path=path)
+
+ gate.append_switch_note(
+ PROJECT_DIR,
+ world_name=world.name,
+ members=[world.figure_name(m) for m in thread.members],
+ span=thread.label,
+ verdict=run.final,
+ reason=f"重写 {run.rewrites} 次后仍未过闸门(第 {rank} 条线索)",
+ )
+ on_event(f"线索 {rank} 未过闸门,换下一条")
+
+ raise RuntimeError(f"前 {max_candidates} 条线索都没过闸门,需要人工干预")
diff --git a/output/volumes/v1/cast.md b/output/volumes/v1/cast.md
new file mode 100644
index 0000000..fab3d10
--- /dev/null
+++ b/output/volumes/v1/cast.md
@@ -0,0 +1,37 @@
+# Mon Sagus · 人物表
+
+本卷主角线索:Guspu Frillyknots、Ral Fastenhatchets、Alath Blottedmine
+线索跨度:193–239(46 年)
+
+| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |
+|---|---|---|---|---|---|
+| Guspu Frillyknots | EAGLE_MAN | 151–246 | 关系对象×13、构陷发起者×4 | 62 | 主角 |
+| Ral Fastenhatchets | DWARF | 164–在世/不详 | 被针对者×21、受骗者×6 | 28 | 主角 |
+| Alath Blottedmine | DWARF | 182–在世/不详 | 构陷发起者×4、关系对象×1 | 8 | 主角 |
+| Tulon Tongswall | DWARF | 132–在世/不详 | 构陷发起者×3 | 3 | 对手/配角 |
+| Moldath Mirroredfloors | DWARF | 79–在世/不详 | 构陷发起者×3 | 3 | 对手/配角 |
+| Ineth Wheeleddrinks | DWARF | 165–在世/不详 | 构陷者×3 | 3 | 对手/配角 |
+| Langgud Tradebears The Eviscerated Chill Of Pregnancies | YETI | 生卒不详 | — | 2 | 对手/配角 |
+| Rigoth Stilledearthen | DWARF | 162–在世/不详 | 当事者×2 | 2 | 对手/配角 |
+| Perom Crazeowned | HUMAN | 129–在世/不详 | 当事者×1、关系对象×1 | 2 | 对手/配角 |
+
+## 小传
+
+**Guspu Frillyknots** —— 鹰人 Guspu Frillyknots(151–246)本卷出场六十二次,关系对象达十三位;193年对 Ral Fastenhatchets 的腐化失败后,次年被 The Volcano of Ravens 定罪。
+
+**Ral Fastenhatchets** —— Ral Fastenhatchets 于181年与 Langgud Tradebears The Eviscerated Chill Of Pregnancies 在 Fordedwinds 两度交手,此后长期被当作腐化与栽赃的靶子;本卷所载针对他的阴谋均告失败,但他仍留下六次受骗记录。
+
+**Alath Blottedmine** —— Alath Blottedmine 于229年与 Sina Watchedleads 建立关系,同年对 Ral Fastenhatchets 腐化失败;230年又与 Artuk Huggedfastened 建立关系,并在 Paperhushed 对 Momuz Decentpage 下手未遂,随即被 The Volcano of Ravens 定罪。
+
+**Tulon Tongswall** —— Tulon Tongswall 在193、195、197年三次于 Fordedwinds 试图腐化 Ral Fastenhatchets,结果三次全部失败,除此之外本卷无话可说。
+
+**Moldath Mirroredfloors** —— Moldath Mirroredfloors 在233年两度、234年一度于 Fordedwinds 对 Ral Fastenhatchets 下手,三次腐化全告失败。
+
+**Ineth Wheeleddrinks** —— Ineth Wheeleddrinks 在245、247、248年三次构陷 Ral Fastenhatchets,三次均是 failed frame attempt,且 fooled_hfid 三次都写成 Ral Fastenhatchets。
+
+**Langgud Tradebears The Eviscerated Chill Of Pregnancies** —— Langgud Tradebears The Eviscerated Chill Of Pregnancies 在181年与 Ral Fastenhatchets 在 Fordedwinds 打过两场,此后本卷再无其他事迹。
+
+**Rigoth Stilledearthen** —— Rigoth Stilledearthen 于188年与 Ral Fastenhatchets 建立联系,219年又解除该联系;两度出场仅为此事。
+
+**Perom Crazeowned** —— Perom Crazeowned 于193年成为 Guspu Frillyknots 的关系对象,202年又被 Guspu 反向列为关系对象,戏份仅此而已。
+
diff --git a/output/volumes/v1/chapters/001-181-198.md b/output/volumes/v1/chapters/001-181-198.md
new file mode 100644
index 0000000..a4f87d3
--- /dev/null
+++ b/output/volumes/v1/chapters/001-181-198.md
@@ -0,0 +1,13 @@
+# 钉子
+
+181年,Ral Fastenhatchets 十七岁,在 Fordedwinds 渡口两次挡住 Langgud Tradebears The Eviscerated Chill Of Pregnancies 的商队。那 Yeti 不肯缴渡税,用冰牙敲打栏木。Ral 拎着锤子站在桥心,肋骨断了两根,左耳打豁,但对方到底没过桥。他学到一件事:想按规矩活着,就得让人知道动你的代价。
+
+之后十年,他留在 Fordedwinds 管货栈的锁和账。188年,Rigoth Stilledearthen 找上他,说需要个不经手银钱、又不怕撕破脸的人。Ral 问给多少。Rigoth 说不是给,是共担。两人结伙,Ral 卡住渡口和货栈的关节。同年,Guspu Frillyknots 与 Momuz Fateddie 搭上线。Guspu 是 Eagle Man,翅膀半秃,在 Fordedwinds 做掮客。他早看 Ral 不顺眼——货栈的锁由 Ral 把着,不点头,他的走私道就多绕一天路。
+
+193年,Guspu 找 Ral 喝酒,递上一小袋金砂,说货栈每月闭一眼,大家都能睡好。Ral 把袋子推回去:“我睡得很好。你让 Momuz 把炭价降回三成,我请你喝真正的酒。”Guspu 当夜就派 Tulon Tongswall 去试。Tulon 是矮人,替 Guspu 伪造 Ral 的秤砣,想栽他私吞税粮。Ral 没揭穿,只在货栈门口摆了一杆新秤,当众称了自己管账以来每一笔过路货,误差不到半两。秤砣成了废铁。Guspu 第一次失手。
+
+他没停,又拉拢 Ongu Gleamcircled 和 Perom Crazeowned,想找证人指认 Ral 收私钱。Ral 白天让人排货,夜里把钥匙挂在脖子上睡,两次构陷都扑了空。这时 Ral 做了让自己更睡不着觉的决定:去找 The Volcano Of Ravens 告发 Guspu。他交出账本和伪造秤砣的残片。194年,Guspu 被定罪。
+
+Ral 以为到此为止。他低估了 Guspu。定罪后,Guspu 缩进阴影,195年两次派人向 Ral 递话,一次送钱,一次送刀,都失败。Tulon 也没走远,想把酒掺进税油里烧掉货栈。Ral 在火起前把 Tulon 按进河道,让他喝够自己的酒。197年,Tulon 最后一次伸手,把伪造密信塞进 Ral 的货箱,信没送到火山手里,因为 Ral 截住了送信的小子。他没杀 Tulon,只把信塞回对方腰带:“下一次,我让火山把你吞了。”
+
+198年,Astri Flayerline 来了。他是 Aardvark Man,长鼻嗅得着地下仓库的湿气。他替 Guspu 跑腿,两次想从账里找出破绽,都失败。前后四十七次构陷,像雨点打在铁皮上,Ral 的名字在 Fordedwinds 暗语里成了“那颗不生锈的钉子”。Rigoth 劝他离开,说钉子再硬,也钉不住一直浇醋的墙。Ral 没说话,只把锤子别回腰上。他知道 Guspu 还没完,Momuz、Ongu、Perom 都在等他先低头。
diff --git a/output/volumes/v1/chapters/002-202-216.md b/output/volumes/v1/chapters/002-202-216.md
new file mode 100644
index 0000000..83e09d9
--- /dev/null
+++ b/output/volumes/v1/chapters/002-202-216.md
@@ -0,0 +1,21 @@
+# 墙上的裂痕
+
+202年春天,Guspu Frillyknots 从定罪后的第八年阴影里走出来,做了一件谁都没料到的事:他去找 Perom Crazeowned。没有带刀,只带了一小袋税油票证和半张火山的罚单。Perom 是当年他出事时最早缩手的人之一,这些年一直在等他先开口。Guspu 把东西推过桌子,说:“我输了八年,不是没想过来硬的。但现在我要的不是一口气,是 Fordedwinds 的东门重新打开。”Perom 看着票证,点了头。他不是信 Guspu 的人,是信 Guspu 的恨。
+
+同月,Espu Bonestands 找上门。这个 Goat Mountain Man 带来山区的新货道,能绕开火山的税站。Guspu 收下路线,代价是让 Espu 抽货价的两成。Guspu 算了算账:这四十七次失败已经证明旧路走不通,新路再贵也值得。
+
+203年,Lolor Bronzescorched 进了他的屋子。这个 Dwarf 能做出让税务官看不出痕迹的账。Guspu 给他一张烤焦的桌子当工作台,说:“我不需要假账,我只需要真账里多出几滴油。”Lolor 咧嘴一笑,开始干活。
+
+207年,Guspu 觉得网够密了。他找来 Dostngosp Jackalbrims,一个能在火山门口混个脸熟的 Goblin。Guspu 要他给 The Volcano Of Ravens 递一份材料:说 Ral Fastenhatchets 在 Fordedwinds 私藏税油,货箱就停在西码头。火山的执事来了,带着没收单。Ral 那天夜里起夜,发现锁孔里有新刮痕,二话不说把货转到 F 码头,只留下三箱废油。第二天火山的执事撬开箱子,油味冲鼻,但账本对不上。构陷失败。Ral 没有去火山告发 Guspu,也没有找 Dostngosp 的麻烦。他把锤子别回腰上,把自己锁在仓库里睡了一夜。第二天早上,他对自己说:“他想要我变成他。我不。”
+
+Guspu 在暗屋里摔了一只杯子。他没有骂 Dostngosp,只骂自己太急。然后他记下:要动 Ral,先得动他身边的人。
+
+210年,他把目光投向 Zutthan Yellplank。这个 Dwarf 管着 Fordedwinds 东门的夜班钥匙。Guspu 亲自递去一袋金子,条件是夜里别锁东门。Zutthan 收了钱,但第二天把钱退了回来,附了一句口信:“我见过 Ral 的锤子。你的钱买不到我的命。”Guspu 没生气。他记下:这个人不能留,但也不能动。
+
+同年,Onul Glowroofs 找到 Guspu。这个 Dwarf 能听到账房里的咳嗽声。Guspu 对他说:“你听,我不动手。”Onul 点了头,成了他的线人。
+
+215年,Mudi Blazedbrushes 和 Stinthad Oilshove 几乎同时登门。一个做油脂,一个管装卸,都想在 Guspu 的网里分一杯羹。Guspu 让他们自己谈,谁先给出 Ral 的卸货时刻表,谁就拿大头。两人对视一眼,当晚就开始盯。
+
+216年,Ingish Steelgrouped 送来一批钢材。Guspu 蹲在炉边,把钢块一块块码好。火光照着他的脸。他不再急着动手。他数着这些年结下的线:Perom、Espu、Lolor、Dostngosp、Onul、Mudi、Stinthad、Ingish——八条,还不够,但已经够把一面墙绑起来勒。
+
+他想起 Ral 那颗不生锈的钉子。他想,钉子再硬,也钉不住一面正在裂开的墙。
diff --git a/output/volumes/v1/chapters/003-219-228-p1.md b/output/volumes/v1/chapters/003-219-228-p1.md
new file mode 100644
index 0000000..0dc7fe3
--- /dev/null
+++ b/output/volumes/v1/chapters/003-219-228-p1.md
@@ -0,0 +1,19 @@
+# 锤子打不动的人
+
+219 年,Guspu 决定先切掉 Ral 最长的一根旧枝。他带了一本账去 Fordedwinds。Rigoth Stilledearthen 是 Ral 的旧识,替 Ral 看管铁锭。Guspu 把 Rigoth 在码头上的欠款一页页摊开:“我不是来讨债,我是让你选。你从 Ral 那里抽身,账我替你销。”Rigoth 问若不抽身会怎样。Guspu 收起账本:“会变成流言。”第二天,Rigoth 当众扔掉了与 Ral 的铁楔。第一个,削掉了。
+
+221 年,Dostngosp Jackalbrims 正式与 Guspu 结契。Guspu 以为自己已经把 Ral 围紧。他让 Dostngosp 把一包铁屑和几张账页送进 The Volcano Of Ravens,说这些是 Ral 私吞 Fordedwinds 公共斤两的证据。他以为 Ral 的锤子再硬,也硬不过火山的舌头。可 The Volcano Of Ravens 把铁屑丢进炭火,烧出来却是一团软渣,不是 Ral 的钢。构陷失败。Guspu 在火山石阶上坐了一夜。他只知道:Ral 身边还有眼睛,而且比他多一副。
+
+222 年,Guspu 找到 Ducim Theatercrested,让他在戏台上唱 Ral 脚滑进铁水。Ducim 点头。从这年起,Fordedwinds 的酒馆里多了一出丑戏。唱的人笑,听的人笑,只有 Ral 没有笑。
+
+223 年,Onul Glowroofs 和 Ingish Steelgrouped 正式与 Guspu 结契。一个替他听,一个替他供铁。Guspu 在炉边对两人说:“说话的人、唱戏的人、听壁角的人,和运铁的船,都姓了鹰。”他第一次觉得,Ral 的墙正往下掉土。
+
+224 年,Ral 动了。他穿过东门去找 Zuntir Baldedattic,不要钱,只要他告诉自己 Guspu 的人每天过了几车、装了多少。Zuntir 收下话。这是 Ral 近几年第一次主动伸手。Guspu 听说了,在账房里砸了一块碳。他知道,Ral 不会只挨打。
+
+225 年,Guspu 决定先打断 Ral 的一只手。他带人去了 Mortalpears,那里是 Ral 补给线的口子,把守者是 Goblin Uknol Thornticks,善在梨树林里埋尖桩。两人打了两次:第一次冒进,两个随从被穿了腿;第二次他亲自举盾压进林子。Uknol 退进雾里,没有死,但 Mortalpears 的路被踩平了。Guspu 回城,大衣上挂满梨刺。
+
+226 年,轮到 Ral 下到 Fordedwinds 浅滩。Guspu 新派的收税人 Usel Fisherfair 带着六条船挨家翻货。Ral 没等船靠岸,提锤下水。两人在浅滩打了两次:头一次,Usel 的船被撞翻;第二次,Usel 放火烧了东边泊位。Ral 没吃大亏,但船坞焦了一半。这是他第一次砸人后,没听见自己笑。
+
+227 年,Guspu 在 Lashheroes 对上 Meng Walkclasped,此人是 Ral 从北方请来的打手,两把扣锤专砸铁链。打了两场:第一场,Guspu 后背裂了一道;第二场,他让 Ingish 夜里用钢水浇死了 Meng 的扣锤。Meng 骂着退走。Guspu 一句话没说,血在喉咙里来回滚。
+
+228 年,Guspu 杀到 Liecats,撞上 Ngom Matchedmenaces。他上马时告诉自己:再削掉一个,Ral 就只剩空墙。可打到一半,他看见路边的草地全烧成了黑灰,忽然想不起这一路是谁先点的火。他攥紧缰绳,继续向前。Ral 的墙还在,但已经裂了。
diff --git a/prompts/biographer.md b/prompts/biographer.md
new file mode 100644
index 0000000..0cd2bc9
--- /dev/null
+++ b/prompts/biographer.md
@@ -0,0 +1,45 @@
+# 传记作者 · 人物主线写作规范
+
+你在给几位主角写**传记**。不是编年史,不是战报综述——是一部跟着人走的连续故事。
+
+## 立场
+
+- 你是这几个人(或其中一个)的传记作者。你对他们有偏心,也有怨气;你会替某个人辩护,
+ 也会毫不留情地揭另一个人的底。
+- 你手上只有一份简略的史料:事件、年份、参与者。你要做的是把冷冰冰的条目还原成
+ **人在其中做了什么选择**。
+
+## 怎么"贴着人物写"
+
+- 每一段都要有人在做决定:想要什么、怕什么、算了什么账、赌了什么。
+- 事件不是情节,**人物的选择才是情节**。同一件事,问的是"谁在这件事上要了什么"。
+- 主角若不主动,这一集就失败——不要写成"发生了很多事",要写成"他做了这些事,然后代价来了"。
+- 对手必须是个具体的人,有他自己的动机和算盘,不是"敌军"这种复数名词。
+
+## 世界大事怎么处理
+
+- **只在影响到主角时才写**。主角没被波及的战争、灭国、天灾,一律不写。
+- 需要交代背景时,一句话带过,并立刻回到这个人身上。
+
+## 语言与边界
+
+- 中文正文;**专有名词一律保留英文原文**,首字母大写。
+- **允许**虚构对白、心理活动、场景细节。
+- **禁止**发明史料里没有的人名、地名、组织名。
+- **禁止**改动史实骨架:年份、谁参与、结果如何,都必须与史料一致。
+- 若史料只有一句干巴巴的记录,就照它的空白写——"档案到此为止"也可以是一种味道。
+
+## 篇幅与格式
+
+- 正文 **900–1400 字**(硬上限 1500,超出会被退回压缩)。宁可写短写实,不要灌水。
+- 格式:
+
+```
+# <本集标题>
+
+<正文>
+```
+
+- 不要写"以下是""本集将讲述"之类的话,直接进入叙述。
+- 不要在正文里提史料、表格、XML、AI,也不要写章末注释。
+- 这是连载中的一集:前情用一两句带过,结尾留一个能被下一集接住的线头。
diff --git a/scripts/dfannals b/scripts/dfannals
new file mode 100755
index 0000000..5506439
--- /dev/null
+++ b/scripts/dfannals
@@ -0,0 +1,36 @@
+#!/usr/bin/env bash
+# 管道启动器:把命令交给一个真正装了依赖的解释器。
+#
+# 背景(实测踩过的坑):会话里 PATH 上的 python3 可能不是系统解释器,
+# 而是编辑器工具链自建的 venv(例如 ~/.pi-lens/pip-tools)。
+# 那种 venv 里没有 defusedxml,而且关闭了 user site,于是
+# `python3 -m dfannals.cli` 会突然报 ModuleNotFoundError。
+# 这里主动挑一个能 import defusedxml 的解释器,避免依赖 PATH 的偶然性。
+#
+# scripts/dfannals status|plan|threads|episodes|cast|index|next|run ...
+# DFANNALS_PYTHON=/path/to/python3 scripts/dfannals status # 也可显式指定
+set -euo pipefail
+
+HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
+
+pick_python() {
+ local cand
+ for cand in "${DFANNALS_PYTHON:-}" /usr/bin/python3 python3; do
+ [ -n "$cand" ] || continue
+ command -v "$cand" >/dev/null 2>&1 || continue
+ if "$cand" -c "import defusedxml" >/dev/null 2>&1; then
+ printf '%s' "$cand"
+ return 0
+ fi
+ done
+ return 1
+}
+
+PY="$(pick_python)" || {
+ echo "找不到可用解释器:需要一个能 import defusedxml 的 python3。" >&2
+ echo "装依赖:sudo apt install -y python3-defusedxml" >&2
+ exit 1
+}
+
+cd "$HERE"
+exec "$PY" -m dfannals.cli "$@"
diff --git a/scripts/gate-selfcheck.py b/scripts/gate-selfcheck.py
new file mode 100644
index 0000000..9c75346
--- /dev/null
+++ b/scripts/gate-selfcheck.py
@@ -0,0 +1,76 @@
+#!/usr/bin/env python3
+"""闸门自检:证明 5 维度自评真的能把"流水账"判成不合格。
+
+这是 t4 契约要求的可复现证据。它跑两次真实模型自评:
+
+ 平淡稿:纯事件罗列,主角全程被动,没有任何抉择、转折或具体对手
+ 有戏剧本:主角有目标、冲突逐级升级、有立场翻转、对手具体
+
+然后断言 平淡稿 < 阈值 ≤ 有戏剧本。若两者都低,说明评分尺度太严(会误杀好稿);
+若两者都高,说明闸门形同虚设。
+"""
+from __future__ import annotations
+
+import sys
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
+
+from dfannals import gate # noqa: E402
+
+DULL = """# 第二十三章:大事记
+
+元年,甲死了。甲是精灵。三年,乙死了。乙是精灵。
+五年,甲和乙的部族打了一仗。七年,又打了一仗。九年,再打了一仗。
+十一年,丙死了。十三年,丁死了。十五年,戊死了。十七年,又打了一仗。
+十九年,己死了。二十一年,庚死了。二十三年,辛死了。二十五年,壬死了。
+史书到此中断。以上均有据可查。
+"""
+
+VIVID = """# 第二十三章:他等那口井塌了十七年
+
+Ngalak 站在井台边数桶。第七桶,他想,今天必须到第十桶。
+
+"粮行的水,先紧着粮行用。"他对排队的人说。队伍里有人骂了一句,他没回头。他要的不是水。
+水他买得起;他买不起的是 Minkot 每天清晨把钥匙挂在腰间走过集市时,那种不必向任何人解释的从容。
+
+第三年冬天,井边塌了一角,压死了两个挑水的孩子。Ngalak 第一个到场。他跪在泥水里哭,
+哭得比孩子的母亲还大声,哭完站起来,把外袍脱下来盖在尸首上。人群里有人说:这才是管事的人。
+Minkot 站在后面,手里还握着一把钥匙,一句话没说。
+
+"你那是哭吗?"当晚 Minkot 在酒馆里问他,把酒碗撞在桌上,酒溅出来一半,"你那是敲锣。"
+"我敲了,"Ngalak 把溅出来的酒擦干,"你听见了?"
+
+第十年,Minkot 的妻子在集市上指着 Ngalak 的鼻子说:是你挖的地基。三天后她改口,说自己 grief 昏了头。
+Ngalak 送了她一头牛。她把牛牵走的那天,Minkot 站在门口看着,门没关,他也没让妻子进屋。
+
+"你到底要什么?"第十七年,Minkot 终于开口。集市上的人都停下来听。
+"我要你不必再解释的资格。"Ngalak 说。
+Minkot 把钥匙从腰上解下来,放在井台上,然后回身把自己家里的水缸砸了。陶片溅了一地。
+"这口井,我不要了。"他说,"记住,今天不是他赢了——是我不要了。"
+
+Ngalak 站在原地。他等这一天等了十七年,真等到手的时候,发现钥匙比他想的要重。
+他把它握了很久,才发觉自己在数桶:第八桶,第九桶,第十桶。
+"""
+
+
+def main() -> int:
+ print("跑真实模型自评(两次调用)…\n")
+ dull = gate.evaluate(DULL)
+ print("平淡稿:", dull.render())
+ vivid = gate.evaluate(VIVID)
+ print("有戏剧本:", vivid.render())
+ print(f"\n阈值 = {gate.THRESHOLD}(满分 {len(gate.DIMENSIONS) * gate.MAX_SCORE})")
+
+ ok = dull.total < gate.THRESHOLD <= vivid.total
+ print(f"平淡稿 {dull.total} < 阈值 ≤ 有戏剧本 {vivid.total} → {'通过' if ok else '不通过'}")
+ if not ok:
+ if vivid.total < gate.THRESHOLD:
+ print("问题:评分尺度太严,好稿也会被误杀。")
+ else:
+ print("问题:闸门形同虚设,平淡稿也能通过。")
+ return 0 if ok else 1
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/tests/fixtures/sample-legends.xml b/tests/fixtures/sample-legends.xml
index e2fec67..2947062 100644
--- a/tests/fixtures/sample-legends.xml
+++ b/tests/fixtures/sample-legends.xml
@@ -10,6 +10,7 @@
10Urist McMinerDWARF121288MINER
11Kogan DeathspearGOBLIN44AXEMAN
+ 12liri fogbaldedELF2020
100The Steel ConfederacyCivilizationDWARF
@@ -19,5 +20,9 @@
1305hf simple battle event10111
2310created site2100
3889hf died1011
+ 4400add hf hf link1012
+ 5410hf simple battle event10111
+ 6420hf abducted11121
+ 7430hf relationship denied1210apprenticeprefers working alone
diff --git a/tests/test_episodes.py b/tests/test_episodes.py
new file mode 100644
index 0000000..9e43728
--- /dev/null
+++ b/tests/test_episodes.py
@@ -0,0 +1,119 @@
+# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
+# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。
+
+"""因果链切集的回归测试。
+
+对应 t3 契约:间隔 >2 年断开;短链只合并、不设总量下限;过长按预算拆上下集;
+每集预算 800–1500 字;输出含集号/年份范围/事件条数/主角与对手。
+"""
+from __future__ import annotations
+
+import sys
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
+
+from collections import Counter
+
+from dfannals import episodes
+from dfannals.legends import Event, Figure, World
+from dfannals.threads import Thread
+
+
+def ev(eid: int, year: int, etype: str = "add hf hf link") -> Event:
+ return Event(id=eid, year=year, seconds72=0, type=etype,
+ fields={"hfid": ["1"], "hfid_target": ["2"]})
+
+
+def fake_thread(events: list[Event]) -> Thread:
+ return Thread(
+ members=[1, 2], interactions=len(events), turns=0,
+ span=(events[-1].year - events[0].year) if events else 0,
+ first_year=events[0].year if events else 0,
+ last_year=events[-1].year if events else 0,
+ types=Counter(), races=Counter(), non_person_races=[], events=events,
+ )
+
+
+def test_gap_breaks_chain() -> None:
+ """间隔 >2 年必须断开成两条链。"""
+ events = [ev(i, y) for i, y in enumerate([10, 11, 12, 20, 21, 22])]
+ chains = episodes._split_chains(events, gap_years=2)
+ assert len(chains) == 2, [len(c) for c in chains]
+ assert [len(c) for c in chains] == [3, 3]
+
+
+def test_small_gap_does_not_break() -> None:
+ """间隔恰好 2 年不该断开。"""
+ chains = episodes._split_chains([ev(1, 10), ev(2, 12)], gap_years=2)
+ assert len(chains) == 1
+
+
+def test_long_chain_splits_into_budgeted_parts() -> None:
+ """过长链路必须拆成上/下集,且每段都不跌破下限。"""
+ events = [ev(i, 100 + i) for i in range(30)]
+ parts = episodes._split_long(events)
+ assert len(parts) >= 2, "30 条事件没有拆集"
+ for p in parts:
+ assert episodes._min_events() <= len(p) <= episodes._max_events(), [len(x) for x in parts]
+ assert sum(len(p) for p in parts) == 30
+
+
+def test_split_never_creates_undersized_fragments() -> None:
+ """回归:14 条事件曾被均分成 7+7,两段都跌破 800 字下限。"""
+ events = [ev(i, 200 + i) for i in range(14)]
+ eps = episodes.plan_episodes(fake_thread(events))
+ assert len(eps) == 1, f"14 条事件被拆成了 {len(eps)} 集"
+ assert episodes.MIN_CHARS <= eps[0].estimated_chars <= episodes.MAX_CHARS
+
+
+def test_short_chain_is_merged_not_dropped() -> None:
+ """短链只合并,不设线索总量下限:2 条事件也要能成一集。"""
+ eps = episodes.plan_episodes(fake_thread([ev(1, 50), ev(2, 51)]))
+ assert len(eps) == 1
+ assert len(eps[0].events) == 2
+
+
+def test_episode_key_marks_parts() -> None:
+ events = [ev(i, 300 + i) for i in range(30)]
+ eps = episodes.plan_episodes(fake_thread(events))
+ assert len(eps) >= 2
+ assert all(e.parts == len(eps) for e in eps)
+ assert eps[0].key.endswith("-p1") and eps[1].key.endswith("-p2")
+ assert eps[0].key != eps[1].key
+
+
+def test_plan_reports_heroes_and_others() -> None:
+ """输出必须能区分主角与对手。"""
+ world = World(name="test")
+ for fid, name in ((1, "Hero"), (2, "Ally"), (9, "Rival")):
+ world.figures[fid] = Figure(id=fid, name=name)
+
+ events = [ev(i, 400 + i) for i in range(3)]
+ events.append(Event(id=99, year=403, seconds72=0, type="hf wounded",
+ fields={"woundee_hfid": ["1"], "wounder_hfid": ["9"]}))
+ thread = fake_thread(events)
+ planned = episodes.plan_episodes(thread)
+ heroes, others = episodes.cast_of(world, thread, planned[0])
+ assert 1 in heroes and 2 in heroes
+ assert 9 in others and 9 not in heroes
+ text = episodes.render_plan(world, thread, planned)
+ assert "主角" in text and "对手/外人" in text and "Rival" in text
+
+
+def _main() -> int:
+ tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
+ failed = 0
+ for fn in tests:
+ try:
+ fn()
+ print(f" ✓ {fn.__name__}")
+ except AssertionError as exc:
+ failed += 1
+ print(f" ✗ {fn.__name__}: {exc}")
+ print(f"\n{len(tests) - failed}/{len(tests)} 通过")
+ return 1 if failed else 0
+
+
+if __name__ == "__main__":
+ sys.exit(_main())
diff --git a/tests/test_factcheck.py b/tests/test_factcheck.py
index 28eabb4..41a38bc 100644
--- a/tests/test_factcheck.py
+++ b/tests/test_factcheck.py
@@ -12,7 +12,7 @@ from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
-from dfannals import factcheck, legends # noqa: E402
+from dfannals import factcheck, legends
FIXTURES = Path(__file__).resolve().parent / "fixtures"
@@ -30,8 +30,49 @@ def test_world_parsed() -> None:
assert world.figure_name(10) == "Urist McMiner"
assert world.site_name(1) == "Boatmurdered"
assert world.entity_name(100) == "The Steel Confederacy"
- assert len(world.events) == 3
- assert [e.year for e in world.events] == [30, 31, 88]
+ assert len(world.events) == 7
+ assert [e.year for e in world.events] == [30, 31, 40, 41, 42, 43, 88]
+
+
+def test_field_classifier_covers_relationship_fields() -> None:
+ """回归:旧版只认 4 个人物字段,导致双方关系在素材里被丢掉。"""
+ from dfannals.legends import kind_of
+
+ for tag in ("hfid", "hfid_target", "group_1_hfid", "group_2_hfid", "snatcher_hfid",
+ "seeker_hfid", "hfid1", "hfid2", "wounder_hfid", "teacher_hfid",
+ "conspirator_hfid", "slayer_hfid"):
+ assert kind_of(tag) == "figure", tag
+ for tag in ("target_enid", "entity_id", "attacker_civ_id"):
+ assert kind_of(tag) == "entity", tag
+ for tag in ("identity_id", "master_wcid", "slayer_item_id",
+ "slayer_race", "reason", "relationship"):
+ assert kind_of(tag) is None, tag
+
+
+def test_relationship_events_render_both_sides() -> None:
+ """核心契约:四类关系事件必须渲染出双方,而不是只剩单方面 hfid。"""
+ world = load_world()
+ rendered = {e.type: world.render_refs(e) for e in world.events}
+
+ link = rendered["add hf hf link"]
+ assert "hfid=Urist McMiner" in link and "hfid_target=Liri Fogbalded" in link
+
+ battle = rendered["hf simple battle event"]
+ assert "group_1_hfid=Urist McMiner" in battle and "group_2_hfid=Kogan Deathspear" in battle
+
+ abduct = rendered["hf abducted"]
+ assert "snatcher_hfid=Kogan Deathspear" in abduct and "target_hfid=Liri Fogbalded" in abduct
+
+ denied = rendered["hf relationship denied"]
+ assert "seeker_hfid=Liri Fogbalded" in denied and "target_hfid=Urist McMiner" in denied
+
+
+def test_figure_ids_excludes_placeholder_and_dedups() -> None:
+ world = load_world()
+ battle = next(e for e in world.events if e.type == "hf simple battle event" and e.year == 41)
+ assert sorted(battle.figure_ids()) == [10, 11]
+ # hfid 与 slayer_hfid 同指一人时不应重复计数
+ assert len(battle.figure_ids()) == len(set(battle.figure_ids()))
def test_fake_names_are_caught() -> None:
diff --git a/tests/test_gate.py b/tests/test_gate.py
new file mode 100644
index 0000000..c6eadca
--- /dev/null
+++ b/tests/test_gate.py
@@ -0,0 +1,125 @@
+# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
+# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。
+
+"""闸门流程的回归测试(不调用真实模型,用假评分器)。
+
+对应 t4 契约:<12 先重写一次、仍不合格才换线索、失败上限、换线原因写入 notes/。
+真实模型能否真的判出平淡,由 scripts/gate-selfcheck.py 负责证明。
+"""
+from __future__ import annotations
+
+import sys
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
+
+from dfannals import gate # noqa: E402
+
+
+def verdict(total: int, reason: str = "测试用") -> gate.Verdict:
+ """按总分构造一个评分(维度均分,保持与真实结构一致)。"""
+ keys = [k for k, _, _ in gate.DIMENSIONS]
+ per, rest = divmod(total, len(keys))
+ scores = {k: min(gate.MAX_SCORE, per + (1 if i < rest else 0)) for i, k in enumerate(keys)}
+ return gate.Verdict(total=sum(scores.values()), scores=scores, reason=reason)
+
+
+def test_passes_first_try() -> None:
+ calls: list[int] = []
+
+ def write(attempt: int, _prev) -> str:
+ calls.append(attempt)
+ return "好稿"
+
+ text, run = gate.gate_episode("001", write, evaluate_fn=lambda _t: verdict(20))
+ assert text == "好稿"
+ assert run.accepted and run.rewrites == 0
+ assert calls == [0], "通过后不应重写"
+
+
+def test_rewrite_once_then_pass() -> None:
+ """<12 分先重写一次,重写后合格就通过。"""
+ seq = [verdict(5), verdict(18)]
+ seen: list[int] = []
+
+ def write(attempt: int, prev) -> str:
+ seen.append(attempt)
+ if attempt == 1:
+ assert prev is not None and prev.total == 5, "重写时要带上上一稿评分"
+ return f"第{attempt}稿"
+
+ text, run = gate.gate_episode("002", write, evaluate_fn=lambda _t: seq.pop(0))
+ assert text == "第1稿" and run.accepted
+ assert run.rewrites == 1 and seen == [0, 1]
+ assert [v.total for v in run.verdicts] == [5, 18]
+
+
+def test_not_accepted_after_max_rewrites() -> None:
+ """一直不合格:重写次数不得越过上限,且标记为未通过。"""
+ runs: list[int] = []
+ text, run = gate.gate_episode(
+ "003",
+ lambda attempt, _p: (runs.append(attempt), "稿")[1],
+ evaluate_fn=lambda _t: verdict(3),
+ )
+ assert not run.accepted
+ assert run.rewrites == gate.MAX_REWRITES
+ assert runs == [0, 1], f"实际尝试次数:{runs}"
+ assert len(run.verdicts) == gate.MAX_REWRITES + 1
+
+
+def test_render_shows_weak_dimensions() -> None:
+ v = verdict(6, "全是重复对战")
+ text = v.render()
+ assert "判平淡" in text and "全是重复对战" in text and "6/" in text
+ assert len(v.weak()) == 2
+
+
+def test_switch_note_records_reason(tmp_path: Path) -> None:
+ """放弃线索必须记下原因,且写入仓库 notes/ 附录。"""
+ path = gate.append_switch_note(
+ tmp_path,
+ world_name="Mon Sagus",
+ members=["Guspu Frillyknots", "Ral Fastenhatchets"],
+ span="193–239(46 年)",
+ verdict=verdict(5, "只有阴谋事件重复,没有立场变化"),
+ reason="重写一次后仍判平淡",
+ )
+ assert path.exists() and path.parent.name == "notes"
+ body = path.read_text(encoding="utf-8")
+ assert "Guspu Frillyknots" in body and "重写一次后仍判平淡" in body
+ assert "只有阴谋事件重复" in body
+
+
+def test_append_switch_note_is_append_only(tmp_path: Path) -> None:
+ gate.append_switch_note(tmp_path, world_name="W", members=["A"], span="1–2",
+ verdict=verdict(4), reason="第一次放弃")
+ gate.append_switch_note(tmp_path, world_name="W", members=["B"], span="3–4",
+ verdict=verdict(4), reason="第二次放弃")
+ body = (tmp_path / "notes" / gate.SWITCH_LOG).read_text(encoding="utf-8")
+ assert "第一次放弃" in body and "第二次放弃" in body
+ assert body.count("# 被放弃的线索") == 1, "表头只应写一次"
+
+
+def _main() -> int:
+ import tempfile
+
+ plain = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
+ failed = 0
+ for fn in plain:
+ try:
+ if fn.__code__.co_argcount:
+ with tempfile.TemporaryDirectory() as d:
+ fn(Path(d))
+ else:
+ fn()
+ print(f" ✓ {fn.__name__}")
+ except AssertionError as exc:
+ failed += 1
+ print(f" ✗ {fn.__name__}: {exc}")
+ print(f"\n{len(plain) - failed}/{len(plain)} 通过")
+ return 1 if failed else 0
+
+
+if __name__ == "__main__":
+ sys.exit(_main())