From e0073e90bd420844bcdc8058f4598937c3ea68c2 Mon Sep 17 00:00:00 2001 From: Chen Yi <466354947@qq.com> Date: Mon, 5 Oct 2026 19:28:49 +0800 Subject: [PATCH] =?UTF-8?q?=E7=AC=AC=201=20=E5=8D=B7=EF=BC=88=E4=BA=BA?= =?UTF-8?q?=E7=89=A9=E4=B8=BB=E7=BA=BF=EF=BC=89=EF=BC=9A=E7=AC=AC=201?= =?UTF-8?q?=E2=80=933=20=E9=9B=86=20+=20=E4=BA=BA=E7=89=A9=E8=A1=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 主角线:Guspu Frillyknots / Ral Fastenhatchets / Alath Blottedmine 由线索挖掘器自动选定,按因果链切集,经 5 维度自评闸门(≥12 分)后入库。 --- dfannals/cast.py | 225 +++++++++++++ dfannals/chronicle.py | 4 +- dfannals/cli.py | 156 ++++++++- dfannals/config.py | 5 +- dfannals/episodes.py | 192 +++++++++++ dfannals/gate.py | 223 ++++++++++++ dfannals/legends.py | 101 ++++-- dfannals/publish.py | 1 + dfannals/score.py | 21 +- dfannals/slice.py | 27 +- dfannals/threads.py | 287 ++++++++++++++++ dfannals/timeline.py | 28 +- dfannals/volume.py | 336 +++++++++++++++++++ output/volumes/v1/cast.md | 37 ++ output/volumes/v1/chapters/001-181-198.md | 13 + output/volumes/v1/chapters/002-202-216.md | 21 ++ output/volumes/v1/chapters/003-219-228-p1.md | 19 ++ prompts/biographer.md | 45 +++ scripts/dfannals | 36 ++ scripts/gate-selfcheck.py | 76 +++++ tests/fixtures/sample-legends.xml | 5 + tests/test_episodes.py | 119 +++++++ tests/test_factcheck.py | 47 ++- tests/test_gate.py | 125 +++++++ 24 files changed, 2070 insertions(+), 79 deletions(-) create mode 100644 dfannals/cast.py create mode 100644 dfannals/episodes.py create mode 100644 dfannals/gate.py create mode 100644 dfannals/threads.py create mode 100644 dfannals/volume.py create mode 100644 output/volumes/v1/cast.md create mode 100644 output/volumes/v1/chapters/001-181-198.md create mode 100644 output/volumes/v1/chapters/002-202-216.md create mode 100644 output/volumes/v1/chapters/003-219-228-p1.md create mode 100644 prompts/biographer.md create mode 100755 scripts/dfannals create mode 100644 scripts/gate-selfcheck.py create mode 100644 tests/test_episodes.py create mode 100644 tests/test_gate.py diff --git a/dfannals/cast.py b/dfannals/cast.py new file mode 100644 index 0000000..8e406d6 --- /dev/null +++ b/dfannals/cast.py @@ -0,0 +1,225 @@ +"""每卷人物表:Markdown 表格 + 每人一两句小传。 + +访谈决策: +- 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库 +- 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册 + +表格部分完全由史料推导(种族、生卒、出场次数、身份); +"身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上, +就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。 +""" +from __future__ import annotations + +import json +import re +from dataclasses import dataclass +from pathlib import Path + +from dfannals import factcheck +from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway +from dfannals.episodes import Episode +from dfannals.legends import Event, World +from dfannals.llm import chat +from dfannals.threads import Thread + +# 事件字段 → 中文角色名(用于反推"身份") +ROLE_LABELS: dict[str, str] = { + "corruptor_hfid": "构陷发起者", + "target_hfid": "被针对者", + "wounder_hfid": "行凶者", + "woundee_hfid": "受伤者", + "snatcher_hfid": "绑走者", + "seeker_hfid": "求关系者", + "winner_hfid": "胜者", + "competitor_hfid": "参赛者", + "slayer_hfid": "凶手", + "hfid": "当事者", + "hfid_target": "关系对象", + "teacher_hfid": "师长", + "student_hfid": "学徒", + "persecutor_hfid": "迫害者", + "expelled_hfid": "被逐者", + "convicted_hfid": "被定罪者", + "interrogator_hfid": "审讯者", + "framer_hfid": "构陷者", + "fooled_hfid": "受骗者", +} + +BIO_LIMIT = 10 # 小传最多覆盖多少人 +EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE) + + +@dataclass +class CastRow: + fid: int + name: str + race: str + span: str + role: str # 主角 / 对手 / 配角 + appearances: int + identity: str + + def as_row(self) -> str: + return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | " + f"{self.appearances} | {self.role} |") + + +def _identity_of(world: World, fid: int, events: list[Event]) -> str: + """从事件字段反推这个人在这段历史里扮演什么。""" + counts: dict[str, int] = {} + for e in events: + for tag, vals in e.fields.items(): + if tag not in ROLE_LABELS: + continue + for raw in vals: + try: + if int(raw) == fid: + counts[tag] = counts.get(tag, 0) + 1 + except ValueError: + continue + if not counts: + return "" + ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2] + return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked) + + +def _span_of(thread: Thread, episodes: list[Episode]) -> str: + """表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。""" + if not episodes: + return thread.label + return f"{episodes[0].start_year}–{episodes[-1].end_year}" + + +def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]: + """主角优先,其次是出场最多的对手/配角。""" + events = [e for ep in episodes for e in ep.events] or thread.events + appearances: dict[int, int] = {} + for e in events: + for fid in e.figure_ids(): + appearances[fid] = appearances.get(fid, 0) + 1 + + members = set(thread.members) + rows: list[CastRow] = [] + + def make(fid: int, role: str) -> CastRow: + fig = world.figures.get(fid) + return CastRow( + fid=fid, + name=world.figure_name(fid), + race=fig.race if fig else "", + span=fig.alive_span if fig else "生卒不详", + role=role, + appearances=appearances.get(fid, 0), + identity=_identity_of(world, fid, events), + ) + + for fid in sorted(members, key=lambda f: -appearances.get(f, 0)): + rows.append(make(fid, "主角")) + others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f]) + for fid in others[:max_others]: + rows.append(make(fid, "对手/配角")) + return rows + + +BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。 + +规则: +- 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。** +- 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。 +- 姓名一律保留英文原文。 +- 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...} +""" + + +def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str: + lines = [BIO_INSTRUCTIONS, "", "材料:"] + for row in rows[:BIO_LIMIT]: + facts = [ + f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次," + f"身份:{row.identity or '不详'},在故事里是{row.role})" + ] + for e in events: + if row.fid not in e.figure_ids(): + continue + refs = " · ".join(world.render_refs(e)) + facts.append(f" {e.year}年 {e.type} {refs}") + if len(facts) > 5: + break + lines.extend(facts) + return "\n".join(lines) + + +def _parse_bios(raw: str) -> dict[str, str]: + text = EDGE_RE.sub("", raw).strip() + start, end = text.find("{"), text.rfind("}") + if start < 0 or end <= start: + raise ValueError(f"小传没有返回 JSON:{raw[:200]}") + try: + data = json.loads(text[start:end + 1]) + except json.JSONDecodeError as exc: + raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None + if not isinstance(data, dict): + raise ValueError("小传返回的不是对象") + return {str(k): str(v) for k, v in data.items()} + + +def build_cast( + world: World, + thread: Thread, + episodes: list[Episode], + *, + gateway: Gateway | None = None, + with_bios: bool = True, +) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]: + """返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。""" + rows = collect_rows(world, thread, episodes) + events = [e for ep in episodes for e in ep.events] or thread.events + + bios: dict[str, str] = {} + if with_bios and rows: + reply = chat( + [{"role": "user", "content": _bio_prompt(world, rows, events)}], + gateway=gateway, + max_tokens=MAX_TOKENS_STRUCTURED, + ) + bios = _parse_bios(reply.content) + + suspicions: list[factcheck.Suspicion] = [] + bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items()) + if bios_text: + suspicions = factcheck.check(bios_text, world) + + lines = [ + f"# {world.name} · 人物表", + "", + f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}", + f"线索跨度:{_span_of(thread, episodes)}", + "", + "| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |", + "|---|---|---|---|---|---|", + ] + lines.extend(row.as_row() for row in rows) + + if bios: + lines += ["", "## 小传", ""] + for row in rows: + bio = bios.get(row.name) or bios.get(row.name.lower()) + if bio: + lines.append(f"**{row.name}** —— {bio}") + lines.append("") + + if suspicions: + lines += [ + "", + "> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:" + + "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]), + "", + ] + + return "\n".join(lines) + "\n", rows, suspicions + + +def write_cast(path: Path, text: str) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + return path diff --git a/dfannals/chronicle.py b/dfannals/chronicle.py index db9181d..08844a0 100644 --- a/dfannals/chronicle.py +++ b/dfannals/chronicle.py @@ -62,8 +62,8 @@ class State: def cjk_length(text: str) -> int: """按中文习惯计字数:CJK 字符与拉丁单词都算 1。""" - without_code = re.sub(r"```.*?```", "", text, flags=re.S) - without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.M) + without_code = re.sub(r"```.*?```", "", text, flags=re.DOTALL) + without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.MULTILINE) n = 0 for ch in without_meta: if ( diff --git a/dfannals/cli.py b/dfannals/cli.py index 56b0104..9556ec6 100644 --- a/dfannals/cli.py +++ b/dfannals/cli.py @@ -16,9 +16,26 @@ import argparse import sys from pathlib import Path -from dfannals import chronicle, factcheck, legends, publish, timeline +from dfannals import ( + cast, + chronicle, + episodes, + factcheck, + legends, + publish, + threads, + timeline, +) from dfannals import slice as chapter_slice -from dfannals.config import EXPORT_DIR, GITEA_WEB, chapter_dir, ensure_dirs, volume_dir +from dfannals import volume as volume_mod +from dfannals.config import ( + DATA_DIR, + EXPORT_DIR, + GITEA_WEB, + chapter_dir, + ensure_dirs, + volume_dir, +) def find_exports(export_dir: Path) -> list[Path]: @@ -142,6 +159,109 @@ def cmd_next(args: argparse.Namespace) -> int: return 0 +def cmd_threads(args: argparse.Namespace) -> int: + ensure_dirs() + world = load_world(EXPORT_DIR) + found = threads.find_threads( + world, + min_interactions=args.min_interactions, + min_span=args.min_span, + limit=args.top, + ) + print(threads.render_report(world, found, per_thread=args.samples)) + path = threads.save_json(world, found, DATA_DIR / "threads.json") + print(f"共 {len(found)} 条候选;明细已写入 {path}") + return 0 + + +def cmd_episodes(args: argparse.Namespace) -> int: + ensure_dirs() + world = load_world(EXPORT_DIR) + found = threads.find_threads( + world, min_interactions=args.min_interactions, min_span=args.min_span + ) + if not found: + print("没有候选线索") + return 1 + if not 1 <= args.thread <= len(found): + print(f"候选只有 {len(found)} 条,--thread 超出范围") + return 1 + thread = found[args.thread - 1] + _, planned = volume_mod.extended_plan(world, thread) + print(episodes.render_plan(world, thread, planned)) + print() + print(episodes.budget_report(planned)) + if args.samples: + print("\n=== 前两集事件样例 ===") + for ep in planned[:2]: + print(f" 第 {ep.index} 集({ep.span})") + for e in ep.events[: args.samples]: + print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e))) + return 0 + + +def cmd_cast(args: argparse.Namespace) -> int: + ensure_dirs() + world = load_world(EXPORT_DIR) + found = threads.find_threads( + world, min_interactions=args.min_interactions, min_span=args.min_span + ) + if not found: + print("没有候选线索") + return 1 + if not 1 <= args.thread <= len(found): + print(f"候选只有 {len(found)} 条,--thread 超出范围") + return 1 + + thread = found[args.thread - 1] + _, planned = volume_mod.extended_plan(world, thread) + text, rows, suspicions = cast.build_cast( + world, thread, planned, with_bios=not args.no_bios + ) + if args.dry_run: + print(text[:2000]) + return 0 + + state = chronicle.State.load() + volume = state.volume if state.world == world.name else 1 + path = cast.write_cast(volume_dir(volume) / "cast.md", text) + print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}") + print(factcheck.report(suspicions) if suspicions else "专名校验:小传里的专名全部能在史料中找到。") + return 0 + + +def cmd_episode(args: argparse.Namespace) -> int: + ensure_dirs() + world = load_world(EXPORT_DIR) + found = threads.find_threads( + world, min_interactions=args.min_interactions, min_span=args.min_span + ) + if not found: + print("没有候选线索") + return 1 + + out_dir = None if args.dry_run else chapter_dir(1) + produced = volume_mod.produce_episode( + world, found, episode_index=args.index, out_dir=out_dir + ) + + print("选择依据:" + produced.selection.reason) + print(f"本集字数:{chronicle.cjk_length(produced.text)}") + if args.dry_run: + print("\n--- 试运行,不落盘 ---\n") + print(produced.text[:1600]) + return 0 + + print(f"已写入 {produced.path}") + if not args.no_push: + publish.ensure_repo() + head = publish.commit_and_push( + f"第 1 卷第 {produced.episode.index} 集(人物主线):{produced.episode.span}" + ) + print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}") + return 0 + + def cmd_run(args: argparse.Namespace) -> int: rc = cmd_index(args) if rc: @@ -163,6 +283,38 @@ def main(argv: list[str] | None = None) -> int: p = sub.add_parser("plan", help="显示切章方案") p.set_defaults(func=cmd_plan) + p = sub.add_parser("threads", help="挖掘人物线索候选") + p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS, + help="强边阈值:一对人物至少反复互动多少次(默认 5)") + p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN, + help="线索最小跨度年数(默认 30)") + p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)") + p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例") + p.set_defaults(func=cmd_threads) + + p = sub.add_parser("episodes", help="按人物线索切出剧集") + p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)") + p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS) + p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN) + p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件") + p.set_defaults(func=cmd_episodes) + + p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)") + p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)") + p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS) + p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN) + p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传") + p.add_argument("--dry-run", action="store_true", help="只打印不落盘") + p.set_defaults(func=cmd_cast) + + p = sub.add_parser("episode", help="按人物主线产出下一集") + p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)") + p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS) + p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN) + p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送") + p.add_argument("--no-push", action="store_true", help="落盘但不推送") + p.set_defaults(func=cmd_episode) + p = sub.add_parser("next", help="生成下一章") p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送") p.set_defaults(func=cmd_next) diff --git a/dfannals/config.py b/dfannals/config.py index 25ae117..34e179a 100644 --- a/dfannals/config.py +++ b/dfannals/config.py @@ -47,6 +47,9 @@ MODEL = "cn:deepseek-v4-pro" # 该模型是推理模型:思考与正文共用 max_tokens,必须留足 MAX_TOKENS_CHAPTER = 16000 MAX_TOKENS_UTIL = 4000 +# 凡是「喂长材料 + 要求结构化 JSON 输出」的调用(闸门自评、人物小传)都容易把预算 +# 烧在思考上而返回空正文,实测 4000 不够。 +MAX_TOKENS_STRUCTURED = 12000 # ---------------------------------------------------------------- 发布 GITEA_SSH_HOST = "124.222.29.26" @@ -61,7 +64,7 @@ SSH_KEY = HOME / ".ssh/id_ed25519" WORLD_SIZE = "Medium" WORLD_HISTORY_YEARS = 250 CHAPTER_MIN_CHARS = 800 -CHAPTER_MAX_CHARS = 1600 # 契约上限 1500,留出标点/换行余量 +CHAPTER_MAX_CHARS = 1500 # 严格按访谈定的 800–1500,不留“余量” @dataclass diff --git a/dfannals/episodes.py b/dfannals/episodes.py new file mode 100644 index 0000000..a2443b4 --- /dev/null +++ b/dfannals/episodes.py @@ -0,0 +1,192 @@ +"""因果链切集:把一条人物线索切成"一集一个完整回合"。 + +参数来自访谈决策: + +- 相邻事件间隔 >2 年即断开(比推荐值更紧,节奏更快) +- 短链只合并,不设线索总量下限 +- 链路过长按字数预算拆上下集 +- 每集正文预算落在 800–1500 字 + +字数预算用「事件条数 × 每条展开字数」估算。系数 110 是按此前实测校准的: +12 条史料事件生成的中文正文约 1100–1400 字。 +""" +from __future__ import annotations + +import math +from dataclasses import dataclass, field + +from dfannals.legends import Event, World +from dfannals.threads import Thread + +# 相邻事件间隔超过它即断开 +GAP_YEARS = 2 +# 每条史料在正文里大致展开的字数。实测区间约 85–115(同样 12 条事件, +# 两次生成分别得到 1026 字与 1388 字),取中值偏保守,避免把集拆得过碎。 +CHARS_PER_EVENT = 95 +MIN_CHARS = 800 # 每集预算下限 +MAX_CHARS = 1500 # 每集预算上限 + + +@dataclass +class Episode: + index: int + start_year: int + end_year: int + events: list[Event] = field(default_factory=list) + part: int = 1 + parts: int = 1 + + @property + def key(self) -> str: + base = f"{self.index:03d}-{self.start_year}-{self.end_year}" + return base if self.parts == 1 else f"{base}-p{self.part}" + + @property + def span(self) -> str: + if self.start_year == self.end_year: + return f"{self.start_year} 年" + return f"{self.start_year}–{self.end_year} 年" + + @property + def estimated_chars(self) -> int: + return len(self.events) * CHARS_PER_EVENT + + def in_budget(self) -> bool: + return MIN_CHARS <= self.estimated_chars <= MAX_CHARS or self.parts > 1 + + +def _min_events() -> int: + return max(1, math.ceil(MIN_CHARS / CHARS_PER_EVENT)) + + +def _max_events() -> int: + return max(1, MAX_CHARS // CHARS_PER_EVENT) + + +def _split_chains(events: list[Event], gap_years: int) -> list[list[Event]]: + """按时间间隔把事件串切成因果链。""" + chains: list[list[Event]] = [] + current: list[Event] = [] + prev_year: int | None = None + for e in events: + if current and prev_year is not None and e.year - prev_year > gap_years: + chains.append(current) + current = [] + current.append(e) + prev_year = e.year + if current: + chains.append(current) + return chains + + +def _merge_short(chains: list[list[Event]], min_events: int) -> list[list[Event]]: + """短链只合并:攒够预算就收一集,尾部不足的并入上一集。""" + merged: list[list[Event]] = [] + buffer: list[Event] = [] + for chain in chains: + buffer.extend(chain) + if len(buffer) >= min_events: + merged.append(buffer) + buffer = [] + if buffer: + if merged: + merged[-1].extend(buffer) + else: + merged = [buffer] + return merged + + +def _split_long(chain: list[Event]) -> list[list[Event]]: + """过长链路按预算拆成若干段(对应正文的上/下集)。 + + 注意不能均分:14 条事件均分成 7+7 会得到两个 770 字的碎片, + 两端都跌破 800 字下限。所以先取最小段数,再确保每段不低于下限。 + """ + n = len(chain) + max_events = _max_events() + min_events = _min_events() + if n <= max_events: + return [chain] + parts = math.ceil(n / max_events) + while parts > 1 and math.ceil(n / parts) < min_events: + parts -= 1 + size = math.ceil(n / parts) + return [chain[i:i + size] for i in range(0, n, size)] + + +def plan_episodes( + thread: Thread, + gap_years: int = GAP_YEARS, + min_chars: int = MIN_CHARS, + max_chars: int = MAX_CHARS, +) -> list[Episode]: + """给定主角组,切出剧集序列。""" + if not thread.events: + return [] + + chains = _split_chains(thread.events, gap_years) + merged = _merge_short(chains, _min_events()) + + episodes: list[Episode] = [] + for chain in merged: + segments = _split_long(chain) + for i, segment in enumerate(segments, 1): + episodes.append( + Episode( + index=len(episodes) + 1, + start_year=segment[0].year, + end_year=segment[-1].year, + events=segment, + part=i, + parts=len(segments), + ) + ) + return episodes + + +def cast_of(world: World, thread: Thread, episode: Episode) -> tuple[list[int], list[int]]: + """返回 (本集出场的主角, 本集出现的对手/外人)。""" + counter: dict[int, int] = {} + for e in episode.events: + for fid in e.figure_ids(): + counter[fid] = counter.get(fid, 0) + 1 + members = set(thread.members) + heros = sorted((f for f in counter if f in members), key=lambda f: -counter[f]) + others = sorted((f for f in counter if f not in members), key=lambda f: -counter[f]) + return heros, others + + +def render_plan(world: World, thread: Thread, episodes: list[Episode], top_others: int = 3) -> str: + """剧集清单:集号、年份范围、事件条数、主角与对手。""" + lines = [ + f"线索主角:{'、'.join(world.figure_name(m) for m in thread.members)}", + f"线索跨度:{thread.label} | 事件 {len(thread.events)} 条 | 规划 {len(episodes)} 集", + "", + ] + for ep in episodes: + heros, others = cast_of(world, thread, ep) + hero_names = "、".join(world.figure_name(h) for h in heros[:4]) or "(本集无主角出场)" + other_names = "、".join(world.figure_name(o) for o in others[:top_others]) + budget = f"{ep.estimated_chars} 字估算" + mark = "" if MIN_CHARS <= ep.estimated_chars <= MAX_CHARS else " ⚠ 超出预算" + lines.append( + f" 第 {ep.index:2d} 集({ep.span})| {len(ep.events):2d} 条事件 | {budget}{mark}" + + (f" | 上/下:{ep.part}/{ep.parts}" if ep.parts > 1 else "") + ) + lines.append(f" 主角:{hero_names}") + if other_names: + lines.append(f" 对手/外人:{other_names}") + return "\n".join(lines) + + +def budget_report(episodes: list[Episode]) -> str: + if not episodes: + return "无剧集" + inside = sum(1 for e in episodes if MIN_CHARS <= e.estimated_chars <= MAX_CHARS) + sizes = [len(e.events) for e in episodes] + return ( + f"共 {len(episodes)} 集;事件条数 {min(sizes)}–{max(sizes)};" + f"估算字数 {min(e.estimated_chars for e in episodes)}–{max(e.estimated_chars for e in episodes)};" + f"落在 {MIN_CHARS}–{MAX_CHARS} 预算内的 {inside}/{len(episodes)} 集" + "(估算按每条约 95 字,真实字数在生成时会再校正)" + ) diff --git a/dfannals/gate.py b/dfannals/gate.py new file mode 100644 index 0000000..15dcf77 --- /dev/null +++ b/dfannals/gate.py @@ -0,0 +1,223 @@ +"""5 维度自评闸门:判断一稿是不是"流水账",并决定重写还是换线索。 + +访谈决策: +- 5 个维度各 0–5 分,总分 <12 判平淡 +- 不合格先重写本集一次,仍不合格才换线索 +- 放弃原因写进仓库 notes/ 附录 +- 设失败上限,防止「重写→换线索」空转 + +为什么要这道闸门:上一版章节读起来像流水账,而"平淡"是模型最容易自我感觉良好的东西。 +因此评分必须结构化、维度必须问得具体,且要有一次反例自检证明它真的能判不合格。 +""" +from __future__ import annotations + +import json +import re +from collections.abc import Callable +from dataclasses import dataclass, field +from pathlib import Path + +from dfannals.config import MAX_TOKENS_STRUCTURED, MODEL, Gateway +from dfannals.llm import chat + +# 维度:键、标签、问什么 +DIMENSIONS: tuple[tuple[str, str, str], ...] = ( + ("goal", "主角目标", "主角在本集里有没有明确的目标,并为此做出主动选择?全程被动得分低"), + ("escalation", "冲突升级", "冲突是否逐级升级,而不是同一件事反复发生?"), + ("turning", "真正转折", "本集有没有真正的转折(立场改变、关系破裂、意外后果)?"), + ("rival", "对手具体", "对手是不是一个具体的人,有自己的动机?只是数字或人群得分低"), + ("beyond_bodycount", "超越战斗计数", "抛开对战次数,本集还有没有可讲的内容(算计、情感、抉择)?"), +) + +THRESHOLD = 12 # 总分低于它判平淡 +MAX_SCORE = 5 # 每个维度满分 +EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE) + +# 失败上限:防止反复重写/换线空转 +MAX_REWRITES = 1 # 每集最多重写次数(访谈决定) +MAX_CANDIDATES = 3 # 最多尝试几条线索 + + +@dataclass +class Verdict: + total: int + scores: dict[str, int] + reason: str + + @property + def passed(self) -> bool: + return self.total >= THRESHOLD + + def weak(self, limit: int = 2) -> list[str]: + labels = {k: label for k, label, _ in DIMENSIONS} + ranked = sorted(self.scores.items(), key=lambda kv: kv[1]) + return [f"{labels.get(k, k)}({v})" for k, v in ranked[:limit]] + + def render(self) -> str: + parts = "、".join(f"{label}={self.scores.get(key, '?')}" for key, label, _ in DIMENSIONS) + verdict = "通过" if self.passed else "判平淡" + return f"{verdict} 总分 {self.total}/{len(DIMENSIONS) * MAX_SCORE}({parts})|{self.reason}" + + +DIMENSION_BLOCK = "\n".join( + f'- "{key}": {hint}(0–5 分)' for key, _label, hint in DIMENSIONS +) + +EVAL_INSTRUCTIONS = f"""你是严格的文学编辑,只做一件事:判断下面这集稿子是不是"流水账"。 + +按 5 个维度各打 0–5 分: +{DIMENSION_BLOCK} + +打分要求: +- 不要客气,不要因为文笔通顺就给分;只看叙事本身有没有戏。 +- 主角全程被动、只是"发生了很多事",goal 与 escalation 必须给低分。 +- 通篇只是同一类事件重复(例如反复对战、反复被杀),turning 与 beyond_bodycount 必须给低分。 + +只输出 JSON,不要解释、不要 Markdown 代码块: +{{"goal": 0-5, "escalation": 0-5, "turning": 0-5, "rival": 0-5, "beyond_bodycount": 0-5, "reason": "一句话说明最弱的地方"}} +""" + + +def build_eval_prompt(chapter_text: str) -> str: + """拼接自评提示词。 + + 刻意不用 str.format:提示词里的 JSON 例子含大括号,用 format 会被当成占位符 + (曾直接报 KeyError: '"goal"')。 + """ + return EVAL_INSTRUCTIONS + "\n待评稿件:\n---\n" + chapter_text + "\n---\n" + + +def _parse(raw: str) -> dict: + text = EDGE_RE.sub("", raw).strip() + start, end = text.find("{"), text.rfind("}") + if start < 0 or end <= start: + raise ValueError(f"自评没有返回 JSON:{raw[:200]}") + try: + data = json.loads(text[start:end + 1]) + except json.JSONDecodeError as exc: + raise ValueError(f"自评返回的 JSON 无法解析:{exc}|原文:{raw[:200]}") from None + if not isinstance(data, dict): + raise ValueError(f"自评返回的不是对象:{type(data).__name__}") + return data + + +def evaluate( + chapter_text: str, + *, + gateway: Gateway | None = None, + model: str = MODEL, +) -> Verdict: + """让模型按 5 个维度打分。返回结构化结果。""" + reply = chat( + [{"role": "user", "content": build_eval_prompt(chapter_text)}], + gateway=gateway, + model=model, + max_tokens=MAX_TOKENS_STRUCTURED, + temperature=0.0, + ) + data = _parse(reply.content) + + scores: dict[str, int] = {} + for key, _label, _hint in DIMENSIONS: + value = data.get(key) + if not isinstance(value, (int, float)): + raise ValueError(f"自评维度 {key} 缺失或非数字:{data.get(key)!r}") + scores[key] = max(0, min(MAX_SCORE, int(value))) + + reason = str(data.get("reason", "")).strip() or "(未给出理由)" + return Verdict(total=sum(scores.values()), scores=scores, reason=reason) + + +@dataclass +class GateRun: + """一次闸门流程的完整记录。""" + episode_key: str + verdicts: list[Verdict] = field(default_factory=list) + accepted: bool = False + rewrites: int = 0 + + @property + def final(self) -> Verdict | None: + return self.verdicts[-1] if self.verdicts else None + + def render(self) -> str: + lines = [f"闸门:{self.episode_key},重写 {self.rewrites} 次," + + ("通过" if self.accepted else "未通过")] + for i, v in enumerate(self.verdicts, 1): + lines.append(f" 第 {i} 稿:{v.render()}") + return "\n".join(lines) + + +def gate_episode( + episode_key: str, + write: Callable[[int, Verdict | None], str], + *, + max_rewrites: int = MAX_REWRITES, + evaluate_fn: Callable[[str], Verdict] = evaluate, + on_event: Callable[[str], None] | None = None, +) -> tuple[str, GateRun]: + """写 → 自评 → 不合格则重写一次 → 仍不合格则交给调用方换线索。 + + ``write(attempt, previous_verdict)``:attempt 为 0 表示初稿, + 大于 0 表示重写,previous_verdict 是上一稿的评分(供重写时指出问题)。 + """ + run = GateRun(episode_key=episode_key) + text = write(0, None) + verdict = evaluate_fn(text) + run.verdicts.append(verdict) + + while not verdict.passed and run.rewrites < max_rewrites: + run.rewrites += 1 + if on_event: + on_event(f"{episode_key} 第 {run.rewrites} 次重写:{verdict.weak()}") + text = write(run.rewrites, verdict) + verdict = evaluate_fn(text) + run.verdicts.append(verdict) + + run.accepted = verdict.passed + return text, run + + +NOTES_DIR_NAME = "notes" +SWITCH_LOG = "switched-threads.md" + + +def append_switch_note( + project_dir: Path, + *, + world_name: str, + members: list[str], + span: str, + verdict: Verdict | None, + reason: str, +) -> Path: + """把"为什么放弃这条线索"写进仓库 notes/ 附录(访谈决定:公开可查)。""" + notes = project_dir / NOTES_DIR_NAME + notes.mkdir(parents=True, exist_ok=True) + path = notes / SWITCH_LOG + header = "# 被放弃的线索\n\n记录每一组挑出来、但写不成故事的线索,以及放弃原因。\n" + if not path.is_file(): + path.write_text(header, encoding="utf-8") + + score = verdict.render() if verdict else "(未评分)" + entry = ( + f"\n## {world_name}|{'、'.join(members)}\n\n" + f"- 线索跨度:{span}\n" + f"- 放弃原因:{reason}\n" + f"- 自评:{score}\n" + ) + with path.open("a", encoding="utf-8") as fh: + fh.write(entry) + return path + + +def append_attempt_note(project_dir: Path, text: str) -> Path: + """一般性记录(例如失败上限触发)。""" + notes = project_dir / NOTES_DIR_NAME + notes.mkdir(parents=True, exist_ok=True) + path = notes / SWITCH_LOG + if not path.is_file(): + path.write_text("# 被放弃的线索\n\n", encoding="utf-8") + with path.open("a", encoding="utf-8") as fh: + fh.write(text if text.startswith("\n") else "\n" + text) + return path diff --git a/dfannals/legends.py b/dfannals/legends.py index 117ce0c..bde8b25 100644 --- a/dfannals/legends.py +++ b/dfannals/legends.py @@ -14,26 +14,29 @@ from typing import Any from defusedxml.ElementTree import iterparse -# 事件里指向其他实体的字段 → 指向哪类对象 -ID_FIELDS: dict[str, str] = { - "hfid": "figure", - "hist_figure_id": "figure", - "slayer_hfid": "figure", - "slayer_item_id": "artifact", - "target_hfid": "figure", - "source_hfid": "figure", - "site_id": "site", - "site_civ_id": "entity", - "entity_id": "entity", - "attacker_civ_id": "entity", - "defender_civ_id": "entity", - "artifact_id": "artifact", - "region_id": "region", - "feature_layer_id": "region", - "deity": "figure", - "worshipper_hfid": "figure", - "creature_id": "creature", -} +# 字段分类规则来自对真实史料全部 225 种字段的穷举: +# 含 "hfid" 的字段全部指向历史人物(共 40+ 个:hfid / hfid_target / group_1_hfid / +# slayer_hfid / snatcher_hfid / seeker_hfid / wounder_hfid / teacher_hfid / +# conspirator_hfid ...);而 target_enid、identity_id、master_wcid、slayer_item_id 不是。 +# 旧版手工维护的字典只列了 4 个人物字段,导致关系信息在素材里被丢掉。 +_ENTITY_SUFFIXES = ("_entity_id", "_civ_id", "_enid") +_REGION_FIELDS = {"region_id", "feature_layer_id", "subregion_id"} + + +def kind_of(tag: str) -> str | None: + """事件字段名 → 它指向的对象类别(非引用类字段返回 None)。""" + t = tag.lower() + if "hfid" in t or t == "deity" or t.endswith("_deity"): + return "figure" + if "artifact" in t: + return "artifact" + if t == "site_id" or t.endswith("_site_id"): + return "site" + if t in ("entity_id", "target_enid") or t.endswith(_ENTITY_SUFFIXES): + return "entity" + if t in _REGION_FIELDS: + return "region" + return None @dataclass @@ -81,6 +84,36 @@ class Event: vals = self.fields.get(tag) return vals[0] if vals else "" + def ids_of_kind(self, kind: str) -> list[int]: + """本事件中指向该类对象的全部有效 id(去重、剔除 -1 占位)。""" + out: list[int] = [] + for tag, vals in self.fields.items(): + if kind_of(tag) != kind: + continue + for raw in vals: + i = _int_or_none(raw) + if i is not None and i >= 0 and i not in out: + out.append(i) + return out + + def figure_ids(self) -> list[int]: + return self.ids_of_kind("figure") + + def site_ids(self) -> list[int]: + return self.ids_of_kind("site") + + def entity_ids(self) -> list[int]: + return self.ids_of_kind("entity") + + def field_kinds(self) -> dict[str, list[int]]: + """{类别: [id, ...]},供渲染与建图使用。""" + out: dict[str, list[int]] = {} + for kind in ("figure", "site", "entity", "artifact", "region"): + ids = self.ids_of_kind(kind) + if ids: + out[kind] = ids + return out + @dataclass class Entity: @@ -138,6 +171,34 @@ class World: st = self.sites.get(sid) return pretty(st.name) if st and st.name else f"Site#{sid}" + def render_refs(self, event: Event) -> list[str]: + """把一条事件里的引用字段渲染成 ``tag=名字``。 + + 保留字段名是有意的:slayer_hfid=谁 与 hfid=谁 语义完全不同, + 丢掉字段名就等于丢掉了“谁对谁做了什么”。 + 负 id(DF 的“无此项”占位)与查不到名字的字段直接跳过。 + """ + out: list[str] = [] + for tag, vals in event.fields.items(): + kind = kind_of(tag) + if kind is None: + continue + for raw in vals: + i = _int_or_none(raw) + if i is None or i < 0: + continue + if kind == "figure": + name = self.figure_name(i) + elif kind == "site": + name = self.site_name(i) + elif kind == "entity": + name = self.entity_name(i) + else: + name = "" + if name: + out.append(f"{tag}={name}") + return out + def event_years(self) -> tuple[int, int]: years = [e.year for e in self.events if e.year] return (min(years), max(years)) if years else (0, 0) diff --git a/dfannals/publish.py b/dfannals/publish.py index 3383c64..10e568e 100644 --- a/dfannals/publish.py +++ b/dfannals/publish.py @@ -32,6 +32,7 @@ def _git(*args: str, check: bool = True) -> subprocess.CompletedProcess: env={**_base_env(), **env}, capture_output=True, text=True, + check=False, # 下面是自定义错误处理,故意不用 check=True ) if check and proc.returncode != 0: raise RuntimeError(f"git {' '.join(args)} 失败:{proc.stderr.strip() or proc.stdout.strip()}") diff --git a/dfannals/score.py b/dfannals/score.py index e335dfa..128b0ae 100644 --- a/dfannals/score.py +++ b/dfannals/score.py @@ -93,22 +93,23 @@ def category(event: Event) -> str: def prominence(world: World) -> Counter[int]: - """每个人物被卷入的事件次数——用来衡量他在史料里有多重要。""" + """每个人物被卷入的事件次数——用来衡量他在史料里有多重要。 + + 改用字段分类器后覆盖了 40+ 个人物字段(旧版只数 6 个),数值尺度整体变大, + 因此 PROMINENT 阈值必须按新尺度重校。结果缓存在 World 上,避免反复重算。 + """ + cached = getattr(world, "_prominence", None) + if cached is not None: + return cached counts: Counter[int] = Counter() for e in world.events: - seen: set[int] = set() - for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity", "worshipper_hfid"): - for fid in e.ids(tag): - seen.add(fid) - counts.update(seen) + counts.update(e.figure_ids()) + world._prominence = counts # type: ignore[attr-defined] return counts def _participants(event: Event) -> list[int]: - out: list[int] = [] - for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity"): - out.extend(event.ids(tag)) - return out + return event.figure_ids() # 人物重要度阈值:参与事件数达到这个量级才算"要角" diff --git a/dfannals/slice.py b/dfannals/slice.py index 1824ca9..8c11ee0 100644 --- a/dfannals/slice.py +++ b/dfannals/slice.py @@ -130,23 +130,10 @@ def material(world: World, chapter: Chapter, max_figures: int = 40) -> str: participation: dict[int, int] = {} for e in chapter.events: - chunks = [f"{e.year}年", e.type] - for tag, kind in lg.ID_FIELDS.items(): - for i in e.ids(tag): - # id 为负是“无此项”,不要渲染成 HF#-1 这类噪音 - if kind == "figure": - name = world.figure_name(i) - elif kind == "site": - name = world.site_name(i) - elif kind == "entity": - name = world.entity_name(i) - else: - name = "" - if not name: - continue - chunks.append(f"{tag}={name}") - if kind == "figure": - participation[i] = participation.get(i, 0) + 1 + # render_refs 会带出双方(比如 slayer_hfid=谁),而不是只剩单方面 hfid + chunks = [f"{e.year}年", e.type, *world.render_refs(e)] + for i in e.figure_ids(): + participation[i] = participation.get(i, 0) + 1 extra = [] for field_name in ("state", "reason", "circumstance"): v = e.text(field_name) @@ -184,8 +171,10 @@ def overview(world: World, chapters: list[Chapter], limit: int = 10) -> str: lo, hi = world.event_years() lines = [ f"世界:{world.name}" + (f"({world.altname})" if world.altname else ""), - f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | " - f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}", + ( + f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | " + f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}" + ), f"规划章节:{len(chapters)} 章(每章约 {YEARS_PER_CHAPTER} 年 / 最多 {EVENTS_PER_CHAPTER} 条史料)", ] for c in chapters[:limit]: diff --git a/dfannals/threads.py b/dfannals/threads.py new file mode 100644 index 0000000..b996f74 --- /dev/null +++ b/dfannals/threads.py @@ -0,0 +1,287 @@ +"""人物线索挖掘:从史料里找出「关系密切、有戏」的 3–6 人小圈子。 + +设计依据全部来自对 Mon Sagus 真实数据(403853 条事件)的实测: + +1. **不能用裸互动次数排序。** 实测排名第一的簇 49 次互动里有 35 次是 + ``hf relationship denied``——反复请求建立关系、反复被拒的循环,是统计噪声。 + 因此整类剔除(访谈已定)。 +2. **同一对同一类型反复出现也不是故事。** 修掉上一条之后,候选又变成 + ``对战×27`` 这种「同一对人反复互殴」的机械循环。因此在计数时对 + (人物对, 事件类型) 做封顶,避免刷量取胜——这也正是访谈定的 + 「转折/冲突多样性为主、互动次数为次」。 +3. **主线必须是"人"。** 种族分布实测:ELF/GOBLIN/DWARF/HUMAN/KOBOLD 五族加 + ``*_MAN`` 人形族占全部历史人物的 96.5%,其余 400 多种是夜行怪、野兽、 + 泰坦、实验体。所以这里用白名单,而不是越列越长的黑名单。 +4. 强边阈值 ≥5 次反复互动;簇规模 3–6 人;跨度 ≥30 年。 + 实测:阈值提到 8 会一条候选都不剩,5 是正确档位。 +""" +from __future__ import annotations + +import json +from collections import Counter, defaultdict +from dataclasses import dataclass, field +from pathlib import Path + +from dfannals.legends import Event, World + +# 事件类型 → (权重, 中文标签, 是否算转折点) +SIGNALS: dict[str, tuple[int, str, bool]] = { + "hf simple battle event": (1, "对战", False), + "add hf hf link": (1, "结缘", False), + "hfs formed reputation relationship": (1, "结缘", False), + "competition": (1, "竞争", False), + "hf wounded": (2, "搏杀", True), + "hf confronted": (2, "对峙", True), + "hf interrogated": (2, "审讯", True), + "failed intrigue corruption": (2, "阴谋", True), + "remove hf hf link": (3, "反目", True), + "hf abducted": (3, "绑架", True), + "hf convicted": (3, "定罪", True), + "entity persecuted": (3, "迫害", True), + "hf enslaved": (3, "奴役", True), + "hf ransomed": (3, "贖金", True), + "entity overthrown": (3, "推翻", True), + "failed frame attempt": (3, "构陷", True), +} + +# 明确剔除:重复性请求,实测会刷满排序(访谈已定) +EXCLUDED_SIGNALS = ("hf relationship denied",) + +# 文明种族白名单:实测覆盖 96.5% 的历史人物 +CIVILIZED_RACES = frozenset({"DWARF", "ELF", "HUMAN", "GOBLIN", "KOBOLD"}) + +DEFAULT_MIN_INTERACTIONS = 5 # 强边阈值:≥5 次反复互动 +DEFAULT_SIZE_RANGE = (3, 6) +DEFAULT_MIN_SPAN = 30 # 跨度 ≥30 年才撑得起连载 +PAIR_TYPE_CAP = 4 # 同一对、同一类型最多计 4 次 +MIN_SIGNAL_KINDS = 2 # 至少两种不同信号,否则只是单一类型的重复 +INTERACTION_CAP = 40 # 互动总量封顶,防止刷量取胜 + + +@dataclass +class Edge: + a: int + b: int + raw: int = 0 + types: Counter = field(default_factory=Counter) + first_year: int = 0 + last_year: int = 0 + + @property + def effective(self) -> int: + """封顶后的有效互动次数。""" + return sum(min(n, PAIR_TYPE_CAP) for n in self.types.values()) + + @property + def turns(self) -> int: + """封顶后的转折点次数。""" + return sum(min(n, PAIR_TYPE_CAP) for t, n in self.types.items() if SIGNALS[t][2]) + + +@dataclass +class Thread: + members: list[int] + interactions: int + turns: int + span: int + first_year: int + last_year: int + types: Counter + races: Counter + non_person_races: list[str] + events: list[Event] + + @property + def kinds(self) -> int: + return len(self.types) + + @property + def score(self) -> int: + """转折为主、多样性次之、互动次数封顶计入(访谈定的排序原则)。""" + return self.turns * 10 + self.kinds * 6 + min(self.interactions, INTERACTION_CAP) + + @property + def label(self) -> str: + return f"{self.first_year}–{self.last_year}({self.span} 年)" + + def signal_line(self) -> str: + return "、".join(f"{SIGNALS[t][1]}×{n}" for t, n in self.types.most_common() if t in SIGNALS) + + +def is_person(race: str) -> bool: + """是否属于"可作为主角的人":文明种族或人形族(*_MAN)。""" + r = (race or "").upper() + return r in CIVILIZED_RACES or r.endswith("_MAN") + + +def build_edges(world: World) -> dict[tuple[int, int], Edge]: + """按有叙事含义的信号建立人物之间的边。""" + edges: dict[tuple[int, int], Edge] = {} + for e in world.events: + if e.type not in SIGNALS: + continue + people = e.figure_ids() + if len(people) < 2: + continue + for i in range(len(people)): + for j in range(i + 1, len(people)): + key = (min(people[i], people[j]), max(people[i], people[j])) + edge = edges.get(key) + if edge is None: + edge = Edge(a=key[0], b=key[1], first_year=e.year, last_year=e.year) + edges[key] = edge + edge.raw += 1 + edge.types[e.type] += 1 + edge.first_year = min(edge.first_year, e.year) + edge.last_year = max(edge.last_year, e.year) + return edges + + +def find_threads( + world: World, + min_interactions: int = DEFAULT_MIN_INTERACTIONS, + size_range: tuple[int, int] = DEFAULT_SIZE_RANGE, + min_span: int = DEFAULT_MIN_SPAN, + require_people: bool = True, + min_kinds: int = MIN_SIGNAL_KINDS, + limit: int = 0, +) -> list[Thread]: + """返回按剧情张力排序的候选线索。""" + edges = build_edges(world) + strong = {k: v for k, v in edges.items() if v.effective >= min_interactions} + + parent: dict[int, int] = {} + + def find(x: int) -> int: + parent.setdefault(x, x) + while parent[x] != x: + parent[x] = parent[parent[x]] + x = parent[x] + return x + + for (a, b) in strong: + ra, rb = find(a), find(b) + if ra != rb: + parent[ra] = rb + + members_of: dict[int, set[int]] = defaultdict(set) + for x in parent: + members_of[find(x)].add(x) + + events_by_pair: dict[tuple[int, int], list[Event]] = defaultdict(list) + for e in world.events: + if e.type not in SIGNALS: + continue + people = e.figure_ids() + if len(people) < 2: + continue + for i in range(len(people)): + for j in range(i + 1, len(people)): + events_by_pair[(min(people[i], people[j]), max(people[i], people[j]))].append(e) + + lo_size, hi_size = size_range + threads: list[Thread] = [] + for members in members_of.values(): + if not (lo_size <= len(members) <= hi_size): + continue + + races = Counter(world.figures[m].race for m in members if m in world.figures) + if require_people and not all(is_person(r) for r in races): + continue + + inner = {k: v for k, v in strong.items() if k[0] in members and k[1] in members} + if not inner: + continue + + types: Counter = Counter() + for v in inner.values(): + types.update({t: min(n, PAIR_TYPE_CAP) for t, n in v.types.items()}) + if len(types) < min_kinds: + continue + + first = min(v.first_year for v in inner.values()) + last = max(v.last_year for v in inner.values()) + if last - first < min_span: + continue + + evs: list[Event] = [] + seen_ids: set[int] = set() + for k in inner: + for e in events_by_pair.get(k, []): + if e.id not in seen_ids: + seen_ids.add(e.id) + evs.append(e) + evs.sort(key=lambda e: (e.year, e.seconds72, e.id)) + + threads.append( + Thread( + members=sorted(members), + interactions=sum(v.effective for v in inner.values()), + turns=sum(v.turns for v in inner.values()), + span=last - first, + first_year=first, + last_year=last, + types=types, + races=races, + non_person_races=sorted(r for r in races if not is_person(r)), + events=evs, + ) + ) + + threads.sort(key=lambda t: (-t.score, -t.span)) + return threads[:limit] if limit else threads + + +def render_report(world: World, threads: list[Thread], per_thread: int = 3) -> str: + """给人看的候选明细。""" + lines = [ + f"候选线索 {len(threads)} 条(转折为主、多样性次之;同一对同类事件已封顶)", + "筛选:强边 ≥5 次互动、3–6 人、跨度 ≥30 年、含巨兽的簇已剔除", + "", + ] + for i, t in enumerate(threads, 1): + races = "、".join(f"{r}×{n}" for r, n in t.races.most_common()) + monsters = "、".join(t.non_person_races) if t.non_person_races else "无" + lines.append( + f"{i:2d}. 得分 {t.score:4d} | {len(t.members)} 人 | 有效互动 {t.interactions:3d} | " + f"转折 {t.turns:2d} | 信号 {t.kinds} 种 | 跨度 {t.label} | 巨兽:{monsters}" + ) + lines.append(f" 种族:{races}") + lines.append(f" 信号:{t.signal_line()}") + lines.append(f" 成员:{'、'.join(world.figure_name(m) for m in t.members)}") + for e in t.events[:per_thread]: + lines.append(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e))) + lines.append("") + return "\n".join(lines) + + +def to_json(world: World, threads: list[Thread]) -> dict: + return { + "world": world.name, + "candidates": [ + { + "rank": i, + "score": t.score, + "members": [ + {"id": m, "name": world.figure_name(m), + "race": world.figures[m].race if m in world.figures else ""} + for m in t.members + ], + "interactions": t.interactions, + "turns": t.turns, + "kinds": t.kinds, + "span": t.span, + "first_year": t.first_year, + "last_year": t.last_year, + "signals": {SIGNALS[k][1]: v for k, v in t.types.items() if k in SIGNALS}, + "event_ids": [e.id for e in t.events], + } + for i, t in enumerate(threads, 1) + ], + } + + +def save_json(world: World, threads: list[Thread], path: Path) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(to_json(world, threads), ensure_ascii=False, indent=2), encoding="utf-8") + return path diff --git a/dfannals/timeline.py b/dfannals/timeline.py index d0e4c1c..e2134fc 100644 --- a/dfannals/timeline.py +++ b/dfannals/timeline.py @@ -19,19 +19,7 @@ category = score.category def event_line(world: World, e: Event) -> str: """把一条事件压成一行可读文字。""" - chunks: list[str] = [e.type] - for tag, kind in lg.ID_FIELDS.items(): - for i in e.ids(tag): - if kind == "figure": - name = world.figure_name(i) - elif kind == "site": - name = world.site_name(i) - elif kind == "entity": - name = world.entity_name(i) - else: - name = "" - if name: - chunks.append(f"{tag}={name}") + chunks: list[str] = [e.type, *world.render_refs(e)] return f"- **{e.year}** [{category(e)}] " + " · ".join(chunks) @@ -51,8 +39,10 @@ def build_timeline(world: World, min_score: int = 70, per_decade: int = 14) -> s f"- 事件总数:{len(world.events)},其中重要事件 {len(important)}", f"- 历史人物 {len(world.figures)} 位 · 文明与组织 {len(world.entities)} 个 · 地点 {len(world.sites)} 处", "", - "> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;" - "完整数据留在本地,不入库。", + ( + "> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;" + "完整数据留在本地,不入库。" + ), "", ] for decade in sorted(buckets): @@ -72,13 +62,7 @@ def build_figures(world: World, top: int = 120) -> str: """人物索引:按参与事件数排序,给出身份与生卒。""" involvement: Counter[int] = Counter() for e in world.events: - seen: set[int] = set() - for tag, kind in lg.ID_FIELDS.items(): - if kind != "figure": - continue - for i in e.ids(tag): - seen.add(i) - involvement.update(seen) + involvement.update(e.figure_ids()) lines = [ f"# {world.name} · 人物索引", diff --git a/dfannals/volume.py b/dfannals/volume.py new file mode 100644 index 0000000..471ea2f --- /dev/null +++ b/dfannals/volume.py @@ -0,0 +1,336 @@ +"""人物主线写作与编排:选主角组 → 切集 → 贴着人物写 → 过闸门 → 落盘。 + +与旧的 chronicle.py 的区别:那里的主语是"年代",一章按 10 年窗口罗列事件; +这里的主语是"人",一集是主角线上一段完整回合,世界大事只在影响到主角时提及。 +""" +from __future__ import annotations + +import json +from collections import Counter +from collections.abc import Callable +from dataclasses import dataclass, replace +from pathlib import Path + +from dfannals import episodes as ep_mod +from dfannals import gate +from dfannals.cast import collect_rows +from dfannals.chronicle import cjk_length +from dfannals.config import ( + CHAPTER_MAX_CHARS, + CHAPTER_MIN_CHARS, + MAX_TOKENS_CHAPTER, + PROJECT_DIR, + PROMPT_DIR, +) +from dfannals.legends import Event, World +from dfannals.llm import chat +from dfannals.threads import SIGNALS, Thread + +SPEC_FILE = PROMPT_DIR / "biographer.md" +TAIL_CHARS = 600 +SELECTION_FILE = PROJECT_DIR / "data" / "selection.json" + + +@dataclass +class Selection: + rank: int + thread: Thread + reason: str + + +@dataclass +class Produced: + selection: Selection + episode: ep_mod.Episode + text: str + run: gate.GateRun + path: Path | None = None + + +@dataclass +class Extended: + """主角参与的扩展素材。""" + events: list[Event] + elided: Counter + + +PER_KEY_CAP = 3 # 同一组人 + 同一类型,每集素材最多保留几次 + + +def expand_thread_events(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP) -> Extended: + """把素材从“成员之间”扩到“主角参与的全部有意义事件”。 + + 实测:某 3 人线索成员之间只有 14 条事件(只能切 1 集),但把他们的对外活动 + (构陷、对战、结义、定罪……)算进来有 136 条,足够支撑一卷。 + + 代价是重复:这 136 条里有 64 条是同一类“构陷失败”。所以同一组人 + 同一类型 + 最多保留 per_key_cap 条,其余计入 elided 供写作时一笔带过——否则又会变流水账。 + """ + members = set(thread.members) + seen: Counter = Counter() + keep: list[Event] = [] + elided: Counter = Counter() + + relevant = [ + e for e in world.events + if e.type in SIGNALS and (members & set(e.figure_ids())) + ] + relevant.sort(key=lambda e: (e.year, e.seconds72, e.id)) + + for e in relevant: + key = (e.type, tuple(sorted(e.figure_ids()))) + seen[key] += 1 + if seen[key] <= per_key_cap: + keep.append(e) + else: + elided[e.type] += 1 + return Extended(events=keep, elided=elided) + + +def extended_plan(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP + ) -> tuple[Extended, list[ep_mod.Episode]]: + """用扩展素材切集(保持切集规则不变)。""" + ext = expand_thread_events(world, thread, per_key_cap=per_key_cap) + planned = ep_mod.plan_episodes(replace(thread, events=ext.events)) + return ext, planned + + +def select_thread(world: World, candidates: list[Thread], rank: int = 1) -> Selection: + """按挖掘器排序取第 rank 条,并把选择依据写成可读理由。""" + thread = candidates[rank - 1] + names = "、".join(world.figure_name(m) for m in thread.members) + reason = ( + f"挖掘器排序第 {rank} 名(转折为主、互动封顶):" + f"{len(thread.members)} 人、有效互动 {thread.interactions}、转折 {thread.turns}、" + f"信号 {thread.kinds} 种({thread.signal_line()})、跨度 {thread.label};成员:{names}" + ) + return Selection(rank=rank, thread=thread, reason=reason) + + +def save_selection(selection: Selection, world: World, path: Path = SELECTION_FILE) -> Path: + """把主角组的选定依据落到本地(不入库,供追溯)。""" + path.parent.mkdir(parents=True, exist_ok=True) + thread = selection.thread + path.write_text( + json.dumps( + { + "world": world.name, + "rank": selection.rank, + "reason": selection.reason, + "score": thread.score, + "members": [ + {"id": m, "name": world.figure_name(m), + "race": world.figures[m].race if m in world.figures else ""} + for m in thread.members + ], + "signals": dict(thread.types), + "span": thread.label, + }, + ensure_ascii=False, + indent=2, + ), + encoding="utf-8", + ) + return path + + +def render_material( + world: World, + thread: Thread, + episode: ep_mod.Episode, + total_episodes: int, + prev_tail: str = "", + elided: Counter | None = None, +) -> str: + """把一集素材渲染给传记作者:主角是谁、发生了什么、对手是谁。""" + rows = {r.fid: r for r in collect_rows(world, thread, [episode], max_others=6)} + heros, others = ep_mod.cast_of(world, thread, episode) + + lines = [ + f"## 本集:第 {episode.index}/{total_episodes} 集,{episode.span}", + "", + "### 主角(本集要写的就是他们)", + "", + ] + for fid in heros: + row = rows.get(fid) + if row: + lines.append(f"- {row.name}|{row.race}|{row.span}|身份:{row.identity or '不详'}" + f"|本集出场 {row.appearances} 次") + else: + lines.append(f"- {world.figure_name(fid)}") + + lines += ["", "### 本集史料(按时间排序;专名保留游戏原文)", ""] + for e in episode.events: + refs = " · ".join(world.render_refs(e)) + extra = [f"{k}={e.text(k)}" for k in ("state", "reason", "circumstance") + if e.text(k) and e.text(k) not in ("-1", "")] + tail = ("|" + ";".join(extra)) if extra else "" + lines.append(f"- {e.year}年 {e.type} · {refs}{tail}") + + if others: + lines += ["", "### 对手/相关者(史料里出现,但不是本卷主角)", ""] + for fid in others[:6]: + row = rows.get(fid) + if row: + lines.append(f"- {row.name}|{row.race}|身份:{row.identity or '不详'}" + f"|本集出场 {row.appearances} 次") + else: + lines.append(f"- {world.figure_name(fid)}") + + if elided: + hot = "、".join(f"{t}×{n}" for t, n in elided.most_common(6)) + lines += [ + "", + "### 被省略的同类重复事件", + "", + f"- 本卷还有这些同类事件已被过滤,不要逐条罗列,需要时用一句话带过:{hot}", + ] + + if prev_tail: + lines += ["", "### 上一集结尾(只用于承接状态,不要复述)", "", prev_tail] + + lines += [ + "", + "### 写作要求", + "", + f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,专名保留英文。", + "- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。", + "- 史料之外的世界大事不要写;只有影响到主角时才提一句。", + ] + return "\n".join(lines) + + +def build_messages( + world: World, + thread: Thread, + episode: ep_mod.Episode, + total_episodes: int, + prev_tail: str, + feedback: str | None = None, + elided: Counter | None = None, +) -> list[dict]: + spec = SPEC_FILE.read_text(encoding="utf-8") + material = render_material(world, thread, episode, total_episodes, prev_tail, elided) + user = [material] + if feedback: + user += ["", "### 上一稿的问题(重写时必须针对这些改)", "", feedback] + return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}] + + +def _repair(messages: list[dict], draft: str, length: int, gateway) -> str: + target = ( + f"内容太短({length} 字),请扩写到 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字," + "补充场景与对白,不要注水。" + if length < CHAPTER_MIN_CHARS + else f"内容太长({length} 字),请压缩到 {CHAPTER_MAX_CHARS} 字以内,删次要枝节。" + ) + reply = chat(messages + [{"role": "assistant", "content": draft}, + {"role": "user", "content": target}], gateway=gateway) + return reply.content + + +def write_episode( + world: World, + thread: Thread, + episode: ep_mod.Episode, + total_episodes: int, + prev_tail: str = "", + feedback: str | None = None, + *, + gateway=None, + max_repairs: int = 2, + elided: Counter | None = None, +) -> str: + """生成一集正文,并把长度校正到 800–1500 字。""" + messages = build_messages(world, thread, episode, total_episodes, prev_tail, feedback, elided) + text = chat(messages, gateway=gateway, max_tokens=MAX_TOKENS_CHAPTER).content + for _ in range(max_repairs): + length = cjk_length(text) + if CHAPTER_MIN_CHARS <= length <= CHAPTER_MAX_CHARS: + break + text = _repair(messages, text, length, gateway) + return text.strip() + + +def write_file(directory: Path, episode: ep_mod.Episode, text: str) -> Path: + directory.mkdir(parents=True, exist_ok=True) + path = directory / f"{episode.key}.md" + path.write_text(text.strip() + "\n", encoding="utf-8") + return path + + +def prev_tail_of(directory: Path, planned: list[ep_mod.Episode], index: int) -> str: + """取上一集结尾作为承接线索。""" + if index <= 1: + return "" + path = directory / f"{planned[index - 2].key}.md" + if not path.is_file(): + return "" + return path.read_text(encoding="utf-8").strip()[-TAIL_CHARS:] + + +def produce_episode( + world: World, + candidates: list[Thread], + *, + episode_index: int = 1, + out_dir: Path | None = None, + gateway=None, + max_candidates: int = gate.MAX_CANDIDATES, + on_event: Callable[[str], None] = print, +) -> Produced: + """选线索 → 写这一集 → 过闸门;不过就换下一线索(有上限)。""" + if not candidates: + raise RuntimeError("没有候选线索可用") + + for rank in range(1, min(max_candidates, len(candidates)) + 1): + selection = select_thread(world, candidates, rank=rank) + thread = selection.thread + ext, planned = extended_plan(world, thread) + if episode_index > len(planned): + on_event(f"第 {rank} 条线索只有 {len(planned)} 集,跳过") + continue + + episode = planned[episode_index - 1] + directory = out_dir or (PROJECT_DIR / "output" / "volumes" / "v1" / "chapters") + tail = prev_tail_of(directory, planned, episode_index) + on_event(f"选定线索 {rank}:{'、'.join(world.figure_name(m) for m in thread.members)}") + + # 把循环变量绑定成默认参数:否则闭包会绑定到循环变量本身(后面还会变) + def write( + attempt: int, + prev: gate.Verdict | None, + _thread: Thread = thread, + _episode: ep_mod.Episode = episode, + _total: int = len(planned), + _tail: str = tail, + _elided: Counter = ext.elided, + ) -> str: + feedback = None + if prev is not None: + feedback = (f"上一稿被判平淡:{prev.render()}。" + f"最弱环节是 {'、'.join(prev.weak())},请围绕这些重写," + "让人物真正做选择。") + return write_episode(world, _thread, _episode, _total, _tail, feedback, + gateway=gateway, elided=_elided) + + text, run = gate.gate_episode(episode.key, write, on_event=on_event) + on_event(run.render()) + + if run.accepted: + save_selection(selection, world) + path = write_file(directory, episode, text) if out_dir is not None else None + return Produced(selection=selection, episode=episode, text=text, run=run, path=path) + + gate.append_switch_note( + PROJECT_DIR, + world_name=world.name, + members=[world.figure_name(m) for m in thread.members], + span=thread.label, + verdict=run.final, + reason=f"重写 {run.rewrites} 次后仍未过闸门(第 {rank} 条线索)", + ) + on_event(f"线索 {rank} 未过闸门,换下一条") + + raise RuntimeError(f"前 {max_candidates} 条线索都没过闸门,需要人工干预") diff --git a/output/volumes/v1/cast.md b/output/volumes/v1/cast.md new file mode 100644 index 0000000..fab3d10 --- /dev/null +++ b/output/volumes/v1/cast.md @@ -0,0 +1,37 @@ +# Mon Sagus · 人物表 + +本卷主角线索:Guspu Frillyknots、Ral Fastenhatchets、Alath Blottedmine +线索跨度:193–239(46 年) + +| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 | +|---|---|---|---|---|---| +| Guspu Frillyknots | EAGLE_MAN | 151–246 | 关系对象×13、构陷发起者×4 | 62 | 主角 | +| Ral Fastenhatchets | DWARF | 164–在世/不详 | 被针对者×21、受骗者×6 | 28 | 主角 | +| Alath Blottedmine | DWARF | 182–在世/不详 | 构陷发起者×4、关系对象×1 | 8 | 主角 | +| Tulon Tongswall | DWARF | 132–在世/不详 | 构陷发起者×3 | 3 | 对手/配角 | +| Moldath Mirroredfloors | DWARF | 79–在世/不详 | 构陷发起者×3 | 3 | 对手/配角 | +| Ineth Wheeleddrinks | DWARF | 165–在世/不详 | 构陷者×3 | 3 | 对手/配角 | +| Langgud Tradebears The Eviscerated Chill Of Pregnancies | YETI | 生卒不详 | — | 2 | 对手/配角 | +| Rigoth Stilledearthen | DWARF | 162–在世/不详 | 当事者×2 | 2 | 对手/配角 | +| Perom Crazeowned | HUMAN | 129–在世/不详 | 当事者×1、关系对象×1 | 2 | 对手/配角 | + +## 小传 + +**Guspu Frillyknots** —— 鹰人 Guspu Frillyknots(151–246)本卷出场六十二次,关系对象达十三位;193年对 Ral Fastenhatchets 的腐化失败后,次年被 The Volcano of Ravens 定罪。 + +**Ral Fastenhatchets** —— Ral Fastenhatchets 于181年与 Langgud Tradebears The Eviscerated Chill Of Pregnancies 在 Fordedwinds 两度交手,此后长期被当作腐化与栽赃的靶子;本卷所载针对他的阴谋均告失败,但他仍留下六次受骗记录。 + +**Alath Blottedmine** —— Alath Blottedmine 于229年与 Sina Watchedleads 建立关系,同年对 Ral Fastenhatchets 腐化失败;230年又与 Artuk Huggedfastened 建立关系,并在 Paperhushed 对 Momuz Decentpage 下手未遂,随即被 The Volcano of Ravens 定罪。 + +**Tulon Tongswall** —— Tulon Tongswall 在193、195、197年三次于 Fordedwinds 试图腐化 Ral Fastenhatchets,结果三次全部失败,除此之外本卷无话可说。 + +**Moldath Mirroredfloors** —— Moldath Mirroredfloors 在233年两度、234年一度于 Fordedwinds 对 Ral Fastenhatchets 下手,三次腐化全告失败。 + +**Ineth Wheeleddrinks** —— Ineth Wheeleddrinks 在245、247、248年三次构陷 Ral Fastenhatchets,三次均是 failed frame attempt,且 fooled_hfid 三次都写成 Ral Fastenhatchets。 + +**Langgud Tradebears The Eviscerated Chill Of Pregnancies** —— Langgud Tradebears The Eviscerated Chill Of Pregnancies 在181年与 Ral Fastenhatchets 在 Fordedwinds 打过两场,此后本卷再无其他事迹。 + +**Rigoth Stilledearthen** —— Rigoth Stilledearthen 于188年与 Ral Fastenhatchets 建立联系,219年又解除该联系;两度出场仅为此事。 + +**Perom Crazeowned** —— Perom Crazeowned 于193年成为 Guspu Frillyknots 的关系对象,202年又被 Guspu 反向列为关系对象,戏份仅此而已。 + diff --git a/output/volumes/v1/chapters/001-181-198.md b/output/volumes/v1/chapters/001-181-198.md new file mode 100644 index 0000000..a4f87d3 --- /dev/null +++ b/output/volumes/v1/chapters/001-181-198.md @@ -0,0 +1,13 @@ +# 钉子 + +181年,Ral Fastenhatchets 十七岁,在 Fordedwinds 渡口两次挡住 Langgud Tradebears The Eviscerated Chill Of Pregnancies 的商队。那 Yeti 不肯缴渡税,用冰牙敲打栏木。Ral 拎着锤子站在桥心,肋骨断了两根,左耳打豁,但对方到底没过桥。他学到一件事:想按规矩活着,就得让人知道动你的代价。 + +之后十年,他留在 Fordedwinds 管货栈的锁和账。188年,Rigoth Stilledearthen 找上他,说需要个不经手银钱、又不怕撕破脸的人。Ral 问给多少。Rigoth 说不是给,是共担。两人结伙,Ral 卡住渡口和货栈的关节。同年,Guspu Frillyknots 与 Momuz Fateddie 搭上线。Guspu 是 Eagle Man,翅膀半秃,在 Fordedwinds 做掮客。他早看 Ral 不顺眼——货栈的锁由 Ral 把着,不点头,他的走私道就多绕一天路。 + +193年,Guspu 找 Ral 喝酒,递上一小袋金砂,说货栈每月闭一眼,大家都能睡好。Ral 把袋子推回去:“我睡得很好。你让 Momuz 把炭价降回三成,我请你喝真正的酒。”Guspu 当夜就派 Tulon Tongswall 去试。Tulon 是矮人,替 Guspu 伪造 Ral 的秤砣,想栽他私吞税粮。Ral 没揭穿,只在货栈门口摆了一杆新秤,当众称了自己管账以来每一笔过路货,误差不到半两。秤砣成了废铁。Guspu 第一次失手。 + +他没停,又拉拢 Ongu Gleamcircled 和 Perom Crazeowned,想找证人指认 Ral 收私钱。Ral 白天让人排货,夜里把钥匙挂在脖子上睡,两次构陷都扑了空。这时 Ral 做了让自己更睡不着觉的决定:去找 The Volcano Of Ravens 告发 Guspu。他交出账本和伪造秤砣的残片。194年,Guspu 被定罪。 + +Ral 以为到此为止。他低估了 Guspu。定罪后,Guspu 缩进阴影,195年两次派人向 Ral 递话,一次送钱,一次送刀,都失败。Tulon 也没走远,想把酒掺进税油里烧掉货栈。Ral 在火起前把 Tulon 按进河道,让他喝够自己的酒。197年,Tulon 最后一次伸手,把伪造密信塞进 Ral 的货箱,信没送到火山手里,因为 Ral 截住了送信的小子。他没杀 Tulon,只把信塞回对方腰带:“下一次,我让火山把你吞了。” + +198年,Astri Flayerline 来了。他是 Aardvark Man,长鼻嗅得着地下仓库的湿气。他替 Guspu 跑腿,两次想从账里找出破绽,都失败。前后四十七次构陷,像雨点打在铁皮上,Ral 的名字在 Fordedwinds 暗语里成了“那颗不生锈的钉子”。Rigoth 劝他离开,说钉子再硬,也钉不住一直浇醋的墙。Ral 没说话,只把锤子别回腰上。他知道 Guspu 还没完,Momuz、Ongu、Perom 都在等他先低头。 diff --git a/output/volumes/v1/chapters/002-202-216.md b/output/volumes/v1/chapters/002-202-216.md new file mode 100644 index 0000000..83e09d9 --- /dev/null +++ b/output/volumes/v1/chapters/002-202-216.md @@ -0,0 +1,21 @@ +# 墙上的裂痕 + +202年春天,Guspu Frillyknots 从定罪后的第八年阴影里走出来,做了一件谁都没料到的事:他去找 Perom Crazeowned。没有带刀,只带了一小袋税油票证和半张火山的罚单。Perom 是当年他出事时最早缩手的人之一,这些年一直在等他先开口。Guspu 把东西推过桌子,说:“我输了八年,不是没想过来硬的。但现在我要的不是一口气,是 Fordedwinds 的东门重新打开。”Perom 看着票证,点了头。他不是信 Guspu 的人,是信 Guspu 的恨。 + +同月,Espu Bonestands 找上门。这个 Goat Mountain Man 带来山区的新货道,能绕开火山的税站。Guspu 收下路线,代价是让 Espu 抽货价的两成。Guspu 算了算账:这四十七次失败已经证明旧路走不通,新路再贵也值得。 + +203年,Lolor Bronzescorched 进了他的屋子。这个 Dwarf 能做出让税务官看不出痕迹的账。Guspu 给他一张烤焦的桌子当工作台,说:“我不需要假账,我只需要真账里多出几滴油。”Lolor 咧嘴一笑,开始干活。 + +207年,Guspu 觉得网够密了。他找来 Dostngosp Jackalbrims,一个能在火山门口混个脸熟的 Goblin。Guspu 要他给 The Volcano Of Ravens 递一份材料:说 Ral Fastenhatchets 在 Fordedwinds 私藏税油,货箱就停在西码头。火山的执事来了,带着没收单。Ral 那天夜里起夜,发现锁孔里有新刮痕,二话不说把货转到 F 码头,只留下三箱废油。第二天火山的执事撬开箱子,油味冲鼻,但账本对不上。构陷失败。Ral 没有去火山告发 Guspu,也没有找 Dostngosp 的麻烦。他把锤子别回腰上,把自己锁在仓库里睡了一夜。第二天早上,他对自己说:“他想要我变成他。我不。” + +Guspu 在暗屋里摔了一只杯子。他没有骂 Dostngosp,只骂自己太急。然后他记下:要动 Ral,先得动他身边的人。 + +210年,他把目光投向 Zutthan Yellplank。这个 Dwarf 管着 Fordedwinds 东门的夜班钥匙。Guspu 亲自递去一袋金子,条件是夜里别锁东门。Zutthan 收了钱,但第二天把钱退了回来,附了一句口信:“我见过 Ral 的锤子。你的钱买不到我的命。”Guspu 没生气。他记下:这个人不能留,但也不能动。 + +同年,Onul Glowroofs 找到 Guspu。这个 Dwarf 能听到账房里的咳嗽声。Guspu 对他说:“你听,我不动手。”Onul 点了头,成了他的线人。 + +215年,Mudi Blazedbrushes 和 Stinthad Oilshove 几乎同时登门。一个做油脂,一个管装卸,都想在 Guspu 的网里分一杯羹。Guspu 让他们自己谈,谁先给出 Ral 的卸货时刻表,谁就拿大头。两人对视一眼,当晚就开始盯。 + +216年,Ingish Steelgrouped 送来一批钢材。Guspu 蹲在炉边,把钢块一块块码好。火光照着他的脸。他不再急着动手。他数着这些年结下的线:Perom、Espu、Lolor、Dostngosp、Onul、Mudi、Stinthad、Ingish——八条,还不够,但已经够把一面墙绑起来勒。 + +他想起 Ral 那颗不生锈的钉子。他想,钉子再硬,也钉不住一面正在裂开的墙。 diff --git a/output/volumes/v1/chapters/003-219-228-p1.md b/output/volumes/v1/chapters/003-219-228-p1.md new file mode 100644 index 0000000..0dc7fe3 --- /dev/null +++ b/output/volumes/v1/chapters/003-219-228-p1.md @@ -0,0 +1,19 @@ +# 锤子打不动的人 + +219 年,Guspu 决定先切掉 Ral 最长的一根旧枝。他带了一本账去 Fordedwinds。Rigoth Stilledearthen 是 Ral 的旧识,替 Ral 看管铁锭。Guspu 把 Rigoth 在码头上的欠款一页页摊开:“我不是来讨债,我是让你选。你从 Ral 那里抽身,账我替你销。”Rigoth 问若不抽身会怎样。Guspu 收起账本:“会变成流言。”第二天,Rigoth 当众扔掉了与 Ral 的铁楔。第一个,削掉了。 + +221 年,Dostngosp Jackalbrims 正式与 Guspu 结契。Guspu 以为自己已经把 Ral 围紧。他让 Dostngosp 把一包铁屑和几张账页送进 The Volcano Of Ravens,说这些是 Ral 私吞 Fordedwinds 公共斤两的证据。他以为 Ral 的锤子再硬,也硬不过火山的舌头。可 The Volcano Of Ravens 把铁屑丢进炭火,烧出来却是一团软渣,不是 Ral 的钢。构陷失败。Guspu 在火山石阶上坐了一夜。他只知道:Ral 身边还有眼睛,而且比他多一副。 + +222 年,Guspu 找到 Ducim Theatercrested,让他在戏台上唱 Ral 脚滑进铁水。Ducim 点头。从这年起,Fordedwinds 的酒馆里多了一出丑戏。唱的人笑,听的人笑,只有 Ral 没有笑。 + +223 年,Onul Glowroofs 和 Ingish Steelgrouped 正式与 Guspu 结契。一个替他听,一个替他供铁。Guspu 在炉边对两人说:“说话的人、唱戏的人、听壁角的人,和运铁的船,都姓了鹰。”他第一次觉得,Ral 的墙正往下掉土。 + +224 年,Ral 动了。他穿过东门去找 Zuntir Baldedattic,不要钱,只要他告诉自己 Guspu 的人每天过了几车、装了多少。Zuntir 收下话。这是 Ral 近几年第一次主动伸手。Guspu 听说了,在账房里砸了一块碳。他知道,Ral 不会只挨打。 + +225 年,Guspu 决定先打断 Ral 的一只手。他带人去了 Mortalpears,那里是 Ral 补给线的口子,把守者是 Goblin Uknol Thornticks,善在梨树林里埋尖桩。两人打了两次:第一次冒进,两个随从被穿了腿;第二次他亲自举盾压进林子。Uknol 退进雾里,没有死,但 Mortalpears 的路被踩平了。Guspu 回城,大衣上挂满梨刺。 + +226 年,轮到 Ral 下到 Fordedwinds 浅滩。Guspu 新派的收税人 Usel Fisherfair 带着六条船挨家翻货。Ral 没等船靠岸,提锤下水。两人在浅滩打了两次:头一次,Usel 的船被撞翻;第二次,Usel 放火烧了东边泊位。Ral 没吃大亏,但船坞焦了一半。这是他第一次砸人后,没听见自己笑。 + +227 年,Guspu 在 Lashheroes 对上 Meng Walkclasped,此人是 Ral 从北方请来的打手,两把扣锤专砸铁链。打了两场:第一场,Guspu 后背裂了一道;第二场,他让 Ingish 夜里用钢水浇死了 Meng 的扣锤。Meng 骂着退走。Guspu 一句话没说,血在喉咙里来回滚。 + +228 年,Guspu 杀到 Liecats,撞上 Ngom Matchedmenaces。他上马时告诉自己:再削掉一个,Ral 就只剩空墙。可打到一半,他看见路边的草地全烧成了黑灰,忽然想不起这一路是谁先点的火。他攥紧缰绳,继续向前。Ral 的墙还在,但已经裂了。 diff --git a/prompts/biographer.md b/prompts/biographer.md new file mode 100644 index 0000000..0cd2bc9 --- /dev/null +++ b/prompts/biographer.md @@ -0,0 +1,45 @@ +# 传记作者 · 人物主线写作规范 + +你在给几位主角写**传记**。不是编年史,不是战报综述——是一部跟着人走的连续故事。 + +## 立场 + +- 你是这几个人(或其中一个)的传记作者。你对他们有偏心,也有怨气;你会替某个人辩护, + 也会毫不留情地揭另一个人的底。 +- 你手上只有一份简略的史料:事件、年份、参与者。你要做的是把冷冰冰的条目还原成 + **人在其中做了什么选择**。 + +## 怎么"贴着人物写" + +- 每一段都要有人在做决定:想要什么、怕什么、算了什么账、赌了什么。 +- 事件不是情节,**人物的选择才是情节**。同一件事,问的是"谁在这件事上要了什么"。 +- 主角若不主动,这一集就失败——不要写成"发生了很多事",要写成"他做了这些事,然后代价来了"。 +- 对手必须是个具体的人,有他自己的动机和算盘,不是"敌军"这种复数名词。 + +## 世界大事怎么处理 + +- **只在影响到主角时才写**。主角没被波及的战争、灭国、天灾,一律不写。 +- 需要交代背景时,一句话带过,并立刻回到这个人身上。 + +## 语言与边界 + +- 中文正文;**专有名词一律保留英文原文**,首字母大写。 +- **允许**虚构对白、心理活动、场景细节。 +- **禁止**发明史料里没有的人名、地名、组织名。 +- **禁止**改动史实骨架:年份、谁参与、结果如何,都必须与史料一致。 +- 若史料只有一句干巴巴的记录,就照它的空白写——"档案到此为止"也可以是一种味道。 + +## 篇幅与格式 + +- 正文 **900–1400 字**(硬上限 1500,超出会被退回压缩)。宁可写短写实,不要灌水。 +- 格式: + +``` +# <本集标题> + +<正文> +``` + +- 不要写"以下是""本集将讲述"之类的话,直接进入叙述。 +- 不要在正文里提史料、表格、XML、AI,也不要写章末注释。 +- 这是连载中的一集:前情用一两句带过,结尾留一个能被下一集接住的线头。 diff --git a/scripts/dfannals b/scripts/dfannals new file mode 100755 index 0000000..5506439 --- /dev/null +++ b/scripts/dfannals @@ -0,0 +1,36 @@ +#!/usr/bin/env bash +# 管道启动器:把命令交给一个真正装了依赖的解释器。 +# +# 背景(实测踩过的坑):会话里 PATH 上的 python3 可能不是系统解释器, +# 而是编辑器工具链自建的 venv(例如 ~/.pi-lens/pip-tools)。 +# 那种 venv 里没有 defusedxml,而且关闭了 user site,于是 +# `python3 -m dfannals.cli` 会突然报 ModuleNotFoundError。 +# 这里主动挑一个能 import defusedxml 的解释器,避免依赖 PATH 的偶然性。 +# +# scripts/dfannals status|plan|threads|episodes|cast|index|next|run ... +# DFANNALS_PYTHON=/path/to/python3 scripts/dfannals status # 也可显式指定 +set -euo pipefail + +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" + +pick_python() { + local cand + for cand in "${DFANNALS_PYTHON:-}" /usr/bin/python3 python3; do + [ -n "$cand" ] || continue + command -v "$cand" >/dev/null 2>&1 || continue + if "$cand" -c "import defusedxml" >/dev/null 2>&1; then + printf '%s' "$cand" + return 0 + fi + done + return 1 +} + +PY="$(pick_python)" || { + echo "找不到可用解释器:需要一个能 import defusedxml 的 python3。" >&2 + echo "装依赖:sudo apt install -y python3-defusedxml" >&2 + exit 1 +} + +cd "$HERE" +exec "$PY" -m dfannals.cli "$@" diff --git a/scripts/gate-selfcheck.py b/scripts/gate-selfcheck.py new file mode 100644 index 0000000..9c75346 --- /dev/null +++ b/scripts/gate-selfcheck.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""闸门自检:证明 5 维度自评真的能把"流水账"判成不合格。 + +这是 t4 契约要求的可复现证据。它跑两次真实模型自评: + + 平淡稿:纯事件罗列,主角全程被动,没有任何抉择、转折或具体对手 + 有戏剧本:主角有目标、冲突逐级升级、有立场翻转、对手具体 + +然后断言 平淡稿 < 阈值 ≤ 有戏剧本。若两者都低,说明评分尺度太严(会误杀好稿); +若两者都高,说明闸门形同虚设。 +""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from dfannals import gate # noqa: E402 + +DULL = """# 第二十三章:大事记 + +元年,甲死了。甲是精灵。三年,乙死了。乙是精灵。 +五年,甲和乙的部族打了一仗。七年,又打了一仗。九年,再打了一仗。 +十一年,丙死了。十三年,丁死了。十五年,戊死了。十七年,又打了一仗。 +十九年,己死了。二十一年,庚死了。二十三年,辛死了。二十五年,壬死了。 +史书到此中断。以上均有据可查。 +""" + +VIVID = """# 第二十三章:他等那口井塌了十七年 + +Ngalak 站在井台边数桶。第七桶,他想,今天必须到第十桶。 + +"粮行的水,先紧着粮行用。"他对排队的人说。队伍里有人骂了一句,他没回头。他要的不是水。 +水他买得起;他买不起的是 Minkot 每天清晨把钥匙挂在腰间走过集市时,那种不必向任何人解释的从容。 + +第三年冬天,井边塌了一角,压死了两个挑水的孩子。Ngalak 第一个到场。他跪在泥水里哭, +哭得比孩子的母亲还大声,哭完站起来,把外袍脱下来盖在尸首上。人群里有人说:这才是管事的人。 +Minkot 站在后面,手里还握着一把钥匙,一句话没说。 + +"你那是哭吗?"当晚 Minkot 在酒馆里问他,把酒碗撞在桌上,酒溅出来一半,"你那是敲锣。" +"我敲了,"Ngalak 把溅出来的酒擦干,"你听见了?" + +第十年,Minkot 的妻子在集市上指着 Ngalak 的鼻子说:是你挖的地基。三天后她改口,说自己 grief 昏了头。 +Ngalak 送了她一头牛。她把牛牵走的那天,Minkot 站在门口看着,门没关,他也没让妻子进屋。 + +"你到底要什么?"第十七年,Minkot 终于开口。集市上的人都停下来听。 +"我要你不必再解释的资格。"Ngalak 说。 +Minkot 把钥匙从腰上解下来,放在井台上,然后回身把自己家里的水缸砸了。陶片溅了一地。 +"这口井,我不要了。"他说,"记住,今天不是他赢了——是我不要了。" + +Ngalak 站在原地。他等这一天等了十七年,真等到手的时候,发现钥匙比他想的要重。 +他把它握了很久,才发觉自己在数桶:第八桶,第九桶,第十桶。 +""" + + +def main() -> int: + print("跑真实模型自评(两次调用)…\n") + dull = gate.evaluate(DULL) + print("平淡稿:", dull.render()) + vivid = gate.evaluate(VIVID) + print("有戏剧本:", vivid.render()) + print(f"\n阈值 = {gate.THRESHOLD}(满分 {len(gate.DIMENSIONS) * gate.MAX_SCORE})") + + ok = dull.total < gate.THRESHOLD <= vivid.total + print(f"平淡稿 {dull.total} < 阈值 ≤ 有戏剧本 {vivid.total} → {'通过' if ok else '不通过'}") + if not ok: + if vivid.total < gate.THRESHOLD: + print("问题:评分尺度太严,好稿也会被误杀。") + else: + print("问题:闸门形同虚设,平淡稿也能通过。") + return 0 if ok else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/fixtures/sample-legends.xml b/tests/fixtures/sample-legends.xml index e2fec67..2947062 100644 --- a/tests/fixtures/sample-legends.xml +++ b/tests/fixtures/sample-legends.xml @@ -10,6 +10,7 @@ 10Urist McMinerDWARF121288MINER 11Kogan DeathspearGOBLIN44AXEMAN + 12liri fogbaldedELF2020 100The Steel ConfederacyCivilizationDWARF @@ -19,5 +20,9 @@ 1305hf simple battle event10111 2310created site2100 3889hf died1011 + 4400add hf hf link1012 + 5410hf simple battle event10111 + 6420hf abducted11121 + 7430hf relationship denied1210apprenticeprefers working alone diff --git a/tests/test_episodes.py b/tests/test_episodes.py new file mode 100644 index 0000000..9e43728 --- /dev/null +++ b/tests/test_episodes.py @@ -0,0 +1,119 @@ +# pyright: reportMissingImports=false, reportAttributeAccessIssue=false +# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。 + +"""因果链切集的回归测试。 + +对应 t3 契约:间隔 >2 年断开;短链只合并、不设总量下限;过长按预算拆上下集; +每集预算 800–1500 字;输出含集号/年份范围/事件条数/主角与对手。 +""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from collections import Counter + +from dfannals import episodes +from dfannals.legends import Event, Figure, World +from dfannals.threads import Thread + + +def ev(eid: int, year: int, etype: str = "add hf hf link") -> Event: + return Event(id=eid, year=year, seconds72=0, type=etype, + fields={"hfid": ["1"], "hfid_target": ["2"]}) + + +def fake_thread(events: list[Event]) -> Thread: + return Thread( + members=[1, 2], interactions=len(events), turns=0, + span=(events[-1].year - events[0].year) if events else 0, + first_year=events[0].year if events else 0, + last_year=events[-1].year if events else 0, + types=Counter(), races=Counter(), non_person_races=[], events=events, + ) + + +def test_gap_breaks_chain() -> None: + """间隔 >2 年必须断开成两条链。""" + events = [ev(i, y) for i, y in enumerate([10, 11, 12, 20, 21, 22])] + chains = episodes._split_chains(events, gap_years=2) + assert len(chains) == 2, [len(c) for c in chains] + assert [len(c) for c in chains] == [3, 3] + + +def test_small_gap_does_not_break() -> None: + """间隔恰好 2 年不该断开。""" + chains = episodes._split_chains([ev(1, 10), ev(2, 12)], gap_years=2) + assert len(chains) == 1 + + +def test_long_chain_splits_into_budgeted_parts() -> None: + """过长链路必须拆成上/下集,且每段都不跌破下限。""" + events = [ev(i, 100 + i) for i in range(30)] + parts = episodes._split_long(events) + assert len(parts) >= 2, "30 条事件没有拆集" + for p in parts: + assert episodes._min_events() <= len(p) <= episodes._max_events(), [len(x) for x in parts] + assert sum(len(p) for p in parts) == 30 + + +def test_split_never_creates_undersized_fragments() -> None: + """回归:14 条事件曾被均分成 7+7,两段都跌破 800 字下限。""" + events = [ev(i, 200 + i) for i in range(14)] + eps = episodes.plan_episodes(fake_thread(events)) + assert len(eps) == 1, f"14 条事件被拆成了 {len(eps)} 集" + assert episodes.MIN_CHARS <= eps[0].estimated_chars <= episodes.MAX_CHARS + + +def test_short_chain_is_merged_not_dropped() -> None: + """短链只合并,不设线索总量下限:2 条事件也要能成一集。""" + eps = episodes.plan_episodes(fake_thread([ev(1, 50), ev(2, 51)])) + assert len(eps) == 1 + assert len(eps[0].events) == 2 + + +def test_episode_key_marks_parts() -> None: + events = [ev(i, 300 + i) for i in range(30)] + eps = episodes.plan_episodes(fake_thread(events)) + assert len(eps) >= 2 + assert all(e.parts == len(eps) for e in eps) + assert eps[0].key.endswith("-p1") and eps[1].key.endswith("-p2") + assert eps[0].key != eps[1].key + + +def test_plan_reports_heroes_and_others() -> None: + """输出必须能区分主角与对手。""" + world = World(name="test") + for fid, name in ((1, "Hero"), (2, "Ally"), (9, "Rival")): + world.figures[fid] = Figure(id=fid, name=name) + + events = [ev(i, 400 + i) for i in range(3)] + events.append(Event(id=99, year=403, seconds72=0, type="hf wounded", + fields={"woundee_hfid": ["1"], "wounder_hfid": ["9"]})) + thread = fake_thread(events) + planned = episodes.plan_episodes(thread) + heroes, others = episodes.cast_of(world, thread, planned[0]) + assert 1 in heroes and 2 in heroes + assert 9 in others and 9 not in heroes + text = episodes.render_plan(world, thread, planned) + assert "主角" in text and "对手/外人" in text and "Rival" in text + + +def _main() -> int: + tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)] + failed = 0 + for fn in tests: + try: + fn() + print(f" ✓ {fn.__name__}") + except AssertionError as exc: + failed += 1 + print(f" ✗ {fn.__name__}: {exc}") + print(f"\n{len(tests) - failed}/{len(tests)} 通过") + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(_main()) diff --git a/tests/test_factcheck.py b/tests/test_factcheck.py index 28eabb4..41a38bc 100644 --- a/tests/test_factcheck.py +++ b/tests/test_factcheck.py @@ -12,7 +12,7 @@ from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1])) -from dfannals import factcheck, legends # noqa: E402 +from dfannals import factcheck, legends FIXTURES = Path(__file__).resolve().parent / "fixtures" @@ -30,8 +30,49 @@ def test_world_parsed() -> None: assert world.figure_name(10) == "Urist McMiner" assert world.site_name(1) == "Boatmurdered" assert world.entity_name(100) == "The Steel Confederacy" - assert len(world.events) == 3 - assert [e.year for e in world.events] == [30, 31, 88] + assert len(world.events) == 7 + assert [e.year for e in world.events] == [30, 31, 40, 41, 42, 43, 88] + + +def test_field_classifier_covers_relationship_fields() -> None: + """回归:旧版只认 4 个人物字段,导致双方关系在素材里被丢掉。""" + from dfannals.legends import kind_of + + for tag in ("hfid", "hfid_target", "group_1_hfid", "group_2_hfid", "snatcher_hfid", + "seeker_hfid", "hfid1", "hfid2", "wounder_hfid", "teacher_hfid", + "conspirator_hfid", "slayer_hfid"): + assert kind_of(tag) == "figure", tag + for tag in ("target_enid", "entity_id", "attacker_civ_id"): + assert kind_of(tag) == "entity", tag + for tag in ("identity_id", "master_wcid", "slayer_item_id", + "slayer_race", "reason", "relationship"): + assert kind_of(tag) is None, tag + + +def test_relationship_events_render_both_sides() -> None: + """核心契约:四类关系事件必须渲染出双方,而不是只剩单方面 hfid。""" + world = load_world() + rendered = {e.type: world.render_refs(e) for e in world.events} + + link = rendered["add hf hf link"] + assert "hfid=Urist McMiner" in link and "hfid_target=Liri Fogbalded" in link + + battle = rendered["hf simple battle event"] + assert "group_1_hfid=Urist McMiner" in battle and "group_2_hfid=Kogan Deathspear" in battle + + abduct = rendered["hf abducted"] + assert "snatcher_hfid=Kogan Deathspear" in abduct and "target_hfid=Liri Fogbalded" in abduct + + denied = rendered["hf relationship denied"] + assert "seeker_hfid=Liri Fogbalded" in denied and "target_hfid=Urist McMiner" in denied + + +def test_figure_ids_excludes_placeholder_and_dedups() -> None: + world = load_world() + battle = next(e for e in world.events if e.type == "hf simple battle event" and e.year == 41) + assert sorted(battle.figure_ids()) == [10, 11] + # hfid 与 slayer_hfid 同指一人时不应重复计数 + assert len(battle.figure_ids()) == len(set(battle.figure_ids())) def test_fake_names_are_caught() -> None: diff --git a/tests/test_gate.py b/tests/test_gate.py new file mode 100644 index 0000000..c6eadca --- /dev/null +++ b/tests/test_gate.py @@ -0,0 +1,125 @@ +# pyright: reportMissingImports=false, reportAttributeAccessIssue=false +# 说明:LSP 的 Python 环境看不到本项目包,会把包内 import 误报为缺失模块。 + +"""闸门流程的回归测试(不调用真实模型,用假评分器)。 + +对应 t4 契约:<12 先重写一次、仍不合格才换线索、失败上限、换线原因写入 notes/。 +真实模型能否真的判出平淡,由 scripts/gate-selfcheck.py 负责证明。 +""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from dfannals import gate # noqa: E402 + + +def verdict(total: int, reason: str = "测试用") -> gate.Verdict: + """按总分构造一个评分(维度均分,保持与真实结构一致)。""" + keys = [k for k, _, _ in gate.DIMENSIONS] + per, rest = divmod(total, len(keys)) + scores = {k: min(gate.MAX_SCORE, per + (1 if i < rest else 0)) for i, k in enumerate(keys)} + return gate.Verdict(total=sum(scores.values()), scores=scores, reason=reason) + + +def test_passes_first_try() -> None: + calls: list[int] = [] + + def write(attempt: int, _prev) -> str: + calls.append(attempt) + return "好稿" + + text, run = gate.gate_episode("001", write, evaluate_fn=lambda _t: verdict(20)) + assert text == "好稿" + assert run.accepted and run.rewrites == 0 + assert calls == [0], "通过后不应重写" + + +def test_rewrite_once_then_pass() -> None: + """<12 分先重写一次,重写后合格就通过。""" + seq = [verdict(5), verdict(18)] + seen: list[int] = [] + + def write(attempt: int, prev) -> str: + seen.append(attempt) + if attempt == 1: + assert prev is not None and prev.total == 5, "重写时要带上上一稿评分" + return f"第{attempt}稿" + + text, run = gate.gate_episode("002", write, evaluate_fn=lambda _t: seq.pop(0)) + assert text == "第1稿" and run.accepted + assert run.rewrites == 1 and seen == [0, 1] + assert [v.total for v in run.verdicts] == [5, 18] + + +def test_not_accepted_after_max_rewrites() -> None: + """一直不合格:重写次数不得越过上限,且标记为未通过。""" + runs: list[int] = [] + text, run = gate.gate_episode( + "003", + lambda attempt, _p: (runs.append(attempt), "稿")[1], + evaluate_fn=lambda _t: verdict(3), + ) + assert not run.accepted + assert run.rewrites == gate.MAX_REWRITES + assert runs == [0, 1], f"实际尝试次数:{runs}" + assert len(run.verdicts) == gate.MAX_REWRITES + 1 + + +def test_render_shows_weak_dimensions() -> None: + v = verdict(6, "全是重复对战") + text = v.render() + assert "判平淡" in text and "全是重复对战" in text and "6/" in text + assert len(v.weak()) == 2 + + +def test_switch_note_records_reason(tmp_path: Path) -> None: + """放弃线索必须记下原因,且写入仓库 notes/ 附录。""" + path = gate.append_switch_note( + tmp_path, + world_name="Mon Sagus", + members=["Guspu Frillyknots", "Ral Fastenhatchets"], + span="193–239(46 年)", + verdict=verdict(5, "只有阴谋事件重复,没有立场变化"), + reason="重写一次后仍判平淡", + ) + assert path.exists() and path.parent.name == "notes" + body = path.read_text(encoding="utf-8") + assert "Guspu Frillyknots" in body and "重写一次后仍判平淡" in body + assert "只有阴谋事件重复" in body + + +def test_append_switch_note_is_append_only(tmp_path: Path) -> None: + gate.append_switch_note(tmp_path, world_name="W", members=["A"], span="1–2", + verdict=verdict(4), reason="第一次放弃") + gate.append_switch_note(tmp_path, world_name="W", members=["B"], span="3–4", + verdict=verdict(4), reason="第二次放弃") + body = (tmp_path / "notes" / gate.SWITCH_LOG).read_text(encoding="utf-8") + assert "第一次放弃" in body and "第二次放弃" in body + assert body.count("# 被放弃的线索") == 1, "表头只应写一次" + + +def _main() -> int: + import tempfile + + plain = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)] + failed = 0 + for fn in plain: + try: + if fn.__code__.co_argcount: + with tempfile.TemporaryDirectory() as d: + fn(Path(d)) + else: + fn() + print(f" ✓ {fn.__name__}") + except AssertionError as exc: + failed += 1 + print(f" ✗ {fn.__name__}: {exc}") + print(f"\n{len(plain) - failed}/{len(plain)} 通过") + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(_main())