第 1 卷(人物主线):第 1–3 集 + 人物表

主角线:Guspu Frillyknots / Ral Fastenhatchets / Alath Blottedmine
由线索挖掘器自动选定,按因果链切集,经 5 维度自评闸门(≥12 分)后入库。
This commit is contained in:
Chen Yi
2026-10-05 19:28:49 +08:00
parent 594cf48a1a
commit e0073e90bd
24 changed files with 2070 additions and 79 deletions
+225
View File
@@ -0,0 +1,225 @@
"""每卷人物表:Markdown 表格 + 每人一两句小传。
访谈决策:
- 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库
- 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册
表格部分完全由史料推导(种族、生卒、出场次数、身份);
"身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上,
就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。
"""
from __future__ import annotations
import json
import re
from dataclasses import dataclass
from pathlib import Path
from dfannals import factcheck
from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway
from dfannals.episodes import Episode
from dfannals.legends import Event, World
from dfannals.llm import chat
from dfannals.threads import Thread
# 事件字段 → 中文角色名(用于反推"身份")
ROLE_LABELS: dict[str, str] = {
"corruptor_hfid": "构陷发起者",
"target_hfid": "被针对者",
"wounder_hfid": "行凶者",
"woundee_hfid": "受伤者",
"snatcher_hfid": "绑走者",
"seeker_hfid": "求关系者",
"winner_hfid": "胜者",
"competitor_hfid": "参赛者",
"slayer_hfid": "凶手",
"hfid": "当事者",
"hfid_target": "关系对象",
"teacher_hfid": "师长",
"student_hfid": "学徒",
"persecutor_hfid": "迫害者",
"expelled_hfid": "被逐者",
"convicted_hfid": "被定罪者",
"interrogator_hfid": "审讯者",
"framer_hfid": "构陷者",
"fooled_hfid": "受骗者",
}
BIO_LIMIT = 10 # 小传最多覆盖多少人
EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
@dataclass
class CastRow:
fid: int
name: str
race: str
span: str
role: str # 主角 / 对手 / 配角
appearances: int
identity: str
def as_row(self) -> str:
return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | "
f"{self.appearances} | {self.role} |")
def _identity_of(world: World, fid: int, events: list[Event]) -> str:
"""从事件字段反推这个人在这段历史里扮演什么。"""
counts: dict[str, int] = {}
for e in events:
for tag, vals in e.fields.items():
if tag not in ROLE_LABELS:
continue
for raw in vals:
try:
if int(raw) == fid:
counts[tag] = counts.get(tag, 0) + 1
except ValueError:
continue
if not counts:
return ""
ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2]
return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked)
def _span_of(thread: Thread, episodes: list[Episode]) -> str:
"""表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。"""
if not episodes:
return thread.label
return f"{episodes[0].start_year}–{episodes[-1].end_year}"
def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]:
"""主角优先,其次是出场最多的对手/配角。"""
events = [e for ep in episodes for e in ep.events] or thread.events
appearances: dict[int, int] = {}
for e in events:
for fid in e.figure_ids():
appearances[fid] = appearances.get(fid, 0) + 1
members = set(thread.members)
rows: list[CastRow] = []
def make(fid: int, role: str) -> CastRow:
fig = world.figures.get(fid)
return CastRow(
fid=fid,
name=world.figure_name(fid),
race=fig.race if fig else "",
span=fig.alive_span if fig else "生卒不详",
role=role,
appearances=appearances.get(fid, 0),
identity=_identity_of(world, fid, events),
)
for fid in sorted(members, key=lambda f: -appearances.get(f, 0)):
rows.append(make(fid, "主角"))
others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f])
for fid in others[:max_others]:
rows.append(make(fid, "对手/配角"))
return rows
BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。
规则:
- 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。**
- 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。
- 姓名一律保留英文原文。
- 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...}
"""
def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
lines = [BIO_INSTRUCTIONS, "", "材料:"]
for row in rows[:BIO_LIMIT]:
facts = [
f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次,"
f"身份:{row.identity or '不详'},在故事里是{row.role})"
]
for e in events:
if row.fid not in e.figure_ids():
continue
refs = " · ".join(world.render_refs(e))
facts.append(f" {e.year}年 {e.type} {refs}")
if len(facts) > 5:
break
lines.extend(facts)
return "\n".join(lines)
def _parse_bios(raw: str) -> dict[str, str]:
text = EDGE_RE.sub("", raw).strip()
start, end = text.find("{"), text.rfind("}")
if start < 0 or end <= start:
raise ValueError(f"小传没有返回 JSON:{raw[:200]}")
try:
data = json.loads(text[start:end + 1])
except json.JSONDecodeError as exc:
raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
if not isinstance(data, dict):
raise ValueError("小传返回的不是对象")
return {str(k): str(v) for k, v in data.items()}
def build_cast(
world: World,
thread: Thread,
episodes: list[Episode],
*,
gateway: Gateway | None = None,
with_bios: bool = True,
) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]:
"""返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。"""
rows = collect_rows(world, thread, episodes)
events = [e for ep in episodes for e in ep.events] or thread.events
bios: dict[str, str] = {}
if with_bios and rows:
reply = chat(
[{"role": "user", "content": _bio_prompt(world, rows, events)}],
gateway=gateway,
max_tokens=MAX_TOKENS_STRUCTURED,
)
bios = _parse_bios(reply.content)
suspicions: list[factcheck.Suspicion] = []
bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items())
if bios_text:
suspicions = factcheck.check(bios_text, world)
lines = [
f"# {world.name} · 人物表",
"",
f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
f"线索跨度:{_span_of(thread, episodes)}",
"",
"| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |",
"|---|---|---|---|---|---|",
]
lines.extend(row.as_row() for row in rows)
if bios:
lines += ["", "## 小传", ""]
for row in rows:
bio = bios.get(row.name) or bios.get(row.name.lower())
if bio:
lines.append(f"**{row.name}** —— {bio}")
lines.append("")
if suspicions:
lines += [
"",
"> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:"
+ "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]),
"",
]
return "\n".join(lines) + "\n", rows, suspicions
def write_cast(path: Path, text: str) -> Path:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text, encoding="utf-8")
return path
+2 -2
View File
@@ -62,8 +62,8 @@ class State:
def cjk_length(text: str) -> int:
"""按中文习惯计字数:CJK 字符与拉丁单词都算 1。"""
without_code = re.sub(r"```.*?```", "", text, flags=re.S)
without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.M)
without_code = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.MULTILINE)
n = 0
for ch in without_meta:
if (
+154 -2
View File
@@ -16,9 +16,26 @@ import argparse
import sys
from pathlib import Path
from dfannals import chronicle, factcheck, legends, publish, timeline
from dfannals import (
cast,
chronicle,
episodes,
factcheck,
legends,
publish,
threads,
timeline,
)
from dfannals import slice as chapter_slice
from dfannals.config import EXPORT_DIR, GITEA_WEB, chapter_dir, ensure_dirs, volume_dir
from dfannals import volume as volume_mod
from dfannals.config import (
DATA_DIR,
EXPORT_DIR,
GITEA_WEB,
chapter_dir,
ensure_dirs,
volume_dir,
)
def find_exports(export_dir: Path) -> list[Path]:
@@ -142,6 +159,109 @@ def cmd_next(args: argparse.Namespace) -> int:
return 0
def cmd_threads(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world,
min_interactions=args.min_interactions,
min_span=args.min_span,
limit=args.top,
)
print(threads.render_report(world, found, per_thread=args.samples))
path = threads.save_json(world, found, DATA_DIR / "threads.json")
print(f"共 {len(found)} 条候选;明细已写入 {path}")
return 0
def cmd_episodes(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
return 1
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
print(episodes.render_plan(world, thread, planned))
print()
print(episodes.budget_report(planned))
if args.samples:
print("\n=== 前两集事件样例 ===")
for ep in planned[:2]:
print(f" 第 {ep.index} 集({ep.span})")
for e in ep.events[: args.samples]:
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
return 0
def cmd_cast(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
return 1
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
text, rows, suspicions = cast.build_cast(
world, thread, planned, with_bios=not args.no_bios
)
if args.dry_run:
print(text[:2000])
return 0
state = chronicle.State.load()
volume = state.volume if state.world == world.name else 1
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
print(factcheck.report(suspicions) if suspicions else "专名校验:小传里的专名全部能在史料中找到。")
return 0
def cmd_episode(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
return 1
out_dir = None if args.dry_run else chapter_dir(1)
produced = volume_mod.produce_episode(
world, found, episode_index=args.index, out_dir=out_dir
)
print("选择依据:" + produced.selection.reason)
print(f"本集字数:{chronicle.cjk_length(produced.text)}")
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(produced.text[:1600])
return 0
print(f"已写入 {produced.path}")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 1 卷第 {produced.episode.index} 集(人物主线):{produced.episode.span}"
)
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_run(args: argparse.Namespace) -> int:
rc = cmd_index(args)
if rc:
@@ -163,6 +283,38 @@ def main(argv: list[str] | None = None) -> int:
p = sub.add_parser("plan", help="显示切章方案")
p.set_defaults(func=cmd_plan)
p = sub.add_parser("threads", help="挖掘人物线索候选")
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
help="线索最小跨度年数(默认 30)")
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
p.set_defaults(func=cmd_threads)
p = sub.add_parser("episodes", help="按人物线索切出剧集")
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
p.set_defaults(func=cmd_episodes)
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
p.set_defaults(func=cmd_cast)
p = sub.add_parser("episode", help="按人物主线产出下一集")
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
p.set_defaults(func=cmd_episode)
p = sub.add_parser("next", help="生成下一章")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.set_defaults(func=cmd_next)
+4 -1
View File
@@ -47,6 +47,9 @@ MODEL = "cn:deepseek-v4-pro"
# 该模型是推理模型:思考与正文共用 max_tokens,必须留足
MAX_TOKENS_CHAPTER = 16000
MAX_TOKENS_UTIL = 4000
# 凡是「喂长材料 + 要求结构化 JSON 输出」的调用(闸门自评、人物小传)都容易把预算
# 烧在思考上而返回空正文,实测 4000 不够。
MAX_TOKENS_STRUCTURED = 12000
# ---------------------------------------------------------------- 发布
GITEA_SSH_HOST = "124.222.29.26"
@@ -61,7 +64,7 @@ SSH_KEY = HOME / ".ssh/id_ed25519"
WORLD_SIZE = "Medium"
WORLD_HISTORY_YEARS = 250
CHAPTER_MIN_CHARS = 800
CHAPTER_MAX_CHARS = 1600 # 契约上限 1500,留出标点/换行余量
CHAPTER_MAX_CHARS = 1500 # 严格按访谈定的 800–1500,不留“余量”
@dataclass
+192
View File
@@ -0,0 +1,192 @@
"""因果链切集:把一条人物线索切成"一集一个完整回合"。
参数来自访谈决策:
- 相邻事件间隔 >2 年即断开(比推荐值更紧,节奏更快)
- 短链只合并,不设线索总量下限
- 链路过长按字数预算拆上下集
- 每集正文预算落在 800–1500 字
字数预算用「事件条数 × 每条展开字数」估算。系数 110 是按此前实测校准的:
12 条史料事件生成的中文正文约 1100–1400 字。
"""
from __future__ import annotations
import math
from dataclasses import dataclass, field
from dfannals.legends import Event, World
from dfannals.threads import Thread
# 相邻事件间隔超过它即断开
GAP_YEARS = 2
# 每条史料在正文里大致展开的字数。实测区间约 85–115(同样 12 条事件,
# 两次生成分别得到 1026 字与 1388 字),取中值偏保守,避免把集拆得过碎。
CHARS_PER_EVENT = 95
MIN_CHARS = 800 # 每集预算下限
MAX_CHARS = 1500 # 每集预算上限
@dataclass
class Episode:
index: int
start_year: int
end_year: int
events: list[Event] = field(default_factory=list)
part: int = 1
parts: int = 1
@property
def key(self) -> str:
base = f"{self.index:03d}-{self.start_year}-{self.end_year}"
return base if self.parts == 1 else f"{base}-p{self.part}"
@property
def span(self) -> str:
if self.start_year == self.end_year:
return f"{self.start_year} 年"
return f"{self.start_year}–{self.end_year} 年"
@property
def estimated_chars(self) -> int:
return len(self.events) * CHARS_PER_EVENT
def in_budget(self) -> bool:
return MIN_CHARS <= self.estimated_chars <= MAX_CHARS or self.parts > 1
def _min_events() -> int:
return max(1, math.ceil(MIN_CHARS / CHARS_PER_EVENT))
def _max_events() -> int:
return max(1, MAX_CHARS // CHARS_PER_EVENT)
def _split_chains(events: list[Event], gap_years: int) -> list[list[Event]]:
"""按时间间隔把事件串切成因果链。"""
chains: list[list[Event]] = []
current: list[Event] = []
prev_year: int | None = None
for e in events:
if current and prev_year is not None and e.year - prev_year > gap_years:
chains.append(current)
current = []
current.append(e)
prev_year = e.year
if current:
chains.append(current)
return chains
def _merge_short(chains: list[list[Event]], min_events: int) -> list[list[Event]]:
"""短链只合并:攒够预算就收一集,尾部不足的并入上一集。"""
merged: list[list[Event]] = []
buffer: list[Event] = []
for chain in chains:
buffer.extend(chain)
if len(buffer) >= min_events:
merged.append(buffer)
buffer = []
if buffer:
if merged:
merged[-1].extend(buffer)
else:
merged = [buffer]
return merged
def _split_long(chain: list[Event]) -> list[list[Event]]:
"""过长链路按预算拆成若干段(对应正文的上/下集)。
注意不能均分:14 条事件均分成 7+7 会得到两个 770 字的碎片,
两端都跌破 800 字下限。所以先取最小段数,再确保每段不低于下限。
"""
n = len(chain)
max_events = _max_events()
min_events = _min_events()
if n <= max_events:
return [chain]
parts = math.ceil(n / max_events)
while parts > 1 and math.ceil(n / parts) < min_events:
parts -= 1
size = math.ceil(n / parts)
return [chain[i:i + size] for i in range(0, n, size)]
def plan_episodes(
thread: Thread,
gap_years: int = GAP_YEARS,
min_chars: int = MIN_CHARS,
max_chars: int = MAX_CHARS,
) -> list[Episode]:
"""给定主角组,切出剧集序列。"""
if not thread.events:
return []
chains = _split_chains(thread.events, gap_years)
merged = _merge_short(chains, _min_events())
episodes: list[Episode] = []
for chain in merged:
segments = _split_long(chain)
for i, segment in enumerate(segments, 1):
episodes.append(
Episode(
index=len(episodes) + 1,
start_year=segment[0].year,
end_year=segment[-1].year,
events=segment,
part=i,
parts=len(segments),
)
)
return episodes
def cast_of(world: World, thread: Thread, episode: Episode) -> tuple[list[int], list[int]]:
"""返回 (本集出场的主角, 本集出现的对手/外人)。"""
counter: dict[int, int] = {}
for e in episode.events:
for fid in e.figure_ids():
counter[fid] = counter.get(fid, 0) + 1
members = set(thread.members)
heros = sorted((f for f in counter if f in members), key=lambda f: -counter[f])
others = sorted((f for f in counter if f not in members), key=lambda f: -counter[f])
return heros, others
def render_plan(world: World, thread: Thread, episodes: list[Episode], top_others: int = 3) -> str:
"""剧集清单:集号、年份范围、事件条数、主角与对手。"""
lines = [
f"线索主角:{'、'.join(world.figure_name(m) for m in thread.members)}",
f"线索跨度:{thread.label} | 事件 {len(thread.events)} 条 | 规划 {len(episodes)} 集",
"",
]
for ep in episodes:
heros, others = cast_of(world, thread, ep)
hero_names = "、".join(world.figure_name(h) for h in heros[:4]) or "(本集无主角出场)"
other_names = "、".join(world.figure_name(o) for o in others[:top_others])
budget = f"{ep.estimated_chars} 字估算"
mark = "" if MIN_CHARS <= ep.estimated_chars <= MAX_CHARS else " ⚠ 超出预算"
lines.append(
f" 第 {ep.index:2d} 集({ep.span})| {len(ep.events):2d} 条事件 | {budget}{mark}"
+ (f" | 上/下:{ep.part}/{ep.parts}" if ep.parts > 1 else "")
)
lines.append(f" 主角:{hero_names}")
if other_names:
lines.append(f" 对手/外人:{other_names}")
return "\n".join(lines)
def budget_report(episodes: list[Episode]) -> str:
if not episodes:
return "无剧集"
inside = sum(1 for e in episodes if MIN_CHARS <= e.estimated_chars <= MAX_CHARS)
sizes = [len(e.events) for e in episodes]
return (
f"共 {len(episodes)} 集;事件条数 {min(sizes)}–{max(sizes)};"
f"估算字数 {min(e.estimated_chars for e in episodes)}–{max(e.estimated_chars for e in episodes)};"
f"落在 {MIN_CHARS}–{MAX_CHARS} 预算内的 {inside}/{len(episodes)} 集"
"(估算按每条约 95 字,真实字数在生成时会再校正)"
)
+223
View File
@@ -0,0 +1,223 @@
"""5 维度自评闸门:判断一稿是不是"流水账",并决定重写还是换线索。
访谈决策:
- 5 个维度各 0–5 分,总分 <12 判平淡
- 不合格先重写本集一次,仍不合格才换线索
- 放弃原因写进仓库 notes/ 附录
- 设失败上限,防止「重写→换线索」空转
为什么要这道闸门:上一版章节读起来像流水账,而"平淡"是模型最容易自我感觉良好的东西。
因此评分必须结构化、维度必须问得具体,且要有一次反例自检证明它真的能判不合格。
"""
from __future__ import annotations
import json
import re
from collections.abc import Callable
from dataclasses import dataclass, field
from pathlib import Path
from dfannals.config import MAX_TOKENS_STRUCTURED, MODEL, Gateway
from dfannals.llm import chat
# 维度:键、标签、问什么
DIMENSIONS: tuple[tuple[str, str, str], ...] = (
("goal", "主角目标", "主角在本集里有没有明确的目标,并为此做出主动选择?全程被动得分低"),
("escalation", "冲突升级", "冲突是否逐级升级,而不是同一件事反复发生?"),
("turning", "真正转折", "本集有没有真正的转折(立场改变、关系破裂、意外后果)?"),
("rival", "对手具体", "对手是不是一个具体的人,有自己的动机?只是数字或人群得分低"),
("beyond_bodycount", "超越战斗计数", "抛开对战次数,本集还有没有可讲的内容(算计、情感、抉择)?"),
)
THRESHOLD = 12 # 总分低于它判平淡
MAX_SCORE = 5 # 每个维度满分
EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
# 失败上限:防止反复重写/换线空转
MAX_REWRITES = 1 # 每集最多重写次数(访谈决定)
MAX_CANDIDATES = 3 # 最多尝试几条线索
@dataclass
class Verdict:
total: int
scores: dict[str, int]
reason: str
@property
def passed(self) -> bool:
return self.total >= THRESHOLD
def weak(self, limit: int = 2) -> list[str]:
labels = {k: label for k, label, _ in DIMENSIONS}
ranked = sorted(self.scores.items(), key=lambda kv: kv[1])
return [f"{labels.get(k, k)}({v})" for k, v in ranked[:limit]]
def render(self) -> str:
parts = "、".join(f"{label}={self.scores.get(key, '?')}" for key, label, _ in DIMENSIONS)
verdict = "通过" if self.passed else "判平淡"
return f"{verdict} 总分 {self.total}/{len(DIMENSIONS) * MAX_SCORE}({parts})|{self.reason}"
DIMENSION_BLOCK = "\n".join(
f'- "{key}": {hint}(0–5 分)' for key, _label, hint in DIMENSIONS
)
EVAL_INSTRUCTIONS = f"""你是严格的文学编辑,只做一件事:判断下面这集稿子是不是"流水账"。
按 5 个维度各打 0–5 分:
{DIMENSION_BLOCK}
打分要求:
- 不要客气,不要因为文笔通顺就给分;只看叙事本身有没有戏。
- 主角全程被动、只是"发生了很多事",goal 与 escalation 必须给低分。
- 通篇只是同一类事件重复(例如反复对战、反复被杀),turning 与 beyond_bodycount 必须给低分。
只输出 JSON,不要解释、不要 Markdown 代码块:
{{"goal": 0-5, "escalation": 0-5, "turning": 0-5, "rival": 0-5, "beyond_bodycount": 0-5, "reason": "一句话说明最弱的地方"}}
"""
def build_eval_prompt(chapter_text: str) -> str:
"""拼接自评提示词。
刻意不用 str.format:提示词里的 JSON 例子含大括号,用 format 会被当成占位符
(曾直接报 KeyError: '"goal"')。
"""
return EVAL_INSTRUCTIONS + "\n待评稿件:\n---\n" + chapter_text + "\n---\n"
def _parse(raw: str) -> dict:
text = EDGE_RE.sub("", raw).strip()
start, end = text.find("{"), text.rfind("}")
if start < 0 or end <= start:
raise ValueError(f"自评没有返回 JSON:{raw[:200]}")
try:
data = json.loads(text[start:end + 1])
except json.JSONDecodeError as exc:
raise ValueError(f"自评返回的 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
if not isinstance(data, dict):
raise ValueError(f"自评返回的不是对象:{type(data).__name__}")
return data
def evaluate(
chapter_text: str,
*,
gateway: Gateway | None = None,
model: str = MODEL,
) -> Verdict:
"""让模型按 5 个维度打分。返回结构化结果。"""
reply = chat(
[{"role": "user", "content": build_eval_prompt(chapter_text)}],
gateway=gateway,
model=model,
max_tokens=MAX_TOKENS_STRUCTURED,
temperature=0.0,
)
data = _parse(reply.content)
scores: dict[str, int] = {}
for key, _label, _hint in DIMENSIONS:
value = data.get(key)
if not isinstance(value, (int, float)):
raise ValueError(f"自评维度 {key} 缺失或非数字:{data.get(key)!r}")
scores[key] = max(0, min(MAX_SCORE, int(value)))
reason = str(data.get("reason", "")).strip() or "(未给出理由)"
return Verdict(total=sum(scores.values()), scores=scores, reason=reason)
@dataclass
class GateRun:
"""一次闸门流程的完整记录。"""
episode_key: str
verdicts: list[Verdict] = field(default_factory=list)
accepted: bool = False
rewrites: int = 0
@property
def final(self) -> Verdict | None:
return self.verdicts[-1] if self.verdicts else None
def render(self) -> str:
lines = [f"闸门:{self.episode_key},重写 {self.rewrites} 次,"
+ ("通过" if self.accepted else "未通过")]
for i, v in enumerate(self.verdicts, 1):
lines.append(f" 第 {i} 稿:{v.render()}")
return "\n".join(lines)
def gate_episode(
episode_key: str,
write: Callable[[int, Verdict | None], str],
*,
max_rewrites: int = MAX_REWRITES,
evaluate_fn: Callable[[str], Verdict] = evaluate,
on_event: Callable[[str], None] | None = None,
) -> tuple[str, GateRun]:
"""写 → 自评 → 不合格则重写一次 → 仍不合格则交给调用方换线索。
``write(attempt, previous_verdict)``:attempt 为 0 表示初稿,
大于 0 表示重写,previous_verdict 是上一稿的评分(供重写时指出问题)。
"""
run = GateRun(episode_key=episode_key)
text = write(0, None)
verdict = evaluate_fn(text)
run.verdicts.append(verdict)
while not verdict.passed and run.rewrites < max_rewrites:
run.rewrites += 1
if on_event:
on_event(f"{episode_key} 第 {run.rewrites} 次重写:{verdict.weak()}")
text = write(run.rewrites, verdict)
verdict = evaluate_fn(text)
run.verdicts.append(verdict)
run.accepted = verdict.passed
return text, run
NOTES_DIR_NAME = "notes"
SWITCH_LOG = "switched-threads.md"
def append_switch_note(
project_dir: Path,
*,
world_name: str,
members: list[str],
span: str,
verdict: Verdict | None,
reason: str,
) -> Path:
"""把"为什么放弃这条线索"写进仓库 notes/ 附录(访谈决定:公开可查)。"""
notes = project_dir / NOTES_DIR_NAME
notes.mkdir(parents=True, exist_ok=True)
path = notes / SWITCH_LOG
header = "# 被放弃的线索\n\n记录每一组挑出来、但写不成故事的线索,以及放弃原因。\n"
if not path.is_file():
path.write_text(header, encoding="utf-8")
score = verdict.render() if verdict else "(未评分)"
entry = (
f"\n## {world_name}|{'、'.join(members)}\n\n"
f"- 线索跨度:{span}\n"
f"- 放弃原因:{reason}\n"
f"- 自评:{score}\n"
)
with path.open("a", encoding="utf-8") as fh:
fh.write(entry)
return path
def append_attempt_note(project_dir: Path, text: str) -> Path:
"""一般性记录(例如失败上限触发)。"""
notes = project_dir / NOTES_DIR_NAME
notes.mkdir(parents=True, exist_ok=True)
path = notes / SWITCH_LOG
if not path.is_file():
path.write_text("# 被放弃的线索\n\n", encoding="utf-8")
with path.open("a", encoding="utf-8") as fh:
fh.write(text if text.startswith("\n") else "\n" + text)
return path
+81 -20
View File
@@ -14,26 +14,29 @@ from typing import Any
from defusedxml.ElementTree import iterparse
# 事件里指向其他实体的字段 → 指向哪类对象
ID_FIELDS: dict[str, str] = {
"hfid": "figure",
"hist_figure_id": "figure",
"slayer_hfid": "figure",
"slayer_item_id": "artifact",
"target_hfid": "figure",
"source_hfid": "figure",
"site_id": "site",
"site_civ_id": "entity",
"entity_id": "entity",
"attacker_civ_id": "entity",
"defender_civ_id": "entity",
"artifact_id": "artifact",
"region_id": "region",
"feature_layer_id": "region",
"deity": "figure",
"worshipper_hfid": "figure",
"creature_id": "creature",
}
# 字段分类规则来自对真实史料全部 225 种字段的穷举:
# 含 "hfid" 的字段全部指向历史人物(共 40+ 个:hfid / hfid_target / group_1_hfid /
# slayer_hfid / snatcher_hfid / seeker_hfid / wounder_hfid / teacher_hfid /
# conspirator_hfid ...);而 target_enid、identity_id、master_wcid、slayer_item_id 不是。
# 旧版手工维护的字典只列了 4 个人物字段,导致关系信息在素材里被丢掉。
_ENTITY_SUFFIXES = ("_entity_id", "_civ_id", "_enid")
_REGION_FIELDS = {"region_id", "feature_layer_id", "subregion_id"}
def kind_of(tag: str) -> str | None:
"""事件字段名 → 它指向的对象类别(非引用类字段返回 None)。"""
t = tag.lower()
if "hfid" in t or t == "deity" or t.endswith("_deity"):
return "figure"
if "artifact" in t:
return "artifact"
if t == "site_id" or t.endswith("_site_id"):
return "site"
if t in ("entity_id", "target_enid") or t.endswith(_ENTITY_SUFFIXES):
return "entity"
if t in _REGION_FIELDS:
return "region"
return None
@dataclass
@@ -81,6 +84,36 @@ class Event:
vals = self.fields.get(tag)
return vals[0] if vals else ""
def ids_of_kind(self, kind: str) -> list[int]:
"""本事件中指向该类对象的全部有效 id(去重、剔除 -1 占位)。"""
out: list[int] = []
for tag, vals in self.fields.items():
if kind_of(tag) != kind:
continue
for raw in vals:
i = _int_or_none(raw)
if i is not None and i >= 0 and i not in out:
out.append(i)
return out
def figure_ids(self) -> list[int]:
return self.ids_of_kind("figure")
def site_ids(self) -> list[int]:
return self.ids_of_kind("site")
def entity_ids(self) -> list[int]:
return self.ids_of_kind("entity")
def field_kinds(self) -> dict[str, list[int]]:
"""{类别: [id, ...]},供渲染与建图使用。"""
out: dict[str, list[int]] = {}
for kind in ("figure", "site", "entity", "artifact", "region"):
ids = self.ids_of_kind(kind)
if ids:
out[kind] = ids
return out
@dataclass
class Entity:
@@ -138,6 +171,34 @@ class World:
st = self.sites.get(sid)
return pretty(st.name) if st and st.name else f"Site#{sid}"
def render_refs(self, event: Event) -> list[str]:
"""把一条事件里的引用字段渲染成 ``tag=名字``。
保留字段名是有意的:slayer_hfid=谁 与 hfid=谁 语义完全不同,
丢掉字段名就等于丢掉了“谁对谁做了什么”。
负 id(DF 的“无此项”占位)与查不到名字的字段直接跳过。
"""
out: list[str] = []
for tag, vals in event.fields.items():
kind = kind_of(tag)
if kind is None:
continue
for raw in vals:
i = _int_or_none(raw)
if i is None or i < 0:
continue
if kind == "figure":
name = self.figure_name(i)
elif kind == "site":
name = self.site_name(i)
elif kind == "entity":
name = self.entity_name(i)
else:
name = ""
if name:
out.append(f"{tag}={name}")
return out
def event_years(self) -> tuple[int, int]:
years = [e.year for e in self.events if e.year]
return (min(years), max(years)) if years else (0, 0)
+1
View File
@@ -32,6 +32,7 @@ def _git(*args: str, check: bool = True) -> subprocess.CompletedProcess:
env={**_base_env(), **env},
capture_output=True,
text=True,
check=False, # 下面是自定义错误处理,故意不用 check=True
)
if check and proc.returncode != 0:
raise RuntimeError(f"git {' '.join(args)} 失败:{proc.stderr.strip() or proc.stdout.strip()}")
+11 -10
View File
@@ -93,22 +93,23 @@ def category(event: Event) -> str:
def prominence(world: World) -> Counter[int]:
"""每个人物被卷入的事件次数——用来衡量他在史料里有多重要。"""
"""每个人物被卷入的事件次数——用来衡量他在史料里有多重要。
改用字段分类器后覆盖了 40+ 个人物字段(旧版只数 6 个),数值尺度整体变大,
因此 PROMINENT 阈值必须按新尺度重校。结果缓存在 World 上,避免反复重算。
"""
cached = getattr(world, "_prominence", None)
if cached is not None:
return cached
counts: Counter[int] = Counter()
for e in world.events:
seen: set[int] = set()
for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity", "worshipper_hfid"):
for fid in e.ids(tag):
seen.add(fid)
counts.update(seen)
counts.update(e.figure_ids())
world._prominence = counts # type: ignore[attr-defined]
return counts
def _participants(event: Event) -> list[int]:
out: list[int] = []
for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity"):
out.extend(event.ids(tag))
return out
return event.figure_ids()
# 人物重要度阈值:参与事件数达到这个量级才算"要角"
+8 -19
View File
@@ -130,23 +130,10 @@ def material(world: World, chapter: Chapter, max_figures: int = 40) -> str:
participation: dict[int, int] = {}
for e in chapter.events:
chunks = [f"{e.year}年", e.type]
for tag, kind in lg.ID_FIELDS.items():
for i in e.ids(tag):
# id 为负是“无此项”,不要渲染成 HF#-1 这类噪音
if kind == "figure":
name = world.figure_name(i)
elif kind == "site":
name = world.site_name(i)
elif kind == "entity":
name = world.entity_name(i)
else:
name = ""
if not name:
continue
chunks.append(f"{tag}={name}")
if kind == "figure":
participation[i] = participation.get(i, 0) + 1
# render_refs 会带出双方(比如 slayer_hfid=谁),而不是只剩单方面 hfid
chunks = [f"{e.year}年", e.type, *world.render_refs(e)]
for i in e.figure_ids():
participation[i] = participation.get(i, 0) + 1
extra = []
for field_name in ("state", "reason", "circumstance"):
v = e.text(field_name)
@@ -184,8 +171,10 @@ def overview(world: World, chapters: list[Chapter], limit: int = 10) -> str:
lo, hi = world.event_years()
lines = [
f"世界:{world.name}" + (f"({world.altname})" if world.altname else ""),
f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}",
(
f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}"
),
f"规划章节:{len(chapters)} 章(每章约 {YEARS_PER_CHAPTER} 年 / 最多 {EVENTS_PER_CHAPTER} 条史料)",
]
for c in chapters[:limit]:
+287
View File
@@ -0,0 +1,287 @@
"""人物线索挖掘:从史料里找出「关系密切、有戏」的 3–6 人小圈子。
设计依据全部来自对 Mon Sagus 真实数据(403853 条事件)的实测:
1. **不能用裸互动次数排序。** 实测排名第一的簇 49 次互动里有 35 次是
``hf relationship denied``——反复请求建立关系、反复被拒的循环,是统计噪声。
因此整类剔除(访谈已定)。
2. **同一对同一类型反复出现也不是故事。** 修掉上一条之后,候选又变成
``对战×27`` 这种「同一对人反复互殴」的机械循环。因此在计数时对
(人物对, 事件类型) 做封顶,避免刷量取胜——这也正是访谈定的
「转折/冲突多样性为主、互动次数为次」。
3. **主线必须是"人"。** 种族分布实测:ELF/GOBLIN/DWARF/HUMAN/KOBOLD 五族加
``*_MAN`` 人形族占全部历史人物的 96.5%,其余 400 多种是夜行怪、野兽、
泰坦、实验体。所以这里用白名单,而不是越列越长的黑名单。
4. 强边阈值 ≥5 次反复互动;簇规模 3–6 人;跨度 ≥30 年。
实测:阈值提到 8 会一条候选都不剩,5 是正确档位。
"""
from __future__ import annotations
import json
from collections import Counter, defaultdict
from dataclasses import dataclass, field
from pathlib import Path
from dfannals.legends import Event, World
# 事件类型 → (权重, 中文标签, 是否算转折点)
SIGNALS: dict[str, tuple[int, str, bool]] = {
"hf simple battle event": (1, "对战", False),
"add hf hf link": (1, "结缘", False),
"hfs formed reputation relationship": (1, "结缘", False),
"competition": (1, "竞争", False),
"hf wounded": (2, "搏杀", True),
"hf confronted": (2, "对峙", True),
"hf interrogated": (2, "审讯", True),
"failed intrigue corruption": (2, "阴谋", True),
"remove hf hf link": (3, "反目", True),
"hf abducted": (3, "绑架", True),
"hf convicted": (3, "定罪", True),
"entity persecuted": (3, "迫害", True),
"hf enslaved": (3, "奴役", True),
"hf ransomed": (3, "贖金", True),
"entity overthrown": (3, "推翻", True),
"failed frame attempt": (3, "构陷", True),
}
# 明确剔除:重复性请求,实测会刷满排序(访谈已定)
EXCLUDED_SIGNALS = ("hf relationship denied",)
# 文明种族白名单:实测覆盖 96.5% 的历史人物
CIVILIZED_RACES = frozenset({"DWARF", "ELF", "HUMAN", "GOBLIN", "KOBOLD"})
DEFAULT_MIN_INTERACTIONS = 5 # 强边阈值:≥5 次反复互动
DEFAULT_SIZE_RANGE = (3, 6)
DEFAULT_MIN_SPAN = 30 # 跨度 ≥30 年才撑得起连载
PAIR_TYPE_CAP = 4 # 同一对、同一类型最多计 4 次
MIN_SIGNAL_KINDS = 2 # 至少两种不同信号,否则只是单一类型的重复
INTERACTION_CAP = 40 # 互动总量封顶,防止刷量取胜
@dataclass
class Edge:
a: int
b: int
raw: int = 0
types: Counter = field(default_factory=Counter)
first_year: int = 0
last_year: int = 0
@property
def effective(self) -> int:
"""封顶后的有效互动次数。"""
return sum(min(n, PAIR_TYPE_CAP) for n in self.types.values())
@property
def turns(self) -> int:
"""封顶后的转折点次数。"""
return sum(min(n, PAIR_TYPE_CAP) for t, n in self.types.items() if SIGNALS[t][2])
@dataclass
class Thread:
members: list[int]
interactions: int
turns: int
span: int
first_year: int
last_year: int
types: Counter
races: Counter
non_person_races: list[str]
events: list[Event]
@property
def kinds(self) -> int:
return len(self.types)
@property
def score(self) -> int:
"""转折为主、多样性次之、互动次数封顶计入(访谈定的排序原则)。"""
return self.turns * 10 + self.kinds * 6 + min(self.interactions, INTERACTION_CAP)
@property
def label(self) -> str:
return f"{self.first_year}–{self.last_year}({self.span} 年)"
def signal_line(self) -> str:
return "、".join(f"{SIGNALS[t][1]}×{n}" for t, n in self.types.most_common() if t in SIGNALS)
def is_person(race: str) -> bool:
"""是否属于"可作为主角的人":文明种族或人形族(*_MAN)。"""
r = (race or "").upper()
return r in CIVILIZED_RACES or r.endswith("_MAN")
def build_edges(world: World) -> dict[tuple[int, int], Edge]:
"""按有叙事含义的信号建立人物之间的边。"""
edges: dict[tuple[int, int], Edge] = {}
for e in world.events:
if e.type not in SIGNALS:
continue
people = e.figure_ids()
if len(people) < 2:
continue
for i in range(len(people)):
for j in range(i + 1, len(people)):
key = (min(people[i], people[j]), max(people[i], people[j]))
edge = edges.get(key)
if edge is None:
edge = Edge(a=key[0], b=key[1], first_year=e.year, last_year=e.year)
edges[key] = edge
edge.raw += 1
edge.types[e.type] += 1
edge.first_year = min(edge.first_year, e.year)
edge.last_year = max(edge.last_year, e.year)
return edges
def find_threads(
world: World,
min_interactions: int = DEFAULT_MIN_INTERACTIONS,
size_range: tuple[int, int] = DEFAULT_SIZE_RANGE,
min_span: int = DEFAULT_MIN_SPAN,
require_people: bool = True,
min_kinds: int = MIN_SIGNAL_KINDS,
limit: int = 0,
) -> list[Thread]:
"""返回按剧情张力排序的候选线索。"""
edges = build_edges(world)
strong = {k: v for k, v in edges.items() if v.effective >= min_interactions}
parent: dict[int, int] = {}
def find(x: int) -> int:
parent.setdefault(x, x)
while parent[x] != x:
parent[x] = parent[parent[x]]
x = parent[x]
return x
for (a, b) in strong:
ra, rb = find(a), find(b)
if ra != rb:
parent[ra] = rb
members_of: dict[int, set[int]] = defaultdict(set)
for x in parent:
members_of[find(x)].add(x)
events_by_pair: dict[tuple[int, int], list[Event]] = defaultdict(list)
for e in world.events:
if e.type not in SIGNALS:
continue
people = e.figure_ids()
if len(people) < 2:
continue
for i in range(len(people)):
for j in range(i + 1, len(people)):
events_by_pair[(min(people[i], people[j]), max(people[i], people[j]))].append(e)
lo_size, hi_size = size_range
threads: list[Thread] = []
for members in members_of.values():
if not (lo_size <= len(members) <= hi_size):
continue
races = Counter(world.figures[m].race for m in members if m in world.figures)
if require_people and not all(is_person(r) for r in races):
continue
inner = {k: v for k, v in strong.items() if k[0] in members and k[1] in members}
if not inner:
continue
types: Counter = Counter()
for v in inner.values():
types.update({t: min(n, PAIR_TYPE_CAP) for t, n in v.types.items()})
if len(types) < min_kinds:
continue
first = min(v.first_year for v in inner.values())
last = max(v.last_year for v in inner.values())
if last - first < min_span:
continue
evs: list[Event] = []
seen_ids: set[int] = set()
for k in inner:
for e in events_by_pair.get(k, []):
if e.id not in seen_ids:
seen_ids.add(e.id)
evs.append(e)
evs.sort(key=lambda e: (e.year, e.seconds72, e.id))
threads.append(
Thread(
members=sorted(members),
interactions=sum(v.effective for v in inner.values()),
turns=sum(v.turns for v in inner.values()),
span=last - first,
first_year=first,
last_year=last,
types=types,
races=races,
non_person_races=sorted(r for r in races if not is_person(r)),
events=evs,
)
)
threads.sort(key=lambda t: (-t.score, -t.span))
return threads[:limit] if limit else threads
def render_report(world: World, threads: list[Thread], per_thread: int = 3) -> str:
"""给人看的候选明细。"""
lines = [
f"候选线索 {len(threads)} 条(转折为主、多样性次之;同一对同类事件已封顶)",
"筛选:强边 ≥5 次互动、3–6 人、跨度 ≥30 年、含巨兽的簇已剔除",
"",
]
for i, t in enumerate(threads, 1):
races = "、".join(f"{r}×{n}" for r, n in t.races.most_common())
monsters = "、".join(t.non_person_races) if t.non_person_races else "无"
lines.append(
f"{i:2d}. 得分 {t.score:4d} | {len(t.members)} 人 | 有效互动 {t.interactions:3d} | "
f"转折 {t.turns:2d} | 信号 {t.kinds} 种 | 跨度 {t.label} | 巨兽:{monsters}"
)
lines.append(f" 种族:{races}")
lines.append(f" 信号:{t.signal_line()}")
lines.append(f" 成员:{'、'.join(world.figure_name(m) for m in t.members)}")
for e in t.events[:per_thread]:
lines.append(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
lines.append("")
return "\n".join(lines)
def to_json(world: World, threads: list[Thread]) -> dict:
return {
"world": world.name,
"candidates": [
{
"rank": i,
"score": t.score,
"members": [
{"id": m, "name": world.figure_name(m),
"race": world.figures[m].race if m in world.figures else ""}
for m in t.members
],
"interactions": t.interactions,
"turns": t.turns,
"kinds": t.kinds,
"span": t.span,
"first_year": t.first_year,
"last_year": t.last_year,
"signals": {SIGNALS[k][1]: v for k, v in t.types.items() if k in SIGNALS},
"event_ids": [e.id for e in t.events],
}
for i, t in enumerate(threads, 1)
],
}
def save_json(world: World, threads: list[Thread], path: Path) -> Path:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(to_json(world, threads), ensure_ascii=False, indent=2), encoding="utf-8")
return path
+6 -22
View File
@@ -19,19 +19,7 @@ category = score.category
def event_line(world: World, e: Event) -> str:
"""把一条事件压成一行可读文字。"""
chunks: list[str] = [e.type]
for tag, kind in lg.ID_FIELDS.items():
for i in e.ids(tag):
if kind == "figure":
name = world.figure_name(i)
elif kind == "site":
name = world.site_name(i)
elif kind == "entity":
name = world.entity_name(i)
else:
name = ""
if name:
chunks.append(f"{tag}={name}")
chunks: list[str] = [e.type, *world.render_refs(e)]
return f"- **{e.year}** [{category(e)}] " + " · ".join(chunks)
@@ -51,8 +39,10 @@ def build_timeline(world: World, min_score: int = 70, per_decade: int = 14) -> s
f"- 事件总数:{len(world.events)},其中重要事件 {len(important)}",
f"- 历史人物 {len(world.figures)} 位 · 文明与组织 {len(world.entities)} 个 · 地点 {len(world.sites)} 处",
"",
"> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
"完整数据留在本地,不入库。",
(
"> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
"完整数据留在本地,不入库。"
),
"",
]
for decade in sorted(buckets):
@@ -72,13 +62,7 @@ def build_figures(world: World, top: int = 120) -> str:
"""人物索引:按参与事件数排序,给出身份与生卒。"""
involvement: Counter[int] = Counter()
for e in world.events:
seen: set[int] = set()
for tag, kind in lg.ID_FIELDS.items():
if kind != "figure":
continue
for i in e.ids(tag):
seen.add(i)
involvement.update(seen)
involvement.update(e.figure_ids())
lines = [
f"# {world.name} · 人物索引",
+336
View File
@@ -0,0 +1,336 @@
"""人物主线写作与编排:选主角组 → 切集 → 贴着人物写 → 过闸门 → 落盘。
与旧的 chronicle.py 的区别:那里的主语是"年代",一章按 10 年窗口罗列事件;
这里的主语是"人",一集是主角线上一段完整回合,世界大事只在影响到主角时提及。
"""
from __future__ import annotations
import json
from collections import Counter
from collections.abc import Callable
from dataclasses import dataclass, replace
from pathlib import Path
from dfannals import episodes as ep_mod
from dfannals import gate
from dfannals.cast import collect_rows
from dfannals.chronicle import cjk_length
from dfannals.config import (
CHAPTER_MAX_CHARS,
CHAPTER_MIN_CHARS,
MAX_TOKENS_CHAPTER,
PROJECT_DIR,
PROMPT_DIR,
)
from dfannals.legends import Event, World
from dfannals.llm import chat
from dfannals.threads import SIGNALS, Thread
SPEC_FILE = PROMPT_DIR / "biographer.md"
TAIL_CHARS = 600
SELECTION_FILE = PROJECT_DIR / "data" / "selection.json"
@dataclass
class Selection:
rank: int
thread: Thread
reason: str
@dataclass
class Produced:
selection: Selection
episode: ep_mod.Episode
text: str
run: gate.GateRun
path: Path | None = None
@dataclass
class Extended:
"""主角参与的扩展素材。"""
events: list[Event]
elided: Counter
PER_KEY_CAP = 3 # 同一组人 + 同一类型,每集素材最多保留几次
def expand_thread_events(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP) -> Extended:
"""把素材从“成员之间”扩到“主角参与的全部有意义事件”。
实测:某 3 人线索成员之间只有 14 条事件(只能切 1 集),但把他们的对外活动
(构陷、对战、结义、定罪……)算进来有 136 条,足够支撑一卷。
代价是重复:这 136 条里有 64 条是同一类“构陷失败”。所以同一组人 + 同一类型
最多保留 per_key_cap 条,其余计入 elided 供写作时一笔带过——否则又会变流水账。
"""
members = set(thread.members)
seen: Counter = Counter()
keep: list[Event] = []
elided: Counter = Counter()
relevant = [
e for e in world.events
if e.type in SIGNALS and (members & set(e.figure_ids()))
]
relevant.sort(key=lambda e: (e.year, e.seconds72, e.id))
for e in relevant:
key = (e.type, tuple(sorted(e.figure_ids())))
seen[key] += 1
if seen[key] <= per_key_cap:
keep.append(e)
else:
elided[e.type] += 1
return Extended(events=keep, elided=elided)
def extended_plan(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP
) -> tuple[Extended, list[ep_mod.Episode]]:
"""用扩展素材切集(保持切集规则不变)。"""
ext = expand_thread_events(world, thread, per_key_cap=per_key_cap)
planned = ep_mod.plan_episodes(replace(thread, events=ext.events))
return ext, planned
def select_thread(world: World, candidates: list[Thread], rank: int = 1) -> Selection:
"""按挖掘器排序取第 rank 条,并把选择依据写成可读理由。"""
thread = candidates[rank - 1]
names = "、".join(world.figure_name(m) for m in thread.members)
reason = (
f"挖掘器排序第 {rank} 名(转折为主、互动封顶):"
f"{len(thread.members)} 人、有效互动 {thread.interactions}、转折 {thread.turns}、"
f"信号 {thread.kinds} 种({thread.signal_line()})、跨度 {thread.label};成员:{names}"
)
return Selection(rank=rank, thread=thread, reason=reason)
def save_selection(selection: Selection, world: World, path: Path = SELECTION_FILE) -> Path:
"""把主角组的选定依据落到本地(不入库,供追溯)。"""
path.parent.mkdir(parents=True, exist_ok=True)
thread = selection.thread
path.write_text(
json.dumps(
{
"world": world.name,
"rank": selection.rank,
"reason": selection.reason,
"score": thread.score,
"members": [
{"id": m, "name": world.figure_name(m),
"race": world.figures[m].race if m in world.figures else ""}
for m in thread.members
],
"signals": dict(thread.types),
"span": thread.label,
},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
return path
def render_material(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
prev_tail: str = "",
elided: Counter | None = None,
) -> str:
"""把一集素材渲染给传记作者:主角是谁、发生了什么、对手是谁。"""
rows = {r.fid: r for r in collect_rows(world, thread, [episode], max_others=6)}
heros, others = ep_mod.cast_of(world, thread, episode)
lines = [
f"## 本集:第 {episode.index}/{total_episodes} 集,{episode.span}",
"",
"### 主角(本集要写的就是他们)",
"",
]
for fid in heros:
row = rows.get(fid)
if row:
lines.append(f"- {row.name}|{row.race}|{row.span}|身份:{row.identity or '不详'}"
f"|本集出场 {row.appearances} 次")
else:
lines.append(f"- {world.figure_name(fid)}")
lines += ["", "### 本集史料(按时间排序;专名保留游戏原文)", ""]
for e in episode.events:
refs = " · ".join(world.render_refs(e))
extra = [f"{k}={e.text(k)}" for k in ("state", "reason", "circumstance")
if e.text(k) and e.text(k) not in ("-1", "")]
tail = ("|" + ";".join(extra)) if extra else ""
lines.append(f"- {e.year}年 {e.type} · {refs}{tail}")
if others:
lines += ["", "### 对手/相关者(史料里出现,但不是本卷主角)", ""]
for fid in others[:6]:
row = rows.get(fid)
if row:
lines.append(f"- {row.name}|{row.race}|身份:{row.identity or '不详'}"
f"|本集出场 {row.appearances} 次")
else:
lines.append(f"- {world.figure_name(fid)}")
if elided:
hot = "、".join(f"{t}×{n}" for t, n in elided.most_common(6))
lines += [
"",
"### 被省略的同类重复事件",
"",
f"- 本卷还有这些同类事件已被过滤,不要逐条罗列,需要时用一句话带过:{hot}",
]
if prev_tail:
lines += ["", "### 上一集结尾(只用于承接状态,不要复述)", "", prev_tail]
lines += [
"",
"### 写作要求",
"",
f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,专名保留英文。",
"- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。",
"- 史料之外的世界大事不要写;只有影响到主角时才提一句。",
]
return "\n".join(lines)
def build_messages(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
prev_tail: str,
feedback: str | None = None,
elided: Counter | None = None,
) -> list[dict]:
spec = SPEC_FILE.read_text(encoding="utf-8")
material = render_material(world, thread, episode, total_episodes, prev_tail, elided)
user = [material]
if feedback:
user += ["", "### 上一稿的问题(重写时必须针对这些改)", "", feedback]
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
def _repair(messages: list[dict], draft: str, length: int, gateway) -> str:
target = (
f"内容太短({length} 字),请扩写到 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,"
"补充场景与对白,不要注水。"
if length < CHAPTER_MIN_CHARS
else f"内容太长({length} 字),请压缩到 {CHAPTER_MAX_CHARS} 字以内,删次要枝节。"
)
reply = chat(messages + [{"role": "assistant", "content": draft},
{"role": "user", "content": target}], gateway=gateway)
return reply.content
def write_episode(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
prev_tail: str = "",
feedback: str | None = None,
*,
gateway=None,
max_repairs: int = 2,
elided: Counter | None = None,
) -> str:
"""生成一集正文,并把长度校正到 800–1500 字。"""
messages = build_messages(world, thread, episode, total_episodes, prev_tail, feedback, elided)
text = chat(messages, gateway=gateway, max_tokens=MAX_TOKENS_CHAPTER).content
for _ in range(max_repairs):
length = cjk_length(text)
if CHAPTER_MIN_CHARS <= length <= CHAPTER_MAX_CHARS:
break
text = _repair(messages, text, length, gateway)
return text.strip()
def write_file(directory: Path, episode: ep_mod.Episode, text: str) -> Path:
directory.mkdir(parents=True, exist_ok=True)
path = directory / f"{episode.key}.md"
path.write_text(text.strip() + "\n", encoding="utf-8")
return path
def prev_tail_of(directory: Path, planned: list[ep_mod.Episode], index: int) -> str:
"""取上一集结尾作为承接线索。"""
if index <= 1:
return ""
path = directory / f"{planned[index - 2].key}.md"
if not path.is_file():
return ""
return path.read_text(encoding="utf-8").strip()[-TAIL_CHARS:]
def produce_episode(
world: World,
candidates: list[Thread],
*,
episode_index: int = 1,
out_dir: Path | None = None,
gateway=None,
max_candidates: int = gate.MAX_CANDIDATES,
on_event: Callable[[str], None] = print,
) -> Produced:
"""选线索 → 写这一集 → 过闸门;不过就换下一线索(有上限)。"""
if not candidates:
raise RuntimeError("没有候选线索可用")
for rank in range(1, min(max_candidates, len(candidates)) + 1):
selection = select_thread(world, candidates, rank=rank)
thread = selection.thread
ext, planned = extended_plan(world, thread)
if episode_index > len(planned):
on_event(f"第 {rank} 条线索只有 {len(planned)} 集,跳过")
continue
episode = planned[episode_index - 1]
directory = out_dir or (PROJECT_DIR / "output" / "volumes" / "v1" / "chapters")
tail = prev_tail_of(directory, planned, episode_index)
on_event(f"选定线索 {rank}:{'、'.join(world.figure_name(m) for m in thread.members)}")
# 把循环变量绑定成默认参数:否则闭包会绑定到循环变量本身(后面还会变)
def write(
attempt: int,
prev: gate.Verdict | None,
_thread: Thread = thread,
_episode: ep_mod.Episode = episode,
_total: int = len(planned),
_tail: str = tail,
_elided: Counter = ext.elided,
) -> str:
feedback = None
if prev is not None:
feedback = (f"上一稿被判平淡:{prev.render()}。"
f"最弱环节是 {'、'.join(prev.weak())},请围绕这些重写,"
"让人物真正做选择。")
return write_episode(world, _thread, _episode, _total, _tail, feedback,
gateway=gateway, elided=_elided)
text, run = gate.gate_episode(episode.key, write, on_event=on_event)
on_event(run.render())
if run.accepted:
save_selection(selection, world)
path = write_file(directory, episode, text) if out_dir is not None else None
return Produced(selection=selection, episode=episode, text=text, run=run, path=path)
gate.append_switch_note(
PROJECT_DIR,
world_name=world.name,
members=[world.figure_name(m) for m in thread.members],
span=thread.label,
verdict=run.final,
reason=f"重写 {run.rewrites} 次后仍未过闸门(第 {rank} 条线索)",
)
on_event(f"线索 {rank} 未过闸门,换下一条")
raise RuntimeError(f"前 {max_candidates} 条线索都没过闸门,需要人工干预")