第 1 卷(人物主线):第 1–3 集 + 人物表
主角线:Guspu Frillyknots / Ral Fastenhatchets / Alath Blottedmine 由线索挖掘器自动选定,按因果链切集,经 5 维度自评闸门(≥12 分)后入库。
This commit is contained in:
@@ -0,0 +1,225 @@
|
||||
"""每卷人物表:Markdown 表格 + 每人一两句小传。
|
||||
|
||||
访谈决策:
|
||||
- 每卷开头一份人物表,表格 + 一两句小传,随年表一起入库
|
||||
- 人物驱动叙事里,读者记不住 3–6 人加对手配角的英文专名,所以必须有一份可回查的名册
|
||||
|
||||
表格部分完全由史料推导(种族、生卒、出场次数、身份);
|
||||
"身份"不是猜的,而是从事件字段反推——例如某人 8 次出现在 corruptor_hfid 上,
|
||||
就写"构陷者×8"。小传才交给模型写,且必须只使用给定事实。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import factcheck
|
||||
from dfannals.config import MAX_TOKENS_STRUCTURED, Gateway
|
||||
from dfannals.episodes import Episode
|
||||
from dfannals.legends import Event, World
|
||||
from dfannals.llm import chat
|
||||
from dfannals.threads import Thread
|
||||
|
||||
# 事件字段 → 中文角色名(用于反推"身份")
|
||||
ROLE_LABELS: dict[str, str] = {
|
||||
"corruptor_hfid": "构陷发起者",
|
||||
"target_hfid": "被针对者",
|
||||
"wounder_hfid": "行凶者",
|
||||
"woundee_hfid": "受伤者",
|
||||
"snatcher_hfid": "绑走者",
|
||||
"seeker_hfid": "求关系者",
|
||||
"winner_hfid": "胜者",
|
||||
"competitor_hfid": "参赛者",
|
||||
"slayer_hfid": "凶手",
|
||||
"hfid": "当事者",
|
||||
"hfid_target": "关系对象",
|
||||
"teacher_hfid": "师长",
|
||||
"student_hfid": "学徒",
|
||||
"persecutor_hfid": "迫害者",
|
||||
"expelled_hfid": "被逐者",
|
||||
"convicted_hfid": "被定罪者",
|
||||
"interrogator_hfid": "审讯者",
|
||||
"framer_hfid": "构陷者",
|
||||
"fooled_hfid": "受骗者",
|
||||
}
|
||||
|
||||
BIO_LIMIT = 10 # 小传最多覆盖多少人
|
||||
EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
|
||||
|
||||
|
||||
@dataclass
|
||||
class CastRow:
|
||||
fid: int
|
||||
name: str
|
||||
race: str
|
||||
span: str
|
||||
role: str # 主角 / 对手 / 配角
|
||||
appearances: int
|
||||
identity: str
|
||||
|
||||
def as_row(self) -> str:
|
||||
return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | "
|
||||
f"{self.appearances} | {self.role} |")
|
||||
|
||||
|
||||
def _identity_of(world: World, fid: int, events: list[Event]) -> str:
|
||||
"""从事件字段反推这个人在这段历史里扮演什么。"""
|
||||
counts: dict[str, int] = {}
|
||||
for e in events:
|
||||
for tag, vals in e.fields.items():
|
||||
if tag not in ROLE_LABELS:
|
||||
continue
|
||||
for raw in vals:
|
||||
try:
|
||||
if int(raw) == fid:
|
||||
counts[tag] = counts.get(tag, 0) + 1
|
||||
except ValueError:
|
||||
continue
|
||||
if not counts:
|
||||
return ""
|
||||
ranked = sorted(counts.items(), key=lambda kv: -kv[1])[:2]
|
||||
return "、".join(f"{ROLE_LABELS[tag]}×{n}" for tag, n in ranked)
|
||||
|
||||
|
||||
def _span_of(thread: Thread, episodes: list[Episode]) -> str:
|
||||
"""表头跨度以实际切出的剧集为准(线索自身跨度只算成员之间的互动)。"""
|
||||
if not episodes:
|
||||
return thread.label
|
||||
return f"{episodes[0].start_year}–{episodes[-1].end_year}"
|
||||
|
||||
|
||||
def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_others: int = 6) -> list[CastRow]:
|
||||
"""主角优先,其次是出场最多的对手/配角。"""
|
||||
events = [e for ep in episodes for e in ep.events] or thread.events
|
||||
appearances: dict[int, int] = {}
|
||||
for e in events:
|
||||
for fid in e.figure_ids():
|
||||
appearances[fid] = appearances.get(fid, 0) + 1
|
||||
|
||||
members = set(thread.members)
|
||||
rows: list[CastRow] = []
|
||||
|
||||
def make(fid: int, role: str) -> CastRow:
|
||||
fig = world.figures.get(fid)
|
||||
return CastRow(
|
||||
fid=fid,
|
||||
name=world.figure_name(fid),
|
||||
race=fig.race if fig else "",
|
||||
span=fig.alive_span if fig else "生卒不详",
|
||||
role=role,
|
||||
appearances=appearances.get(fid, 0),
|
||||
identity=_identity_of(world, fid, events),
|
||||
)
|
||||
|
||||
for fid in sorted(members, key=lambda f: -appearances.get(f, 0)):
|
||||
rows.append(make(fid, "主角"))
|
||||
others = sorted((f for f in appearances if f not in members), key=lambda f: -appearances[f])
|
||||
for fid in others[:max_others]:
|
||||
rows.append(make(fid, "对手/配角"))
|
||||
return rows
|
||||
|
||||
|
||||
BIO_INSTRUCTIONS = """你在给一份矮人要塞编年史的人物表写小传。
|
||||
|
||||
规则:
|
||||
- 只使用下面给出的事实。**不许编造任何新的人名、地名、组织名。**
|
||||
- 每人 1–2 句中文,口吻像编年史官写的旁注:可以刻薄、可以偏心,但事实必须来自材料。
|
||||
- 姓名一律保留英文原文。
|
||||
- 只输出 JSON,不要解释、不要代码块:{"英文名": "小传", ...}
|
||||
"""
|
||||
|
||||
|
||||
def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
|
||||
lines = [BIO_INSTRUCTIONS, "", "材料:"]
|
||||
for row in rows[:BIO_LIMIT]:
|
||||
facts = [
|
||||
f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次,"
|
||||
f"身份:{row.identity or '不详'},在故事里是{row.role})"
|
||||
]
|
||||
for e in events:
|
||||
if row.fid not in e.figure_ids():
|
||||
continue
|
||||
refs = " · ".join(world.render_refs(e))
|
||||
facts.append(f" {e.year}年 {e.type} {refs}")
|
||||
if len(facts) > 5:
|
||||
break
|
||||
lines.extend(facts)
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _parse_bios(raw: str) -> dict[str, str]:
|
||||
text = EDGE_RE.sub("", raw).strip()
|
||||
start, end = text.find("{"), text.rfind("}")
|
||||
if start < 0 or end <= start:
|
||||
raise ValueError(f"小传没有返回 JSON:{raw[:200]}")
|
||||
try:
|
||||
data = json.loads(text[start:end + 1])
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"小传 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("小传返回的不是对象")
|
||||
return {str(k): str(v) for k, v in data.items()}
|
||||
|
||||
|
||||
def build_cast(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episodes: list[Episode],
|
||||
*,
|
||||
gateway: Gateway | None = None,
|
||||
with_bios: bool = True,
|
||||
) -> tuple[str, list[CastRow], list[factcheck.Suspicion]]:
|
||||
"""返回 (Markdown 文本, 表格行, 小传里查出的可疑专名)。"""
|
||||
rows = collect_rows(world, thread, episodes)
|
||||
events = [e for ep in episodes for e in ep.events] or thread.events
|
||||
|
||||
bios: dict[str, str] = {}
|
||||
if with_bios and rows:
|
||||
reply = chat(
|
||||
[{"role": "user", "content": _bio_prompt(world, rows, events)}],
|
||||
gateway=gateway,
|
||||
max_tokens=MAX_TOKENS_STRUCTURED,
|
||||
)
|
||||
bios = _parse_bios(reply.content)
|
||||
|
||||
suspicions: list[factcheck.Suspicion] = []
|
||||
bios_text = "\n".join(f"{k}:{v}" for k, v in bios.items())
|
||||
if bios_text:
|
||||
suspicions = factcheck.check(bios_text, world)
|
||||
|
||||
lines = [
|
||||
f"# {world.name} · 人物表",
|
||||
"",
|
||||
f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
|
||||
f"线索跨度:{_span_of(thread, episodes)}",
|
||||
"",
|
||||
"| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |",
|
||||
"|---|---|---|---|---|---|",
|
||||
]
|
||||
lines.extend(row.as_row() for row in rows)
|
||||
|
||||
if bios:
|
||||
lines += ["", "## 小传", ""]
|
||||
for row in rows:
|
||||
bio = bios.get(row.name) or bios.get(row.name.lower())
|
||||
if bio:
|
||||
lines.append(f"**{row.name}** —— {bio}")
|
||||
lines.append("")
|
||||
|
||||
if suspicions:
|
||||
lines += [
|
||||
"",
|
||||
"> ⚠️ 小传中以下专名未在史料中找到,可能是演绎时误引入:"
|
||||
+ "、".join(f"{s.name}×{s.count}" for s in suspicions[:8]),
|
||||
"",
|
||||
]
|
||||
|
||||
return "\n".join(lines) + "\n", rows, suspicions
|
||||
|
||||
|
||||
def write_cast(path: Path, text: str) -> Path:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(text, encoding="utf-8")
|
||||
return path
|
||||
@@ -62,8 +62,8 @@ class State:
|
||||
|
||||
def cjk_length(text: str) -> int:
|
||||
"""按中文习惯计字数:CJK 字符与拉丁单词都算 1。"""
|
||||
without_code = re.sub(r"```.*?```", "", text, flags=re.S)
|
||||
without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.M)
|
||||
without_code = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
|
||||
without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.MULTILINE)
|
||||
n = 0
|
||||
for ch in without_meta:
|
||||
if (
|
||||
|
||||
+154
-2
@@ -16,9 +16,26 @@ import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import chronicle, factcheck, legends, publish, timeline
|
||||
from dfannals import (
|
||||
cast,
|
||||
chronicle,
|
||||
episodes,
|
||||
factcheck,
|
||||
legends,
|
||||
publish,
|
||||
threads,
|
||||
timeline,
|
||||
)
|
||||
from dfannals import slice as chapter_slice
|
||||
from dfannals.config import EXPORT_DIR, GITEA_WEB, chapter_dir, ensure_dirs, volume_dir
|
||||
from dfannals import volume as volume_mod
|
||||
from dfannals.config import (
|
||||
DATA_DIR,
|
||||
EXPORT_DIR,
|
||||
GITEA_WEB,
|
||||
chapter_dir,
|
||||
ensure_dirs,
|
||||
volume_dir,
|
||||
)
|
||||
|
||||
|
||||
def find_exports(export_dir: Path) -> list[Path]:
|
||||
@@ -142,6 +159,109 @@ def cmd_next(args: argparse.Namespace) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_threads(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
found = threads.find_threads(
|
||||
world,
|
||||
min_interactions=args.min_interactions,
|
||||
min_span=args.min_span,
|
||||
limit=args.top,
|
||||
)
|
||||
print(threads.render_report(world, found, per_thread=args.samples))
|
||||
path = threads.save_json(world, found, DATA_DIR / "threads.json")
|
||||
print(f"共 {len(found)} 条候选;明细已写入 {path}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_episodes(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
found = threads.find_threads(
|
||||
world, min_interactions=args.min_interactions, min_span=args.min_span
|
||||
)
|
||||
if not found:
|
||||
print("没有候选线索")
|
||||
return 1
|
||||
if not 1 <= args.thread <= len(found):
|
||||
print(f"候选只有 {len(found)} 条,--thread 超出范围")
|
||||
return 1
|
||||
thread = found[args.thread - 1]
|
||||
_, planned = volume_mod.extended_plan(world, thread)
|
||||
print(episodes.render_plan(world, thread, planned))
|
||||
print()
|
||||
print(episodes.budget_report(planned))
|
||||
if args.samples:
|
||||
print("\n=== 前两集事件样例 ===")
|
||||
for ep in planned[:2]:
|
||||
print(f" 第 {ep.index} 集({ep.span})")
|
||||
for e in ep.events[: args.samples]:
|
||||
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_cast(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
found = threads.find_threads(
|
||||
world, min_interactions=args.min_interactions, min_span=args.min_span
|
||||
)
|
||||
if not found:
|
||||
print("没有候选线索")
|
||||
return 1
|
||||
if not 1 <= args.thread <= len(found):
|
||||
print(f"候选只有 {len(found)} 条,--thread 超出范围")
|
||||
return 1
|
||||
|
||||
thread = found[args.thread - 1]
|
||||
_, planned = volume_mod.extended_plan(world, thread)
|
||||
text, rows, suspicions = cast.build_cast(
|
||||
world, thread, planned, with_bios=not args.no_bios
|
||||
)
|
||||
if args.dry_run:
|
||||
print(text[:2000])
|
||||
return 0
|
||||
|
||||
state = chronicle.State.load()
|
||||
volume = state.volume if state.world == world.name else 1
|
||||
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
|
||||
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
|
||||
print(factcheck.report(suspicions) if suspicions else "专名校验:小传里的专名全部能在史料中找到。")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_episode(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
found = threads.find_threads(
|
||||
world, min_interactions=args.min_interactions, min_span=args.min_span
|
||||
)
|
||||
if not found:
|
||||
print("没有候选线索")
|
||||
return 1
|
||||
|
||||
out_dir = None if args.dry_run else chapter_dir(1)
|
||||
produced = volume_mod.produce_episode(
|
||||
world, found, episode_index=args.index, out_dir=out_dir
|
||||
)
|
||||
|
||||
print("选择依据:" + produced.selection.reason)
|
||||
print(f"本集字数:{chronicle.cjk_length(produced.text)}")
|
||||
if args.dry_run:
|
||||
print("\n--- 试运行,不落盘 ---\n")
|
||||
print(produced.text[:1600])
|
||||
return 0
|
||||
|
||||
print(f"已写入 {produced.path}")
|
||||
if not args.no_push:
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 1 卷第 {produced.episode.index} 集(人物主线):{produced.episode.span}"
|
||||
)
|
||||
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_run(args: argparse.Namespace) -> int:
|
||||
rc = cmd_index(args)
|
||||
if rc:
|
||||
@@ -163,6 +283,38 @@ def main(argv: list[str] | None = None) -> int:
|
||||
p = sub.add_parser("plan", help="显示切章方案")
|
||||
p.set_defaults(func=cmd_plan)
|
||||
|
||||
p = sub.add_parser("threads", help="挖掘人物线索候选")
|
||||
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
|
||||
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
|
||||
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
|
||||
help="线索最小跨度年数(默认 30)")
|
||||
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
|
||||
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
|
||||
p.set_defaults(func=cmd_threads)
|
||||
|
||||
p = sub.add_parser("episodes", help="按人物线索切出剧集")
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
|
||||
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
|
||||
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
|
||||
p.set_defaults(func=cmd_episodes)
|
||||
|
||||
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
|
||||
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
|
||||
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
|
||||
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
|
||||
p.set_defaults(func=cmd_cast)
|
||||
|
||||
p = sub.add_parser("episode", help="按人物主线产出下一集")
|
||||
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
|
||||
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS)
|
||||
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN)
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
|
||||
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
|
||||
p.set_defaults(func=cmd_episode)
|
||||
|
||||
p = sub.add_parser("next", help="生成下一章")
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
|
||||
p.set_defaults(func=cmd_next)
|
||||
|
||||
+4
-1
@@ -47,6 +47,9 @@ MODEL = "cn:deepseek-v4-pro"
|
||||
# 该模型是推理模型:思考与正文共用 max_tokens,必须留足
|
||||
MAX_TOKENS_CHAPTER = 16000
|
||||
MAX_TOKENS_UTIL = 4000
|
||||
# 凡是「喂长材料 + 要求结构化 JSON 输出」的调用(闸门自评、人物小传)都容易把预算
|
||||
# 烧在思考上而返回空正文,实测 4000 不够。
|
||||
MAX_TOKENS_STRUCTURED = 12000
|
||||
|
||||
# ---------------------------------------------------------------- 发布
|
||||
GITEA_SSH_HOST = "124.222.29.26"
|
||||
@@ -61,7 +64,7 @@ SSH_KEY = HOME / ".ssh/id_ed25519"
|
||||
WORLD_SIZE = "Medium"
|
||||
WORLD_HISTORY_YEARS = 250
|
||||
CHAPTER_MIN_CHARS = 800
|
||||
CHAPTER_MAX_CHARS = 1600 # 契约上限 1500,留出标点/换行余量
|
||||
CHAPTER_MAX_CHARS = 1500 # 严格按访谈定的 800–1500,不留“余量”
|
||||
|
||||
|
||||
@dataclass
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
"""因果链切集:把一条人物线索切成"一集一个完整回合"。
|
||||
|
||||
参数来自访谈决策:
|
||||
|
||||
- 相邻事件间隔 >2 年即断开(比推荐值更紧,节奏更快)
|
||||
- 短链只合并,不设线索总量下限
|
||||
- 链路过长按字数预算拆上下集
|
||||
- 每集正文预算落在 800–1500 字
|
||||
|
||||
字数预算用「事件条数 × 每条展开字数」估算。系数 110 是按此前实测校准的:
|
||||
12 条史料事件生成的中文正文约 1100–1400 字。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from dfannals.legends import Event, World
|
||||
from dfannals.threads import Thread
|
||||
|
||||
# 相邻事件间隔超过它即断开
|
||||
GAP_YEARS = 2
|
||||
# 每条史料在正文里大致展开的字数。实测区间约 85–115(同样 12 条事件,
|
||||
# 两次生成分别得到 1026 字与 1388 字),取中值偏保守,避免把集拆得过碎。
|
||||
CHARS_PER_EVENT = 95
|
||||
MIN_CHARS = 800 # 每集预算下限
|
||||
MAX_CHARS = 1500 # 每集预算上限
|
||||
|
||||
|
||||
@dataclass
|
||||
class Episode:
|
||||
index: int
|
||||
start_year: int
|
||||
end_year: int
|
||||
events: list[Event] = field(default_factory=list)
|
||||
part: int = 1
|
||||
parts: int = 1
|
||||
|
||||
@property
|
||||
def key(self) -> str:
|
||||
base = f"{self.index:03d}-{self.start_year}-{self.end_year}"
|
||||
return base if self.parts == 1 else f"{base}-p{self.part}"
|
||||
|
||||
@property
|
||||
def span(self) -> str:
|
||||
if self.start_year == self.end_year:
|
||||
return f"{self.start_year} 年"
|
||||
return f"{self.start_year}–{self.end_year} 年"
|
||||
|
||||
@property
|
||||
def estimated_chars(self) -> int:
|
||||
return len(self.events) * CHARS_PER_EVENT
|
||||
|
||||
def in_budget(self) -> bool:
|
||||
return MIN_CHARS <= self.estimated_chars <= MAX_CHARS or self.parts > 1
|
||||
|
||||
|
||||
def _min_events() -> int:
|
||||
return max(1, math.ceil(MIN_CHARS / CHARS_PER_EVENT))
|
||||
|
||||
|
||||
def _max_events() -> int:
|
||||
return max(1, MAX_CHARS // CHARS_PER_EVENT)
|
||||
|
||||
|
||||
def _split_chains(events: list[Event], gap_years: int) -> list[list[Event]]:
|
||||
"""按时间间隔把事件串切成因果链。"""
|
||||
chains: list[list[Event]] = []
|
||||
current: list[Event] = []
|
||||
prev_year: int | None = None
|
||||
for e in events:
|
||||
if current and prev_year is not None and e.year - prev_year > gap_years:
|
||||
chains.append(current)
|
||||
current = []
|
||||
current.append(e)
|
||||
prev_year = e.year
|
||||
if current:
|
||||
chains.append(current)
|
||||
return chains
|
||||
|
||||
|
||||
def _merge_short(chains: list[list[Event]], min_events: int) -> list[list[Event]]:
|
||||
"""短链只合并:攒够预算就收一集,尾部不足的并入上一集。"""
|
||||
merged: list[list[Event]] = []
|
||||
buffer: list[Event] = []
|
||||
for chain in chains:
|
||||
buffer.extend(chain)
|
||||
if len(buffer) >= min_events:
|
||||
merged.append(buffer)
|
||||
buffer = []
|
||||
if buffer:
|
||||
if merged:
|
||||
merged[-1].extend(buffer)
|
||||
else:
|
||||
merged = [buffer]
|
||||
return merged
|
||||
|
||||
|
||||
def _split_long(chain: list[Event]) -> list[list[Event]]:
|
||||
"""过长链路按预算拆成若干段(对应正文的上/下集)。
|
||||
|
||||
注意不能均分:14 条事件均分成 7+7 会得到两个 770 字的碎片,
|
||||
两端都跌破 800 字下限。所以先取最小段数,再确保每段不低于下限。
|
||||
"""
|
||||
n = len(chain)
|
||||
max_events = _max_events()
|
||||
min_events = _min_events()
|
||||
if n <= max_events:
|
||||
return [chain]
|
||||
parts = math.ceil(n / max_events)
|
||||
while parts > 1 and math.ceil(n / parts) < min_events:
|
||||
parts -= 1
|
||||
size = math.ceil(n / parts)
|
||||
return [chain[i:i + size] for i in range(0, n, size)]
|
||||
|
||||
|
||||
def plan_episodes(
|
||||
thread: Thread,
|
||||
gap_years: int = GAP_YEARS,
|
||||
min_chars: int = MIN_CHARS,
|
||||
max_chars: int = MAX_CHARS,
|
||||
) -> list[Episode]:
|
||||
"""给定主角组,切出剧集序列。"""
|
||||
if not thread.events:
|
||||
return []
|
||||
|
||||
chains = _split_chains(thread.events, gap_years)
|
||||
merged = _merge_short(chains, _min_events())
|
||||
|
||||
episodes: list[Episode] = []
|
||||
for chain in merged:
|
||||
segments = _split_long(chain)
|
||||
for i, segment in enumerate(segments, 1):
|
||||
episodes.append(
|
||||
Episode(
|
||||
index=len(episodes) + 1,
|
||||
start_year=segment[0].year,
|
||||
end_year=segment[-1].year,
|
||||
events=segment,
|
||||
part=i,
|
||||
parts=len(segments),
|
||||
)
|
||||
)
|
||||
return episodes
|
||||
|
||||
|
||||
def cast_of(world: World, thread: Thread, episode: Episode) -> tuple[list[int], list[int]]:
|
||||
"""返回 (本集出场的主角, 本集出现的对手/外人)。"""
|
||||
counter: dict[int, int] = {}
|
||||
for e in episode.events:
|
||||
for fid in e.figure_ids():
|
||||
counter[fid] = counter.get(fid, 0) + 1
|
||||
members = set(thread.members)
|
||||
heros = sorted((f for f in counter if f in members), key=lambda f: -counter[f])
|
||||
others = sorted((f for f in counter if f not in members), key=lambda f: -counter[f])
|
||||
return heros, others
|
||||
|
||||
|
||||
def render_plan(world: World, thread: Thread, episodes: list[Episode], top_others: int = 3) -> str:
|
||||
"""剧集清单:集号、年份范围、事件条数、主角与对手。"""
|
||||
lines = [
|
||||
f"线索主角:{'、'.join(world.figure_name(m) for m in thread.members)}",
|
||||
f"线索跨度:{thread.label} | 事件 {len(thread.events)} 条 | 规划 {len(episodes)} 集",
|
||||
"",
|
||||
]
|
||||
for ep in episodes:
|
||||
heros, others = cast_of(world, thread, ep)
|
||||
hero_names = "、".join(world.figure_name(h) for h in heros[:4]) or "(本集无主角出场)"
|
||||
other_names = "、".join(world.figure_name(o) for o in others[:top_others])
|
||||
budget = f"{ep.estimated_chars} 字估算"
|
||||
mark = "" if MIN_CHARS <= ep.estimated_chars <= MAX_CHARS else " ⚠ 超出预算"
|
||||
lines.append(
|
||||
f" 第 {ep.index:2d} 集({ep.span})| {len(ep.events):2d} 条事件 | {budget}{mark}"
|
||||
+ (f" | 上/下:{ep.part}/{ep.parts}" if ep.parts > 1 else "")
|
||||
)
|
||||
lines.append(f" 主角:{hero_names}")
|
||||
if other_names:
|
||||
lines.append(f" 对手/外人:{other_names}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def budget_report(episodes: list[Episode]) -> str:
|
||||
if not episodes:
|
||||
return "无剧集"
|
||||
inside = sum(1 for e in episodes if MIN_CHARS <= e.estimated_chars <= MAX_CHARS)
|
||||
sizes = [len(e.events) for e in episodes]
|
||||
return (
|
||||
f"共 {len(episodes)} 集;事件条数 {min(sizes)}–{max(sizes)};"
|
||||
f"估算字数 {min(e.estimated_chars for e in episodes)}–{max(e.estimated_chars for e in episodes)};"
|
||||
f"落在 {MIN_CHARS}–{MAX_CHARS} 预算内的 {inside}/{len(episodes)} 集"
|
||||
"(估算按每条约 95 字,真实字数在生成时会再校正)"
|
||||
)
|
||||
@@ -0,0 +1,223 @@
|
||||
"""5 维度自评闸门:判断一稿是不是"流水账",并决定重写还是换线索。
|
||||
|
||||
访谈决策:
|
||||
- 5 个维度各 0–5 分,总分 <12 判平淡
|
||||
- 不合格先重写本集一次,仍不合格才换线索
|
||||
- 放弃原因写进仓库 notes/ 附录
|
||||
- 设失败上限,防止「重写→换线索」空转
|
||||
|
||||
为什么要这道闸门:上一版章节读起来像流水账,而"平淡"是模型最容易自我感觉良好的东西。
|
||||
因此评分必须结构化、维度必须问得具体,且要有一次反例自检证明它真的能判不合格。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals.config import MAX_TOKENS_STRUCTURED, MODEL, Gateway
|
||||
from dfannals.llm import chat
|
||||
|
||||
# 维度:键、标签、问什么
|
||||
DIMENSIONS: tuple[tuple[str, str, str], ...] = (
|
||||
("goal", "主角目标", "主角在本集里有没有明确的目标,并为此做出主动选择?全程被动得分低"),
|
||||
("escalation", "冲突升级", "冲突是否逐级升级,而不是同一件事反复发生?"),
|
||||
("turning", "真正转折", "本集有没有真正的转折(立场改变、关系破裂、意外后果)?"),
|
||||
("rival", "对手具体", "对手是不是一个具体的人,有自己的动机?只是数字或人群得分低"),
|
||||
("beyond_bodycount", "超越战斗计数", "抛开对战次数,本集还有没有可讲的内容(算计、情感、抉择)?"),
|
||||
)
|
||||
|
||||
THRESHOLD = 12 # 总分低于它判平淡
|
||||
MAX_SCORE = 5 # 每个维度满分
|
||||
EDGE_RE = re.compile(r"^```(?:json)?|```$", re.MULTILINE)
|
||||
|
||||
# 失败上限:防止反复重写/换线空转
|
||||
MAX_REWRITES = 1 # 每集最多重写次数(访谈决定)
|
||||
MAX_CANDIDATES = 3 # 最多尝试几条线索
|
||||
|
||||
|
||||
@dataclass
|
||||
class Verdict:
|
||||
total: int
|
||||
scores: dict[str, int]
|
||||
reason: str
|
||||
|
||||
@property
|
||||
def passed(self) -> bool:
|
||||
return self.total >= THRESHOLD
|
||||
|
||||
def weak(self, limit: int = 2) -> list[str]:
|
||||
labels = {k: label for k, label, _ in DIMENSIONS}
|
||||
ranked = sorted(self.scores.items(), key=lambda kv: kv[1])
|
||||
return [f"{labels.get(k, k)}({v})" for k, v in ranked[:limit]]
|
||||
|
||||
def render(self) -> str:
|
||||
parts = "、".join(f"{label}={self.scores.get(key, '?')}" for key, label, _ in DIMENSIONS)
|
||||
verdict = "通过" if self.passed else "判平淡"
|
||||
return f"{verdict} 总分 {self.total}/{len(DIMENSIONS) * MAX_SCORE}({parts})|{self.reason}"
|
||||
|
||||
|
||||
DIMENSION_BLOCK = "\n".join(
|
||||
f'- "{key}": {hint}(0–5 分)' for key, _label, hint in DIMENSIONS
|
||||
)
|
||||
|
||||
EVAL_INSTRUCTIONS = f"""你是严格的文学编辑,只做一件事:判断下面这集稿子是不是"流水账"。
|
||||
|
||||
按 5 个维度各打 0–5 分:
|
||||
{DIMENSION_BLOCK}
|
||||
|
||||
打分要求:
|
||||
- 不要客气,不要因为文笔通顺就给分;只看叙事本身有没有戏。
|
||||
- 主角全程被动、只是"发生了很多事",goal 与 escalation 必须给低分。
|
||||
- 通篇只是同一类事件重复(例如反复对战、反复被杀),turning 与 beyond_bodycount 必须给低分。
|
||||
|
||||
只输出 JSON,不要解释、不要 Markdown 代码块:
|
||||
{{"goal": 0-5, "escalation": 0-5, "turning": 0-5, "rival": 0-5, "beyond_bodycount": 0-5, "reason": "一句话说明最弱的地方"}}
|
||||
"""
|
||||
|
||||
|
||||
def build_eval_prompt(chapter_text: str) -> str:
|
||||
"""拼接自评提示词。
|
||||
|
||||
刻意不用 str.format:提示词里的 JSON 例子含大括号,用 format 会被当成占位符
|
||||
(曾直接报 KeyError: '"goal"')。
|
||||
"""
|
||||
return EVAL_INSTRUCTIONS + "\n待评稿件:\n---\n" + chapter_text + "\n---\n"
|
||||
|
||||
|
||||
def _parse(raw: str) -> dict:
|
||||
text = EDGE_RE.sub("", raw).strip()
|
||||
start, end = text.find("{"), text.rfind("}")
|
||||
if start < 0 or end <= start:
|
||||
raise ValueError(f"自评没有返回 JSON:{raw[:200]}")
|
||||
try:
|
||||
data = json.loads(text[start:end + 1])
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"自评返回的 JSON 无法解析:{exc}|原文:{raw[:200]}") from None
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError(f"自评返回的不是对象:{type(data).__name__}")
|
||||
return data
|
||||
|
||||
|
||||
def evaluate(
|
||||
chapter_text: str,
|
||||
*,
|
||||
gateway: Gateway | None = None,
|
||||
model: str = MODEL,
|
||||
) -> Verdict:
|
||||
"""让模型按 5 个维度打分。返回结构化结果。"""
|
||||
reply = chat(
|
||||
[{"role": "user", "content": build_eval_prompt(chapter_text)}],
|
||||
gateway=gateway,
|
||||
model=model,
|
||||
max_tokens=MAX_TOKENS_STRUCTURED,
|
||||
temperature=0.0,
|
||||
)
|
||||
data = _parse(reply.content)
|
||||
|
||||
scores: dict[str, int] = {}
|
||||
for key, _label, _hint in DIMENSIONS:
|
||||
value = data.get(key)
|
||||
if not isinstance(value, (int, float)):
|
||||
raise ValueError(f"自评维度 {key} 缺失或非数字:{data.get(key)!r}")
|
||||
scores[key] = max(0, min(MAX_SCORE, int(value)))
|
||||
|
||||
reason = str(data.get("reason", "")).strip() or "(未给出理由)"
|
||||
return Verdict(total=sum(scores.values()), scores=scores, reason=reason)
|
||||
|
||||
|
||||
@dataclass
|
||||
class GateRun:
|
||||
"""一次闸门流程的完整记录。"""
|
||||
episode_key: str
|
||||
verdicts: list[Verdict] = field(default_factory=list)
|
||||
accepted: bool = False
|
||||
rewrites: int = 0
|
||||
|
||||
@property
|
||||
def final(self) -> Verdict | None:
|
||||
return self.verdicts[-1] if self.verdicts else None
|
||||
|
||||
def render(self) -> str:
|
||||
lines = [f"闸门:{self.episode_key},重写 {self.rewrites} 次,"
|
||||
+ ("通过" if self.accepted else "未通过")]
|
||||
for i, v in enumerate(self.verdicts, 1):
|
||||
lines.append(f" 第 {i} 稿:{v.render()}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def gate_episode(
|
||||
episode_key: str,
|
||||
write: Callable[[int, Verdict | None], str],
|
||||
*,
|
||||
max_rewrites: int = MAX_REWRITES,
|
||||
evaluate_fn: Callable[[str], Verdict] = evaluate,
|
||||
on_event: Callable[[str], None] | None = None,
|
||||
) -> tuple[str, GateRun]:
|
||||
"""写 → 自评 → 不合格则重写一次 → 仍不合格则交给调用方换线索。
|
||||
|
||||
``write(attempt, previous_verdict)``:attempt 为 0 表示初稿,
|
||||
大于 0 表示重写,previous_verdict 是上一稿的评分(供重写时指出问题)。
|
||||
"""
|
||||
run = GateRun(episode_key=episode_key)
|
||||
text = write(0, None)
|
||||
verdict = evaluate_fn(text)
|
||||
run.verdicts.append(verdict)
|
||||
|
||||
while not verdict.passed and run.rewrites < max_rewrites:
|
||||
run.rewrites += 1
|
||||
if on_event:
|
||||
on_event(f"{episode_key} 第 {run.rewrites} 次重写:{verdict.weak()}")
|
||||
text = write(run.rewrites, verdict)
|
||||
verdict = evaluate_fn(text)
|
||||
run.verdicts.append(verdict)
|
||||
|
||||
run.accepted = verdict.passed
|
||||
return text, run
|
||||
|
||||
|
||||
NOTES_DIR_NAME = "notes"
|
||||
SWITCH_LOG = "switched-threads.md"
|
||||
|
||||
|
||||
def append_switch_note(
|
||||
project_dir: Path,
|
||||
*,
|
||||
world_name: str,
|
||||
members: list[str],
|
||||
span: str,
|
||||
verdict: Verdict | None,
|
||||
reason: str,
|
||||
) -> Path:
|
||||
"""把"为什么放弃这条线索"写进仓库 notes/ 附录(访谈决定:公开可查)。"""
|
||||
notes = project_dir / NOTES_DIR_NAME
|
||||
notes.mkdir(parents=True, exist_ok=True)
|
||||
path = notes / SWITCH_LOG
|
||||
header = "# 被放弃的线索\n\n记录每一组挑出来、但写不成故事的线索,以及放弃原因。\n"
|
||||
if not path.is_file():
|
||||
path.write_text(header, encoding="utf-8")
|
||||
|
||||
score = verdict.render() if verdict else "(未评分)"
|
||||
entry = (
|
||||
f"\n## {world_name}|{'、'.join(members)}\n\n"
|
||||
f"- 线索跨度:{span}\n"
|
||||
f"- 放弃原因:{reason}\n"
|
||||
f"- 自评:{score}\n"
|
||||
)
|
||||
with path.open("a", encoding="utf-8") as fh:
|
||||
fh.write(entry)
|
||||
return path
|
||||
|
||||
|
||||
def append_attempt_note(project_dir: Path, text: str) -> Path:
|
||||
"""一般性记录(例如失败上限触发)。"""
|
||||
notes = project_dir / NOTES_DIR_NAME
|
||||
notes.mkdir(parents=True, exist_ok=True)
|
||||
path = notes / SWITCH_LOG
|
||||
if not path.is_file():
|
||||
path.write_text("# 被放弃的线索\n\n", encoding="utf-8")
|
||||
with path.open("a", encoding="utf-8") as fh:
|
||||
fh.write(text if text.startswith("\n") else "\n" + text)
|
||||
return path
|
||||
+81
-20
@@ -14,26 +14,29 @@ from typing import Any
|
||||
|
||||
from defusedxml.ElementTree import iterparse
|
||||
|
||||
# 事件里指向其他实体的字段 → 指向哪类对象
|
||||
ID_FIELDS: dict[str, str] = {
|
||||
"hfid": "figure",
|
||||
"hist_figure_id": "figure",
|
||||
"slayer_hfid": "figure",
|
||||
"slayer_item_id": "artifact",
|
||||
"target_hfid": "figure",
|
||||
"source_hfid": "figure",
|
||||
"site_id": "site",
|
||||
"site_civ_id": "entity",
|
||||
"entity_id": "entity",
|
||||
"attacker_civ_id": "entity",
|
||||
"defender_civ_id": "entity",
|
||||
"artifact_id": "artifact",
|
||||
"region_id": "region",
|
||||
"feature_layer_id": "region",
|
||||
"deity": "figure",
|
||||
"worshipper_hfid": "figure",
|
||||
"creature_id": "creature",
|
||||
}
|
||||
# 字段分类规则来自对真实史料全部 225 种字段的穷举:
|
||||
# 含 "hfid" 的字段全部指向历史人物(共 40+ 个:hfid / hfid_target / group_1_hfid /
|
||||
# slayer_hfid / snatcher_hfid / seeker_hfid / wounder_hfid / teacher_hfid /
|
||||
# conspirator_hfid ...);而 target_enid、identity_id、master_wcid、slayer_item_id 不是。
|
||||
# 旧版手工维护的字典只列了 4 个人物字段,导致关系信息在素材里被丢掉。
|
||||
_ENTITY_SUFFIXES = ("_entity_id", "_civ_id", "_enid")
|
||||
_REGION_FIELDS = {"region_id", "feature_layer_id", "subregion_id"}
|
||||
|
||||
|
||||
def kind_of(tag: str) -> str | None:
|
||||
"""事件字段名 → 它指向的对象类别(非引用类字段返回 None)。"""
|
||||
t = tag.lower()
|
||||
if "hfid" in t or t == "deity" or t.endswith("_deity"):
|
||||
return "figure"
|
||||
if "artifact" in t:
|
||||
return "artifact"
|
||||
if t == "site_id" or t.endswith("_site_id"):
|
||||
return "site"
|
||||
if t in ("entity_id", "target_enid") or t.endswith(_ENTITY_SUFFIXES):
|
||||
return "entity"
|
||||
if t in _REGION_FIELDS:
|
||||
return "region"
|
||||
return None
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -81,6 +84,36 @@ class Event:
|
||||
vals = self.fields.get(tag)
|
||||
return vals[0] if vals else ""
|
||||
|
||||
def ids_of_kind(self, kind: str) -> list[int]:
|
||||
"""本事件中指向该类对象的全部有效 id(去重、剔除 -1 占位)。"""
|
||||
out: list[int] = []
|
||||
for tag, vals in self.fields.items():
|
||||
if kind_of(tag) != kind:
|
||||
continue
|
||||
for raw in vals:
|
||||
i = _int_or_none(raw)
|
||||
if i is not None and i >= 0 and i not in out:
|
||||
out.append(i)
|
||||
return out
|
||||
|
||||
def figure_ids(self) -> list[int]:
|
||||
return self.ids_of_kind("figure")
|
||||
|
||||
def site_ids(self) -> list[int]:
|
||||
return self.ids_of_kind("site")
|
||||
|
||||
def entity_ids(self) -> list[int]:
|
||||
return self.ids_of_kind("entity")
|
||||
|
||||
def field_kinds(self) -> dict[str, list[int]]:
|
||||
"""{类别: [id, ...]},供渲染与建图使用。"""
|
||||
out: dict[str, list[int]] = {}
|
||||
for kind in ("figure", "site", "entity", "artifact", "region"):
|
||||
ids = self.ids_of_kind(kind)
|
||||
if ids:
|
||||
out[kind] = ids
|
||||
return out
|
||||
|
||||
|
||||
@dataclass
|
||||
class Entity:
|
||||
@@ -138,6 +171,34 @@ class World:
|
||||
st = self.sites.get(sid)
|
||||
return pretty(st.name) if st and st.name else f"Site#{sid}"
|
||||
|
||||
def render_refs(self, event: Event) -> list[str]:
|
||||
"""把一条事件里的引用字段渲染成 ``tag=名字``。
|
||||
|
||||
保留字段名是有意的:slayer_hfid=谁 与 hfid=谁 语义完全不同,
|
||||
丢掉字段名就等于丢掉了“谁对谁做了什么”。
|
||||
负 id(DF 的“无此项”占位)与查不到名字的字段直接跳过。
|
||||
"""
|
||||
out: list[str] = []
|
||||
for tag, vals in event.fields.items():
|
||||
kind = kind_of(tag)
|
||||
if kind is None:
|
||||
continue
|
||||
for raw in vals:
|
||||
i = _int_or_none(raw)
|
||||
if i is None or i < 0:
|
||||
continue
|
||||
if kind == "figure":
|
||||
name = self.figure_name(i)
|
||||
elif kind == "site":
|
||||
name = self.site_name(i)
|
||||
elif kind == "entity":
|
||||
name = self.entity_name(i)
|
||||
else:
|
||||
name = ""
|
||||
if name:
|
||||
out.append(f"{tag}={name}")
|
||||
return out
|
||||
|
||||
def event_years(self) -> tuple[int, int]:
|
||||
years = [e.year for e in self.events if e.year]
|
||||
return (min(years), max(years)) if years else (0, 0)
|
||||
|
||||
@@ -32,6 +32,7 @@ def _git(*args: str, check: bool = True) -> subprocess.CompletedProcess:
|
||||
env={**_base_env(), **env},
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False, # 下面是自定义错误处理,故意不用 check=True
|
||||
)
|
||||
if check and proc.returncode != 0:
|
||||
raise RuntimeError(f"git {' '.join(args)} 失败:{proc.stderr.strip() or proc.stdout.strip()}")
|
||||
|
||||
+11
-10
@@ -93,22 +93,23 @@ def category(event: Event) -> str:
|
||||
|
||||
|
||||
def prominence(world: World) -> Counter[int]:
|
||||
"""每个人物被卷入的事件次数——用来衡量他在史料里有多重要。"""
|
||||
"""每个人物被卷入的事件次数——用来衡量他在史料里有多重要。
|
||||
|
||||
改用字段分类器后覆盖了 40+ 个人物字段(旧版只数 6 个),数值尺度整体变大,
|
||||
因此 PROMINENT 阈值必须按新尺度重校。结果缓存在 World 上,避免反复重算。
|
||||
"""
|
||||
cached = getattr(world, "_prominence", None)
|
||||
if cached is not None:
|
||||
return cached
|
||||
counts: Counter[int] = Counter()
|
||||
for e in world.events:
|
||||
seen: set[int] = set()
|
||||
for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity", "worshipper_hfid"):
|
||||
for fid in e.ids(tag):
|
||||
seen.add(fid)
|
||||
counts.update(seen)
|
||||
counts.update(e.figure_ids())
|
||||
world._prominence = counts # type: ignore[attr-defined]
|
||||
return counts
|
||||
|
||||
|
||||
def _participants(event: Event) -> list[int]:
|
||||
out: list[int] = []
|
||||
for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity"):
|
||||
out.extend(event.ids(tag))
|
||||
return out
|
||||
return event.figure_ids()
|
||||
|
||||
|
||||
# 人物重要度阈值:参与事件数达到这个量级才算"要角"
|
||||
|
||||
+8
-19
@@ -130,23 +130,10 @@ def material(world: World, chapter: Chapter, max_figures: int = 40) -> str:
|
||||
|
||||
participation: dict[int, int] = {}
|
||||
for e in chapter.events:
|
||||
chunks = [f"{e.year}年", e.type]
|
||||
for tag, kind in lg.ID_FIELDS.items():
|
||||
for i in e.ids(tag):
|
||||
# id 为负是“无此项”,不要渲染成 HF#-1 这类噪音
|
||||
if kind == "figure":
|
||||
name = world.figure_name(i)
|
||||
elif kind == "site":
|
||||
name = world.site_name(i)
|
||||
elif kind == "entity":
|
||||
name = world.entity_name(i)
|
||||
else:
|
||||
name = ""
|
||||
if not name:
|
||||
continue
|
||||
chunks.append(f"{tag}={name}")
|
||||
if kind == "figure":
|
||||
participation[i] = participation.get(i, 0) + 1
|
||||
# render_refs 会带出双方(比如 slayer_hfid=谁),而不是只剩单方面 hfid
|
||||
chunks = [f"{e.year}年", e.type, *world.render_refs(e)]
|
||||
for i in e.figure_ids():
|
||||
participation[i] = participation.get(i, 0) + 1
|
||||
extra = []
|
||||
for field_name in ("state", "reason", "circumstance"):
|
||||
v = e.text(field_name)
|
||||
@@ -184,8 +171,10 @@ def overview(world: World, chapters: list[Chapter], limit: int = 10) -> str:
|
||||
lo, hi = world.event_years()
|
||||
lines = [
|
||||
f"世界:{world.name}" + (f"({world.altname})" if world.altname else ""),
|
||||
f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
|
||||
f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}",
|
||||
(
|
||||
f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
|
||||
f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}"
|
||||
),
|
||||
f"规划章节:{len(chapters)} 章(每章约 {YEARS_PER_CHAPTER} 年 / 最多 {EVENTS_PER_CHAPTER} 条史料)",
|
||||
]
|
||||
for c in chapters[:limit]:
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
"""人物线索挖掘:从史料里找出「关系密切、有戏」的 3–6 人小圈子。
|
||||
|
||||
设计依据全部来自对 Mon Sagus 真实数据(403853 条事件)的实测:
|
||||
|
||||
1. **不能用裸互动次数排序。** 实测排名第一的簇 49 次互动里有 35 次是
|
||||
``hf relationship denied``——反复请求建立关系、反复被拒的循环,是统计噪声。
|
||||
因此整类剔除(访谈已定)。
|
||||
2. **同一对同一类型反复出现也不是故事。** 修掉上一条之后,候选又变成
|
||||
``对战×27`` 这种「同一对人反复互殴」的机械循环。因此在计数时对
|
||||
(人物对, 事件类型) 做封顶,避免刷量取胜——这也正是访谈定的
|
||||
「转折/冲突多样性为主、互动次数为次」。
|
||||
3. **主线必须是"人"。** 种族分布实测:ELF/GOBLIN/DWARF/HUMAN/KOBOLD 五族加
|
||||
``*_MAN`` 人形族占全部历史人物的 96.5%,其余 400 多种是夜行怪、野兽、
|
||||
泰坦、实验体。所以这里用白名单,而不是越列越长的黑名单。
|
||||
4. 强边阈值 ≥5 次反复互动;簇规模 3–6 人;跨度 ≥30 年。
|
||||
实测:阈值提到 8 会一条候选都不剩,5 是正确档位。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import Counter, defaultdict
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals.legends import Event, World
|
||||
|
||||
# 事件类型 → (权重, 中文标签, 是否算转折点)
|
||||
SIGNALS: dict[str, tuple[int, str, bool]] = {
|
||||
"hf simple battle event": (1, "对战", False),
|
||||
"add hf hf link": (1, "结缘", False),
|
||||
"hfs formed reputation relationship": (1, "结缘", False),
|
||||
"competition": (1, "竞争", False),
|
||||
"hf wounded": (2, "搏杀", True),
|
||||
"hf confronted": (2, "对峙", True),
|
||||
"hf interrogated": (2, "审讯", True),
|
||||
"failed intrigue corruption": (2, "阴谋", True),
|
||||
"remove hf hf link": (3, "反目", True),
|
||||
"hf abducted": (3, "绑架", True),
|
||||
"hf convicted": (3, "定罪", True),
|
||||
"entity persecuted": (3, "迫害", True),
|
||||
"hf enslaved": (3, "奴役", True),
|
||||
"hf ransomed": (3, "贖金", True),
|
||||
"entity overthrown": (3, "推翻", True),
|
||||
"failed frame attempt": (3, "构陷", True),
|
||||
}
|
||||
|
||||
# 明确剔除:重复性请求,实测会刷满排序(访谈已定)
|
||||
EXCLUDED_SIGNALS = ("hf relationship denied",)
|
||||
|
||||
# 文明种族白名单:实测覆盖 96.5% 的历史人物
|
||||
CIVILIZED_RACES = frozenset({"DWARF", "ELF", "HUMAN", "GOBLIN", "KOBOLD"})
|
||||
|
||||
DEFAULT_MIN_INTERACTIONS = 5 # 强边阈值:≥5 次反复互动
|
||||
DEFAULT_SIZE_RANGE = (3, 6)
|
||||
DEFAULT_MIN_SPAN = 30 # 跨度 ≥30 年才撑得起连载
|
||||
PAIR_TYPE_CAP = 4 # 同一对、同一类型最多计 4 次
|
||||
MIN_SIGNAL_KINDS = 2 # 至少两种不同信号,否则只是单一类型的重复
|
||||
INTERACTION_CAP = 40 # 互动总量封顶,防止刷量取胜
|
||||
|
||||
|
||||
@dataclass
|
||||
class Edge:
|
||||
a: int
|
||||
b: int
|
||||
raw: int = 0
|
||||
types: Counter = field(default_factory=Counter)
|
||||
first_year: int = 0
|
||||
last_year: int = 0
|
||||
|
||||
@property
|
||||
def effective(self) -> int:
|
||||
"""封顶后的有效互动次数。"""
|
||||
return sum(min(n, PAIR_TYPE_CAP) for n in self.types.values())
|
||||
|
||||
@property
|
||||
def turns(self) -> int:
|
||||
"""封顶后的转折点次数。"""
|
||||
return sum(min(n, PAIR_TYPE_CAP) for t, n in self.types.items() if SIGNALS[t][2])
|
||||
|
||||
|
||||
@dataclass
|
||||
class Thread:
|
||||
members: list[int]
|
||||
interactions: int
|
||||
turns: int
|
||||
span: int
|
||||
first_year: int
|
||||
last_year: int
|
||||
types: Counter
|
||||
races: Counter
|
||||
non_person_races: list[str]
|
||||
events: list[Event]
|
||||
|
||||
@property
|
||||
def kinds(self) -> int:
|
||||
return len(self.types)
|
||||
|
||||
@property
|
||||
def score(self) -> int:
|
||||
"""转折为主、多样性次之、互动次数封顶计入(访谈定的排序原则)。"""
|
||||
return self.turns * 10 + self.kinds * 6 + min(self.interactions, INTERACTION_CAP)
|
||||
|
||||
@property
|
||||
def label(self) -> str:
|
||||
return f"{self.first_year}–{self.last_year}({self.span} 年)"
|
||||
|
||||
def signal_line(self) -> str:
|
||||
return "、".join(f"{SIGNALS[t][1]}×{n}" for t, n in self.types.most_common() if t in SIGNALS)
|
||||
|
||||
|
||||
def is_person(race: str) -> bool:
|
||||
"""是否属于"可作为主角的人":文明种族或人形族(*_MAN)。"""
|
||||
r = (race or "").upper()
|
||||
return r in CIVILIZED_RACES or r.endswith("_MAN")
|
||||
|
||||
|
||||
def build_edges(world: World) -> dict[tuple[int, int], Edge]:
|
||||
"""按有叙事含义的信号建立人物之间的边。"""
|
||||
edges: dict[tuple[int, int], Edge] = {}
|
||||
for e in world.events:
|
||||
if e.type not in SIGNALS:
|
||||
continue
|
||||
people = e.figure_ids()
|
||||
if len(people) < 2:
|
||||
continue
|
||||
for i in range(len(people)):
|
||||
for j in range(i + 1, len(people)):
|
||||
key = (min(people[i], people[j]), max(people[i], people[j]))
|
||||
edge = edges.get(key)
|
||||
if edge is None:
|
||||
edge = Edge(a=key[0], b=key[1], first_year=e.year, last_year=e.year)
|
||||
edges[key] = edge
|
||||
edge.raw += 1
|
||||
edge.types[e.type] += 1
|
||||
edge.first_year = min(edge.first_year, e.year)
|
||||
edge.last_year = max(edge.last_year, e.year)
|
||||
return edges
|
||||
|
||||
|
||||
def find_threads(
|
||||
world: World,
|
||||
min_interactions: int = DEFAULT_MIN_INTERACTIONS,
|
||||
size_range: tuple[int, int] = DEFAULT_SIZE_RANGE,
|
||||
min_span: int = DEFAULT_MIN_SPAN,
|
||||
require_people: bool = True,
|
||||
min_kinds: int = MIN_SIGNAL_KINDS,
|
||||
limit: int = 0,
|
||||
) -> list[Thread]:
|
||||
"""返回按剧情张力排序的候选线索。"""
|
||||
edges = build_edges(world)
|
||||
strong = {k: v for k, v in edges.items() if v.effective >= min_interactions}
|
||||
|
||||
parent: dict[int, int] = {}
|
||||
|
||||
def find(x: int) -> int:
|
||||
parent.setdefault(x, x)
|
||||
while parent[x] != x:
|
||||
parent[x] = parent[parent[x]]
|
||||
x = parent[x]
|
||||
return x
|
||||
|
||||
for (a, b) in strong:
|
||||
ra, rb = find(a), find(b)
|
||||
if ra != rb:
|
||||
parent[ra] = rb
|
||||
|
||||
members_of: dict[int, set[int]] = defaultdict(set)
|
||||
for x in parent:
|
||||
members_of[find(x)].add(x)
|
||||
|
||||
events_by_pair: dict[tuple[int, int], list[Event]] = defaultdict(list)
|
||||
for e in world.events:
|
||||
if e.type not in SIGNALS:
|
||||
continue
|
||||
people = e.figure_ids()
|
||||
if len(people) < 2:
|
||||
continue
|
||||
for i in range(len(people)):
|
||||
for j in range(i + 1, len(people)):
|
||||
events_by_pair[(min(people[i], people[j]), max(people[i], people[j]))].append(e)
|
||||
|
||||
lo_size, hi_size = size_range
|
||||
threads: list[Thread] = []
|
||||
for members in members_of.values():
|
||||
if not (lo_size <= len(members) <= hi_size):
|
||||
continue
|
||||
|
||||
races = Counter(world.figures[m].race for m in members if m in world.figures)
|
||||
if require_people and not all(is_person(r) for r in races):
|
||||
continue
|
||||
|
||||
inner = {k: v for k, v in strong.items() if k[0] in members and k[1] in members}
|
||||
if not inner:
|
||||
continue
|
||||
|
||||
types: Counter = Counter()
|
||||
for v in inner.values():
|
||||
types.update({t: min(n, PAIR_TYPE_CAP) for t, n in v.types.items()})
|
||||
if len(types) < min_kinds:
|
||||
continue
|
||||
|
||||
first = min(v.first_year for v in inner.values())
|
||||
last = max(v.last_year for v in inner.values())
|
||||
if last - first < min_span:
|
||||
continue
|
||||
|
||||
evs: list[Event] = []
|
||||
seen_ids: set[int] = set()
|
||||
for k in inner:
|
||||
for e in events_by_pair.get(k, []):
|
||||
if e.id not in seen_ids:
|
||||
seen_ids.add(e.id)
|
||||
evs.append(e)
|
||||
evs.sort(key=lambda e: (e.year, e.seconds72, e.id))
|
||||
|
||||
threads.append(
|
||||
Thread(
|
||||
members=sorted(members),
|
||||
interactions=sum(v.effective for v in inner.values()),
|
||||
turns=sum(v.turns for v in inner.values()),
|
||||
span=last - first,
|
||||
first_year=first,
|
||||
last_year=last,
|
||||
types=types,
|
||||
races=races,
|
||||
non_person_races=sorted(r for r in races if not is_person(r)),
|
||||
events=evs,
|
||||
)
|
||||
)
|
||||
|
||||
threads.sort(key=lambda t: (-t.score, -t.span))
|
||||
return threads[:limit] if limit else threads
|
||||
|
||||
|
||||
def render_report(world: World, threads: list[Thread], per_thread: int = 3) -> str:
|
||||
"""给人看的候选明细。"""
|
||||
lines = [
|
||||
f"候选线索 {len(threads)} 条(转折为主、多样性次之;同一对同类事件已封顶)",
|
||||
"筛选:强边 ≥5 次互动、3–6 人、跨度 ≥30 年、含巨兽的簇已剔除",
|
||||
"",
|
||||
]
|
||||
for i, t in enumerate(threads, 1):
|
||||
races = "、".join(f"{r}×{n}" for r, n in t.races.most_common())
|
||||
monsters = "、".join(t.non_person_races) if t.non_person_races else "无"
|
||||
lines.append(
|
||||
f"{i:2d}. 得分 {t.score:4d} | {len(t.members)} 人 | 有效互动 {t.interactions:3d} | "
|
||||
f"转折 {t.turns:2d} | 信号 {t.kinds} 种 | 跨度 {t.label} | 巨兽:{monsters}"
|
||||
)
|
||||
lines.append(f" 种族:{races}")
|
||||
lines.append(f" 信号:{t.signal_line()}")
|
||||
lines.append(f" 成员:{'、'.join(world.figure_name(m) for m in t.members)}")
|
||||
for e in t.events[:per_thread]:
|
||||
lines.append(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def to_json(world: World, threads: list[Thread]) -> dict:
|
||||
return {
|
||||
"world": world.name,
|
||||
"candidates": [
|
||||
{
|
||||
"rank": i,
|
||||
"score": t.score,
|
||||
"members": [
|
||||
{"id": m, "name": world.figure_name(m),
|
||||
"race": world.figures[m].race if m in world.figures else ""}
|
||||
for m in t.members
|
||||
],
|
||||
"interactions": t.interactions,
|
||||
"turns": t.turns,
|
||||
"kinds": t.kinds,
|
||||
"span": t.span,
|
||||
"first_year": t.first_year,
|
||||
"last_year": t.last_year,
|
||||
"signals": {SIGNALS[k][1]: v for k, v in t.types.items() if k in SIGNALS},
|
||||
"event_ids": [e.id for e in t.events],
|
||||
}
|
||||
for i, t in enumerate(threads, 1)
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def save_json(world: World, threads: list[Thread], path: Path) -> Path:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(to_json(world, threads), ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
return path
|
||||
+6
-22
@@ -19,19 +19,7 @@ category = score.category
|
||||
|
||||
def event_line(world: World, e: Event) -> str:
|
||||
"""把一条事件压成一行可读文字。"""
|
||||
chunks: list[str] = [e.type]
|
||||
for tag, kind in lg.ID_FIELDS.items():
|
||||
for i in e.ids(tag):
|
||||
if kind == "figure":
|
||||
name = world.figure_name(i)
|
||||
elif kind == "site":
|
||||
name = world.site_name(i)
|
||||
elif kind == "entity":
|
||||
name = world.entity_name(i)
|
||||
else:
|
||||
name = ""
|
||||
if name:
|
||||
chunks.append(f"{tag}={name}")
|
||||
chunks: list[str] = [e.type, *world.render_refs(e)]
|
||||
return f"- **{e.year}** [{category(e)}] " + " · ".join(chunks)
|
||||
|
||||
|
||||
@@ -51,8 +39,10 @@ def build_timeline(world: World, min_score: int = 70, per_decade: int = 14) -> s
|
||||
f"- 事件总数:{len(world.events)},其中重要事件 {len(important)}",
|
||||
f"- 历史人物 {len(world.figures)} 位 · 文明与组织 {len(world.entities)} 个 · 地点 {len(world.sites)} 处",
|
||||
"",
|
||||
"> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
|
||||
"完整数据留在本地,不入库。",
|
||||
(
|
||||
"> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
|
||||
"完整数据留在本地,不入库。"
|
||||
),
|
||||
"",
|
||||
]
|
||||
for decade in sorted(buckets):
|
||||
@@ -72,13 +62,7 @@ def build_figures(world: World, top: int = 120) -> str:
|
||||
"""人物索引:按参与事件数排序,给出身份与生卒。"""
|
||||
involvement: Counter[int] = Counter()
|
||||
for e in world.events:
|
||||
seen: set[int] = set()
|
||||
for tag, kind in lg.ID_FIELDS.items():
|
||||
if kind != "figure":
|
||||
continue
|
||||
for i in e.ids(tag):
|
||||
seen.add(i)
|
||||
involvement.update(seen)
|
||||
involvement.update(e.figure_ids())
|
||||
|
||||
lines = [
|
||||
f"# {world.name} · 人物索引",
|
||||
|
||||
@@ -0,0 +1,336 @@
|
||||
"""人物主线写作与编排:选主角组 → 切集 → 贴着人物写 → 过闸门 → 落盘。
|
||||
|
||||
与旧的 chronicle.py 的区别:那里的主语是"年代",一章按 10 年窗口罗列事件;
|
||||
这里的主语是"人",一集是主角线上一段完整回合,世界大事只在影响到主角时提及。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import Counter
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass, replace
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import episodes as ep_mod
|
||||
from dfannals import gate
|
||||
from dfannals.cast import collect_rows
|
||||
from dfannals.chronicle import cjk_length
|
||||
from dfannals.config import (
|
||||
CHAPTER_MAX_CHARS,
|
||||
CHAPTER_MIN_CHARS,
|
||||
MAX_TOKENS_CHAPTER,
|
||||
PROJECT_DIR,
|
||||
PROMPT_DIR,
|
||||
)
|
||||
from dfannals.legends import Event, World
|
||||
from dfannals.llm import chat
|
||||
from dfannals.threads import SIGNALS, Thread
|
||||
|
||||
SPEC_FILE = PROMPT_DIR / "biographer.md"
|
||||
TAIL_CHARS = 600
|
||||
SELECTION_FILE = PROJECT_DIR / "data" / "selection.json"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Selection:
|
||||
rank: int
|
||||
thread: Thread
|
||||
reason: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class Produced:
|
||||
selection: Selection
|
||||
episode: ep_mod.Episode
|
||||
text: str
|
||||
run: gate.GateRun
|
||||
path: Path | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class Extended:
|
||||
"""主角参与的扩展素材。"""
|
||||
events: list[Event]
|
||||
elided: Counter
|
||||
|
||||
|
||||
PER_KEY_CAP = 3 # 同一组人 + 同一类型,每集素材最多保留几次
|
||||
|
||||
|
||||
def expand_thread_events(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP) -> Extended:
|
||||
"""把素材从“成员之间”扩到“主角参与的全部有意义事件”。
|
||||
|
||||
实测:某 3 人线索成员之间只有 14 条事件(只能切 1 集),但把他们的对外活动
|
||||
(构陷、对战、结义、定罪……)算进来有 136 条,足够支撑一卷。
|
||||
|
||||
代价是重复:这 136 条里有 64 条是同一类“构陷失败”。所以同一组人 + 同一类型
|
||||
最多保留 per_key_cap 条,其余计入 elided 供写作时一笔带过——否则又会变流水账。
|
||||
"""
|
||||
members = set(thread.members)
|
||||
seen: Counter = Counter()
|
||||
keep: list[Event] = []
|
||||
elided: Counter = Counter()
|
||||
|
||||
relevant = [
|
||||
e for e in world.events
|
||||
if e.type in SIGNALS and (members & set(e.figure_ids()))
|
||||
]
|
||||
relevant.sort(key=lambda e: (e.year, e.seconds72, e.id))
|
||||
|
||||
for e in relevant:
|
||||
key = (e.type, tuple(sorted(e.figure_ids())))
|
||||
seen[key] += 1
|
||||
if seen[key] <= per_key_cap:
|
||||
keep.append(e)
|
||||
else:
|
||||
elided[e.type] += 1
|
||||
return Extended(events=keep, elided=elided)
|
||||
|
||||
|
||||
def extended_plan(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP
|
||||
) -> tuple[Extended, list[ep_mod.Episode]]:
|
||||
"""用扩展素材切集(保持切集规则不变)。"""
|
||||
ext = expand_thread_events(world, thread, per_key_cap=per_key_cap)
|
||||
planned = ep_mod.plan_episodes(replace(thread, events=ext.events))
|
||||
return ext, planned
|
||||
|
||||
|
||||
def select_thread(world: World, candidates: list[Thread], rank: int = 1) -> Selection:
|
||||
"""按挖掘器排序取第 rank 条,并把选择依据写成可读理由。"""
|
||||
thread = candidates[rank - 1]
|
||||
names = "、".join(world.figure_name(m) for m in thread.members)
|
||||
reason = (
|
||||
f"挖掘器排序第 {rank} 名(转折为主、互动封顶):"
|
||||
f"{len(thread.members)} 人、有效互动 {thread.interactions}、转折 {thread.turns}、"
|
||||
f"信号 {thread.kinds} 种({thread.signal_line()})、跨度 {thread.label};成员:{names}"
|
||||
)
|
||||
return Selection(rank=rank, thread=thread, reason=reason)
|
||||
|
||||
|
||||
def save_selection(selection: Selection, world: World, path: Path = SELECTION_FILE) -> Path:
|
||||
"""把主角组的选定依据落到本地(不入库,供追溯)。"""
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
thread = selection.thread
|
||||
path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"world": world.name,
|
||||
"rank": selection.rank,
|
||||
"reason": selection.reason,
|
||||
"score": thread.score,
|
||||
"members": [
|
||||
{"id": m, "name": world.figure_name(m),
|
||||
"race": world.figures[m].race if m in world.figures else ""}
|
||||
for m in thread.members
|
||||
],
|
||||
"signals": dict(thread.types),
|
||||
"span": thread.label,
|
||||
},
|
||||
ensure_ascii=False,
|
||||
indent=2,
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return path
|
||||
|
||||
|
||||
def render_material(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
total_episodes: int,
|
||||
prev_tail: str = "",
|
||||
elided: Counter | None = None,
|
||||
) -> str:
|
||||
"""把一集素材渲染给传记作者:主角是谁、发生了什么、对手是谁。"""
|
||||
rows = {r.fid: r for r in collect_rows(world, thread, [episode], max_others=6)}
|
||||
heros, others = ep_mod.cast_of(world, thread, episode)
|
||||
|
||||
lines = [
|
||||
f"## 本集:第 {episode.index}/{total_episodes} 集,{episode.span}",
|
||||
"",
|
||||
"### 主角(本集要写的就是他们)",
|
||||
"",
|
||||
]
|
||||
for fid in heros:
|
||||
row = rows.get(fid)
|
||||
if row:
|
||||
lines.append(f"- {row.name}|{row.race}|{row.span}|身份:{row.identity or '不详'}"
|
||||
f"|本集出场 {row.appearances} 次")
|
||||
else:
|
||||
lines.append(f"- {world.figure_name(fid)}")
|
||||
|
||||
lines += ["", "### 本集史料(按时间排序;专名保留游戏原文)", ""]
|
||||
for e in episode.events:
|
||||
refs = " · ".join(world.render_refs(e))
|
||||
extra = [f"{k}={e.text(k)}" for k in ("state", "reason", "circumstance")
|
||||
if e.text(k) and e.text(k) not in ("-1", "")]
|
||||
tail = ("|" + ";".join(extra)) if extra else ""
|
||||
lines.append(f"- {e.year}年 {e.type} · {refs}{tail}")
|
||||
|
||||
if others:
|
||||
lines += ["", "### 对手/相关者(史料里出现,但不是本卷主角)", ""]
|
||||
for fid in others[:6]:
|
||||
row = rows.get(fid)
|
||||
if row:
|
||||
lines.append(f"- {row.name}|{row.race}|身份:{row.identity or '不详'}"
|
||||
f"|本集出场 {row.appearances} 次")
|
||||
else:
|
||||
lines.append(f"- {world.figure_name(fid)}")
|
||||
|
||||
if elided:
|
||||
hot = "、".join(f"{t}×{n}" for t, n in elided.most_common(6))
|
||||
lines += [
|
||||
"",
|
||||
"### 被省略的同类重复事件",
|
||||
"",
|
||||
f"- 本卷还有这些同类事件已被过滤,不要逐条罗列,需要时用一句话带过:{hot}",
|
||||
]
|
||||
|
||||
if prev_tail:
|
||||
lines += ["", "### 上一集结尾(只用于承接状态,不要复述)", "", prev_tail]
|
||||
|
||||
lines += [
|
||||
"",
|
||||
"### 写作要求",
|
||||
"",
|
||||
f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,专名保留英文。",
|
||||
"- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。",
|
||||
"- 史料之外的世界大事不要写;只有影响到主角时才提一句。",
|
||||
]
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def build_messages(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
total_episodes: int,
|
||||
prev_tail: str,
|
||||
feedback: str | None = None,
|
||||
elided: Counter | None = None,
|
||||
) -> list[dict]:
|
||||
spec = SPEC_FILE.read_text(encoding="utf-8")
|
||||
material = render_material(world, thread, episode, total_episodes, prev_tail, elided)
|
||||
user = [material]
|
||||
if feedback:
|
||||
user += ["", "### 上一稿的问题(重写时必须针对这些改)", "", feedback]
|
||||
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
|
||||
|
||||
|
||||
def _repair(messages: list[dict], draft: str, length: int, gateway) -> str:
|
||||
target = (
|
||||
f"内容太短({length} 字),请扩写到 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,"
|
||||
"补充场景与对白,不要注水。"
|
||||
if length < CHAPTER_MIN_CHARS
|
||||
else f"内容太长({length} 字),请压缩到 {CHAPTER_MAX_CHARS} 字以内,删次要枝节。"
|
||||
)
|
||||
reply = chat(messages + [{"role": "assistant", "content": draft},
|
||||
{"role": "user", "content": target}], gateway=gateway)
|
||||
return reply.content
|
||||
|
||||
|
||||
def write_episode(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
total_episodes: int,
|
||||
prev_tail: str = "",
|
||||
feedback: str | None = None,
|
||||
*,
|
||||
gateway=None,
|
||||
max_repairs: int = 2,
|
||||
elided: Counter | None = None,
|
||||
) -> str:
|
||||
"""生成一集正文,并把长度校正到 800–1500 字。"""
|
||||
messages = build_messages(world, thread, episode, total_episodes, prev_tail, feedback, elided)
|
||||
text = chat(messages, gateway=gateway, max_tokens=MAX_TOKENS_CHAPTER).content
|
||||
for _ in range(max_repairs):
|
||||
length = cjk_length(text)
|
||||
if CHAPTER_MIN_CHARS <= length <= CHAPTER_MAX_CHARS:
|
||||
break
|
||||
text = _repair(messages, text, length, gateway)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def write_file(directory: Path, episode: ep_mod.Episode, text: str) -> Path:
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
path = directory / f"{episode.key}.md"
|
||||
path.write_text(text.strip() + "\n", encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
def prev_tail_of(directory: Path, planned: list[ep_mod.Episode], index: int) -> str:
|
||||
"""取上一集结尾作为承接线索。"""
|
||||
if index <= 1:
|
||||
return ""
|
||||
path = directory / f"{planned[index - 2].key}.md"
|
||||
if not path.is_file():
|
||||
return ""
|
||||
return path.read_text(encoding="utf-8").strip()[-TAIL_CHARS:]
|
||||
|
||||
|
||||
def produce_episode(
|
||||
world: World,
|
||||
candidates: list[Thread],
|
||||
*,
|
||||
episode_index: int = 1,
|
||||
out_dir: Path | None = None,
|
||||
gateway=None,
|
||||
max_candidates: int = gate.MAX_CANDIDATES,
|
||||
on_event: Callable[[str], None] = print,
|
||||
) -> Produced:
|
||||
"""选线索 → 写这一集 → 过闸门;不过就换下一线索(有上限)。"""
|
||||
if not candidates:
|
||||
raise RuntimeError("没有候选线索可用")
|
||||
|
||||
for rank in range(1, min(max_candidates, len(candidates)) + 1):
|
||||
selection = select_thread(world, candidates, rank=rank)
|
||||
thread = selection.thread
|
||||
ext, planned = extended_plan(world, thread)
|
||||
if episode_index > len(planned):
|
||||
on_event(f"第 {rank} 条线索只有 {len(planned)} 集,跳过")
|
||||
continue
|
||||
|
||||
episode = planned[episode_index - 1]
|
||||
directory = out_dir or (PROJECT_DIR / "output" / "volumes" / "v1" / "chapters")
|
||||
tail = prev_tail_of(directory, planned, episode_index)
|
||||
on_event(f"选定线索 {rank}:{'、'.join(world.figure_name(m) for m in thread.members)}")
|
||||
|
||||
# 把循环变量绑定成默认参数:否则闭包会绑定到循环变量本身(后面还会变)
|
||||
def write(
|
||||
attempt: int,
|
||||
prev: gate.Verdict | None,
|
||||
_thread: Thread = thread,
|
||||
_episode: ep_mod.Episode = episode,
|
||||
_total: int = len(planned),
|
||||
_tail: str = tail,
|
||||
_elided: Counter = ext.elided,
|
||||
) -> str:
|
||||
feedback = None
|
||||
if prev is not None:
|
||||
feedback = (f"上一稿被判平淡:{prev.render()}。"
|
||||
f"最弱环节是 {'、'.join(prev.weak())},请围绕这些重写,"
|
||||
"让人物真正做选择。")
|
||||
return write_episode(world, _thread, _episode, _total, _tail, feedback,
|
||||
gateway=gateway, elided=_elided)
|
||||
|
||||
text, run = gate.gate_episode(episode.key, write, on_event=on_event)
|
||||
on_event(run.render())
|
||||
|
||||
if run.accepted:
|
||||
save_selection(selection, world)
|
||||
path = write_file(directory, episode, text) if out_dir is not None else None
|
||||
return Produced(selection=selection, episode=episode, text=text, run=run, path=path)
|
||||
|
||||
gate.append_switch_note(
|
||||
PROJECT_DIR,
|
||||
world_name=world.name,
|
||||
members=[world.figure_name(m) for m in thread.members],
|
||||
span=thread.label,
|
||||
verdict=run.final,
|
||||
reason=f"重写 {run.rewrites} 次后仍未过闸门(第 {rank} 条线索)",
|
||||
)
|
||||
on_event(f"线索 {rank} 未过闸门,换下一条")
|
||||
|
||||
raise RuntimeError(f"前 {max_candidates} 条线索都没过闸门,需要人工干预")
|
||||
Reference in New Issue
Block a user