Files
Chen Yi 16b476167e 专名本土化(信达雅):冻结译名表 + 渲染层,并修掉校验器的正则盲点
- 新增 dfannals/glossary.py(译名表)与 dfannals/localize.py(渲染层)
  人名=音译+绰号意译(Ral Fastenhatchets→拉尔·缚斧、朗古德·商熊「剖胎之寒」),
  地名整体意译(Fordedwinds→涉风渡、The Volcano Of Ravens→渡鸦火山)
- 生成链路本土化:素材渲染、人物表、扩写白名单全部走译名;已有产物做确定性替换
- 修复 factcheck 的 \b 盲点:Unicode 下汉字也算词字符,紧贴汉字的英文名不被检查
  (曾因此漏掉编造专名)。修好后抓出并定点修掉长版第 1 集的 2 处编造
- 修正 title_form 大小写(YETI→Yeti),此前 Yeti/Goblin 一直漏译
- 新增 tests/test_localize.py(13 项回归)
2026-10-05 22:39:53 +08:00

339 lines
13 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""人物主线写作与编排:选主角组 → 切集 → 贴着人物写 → 过闸门 → 落盘。
与旧的 chronicle.py 的区别:那里的主语是"年代",一章按 10 年窗口罗列事件;
这里的主语是"人",一集是主角线上一段完整回合,世界大事只在影响到主角时提及。
"""
from __future__ import annotations
import json
from collections import Counter
from collections.abc import Callable
from dataclasses import dataclass, replace
from pathlib import Path
from dfannals import episodes as ep_mod
from dfannals import gate, localize
from dfannals.cast import collect_rows
from dfannals.chronicle import cjk_length
from dfannals.config import (
CHAPTER_MAX_CHARS,
CHAPTER_MIN_CHARS,
MAX_TOKENS_CHAPTER,
PROJECT_DIR,
PROMPT_DIR,
)
from dfannals.legends import Event, World
from dfannals.llm import chat
from dfannals.threads import SIGNALS, Thread
SPEC_FILE = PROMPT_DIR / "biographer.md"
TAIL_CHARS = 600
SELECTION_FILE = PROJECT_DIR / "data" / "selection.json"
@dataclass
class Selection:
rank: int
thread: Thread
reason: str
@dataclass
class Produced:
selection: Selection
episode: ep_mod.Episode
text: str
run: gate.GateRun
path: Path | None = None
@dataclass
class Extended:
"""主角参与的扩展素材。"""
events: list[Event]
elided: Counter
PER_KEY_CAP = 3 # 同一组人 + 同一类型,每集素材最多保留几次
def expand_thread_events(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP) -> Extended:
"""把素材从“成员之间”扩到“主角参与的全部有意义事件”。
实测:某 3 人线索成员之间只有 14 条事件(只能切 1 集),但把他们的对外活动
(构陷、对战、结义、定罪……)算进来有 136 条,足够支撑一卷。
代价是重复:这 136 条里有 64 条是同一类“构陷失败”。所以同一组人 + 同一类型
最多保留 per_key_cap 条,其余计入 elided 供写作时一笔带过——否则又会变流水账。
"""
members = set(thread.members)
seen: Counter = Counter()
keep: list[Event] = []
elided: Counter = Counter()
relevant = [
e for e in world.events
if e.type in SIGNALS and (members & set(e.figure_ids()))
]
relevant.sort(key=lambda e: (e.year, e.seconds72, e.id))
for e in relevant:
key = (e.type, tuple(sorted(e.figure_ids())))
seen[key] += 1
if seen[key] <= per_key_cap:
keep.append(e)
else:
elided[e.type] += 1
return Extended(events=keep, elided=elided)
def extended_plan(world: World, thread: Thread, per_key_cap: int = PER_KEY_CAP
) -> tuple[Extended, list[ep_mod.Episode]]:
"""用扩展素材切集(保持切集规则不变)。"""
ext = expand_thread_events(world, thread, per_key_cap=per_key_cap)
planned = ep_mod.plan_episodes(replace(thread, events=ext.events))
return ext, planned
def select_thread(world: World, candidates: list[Thread], rank: int = 1) -> Selection:
"""按挖掘器排序取第 rank 条,并把选择依据写成可读理由。"""
thread = candidates[rank - 1]
names = "、".join(world.figure_name(m) for m in thread.members)
reason = (
f"挖掘器排序第 {rank} 名(转折为主、互动封顶):"
f"{len(thread.members)} 人、有效互动 {thread.interactions}、转折 {thread.turns}、"
f"信号 {thread.kinds} 种({thread.signal_line()})、跨度 {thread.label};成员:{names}"
)
return Selection(rank=rank, thread=thread, reason=reason)
def save_selection(selection: Selection, world: World, path: Path = SELECTION_FILE) -> Path:
"""把主角组的选定依据落到本地(不入库,供追溯)。"""
path.parent.mkdir(parents=True, exist_ok=True)
thread = selection.thread
path.write_text(
json.dumps(
{
"world": world.name,
"rank": selection.rank,
"reason": selection.reason,
"score": thread.score,
"members": [
{"id": m, "name": world.figure_name(m),
"race": world.figures[m].race if m in world.figures else ""}
for m in thread.members
],
"signals": dict(thread.types),
"span": thread.label,
},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
return path
def render_material(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
prev_tail: str = "",
elided: Counter | None = None,
) -> str:
"""把一集素材渲染给传记作者:主角是谁、发生了什么、对手是谁。"""
rows = {r.fid: r for r in collect_rows(world, thread, [episode], max_others=6)}
heros, others = ep_mod.cast_of(world, thread, episode)
lines = [
f"## 本集:第 {episode.index}/{total_episodes} 集,{episode.span}",
"",
"### 主角(本集要写的就是他们)",
"",
]
for fid in heros:
row = rows.get(fid)
if row:
lines.append(f"- {row.name}|{row.race}|性别:{row.gender or '史料未载'}"
f"|{row.span}|身份:{row.identity or '不详'}|本集出场 {row.appearances} 次")
else:
lines.append(f"- {world.figure_name(fid)}")
lines += ["", "### 本集史料(按时间排序;专名保留游戏原文)", ""]
for e in episode.events:
refs = " · ".join(world.render_refs(e))
extra = [f"{k}={e.text(k)}" for k in ("state", "reason", "circumstance")
if e.text(k) and e.text(k) not in ("-1", "")]
tail = ("|" + ";".join(extra)) if extra else ""
lines.append(f"- {e.year}年 {e.type} · {refs}{tail}")
if others:
lines += ["", "### 对手/相关者(史料里出现,但不是本卷主角)", ""]
for fid in others[:6]:
row = rows.get(fid)
if row:
lines.append(f"- {row.name}|{row.race}|性别:{row.gender or '史料未载'}"
f"|身份:{row.identity or '不详'}|本集出场 {row.appearances} 次")
else:
lines.append(f"- {world.figure_name(fid)}")
if elided:
hot = "、".join(f"{t}×{n}" for t, n in elided.most_common(6))
lines += [
"",
"### 被省略的同类重复事件",
"",
f"- 本卷还有这些同类事件已被过滤,不要逐条罗列,需要时用一句话带过:{hot}",
]
if prev_tail:
lines += ["", "### 上一集结尾(只用于承接状态,不要复述)", "", prev_tail]
lines += [
"",
"### 写作要求",
"",
f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,人名地名用上面给出的中文译名。",
"- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。",
"- 性别按上面卡片写,不要默认“他”;史料未载性别时用名字或身份称呼。",
"- 史料之外的世界大事不要写;只有影响到主角时才提一句。",
]
# 专名本土化:素材里的英文原名换成中文译名,让模型直接用中文写
return localize.for_world(world).render("\n".join(lines))
def build_messages(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
prev_tail: str,
feedback: str | None = None,
elided: Counter | None = None,
) -> list[dict]:
spec = SPEC_FILE.read_text(encoding="utf-8")
material = render_material(world, thread, episode, total_episodes, prev_tail, elided)
user = [material]
if feedback:
user += ["", "### 上一稿的问题(重写时必须针对这些改)", "", feedback]
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
def _repair(messages: list[dict], draft: str, length: int, gateway) -> str:
target = (
f"内容太短({length} 字),请扩写到 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,"
"补充场景与对白,不要注水。"
if length < CHAPTER_MIN_CHARS
else f"内容太长({length} 字),请压缩到 {CHAPTER_MAX_CHARS} 字以内,删次要枝节。"
)
reply = chat(messages + [{"role": "assistant", "content": draft},
{"role": "user", "content": target}], gateway=gateway)
return reply.content
def write_episode(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
prev_tail: str = "",
feedback: str | None = None,
*,
gateway=None,
max_repairs: int = 2,
elided: Counter | None = None,
) -> str:
"""生成一集正文,并把长度校正到 800–1500 字。"""
messages = build_messages(world, thread, episode, total_episodes, prev_tail, feedback, elided)
text = chat(messages, gateway=gateway, max_tokens=MAX_TOKENS_CHAPTER).content
for _ in range(max_repairs):
length = cjk_length(text)
if CHAPTER_MIN_CHARS <= length <= CHAPTER_MAX_CHARS:
break
text = _repair(messages, text, length, gateway)
return text.strip()
def write_file(directory: Path, episode: ep_mod.Episode, text: str) -> Path:
directory.mkdir(parents=True, exist_ok=True)
path = directory / f"{episode.key}.md"
path.write_text(text.strip() + "\n", encoding="utf-8")
return path
def prev_tail_of(directory: Path, planned: list[ep_mod.Episode], index: int) -> str:
"""取上一集结尾作为承接线索。"""
if index <= 1:
return ""
path = directory / f"{planned[index - 2].key}.md"
if not path.is_file():
return ""
return path.read_text(encoding="utf-8").strip()[-TAIL_CHARS:]
def produce_episode(
world: World,
candidates: list[Thread],
*,
episode_index: int = 1,
out_dir: Path | None = None,
gateway=None,
max_candidates: int = gate.MAX_CANDIDATES,
on_event: Callable[[str], None] = print,
) -> Produced:
"""选线索 → 写这一集 → 过闸门;不过就换下一线索(有上限)。"""
if not candidates:
raise RuntimeError("没有候选线索可用")
for rank in range(1, min(max_candidates, len(candidates)) + 1):
selection = select_thread(world, candidates, rank=rank)
thread = selection.thread
ext, planned = extended_plan(world, thread)
if episode_index > len(planned):
on_event(f"第 {rank} 条线索只有 {len(planned)} 集,跳过")
continue
episode = planned[episode_index - 1]
directory = out_dir or (PROJECT_DIR / "output" / "volumes" / "v1" / "chapters")
tail = prev_tail_of(directory, planned, episode_index)
on_event(f"选定线索 {rank}:{'、'.join(world.figure_name(m) for m in thread.members)}")
# 把循环变量绑定成默认参数:否则闭包会绑定到循环变量本身(后面还会变)
def write(
attempt: int,
prev: gate.Verdict | None,
_thread: Thread = thread,
_episode: ep_mod.Episode = episode,
_total: int = len(planned),
_tail: str = tail,
_elided: Counter = ext.elided,
) -> str:
feedback = None
if prev is not None:
feedback = (f"上一稿被判平淡:{prev.render()}。"
f"最弱环节是 {'、'.join(prev.weak())},请围绕这些重写,"
"让人物真正做选择。")
return write_episode(world, _thread, _episode, _total, _tail, feedback,
gateway=gateway, elided=_elided)
text, run = gate.gate_episode(episode.key, write, on_event=on_event)
on_event(run.render())
if run.accepted:
save_selection(selection, world)
path = write_file(directory, episode, text) if out_dir is not None else None
return Produced(selection=selection, episode=episode, text=text, run=run, path=path)
gate.append_switch_note(
PROJECT_DIR,
world_name=world.name,
members=[world.figure_name(m) for m in thread.members],
span=thread.label,
verdict=run.final,
reason=f"重写 {run.rewrites} 次后仍未过闸门(第 {rank} 条线索)",
)
on_event(f"线索 {rank} 未过闸门,换下一条")
raise RuntimeError(f"前 {max_candidates} 条线索都没过闸门,需要人工干预")