Files
dwarf-fortress-annals/dfannals/cli.py
T
Chen Yi 70ce758e8d 第 1 卷三集全部转为成稿(分块扩写 + AI 腔定点返修),文档同步
- 每集两稿:NNN-….md 为成稿(9492/11260/11869 字),NNN-….skeleton.md 为过闸门的骨架稿
- 新增 expand.repair_style:只重写含 AI 腔标记的段落(本例出现了 14 处破折号),
  三集硬伤密度 1.2/1.0/1.5 → 0.1/0.3/0.0(每千字)
- episode 命令接入定点返修;expand 命令保留 A/B 对照用法
- README 同步文笔层、本土化、子命令、以及本轮新踩的坑(\b 盲点、枚举大小写、自制小节标记)
- 第 3 集骨架连换三条线索仍未过闸门(真实拒绝),其成稿基于既有骨架扩写
2026-10-05 23:14:24 +08:00

409 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""命令行入口。
人物主线流程:
scripts/dfannals status 查看当前进度
scripts/dfannals index 解析 exports → 年表 + 人物索引
scripts/dfannals threads 挖掘人物线索候选(含信号明细)
scripts/dfannals episodes --thread 1 查看某条线索切出的剧集
scripts/dfannals cast --thread 1 生成人物表(表格 + 小传)
scripts/dfannals episode --index 1 写出某一集(自评闸门 + 文笔层扩写)
scripts/dfannals expand --index 1 只跑文笔层:把已有的一集大量扩写(A/B 对照)
用 scripts/dfannals 启动,不要直接 python3 -m:会话 PATH 上的 python3 可能是
编辑器工具链的 venv,里面没有 defusedxml。
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from dfannals import (
cast,
chronicle,
deslop,
episodes,
factcheck,
legends,
publish,
threads,
timeline,
)
from dfannals import expand as expand_mod
from dfannals import volume as volume_mod
from dfannals.config import (
DATA_DIR,
EXPAND_TARGET_CHARS,
EXPORT_DIR,
GITEA_WEB,
LITERARY_MODEL,
chapter_dir,
ensure_dirs,
volume_dir,
)
def find_exports(export_dir: Path) -> list[Path]:
"""找出 legends 导出文件。
原生 legends.xml 与 DFHack 的 legends_plus.xml 都要;
预处理缓存放在子目录,所以这里不会重复收到同一份数据。
"""
if not export_dir.is_dir():
return []
files = sorted(export_dir.glob("*.xml"))
base = [f for f in files if "legends_plus" not in f.name]
plus = [f for f in files if "legends_plus" in f.name]
return base + plus
def load_world(export_dir: Path) -> legends.World:
paths = find_exports(export_dir)
if not paths:
raise SystemExit(
f"在 {export_dir} 里没找到 legends XML。\n"
"请先导出(游戏内 Export XML + DFHack 的 exportlegends),再复制进这个目录。"
)
return legends.load(paths)
def _current_volume(state: chronicle.State, world: legends.World) -> int:
"""世界换了就开新卷。"""
if state.world != world.name:
if state.world:
state.volume += 1
state.done = []
state.world = world.name
return state.volume
def _volume_for(state: chronicle.State, world: legends.World) -> int:
return state.volume if state.world == world.name else 1
def _load_candidates(args: argparse.Namespace) -> tuple[legends.World, list[threads.Thread]]:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
raise SystemExit(1)
return world, found
def cmd_status(args: argparse.Namespace) -> int:
state = chronicle.State.load()
if not state.world:
print("尚未开始。先运行 index。")
return 0
vdir = volume_dir(state.volume)
print(f"世界:{state.world}")
print(f"卷号:第 {state.volume} 卷")
print(f"已完成集数:{len(state.done)}")
for key in state.done:
print(f" · {key}")
print(f"成品目录:{vdir}")
return 0
def cmd_index(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
state = chronicle.State.load()
volume = _current_volume(state, world)
vdir = volume_dir(volume)
timeline.write_outputs(world, vdir)
state.save()
print(legends.describe(world))
print()
print(f"年表与人物索引已写入 {vdir}")
print("下一步:threads 看候选线索,episodes 看剧集规划。")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷索引:{world.name}", [vdir / "timeline.md", vdir / "figures.md"]
)
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_threads(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
report = threads.render_report(world, found, per_thread=args.samples)
print(report)
path = threads.save_json(world, found, DATA_DIR / "threads.json")
print(f"共 {len(found)} 条候选;明细已写入 {path}")
return 0
def cmd_episodes(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
print(episodes.render_plan(world, thread, planned))
print()
print(episodes.budget_report(planned))
if args.samples:
print("\n=== 前两集事件样例 ===")
for ep in planned[:2]:
print(f" 第 {ep.index} 集({ep.span})")
for e in ep.events[: args.samples]:
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
return 0
def cmd_cast(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=not args.no_bios)
if args.dry_run:
print(text[:2000])
return 0
state = chronicle.State.load()
volume = _volume_for(state, world)
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
print(factcheck.report(suspicions) if suspicions
else "专名校验:小传里的专名全部能在史料中找到。")
return 0
def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
"""把刚写好的骨架稿大量扩写成成稿,覆盖本集文件;骨架另存 `.skeleton.md`。
骨架仍然保留:它过了自评闸门、字数合规,是溯因和对照的依据。
"""
ext, planned = volume_mod.extended_plan(world, produced.selection.thread)
directory = chapter_dir(volume)
skeleton = produced.path.with_name(produced.path.stem + ".skeleton.md")
skeleton.write_text(produced.text.strip() + "\n", encoding="utf-8")
print(f"--- 文笔层扩写中({args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, produced.selection.thread, produced.episode, len(planned), produced.text,
prev_tail=volume_mod.prev_tail_of(directory, planned, produced.episode.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
suspects = factcheck.check(text, world)
if suspects:
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, produced.selection.thread, produced.episode,
text, suspects)
text = fixed.text
for note in fixed.notes:
print(" " + note)
text = _polish(text, world, produced.selection.thread, produced.episode, args.model)
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
remaining = factcheck.check(text, world)
print(expand_mod.compare(produced.text, text))
print(f"成稿专名校验:可疑 {len(remaining)} 处|骨架稿另存 {skeleton.name}")
if remaining:
print(factcheck.report(remaining, top=5))
return text
def _polish(text: str, world, thread, episode, model: str, max_rounds: int = 2) -> str:
"""定点返修 AI 腔标记(破折号、「不是A而是B」等),只改命中的段落。"""
for round_no in range(1, max_rounds + 1):
if not expand_mod.style_targets(text):
return text
result = expand_mod.repair_style(world, thread, episode, text, model=model)
for note in result.notes:
print(f" (第 {round_no} 轮){note}")
if result.text == text:
return text
text = result.text
return text
def cmd_episode(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
out_dir = None if args.dry_run else chapter_dir(volume)
produced = volume_mod.produce_episode(
world, found, episode_index=args.index, out_dir=out_dir
)
print("选择依据:" + produced.selection.reason)
print(f"骨架稿字数:{chronicle.cjk_length(produced.text)}")
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(produced.text[:1600])
return 0
if args.expand:
_expand_in_place(volume, world, produced, args)
state.mark_done(produced.episode.key)
state.save()
print(f"已写入 {produced.path}")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷第 {produced.episode.index} 集(人物主线):"
f"{produced.episode.span}"
)
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_expand(args: argparse.Namespace) -> int:
"""文笔层:拿已写好的那一集当骨架,大量扩写成场景化长文。
故意不覆盖原稿:产物写 `*.literary.md`,两稿并存才能比对。也不推送。
"""
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
selection = volume_mod.select_thread(world, found, rank=args.thread)
ext, planned = volume_mod.extended_plan(world, selection.thread)
if args.index > len(planned):
print(f"第 {args.thread} 条线索只有 {len(planned)} 集,没有第 {args.index} 集")
return 1
episode = planned[args.index - 1]
directory = chapter_dir(volume)
draft_path = directory / f"{episode.key}.md"
if not draft_path.is_file():
print(f"原稿不存在:{draft_path}")
print(f"先跑:scripts/dfannals episode --index {args.index}")
return 1
draft = draft_path.read_text(encoding="utf-8")
print(f"线索:{'、'.join(world.figure_name(m) for m in selection.thread.members)}")
print(f"本集:{episode.span}|原稿 {draft_path.name}({chronicle.cjk_length(draft)} 字)")
print()
print(deslop.report(draft))
print()
print(f"--- 文笔层扩写中(模型 {args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, selection.thread, episode, len(planned), draft,
prev_tail=volume_mod.prev_tail_of(directory, planned, args.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
print()
print(expand_mod.full_report(draft, text))
print()
before = factcheck.check(draft, world)
suspects = factcheck.check(text, world)
print(f"专名校验:原稿 {len(before)} 处|扩写稿 {len(suspects)} 处")
if suspects:
print(factcheck.report(suspects, top=5))
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, selection.thread, episode, text, suspects)
for note in fixed.notes:
print(" " + note)
text = fixed.text
left = factcheck.check(text, world)
print(f" 复检:剩余可疑专名 {len(left)} 处")
if left:
print(factcheck.report(left, top=5))
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(text)
return 0
path = expand_mod.literary_path(directory, episode)
path.write_text(text.strip() + "\n", encoding="utf-8")
print(f"\n已写入 {path}")
print("原稿未改动、未推送(A/B 对照用)")
return 0
def _add_pipeline_args(p: argparse.ArgumentParser) -> None:
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
help="线索最小跨度年数(默认 30)")
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞人物主线连载管道")
sub = parser.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("status", help="查看当前进度")
p.set_defaults(func=cmd_status)
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
p.set_defaults(func=cmd_index)
p = sub.add_parser("threads", help="挖掘人物线索候选")
_add_pipeline_args(p)
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
p.set_defaults(func=cmd_threads)
p = sub.add_parser("episodes", help="按人物线索切出剧集")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
p.set_defaults(func=cmd_episodes)
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
p.set_defaults(func=cmd_cast)
p = sub.add_parser("episode", help="产出某一集(自评闸门 + 文笔层扩写)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
p.add_argument("--no-expand", dest="expand", action="store_false",
help="只出骨架稿,不做文笔层扩写")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS,
help=f"扩写目标字数(默认 {EXPAND_TARGET_CHARS})")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
p.set_defaults(func=cmd_episode, expand=True)
p = sub.add_parser("expand", help="文笔层:把某一集大量扩写成场景化长文(不覆盖原稿)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="扩写哪一集(默认 1)")
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS, help="目标字数")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘")
p.set_defaults(func=cmd_expand)
args = parser.parse_args(argv)
ensure_dirs()
return args.func(args)
if __name__ == "__main__":
sys.exit(main())