Files
dwarf-fortress-annals/dfannals/cli.py
T
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

393 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""命令行入口。
人物主线流程:
scripts/dfannals status 查看当前进度
scripts/dfannals index 解析 exports → 年表 + 人物索引
scripts/dfannals threads 挖掘人物线索候选(含信号明细)
scripts/dfannals episodes --thread 1 查看某条线索切出的剧集
scripts/dfannals cast --thread 1 生成人物表(表格 + 小传)
scripts/dfannals episode --index 1 写出某一集(自评闸门 + 文笔层扩写)
scripts/dfannals expand --index 1 只跑文笔层:把已有的一集大量扩写(A/B 对照)
用 scripts/dfannals 启动,不要直接 python3 -m:会话 PATH 上的 python3 可能是
编辑器工具链的 venv,里面没有 defusedxml。
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from dfannals import (
cast,
chronicle,
deslop,
episodes,
factcheck,
legends,
publish,
threads,
timeline,
)
from dfannals import expand as expand_mod
from dfannals import volume as volume_mod
from dfannals.config import (
DATA_DIR,
EXPAND_TARGET_CHARS,
EXPORT_DIR,
GITEA_WEB,
LITERARY_MODEL,
chapter_dir,
ensure_dirs,
volume_dir,
)
def find_exports(export_dir: Path) -> list[Path]:
"""找出 legends 导出文件。
原生 legends.xml 与 DFHack 的 legends_plus.xml 都要;
预处理缓存放在子目录,所以这里不会重复收到同一份数据。
"""
if not export_dir.is_dir():
return []
files = sorted(export_dir.glob("*.xml"))
base = [f for f in files if "legends_plus" not in f.name]
plus = [f for f in files if "legends_plus" in f.name]
return base + plus
def load_world(export_dir: Path) -> legends.World:
paths = find_exports(export_dir)
if not paths:
raise SystemExit(
f"在 {export_dir} 里没找到 legends XML。\n"
"请先导出(游戏内 Export XML + DFHack 的 exportlegends),再复制进这个目录。"
)
return legends.load(paths)
def _current_volume(state: chronicle.State, world: legends.World) -> int:
"""世界换了就开新卷。"""
if state.world != world.name:
if state.world:
state.volume += 1
state.done = []
state.world = world.name
return state.volume
def _volume_for(state: chronicle.State, world: legends.World) -> int:
return state.volume if state.world == world.name else 1
def _load_candidates(args: argparse.Namespace) -> tuple[legends.World, list[threads.Thread]]:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
raise SystemExit(1)
return world, found
def cmd_status(args: argparse.Namespace) -> int:
state = chronicle.State.load()
if not state.world:
print("尚未开始。先运行 index。")
return 0
vdir = volume_dir(state.volume)
print(f"世界:{state.world}")
print(f"卷号:第 {state.volume} 卷")
print(f"已完成集数:{len(state.done)}")
for key in state.done:
print(f" · {key}")
print(f"成品目录:{vdir}")
return 0
def cmd_index(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
state = chronicle.State.load()
volume = _current_volume(state, world)
vdir = volume_dir(volume)
timeline.write_outputs(world, vdir)
state.save()
print(legends.describe(world))
print()
print(f"年表与人物索引已写入 {vdir}")
print("下一步:threads 看候选线索,episodes 看剧集规划。")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷索引:{world.name}", [vdir / "timeline.md", vdir / "figures.md"]
)
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_threads(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
report = threads.render_report(world, found, per_thread=args.samples)
print(report)
path = threads.save_json(world, found, DATA_DIR / "threads.json")
print(f"共 {len(found)} 条候选;明细已写入 {path}")
return 0
def cmd_episodes(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
print(episodes.render_plan(world, thread, planned))
print()
print(episodes.budget_report(planned))
if args.samples:
print("\n=== 前两集事件样例 ===")
for ep in planned[:2]:
print(f" 第 {ep.index} 集({ep.span})")
for e in ep.events[: args.samples]:
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
return 0
def cmd_cast(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=not args.no_bios)
if args.dry_run:
print(text[:2000])
return 0
state = chronicle.State.load()
volume = _volume_for(state, world)
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
print(factcheck.report(suspicions) if suspicions
else "专名校验:小传里的专名全部能在史料中找到。")
return 0
def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
"""把刚写好的骨架稿大量扩写成成稿,覆盖本集文件;骨架另存 `.skeleton.md`。
骨架仍然保留:它过了自评闸门、字数合规,是溯因和对照的依据。
"""
ext, planned = volume_mod.extended_plan(world, produced.selection.thread)
directory = chapter_dir(volume)
skeleton = produced.path.with_name(produced.path.stem + ".skeleton.md")
skeleton.write_text(produced.text.strip() + "\n", encoding="utf-8")
print(f"--- 文笔层扩写中({args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, produced.selection.thread, produced.episode, len(planned), produced.text,
prev_tail=volume_mod.prev_tail_of(directory, planned, produced.episode.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
suspects = factcheck.check(text, world)
if suspects:
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, produced.selection.thread, produced.episode,
text, suspects)
text = fixed.text
for note in fixed.notes:
print(" " + note)
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
remaining = factcheck.check(text, world)
print(expand_mod.compare(produced.text, text))
print(f"成稿专名校验:可疑 {len(remaining)} 处|骨架稿另存 {skeleton.name}")
if remaining:
print(factcheck.report(remaining, top=5))
return text
def cmd_episode(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
out_dir = None if args.dry_run else chapter_dir(volume)
produced = volume_mod.produce_episode(
world, found, episode_index=args.index, out_dir=out_dir
)
print("选择依据:" + produced.selection.reason)
print(f"骨架稿字数:{chronicle.cjk_length(produced.text)}")
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(produced.text[:1600])
return 0
if args.expand:
_expand_in_place(volume, world, produced, args)
state.mark_done(produced.episode.key)
state.save()
print(f"已写入 {produced.path}")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷第 {produced.episode.index} 集(人物主线):"
f"{produced.episode.span}"
)
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_expand(args: argparse.Namespace) -> int:
"""文笔层:拿已写好的那一集当骨架,大量扩写成场景化长文。
故意不覆盖原稿:产物写 `*.literary.md`,两稿并存才能比对。也不推送。
"""
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
selection = volume_mod.select_thread(world, found, rank=args.thread)
ext, planned = volume_mod.extended_plan(world, selection.thread)
if args.index > len(planned):
print(f"第 {args.thread} 条线索只有 {len(planned)} 集,没有第 {args.index} 集")
return 1
episode = planned[args.index - 1]
directory = chapter_dir(volume)
draft_path = directory / f"{episode.key}.md"
if not draft_path.is_file():
print(f"原稿不存在:{draft_path}")
print(f"先跑:scripts/dfannals episode --index {args.index}")
return 1
draft = draft_path.read_text(encoding="utf-8")
print(f"线索:{'、'.join(world.figure_name(m) for m in selection.thread.members)}")
print(f"本集:{episode.span}|原稿 {draft_path.name}({chronicle.cjk_length(draft)} 字)")
print()
print(deslop.report(draft))
print()
print(f"--- 文笔层扩写中(模型 {args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, selection.thread, episode, len(planned), draft,
prev_tail=volume_mod.prev_tail_of(directory, planned, args.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
print()
print(expand_mod.full_report(draft, text))
print()
before = factcheck.check(draft, world)
suspects = factcheck.check(text, world)
print(f"专名校验:原稿 {len(before)} 处|扩写稿 {len(suspects)} 处")
if suspects:
print(factcheck.report(suspects, top=5))
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, selection.thread, episode, text, suspects)
for note in fixed.notes:
print(" " + note)
text = fixed.text
left = factcheck.check(text, world)
print(f" 复检:剩余可疑专名 {len(left)} 处")
if left:
print(factcheck.report(left, top=5))
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(text)
return 0
path = expand_mod.literary_path(directory, episode)
path.write_text(text.strip() + "\n", encoding="utf-8")
print(f"\n已写入 {path}")
print("原稿未改动、未推送(A/B 对照用)")
return 0
def _add_pipeline_args(p: argparse.ArgumentParser) -> None:
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
help="线索最小跨度年数(默认 30)")
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞人物主线连载管道")
sub = parser.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("status", help="查看当前进度")
p.set_defaults(func=cmd_status)
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
p.set_defaults(func=cmd_index)
p = sub.add_parser("threads", help="挖掘人物线索候选")
_add_pipeline_args(p)
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
p.set_defaults(func=cmd_threads)
p = sub.add_parser("episodes", help="按人物线索切出剧集")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
p.set_defaults(func=cmd_episodes)
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
p.set_defaults(func=cmd_cast)
p = sub.add_parser("episode", help="产出某一集(自评闸门 + 文笔层扩写)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
p.add_argument("--no-expand", dest="expand", action="store_false",
help="只出骨架稿,不做文笔层扩写")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS,
help=f"扩写目标字数(默认 {EXPAND_TARGET_CHARS})")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
p.set_defaults(func=cmd_episode, expand=True)
p = sub.add_parser("expand", help="文笔层:把某一集大量扩写成场景化长文(不覆盖原稿)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="扩写哪一集(默认 1)")
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS, help="目标字数")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘")
p.set_defaults(func=cmd_expand)
args = parser.parse_args(argv)
ensure_dirs()
return args.func(args)
if __name__ == "__main__":
sys.exit(main())