文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)

- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
Chen Yi
2026-10-05 22:27:02 +08:00
parent 08a032120a
commit 8630dcad55
23 changed files with 1971 additions and 56 deletions
+392
View File
@@ -0,0 +1,392 @@
"""命令行入口。
人物主线流程:
scripts/dfannals status 查看当前进度
scripts/dfannals index 解析 exports → 年表 + 人物索引
scripts/dfannals threads 挖掘人物线索候选(含信号明细)
scripts/dfannals episodes --thread 1 查看某条线索切出的剧集
scripts/dfannals cast --thread 1 生成人物表(表格 + 小传)
scripts/dfannals episode --index 1 写出某一集(自评闸门 + 文笔层扩写)
scripts/dfannals expand --index 1 只跑文笔层:把已有的一集大量扩写(A/B 对照)
用 scripts/dfannals 启动,不要直接 python3 -m:会话 PATH 上的 python3 可能是
编辑器工具链的 venv,里面没有 defusedxml。
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from dfannals import (
cast,
chronicle,
deslop,
episodes,
factcheck,
legends,
publish,
threads,
timeline,
)
from dfannals import expand as expand_mod
from dfannals import volume as volume_mod
from dfannals.config import (
DATA_DIR,
EXPAND_TARGET_CHARS,
EXPORT_DIR,
GITEA_WEB,
LITERARY_MODEL,
chapter_dir,
ensure_dirs,
volume_dir,
)
def find_exports(export_dir: Path) -> list[Path]:
"""找出 legends 导出文件。
原生 legends.xml 与 DFHack 的 legends_plus.xml 都要;
预处理缓存放在子目录,所以这里不会重复收到同一份数据。
"""
if not export_dir.is_dir():
return []
files = sorted(export_dir.glob("*.xml"))
base = [f for f in files if "legends_plus" not in f.name]
plus = [f for f in files if "legends_plus" in f.name]
return base + plus
def load_world(export_dir: Path) -> legends.World:
paths = find_exports(export_dir)
if not paths:
raise SystemExit(
f"在 {export_dir} 里没找到 legends XML。\n"
"请先导出(游戏内 Export XML + DFHack 的 exportlegends),再复制进这个目录。"
)
return legends.load(paths)
def _current_volume(state: chronicle.State, world: legends.World) -> int:
"""世界换了就开新卷。"""
if state.world != world.name:
if state.world:
state.volume += 1
state.done = []
state.world = world.name
return state.volume
def _volume_for(state: chronicle.State, world: legends.World) -> int:
return state.volume if state.world == world.name else 1
def _load_candidates(args: argparse.Namespace) -> tuple[legends.World, list[threads.Thread]]:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
raise SystemExit(1)
return world, found
def cmd_status(args: argparse.Namespace) -> int:
state = chronicle.State.load()
if not state.world:
print("尚未开始。先运行 index。")
return 0
vdir = volume_dir(state.volume)
print(f"世界:{state.world}")
print(f"卷号:第 {state.volume} 卷")
print(f"已完成集数:{len(state.done)}")
for key in state.done:
print(f" · {key}")
print(f"成品目录:{vdir}")
return 0
def cmd_index(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
state = chronicle.State.load()
volume = _current_volume(state, world)
vdir = volume_dir(volume)
timeline.write_outputs(world, vdir)
state.save()
print(legends.describe(world))
print()
print(f"年表与人物索引已写入 {vdir}")
print("下一步:threads 看候选线索,episodes 看剧集规划。")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷索引:{world.name}", [vdir / "timeline.md", vdir / "figures.md"]
)
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_threads(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
report = threads.render_report(world, found, per_thread=args.samples)
print(report)
path = threads.save_json(world, found, DATA_DIR / "threads.json")
print(f"共 {len(found)} 条候选;明细已写入 {path}")
return 0
def cmd_episodes(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
print(episodes.render_plan(world, thread, planned))
print()
print(episodes.budget_report(planned))
if args.samples:
print("\n=== 前两集事件样例 ===")
for ep in planned[:2]:
print(f" 第 {ep.index} 集({ep.span})")
for e in ep.events[: args.samples]:
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
return 0
def cmd_cast(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=not args.no_bios)
if args.dry_run:
print(text[:2000])
return 0
state = chronicle.State.load()
volume = _volume_for(state, world)
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
print(factcheck.report(suspicions) if suspicions
else "专名校验:小传里的专名全部能在史料中找到。")
return 0
def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
"""把刚写好的骨架稿大量扩写成成稿,覆盖本集文件;骨架另存 `.skeleton.md`。
骨架仍然保留:它过了自评闸门、字数合规,是溯因和对照的依据。
"""
ext, planned = volume_mod.extended_plan(world, produced.selection.thread)
directory = chapter_dir(volume)
skeleton = produced.path.with_name(produced.path.stem + ".skeleton.md")
skeleton.write_text(produced.text.strip() + "\n", encoding="utf-8")
print(f"--- 文笔层扩写中({args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, produced.selection.thread, produced.episode, len(planned), produced.text,
prev_tail=volume_mod.prev_tail_of(directory, planned, produced.episode.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
suspects = factcheck.check(text, world)
if suspects:
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, produced.selection.thread, produced.episode,
text, suspects)
text = fixed.text
for note in fixed.notes:
print(" " + note)
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
remaining = factcheck.check(text, world)
print(expand_mod.compare(produced.text, text))
print(f"成稿专名校验:可疑 {len(remaining)} 处|骨架稿另存 {skeleton.name}")
if remaining:
print(factcheck.report(remaining, top=5))
return text
def cmd_episode(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
out_dir = None if args.dry_run else chapter_dir(volume)
produced = volume_mod.produce_episode(
world, found, episode_index=args.index, out_dir=out_dir
)
print("选择依据:" + produced.selection.reason)
print(f"骨架稿字数:{chronicle.cjk_length(produced.text)}")
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(produced.text[:1600])
return 0
if args.expand:
_expand_in_place(volume, world, produced, args)
state.mark_done(produced.episode.key)
state.save()
print(f"已写入 {produced.path}")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷第 {produced.episode.index} 集(人物主线):"
f"{produced.episode.span}"
)
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_expand(args: argparse.Namespace) -> int:
"""文笔层:拿已写好的那一集当骨架,大量扩写成场景化长文。
故意不覆盖原稿:产物写 `*.literary.md`,两稿并存才能比对。也不推送。
"""
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
selection = volume_mod.select_thread(world, found, rank=args.thread)
ext, planned = volume_mod.extended_plan(world, selection.thread)
if args.index > len(planned):
print(f"第 {args.thread} 条线索只有 {len(planned)} 集,没有第 {args.index} 集")
return 1
episode = planned[args.index - 1]
directory = chapter_dir(volume)
draft_path = directory / f"{episode.key}.md"
if not draft_path.is_file():
print(f"原稿不存在:{draft_path}")
print(f"先跑:scripts/dfannals episode --index {args.index}")
return 1
draft = draft_path.read_text(encoding="utf-8")
print(f"线索:{'、'.join(world.figure_name(m) for m in selection.thread.members)}")
print(f"本集:{episode.span}|原稿 {draft_path.name}({chronicle.cjk_length(draft)} 字)")
print()
print(deslop.report(draft))
print()
print(f"--- 文笔层扩写中(模型 {args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, selection.thread, episode, len(planned), draft,
prev_tail=volume_mod.prev_tail_of(directory, planned, args.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
print()
print(expand_mod.full_report(draft, text))
print()
before = factcheck.check(draft, world)
suspects = factcheck.check(text, world)
print(f"专名校验:原稿 {len(before)} 处|扩写稿 {len(suspects)} 处")
if suspects:
print(factcheck.report(suspects, top=5))
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, selection.thread, episode, text, suspects)
for note in fixed.notes:
print(" " + note)
text = fixed.text
left = factcheck.check(text, world)
print(f" 复检:剩余可疑专名 {len(left)} 处")
if left:
print(factcheck.report(left, top=5))
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(text)
return 0
path = expand_mod.literary_path(directory, episode)
path.write_text(text.strip() + "\n", encoding="utf-8")
print(f"\n已写入 {path}")
print("原稿未改动、未推送(A/B 对照用)")
return 0
def _add_pipeline_args(p: argparse.ArgumentParser) -> None:
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
help="线索最小跨度年数(默认 30)")
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞人物主线连载管道")
sub = parser.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("status", help="查看当前进度")
p.set_defaults(func=cmd_status)
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
p.set_defaults(func=cmd_index)
p = sub.add_parser("threads", help="挖掘人物线索候选")
_add_pipeline_args(p)
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
p.set_defaults(func=cmd_threads)
p = sub.add_parser("episodes", help="按人物线索切出剧集")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
p.set_defaults(func=cmd_episodes)
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
p.set_defaults(func=cmd_cast)
p = sub.add_parser("episode", help="产出某一集(自评闸门 + 文笔层扩写)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
p.add_argument("--no-expand", dest="expand", action="store_false",
help="只出骨架稿,不做文笔层扩写")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS,
help=f"扩写目标字数(默认 {EXPAND_TARGET_CHARS})")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
p.set_defaults(func=cmd_episode, expand=True)
p = sub.add_parser("expand", help="文笔层:把某一集大量扩写成场景化长文(不覆盖原稿)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="扩写哪一集(默认 1)")
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS, help="目标字数")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘")
p.set_defaults(func=cmd_expand)
args = parser.parse_args(argv)
ensure_dirs()
return args.func(args)
if __name__ == "__main__":
sys.exit(main())