文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
+392
@@ -0,0 +1,392 @@
|
||||
"""命令行入口。
|
||||
|
||||
人物主线流程:
|
||||
|
||||
scripts/dfannals status 查看当前进度
|
||||
scripts/dfannals index 解析 exports → 年表 + 人物索引
|
||||
scripts/dfannals threads 挖掘人物线索候选(含信号明细)
|
||||
scripts/dfannals episodes --thread 1 查看某条线索切出的剧集
|
||||
scripts/dfannals cast --thread 1 生成人物表(表格 + 小传)
|
||||
scripts/dfannals episode --index 1 写出某一集(自评闸门 + 文笔层扩写)
|
||||
scripts/dfannals expand --index 1 只跑文笔层:把已有的一集大量扩写(A/B 对照)
|
||||
|
||||
用 scripts/dfannals 启动,不要直接 python3 -m:会话 PATH 上的 python3 可能是
|
||||
编辑器工具链的 venv,里面没有 defusedxml。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import (
|
||||
cast,
|
||||
chronicle,
|
||||
deslop,
|
||||
episodes,
|
||||
factcheck,
|
||||
legends,
|
||||
publish,
|
||||
threads,
|
||||
timeline,
|
||||
)
|
||||
from dfannals import expand as expand_mod
|
||||
from dfannals import volume as volume_mod
|
||||
from dfannals.config import (
|
||||
DATA_DIR,
|
||||
EXPAND_TARGET_CHARS,
|
||||
EXPORT_DIR,
|
||||
GITEA_WEB,
|
||||
LITERARY_MODEL,
|
||||
chapter_dir,
|
||||
ensure_dirs,
|
||||
volume_dir,
|
||||
)
|
||||
|
||||
|
||||
def find_exports(export_dir: Path) -> list[Path]:
|
||||
"""找出 legends 导出文件。
|
||||
|
||||
原生 legends.xml 与 DFHack 的 legends_plus.xml 都要;
|
||||
预处理缓存放在子目录,所以这里不会重复收到同一份数据。
|
||||
"""
|
||||
if not export_dir.is_dir():
|
||||
return []
|
||||
files = sorted(export_dir.glob("*.xml"))
|
||||
base = [f for f in files if "legends_plus" not in f.name]
|
||||
plus = [f for f in files if "legends_plus" in f.name]
|
||||
return base + plus
|
||||
|
||||
|
||||
def load_world(export_dir: Path) -> legends.World:
|
||||
paths = find_exports(export_dir)
|
||||
if not paths:
|
||||
raise SystemExit(
|
||||
f"在 {export_dir} 里没找到 legends XML。\n"
|
||||
"请先导出(游戏内 Export XML + DFHack 的 exportlegends),再复制进这个目录。"
|
||||
)
|
||||
return legends.load(paths)
|
||||
|
||||
|
||||
def _current_volume(state: chronicle.State, world: legends.World) -> int:
|
||||
"""世界换了就开新卷。"""
|
||||
if state.world != world.name:
|
||||
if state.world:
|
||||
state.volume += 1
|
||||
state.done = []
|
||||
state.world = world.name
|
||||
return state.volume
|
||||
|
||||
|
||||
def _volume_for(state: chronicle.State, world: legends.World) -> int:
|
||||
return state.volume if state.world == world.name else 1
|
||||
|
||||
|
||||
def _load_candidates(args: argparse.Namespace) -> tuple[legends.World, list[threads.Thread]]:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
found = threads.find_threads(
|
||||
world, min_interactions=args.min_interactions, min_span=args.min_span
|
||||
)
|
||||
if not found:
|
||||
print("没有候选线索")
|
||||
raise SystemExit(1)
|
||||
return world, found
|
||||
|
||||
|
||||
def cmd_status(args: argparse.Namespace) -> int:
|
||||
state = chronicle.State.load()
|
||||
if not state.world:
|
||||
print("尚未开始。先运行 index。")
|
||||
return 0
|
||||
vdir = volume_dir(state.volume)
|
||||
print(f"世界:{state.world}")
|
||||
print(f"卷号:第 {state.volume} 卷")
|
||||
print(f"已完成集数:{len(state.done)}")
|
||||
for key in state.done:
|
||||
print(f" · {key}")
|
||||
print(f"成品目录:{vdir}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_index(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
state = chronicle.State.load()
|
||||
volume = _current_volume(state, world)
|
||||
|
||||
vdir = volume_dir(volume)
|
||||
timeline.write_outputs(world, vdir)
|
||||
state.save()
|
||||
|
||||
print(legends.describe(world))
|
||||
print()
|
||||
print(f"年表与人物索引已写入 {vdir}")
|
||||
print("下一步:threads 看候选线索,episodes 看剧集规划。")
|
||||
|
||||
if not args.no_push:
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 {volume} 卷索引:{world.name}", [vdir / "timeline.md", vdir / "figures.md"]
|
||||
)
|
||||
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_threads(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
report = threads.render_report(world, found, per_thread=args.samples)
|
||||
print(report)
|
||||
path = threads.save_json(world, found, DATA_DIR / "threads.json")
|
||||
print(f"共 {len(found)} 条候选;明细已写入 {path}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_episodes(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
if not 1 <= args.thread <= len(found):
|
||||
print(f"候选只有 {len(found)} 条,--thread 超出范围")
|
||||
return 1
|
||||
thread = found[args.thread - 1]
|
||||
_, planned = volume_mod.extended_plan(world, thread)
|
||||
print(episodes.render_plan(world, thread, planned))
|
||||
print()
|
||||
print(episodes.budget_report(planned))
|
||||
if args.samples:
|
||||
print("\n=== 前两集事件样例 ===")
|
||||
for ep in planned[:2]:
|
||||
print(f" 第 {ep.index} 集({ep.span})")
|
||||
for e in ep.events[: args.samples]:
|
||||
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_cast(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
if not 1 <= args.thread <= len(found):
|
||||
print(f"候选只有 {len(found)} 条,--thread 超出范围")
|
||||
return 1
|
||||
|
||||
thread = found[args.thread - 1]
|
||||
_, planned = volume_mod.extended_plan(world, thread)
|
||||
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=not args.no_bios)
|
||||
if args.dry_run:
|
||||
print(text[:2000])
|
||||
return 0
|
||||
|
||||
state = chronicle.State.load()
|
||||
volume = _volume_for(state, world)
|
||||
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
|
||||
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
|
||||
print(factcheck.report(suspicions) if suspicions
|
||||
else "专名校验:小传里的专名全部能在史料中找到。")
|
||||
return 0
|
||||
|
||||
|
||||
def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
|
||||
"""把刚写好的骨架稿大量扩写成成稿,覆盖本集文件;骨架另存 `.skeleton.md`。
|
||||
|
||||
骨架仍然保留:它过了自评闸门、字数合规,是溯因和对照的依据。
|
||||
"""
|
||||
ext, planned = volume_mod.extended_plan(world, produced.selection.thread)
|
||||
directory = chapter_dir(volume)
|
||||
skeleton = produced.path.with_name(produced.path.stem + ".skeleton.md")
|
||||
skeleton.write_text(produced.text.strip() + "\n", encoding="utf-8")
|
||||
|
||||
print(f"--- 文笔层扩写中({args.model},目标 {args.target} 字)---")
|
||||
result = expand_mod.expand_episode(
|
||||
world, produced.selection.thread, produced.episode, len(planned), produced.text,
|
||||
prev_tail=volume_mod.prev_tail_of(directory, planned, produced.episode.index),
|
||||
model=args.model, target=args.target, elided=ext.elided,
|
||||
)
|
||||
text = result.text
|
||||
for note in result.notes:
|
||||
print(" " + note)
|
||||
|
||||
suspects = factcheck.check(text, world)
|
||||
if suspects:
|
||||
print("→ 自动修复可疑专名(只改含它们的段落)")
|
||||
fixed = expand_mod.repair_names(world, produced.selection.thread, produced.episode,
|
||||
text, suspects)
|
||||
text = fixed.text
|
||||
for note in fixed.notes:
|
||||
print(" " + note)
|
||||
|
||||
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
|
||||
remaining = factcheck.check(text, world)
|
||||
print(expand_mod.compare(produced.text, text))
|
||||
print(f"成稿专名校验:可疑 {len(remaining)} 处|骨架稿另存 {skeleton.name}")
|
||||
if remaining:
|
||||
print(factcheck.report(remaining, top=5))
|
||||
return text
|
||||
|
||||
|
||||
def cmd_episode(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
state = chronicle.State.load()
|
||||
volume = _volume_for(state, world)
|
||||
|
||||
out_dir = None if args.dry_run else chapter_dir(volume)
|
||||
produced = volume_mod.produce_episode(
|
||||
world, found, episode_index=args.index, out_dir=out_dir
|
||||
)
|
||||
|
||||
print("选择依据:" + produced.selection.reason)
|
||||
print(f"骨架稿字数:{chronicle.cjk_length(produced.text)}")
|
||||
if args.dry_run:
|
||||
print("\n--- 试运行,不落盘 ---\n")
|
||||
print(produced.text[:1600])
|
||||
return 0
|
||||
|
||||
if args.expand:
|
||||
_expand_in_place(volume, world, produced, args)
|
||||
|
||||
state.mark_done(produced.episode.key)
|
||||
state.save()
|
||||
print(f"已写入 {produced.path}")
|
||||
if not args.no_push:
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 {volume} 卷第 {produced.episode.index} 集(人物主线):"
|
||||
f"{produced.episode.span}"
|
||||
)
|
||||
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_expand(args: argparse.Namespace) -> int:
|
||||
"""文笔层:拿已写好的那一集当骨架,大量扩写成场景化长文。
|
||||
|
||||
故意不覆盖原稿:产物写 `*.literary.md`,两稿并存才能比对。也不推送。
|
||||
"""
|
||||
world, found = _load_candidates(args)
|
||||
state = chronicle.State.load()
|
||||
volume = _volume_for(state, world)
|
||||
|
||||
selection = volume_mod.select_thread(world, found, rank=args.thread)
|
||||
ext, planned = volume_mod.extended_plan(world, selection.thread)
|
||||
if args.index > len(planned):
|
||||
print(f"第 {args.thread} 条线索只有 {len(planned)} 集,没有第 {args.index} 集")
|
||||
return 1
|
||||
|
||||
episode = planned[args.index - 1]
|
||||
directory = chapter_dir(volume)
|
||||
draft_path = directory / f"{episode.key}.md"
|
||||
if not draft_path.is_file():
|
||||
print(f"原稿不存在:{draft_path}")
|
||||
print(f"先跑:scripts/dfannals episode --index {args.index}")
|
||||
return 1
|
||||
|
||||
draft = draft_path.read_text(encoding="utf-8")
|
||||
print(f"线索:{'、'.join(world.figure_name(m) for m in selection.thread.members)}")
|
||||
print(f"本集:{episode.span}|原稿 {draft_path.name}({chronicle.cjk_length(draft)} 字)")
|
||||
print()
|
||||
print(deslop.report(draft))
|
||||
print()
|
||||
print(f"--- 文笔层扩写中(模型 {args.model},目标 {args.target} 字)---")
|
||||
|
||||
result = expand_mod.expand_episode(
|
||||
world, selection.thread, episode, len(planned), draft,
|
||||
prev_tail=volume_mod.prev_tail_of(directory, planned, args.index),
|
||||
model=args.model, target=args.target, elided=ext.elided,
|
||||
)
|
||||
text = result.text
|
||||
for note in result.notes:
|
||||
print(" " + note)
|
||||
|
||||
print()
|
||||
print(expand_mod.full_report(draft, text))
|
||||
print()
|
||||
|
||||
before = factcheck.check(draft, world)
|
||||
suspects = factcheck.check(text, world)
|
||||
print(f"专名校验:原稿 {len(before)} 处|扩写稿 {len(suspects)} 处")
|
||||
if suspects:
|
||||
print(factcheck.report(suspects, top=5))
|
||||
print("→ 自动修复可疑专名(只改含它们的段落)")
|
||||
fixed = expand_mod.repair_names(world, selection.thread, episode, text, suspects)
|
||||
for note in fixed.notes:
|
||||
print(" " + note)
|
||||
text = fixed.text
|
||||
left = factcheck.check(text, world)
|
||||
print(f" 复检:剩余可疑专名 {len(left)} 处")
|
||||
if left:
|
||||
print(factcheck.report(left, top=5))
|
||||
|
||||
if args.dry_run:
|
||||
print("\n--- 试运行,不落盘 ---\n")
|
||||
print(text)
|
||||
return 0
|
||||
|
||||
path = expand_mod.literary_path(directory, episode)
|
||||
path.write_text(text.strip() + "\n", encoding="utf-8")
|
||||
print(f"\n已写入 {path}")
|
||||
print("原稿未改动、未推送(A/B 对照用)")
|
||||
return 0
|
||||
|
||||
|
||||
def _add_pipeline_args(p: argparse.ArgumentParser) -> None:
|
||||
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
|
||||
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
|
||||
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
|
||||
help="线索最小跨度年数(默认 30)")
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞人物主线连载管道")
|
||||
sub = parser.add_subparsers(dest="cmd", required=True)
|
||||
|
||||
p = sub.add_parser("status", help="查看当前进度")
|
||||
p.set_defaults(func=cmd_status)
|
||||
|
||||
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
|
||||
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
|
||||
p.set_defaults(func=cmd_index)
|
||||
|
||||
p = sub.add_parser("threads", help="挖掘人物线索候选")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
|
||||
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
|
||||
p.set_defaults(func=cmd_threads)
|
||||
|
||||
p = sub.add_parser("episodes", help="按人物线索切出剧集")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
|
||||
p.set_defaults(func=cmd_episodes)
|
||||
|
||||
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
|
||||
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
|
||||
p.set_defaults(func=cmd_cast)
|
||||
|
||||
p = sub.add_parser("episode", help="产出某一集(自评闸门 + 文笔层扩写)")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
|
||||
p.add_argument("--no-expand", dest="expand", action="store_false",
|
||||
help="只出骨架稿,不做文笔层扩写")
|
||||
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS,
|
||||
help=f"扩写目标字数(默认 {EXPAND_TARGET_CHARS})")
|
||||
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
|
||||
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
|
||||
p.set_defaults(func=cmd_episode, expand=True)
|
||||
|
||||
p = sub.add_parser("expand", help="文笔层:把某一集大量扩写成场景化长文(不覆盖原稿)")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--index", type=int, default=1, help="扩写哪一集(默认 1)")
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS, help="目标字数")
|
||||
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘")
|
||||
p.set_defaults(func=cmd_expand)
|
||||
|
||||
args = parser.parse_args(argv)
|
||||
ensure_dirs()
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
|
||||
Reference in New Issue
Block a user