文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
+9
-6
@@ -54,14 +54,15 @@ class CastRow:
|
||||
fid: int
|
||||
name: str
|
||||
race: str
|
||||
gender: str # 男/女/空(史料未载)
|
||||
span: str
|
||||
role: str # 主角 / 对手 / 配角
|
||||
appearances: int
|
||||
identity: str
|
||||
|
||||
def as_row(self) -> str:
|
||||
return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | "
|
||||
f"{self.appearances} | {self.role} |")
|
||||
return (f"| {self.name} | {self.race or '—'} | {self.gender or '—'} | {self.span} | "
|
||||
f"{self.identity or '—'} | {self.appearances} | {self.role} |")
|
||||
|
||||
|
||||
def _identity_of(world: World, fid: int, events: list[Event]) -> str:
|
||||
@@ -107,6 +108,7 @@ def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_othe
|
||||
fid=fid,
|
||||
name=world.figure_name(fid),
|
||||
race=fig.race if fig else "",
|
||||
gender=fig.gender if fig else "",
|
||||
span=fig.alive_span if fig else "生卒不详",
|
||||
role=role,
|
||||
appearances=appearances.get(fid, 0),
|
||||
@@ -136,8 +138,9 @@ def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
|
||||
for row in rows[:BIO_LIMIT]:
|
||||
facts = [
|
||||
(
|
||||
f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次,"
|
||||
f"身份:{row.identity or '不详'},在故事里是{row.role})"
|
||||
f"{row.name}({row.race or '种族不详'},{row.gender or '性别史料未载'},{row.span},"
|
||||
f"本卷出场 {row.appearances} 次,身份:{row.identity or '不详'},"
|
||||
f"在故事里是{row.role})"
|
||||
)
|
||||
]
|
||||
for e in events:
|
||||
@@ -197,8 +200,8 @@ def build_cast(
|
||||
f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
|
||||
f"线索跨度:{_span_of(thread, episodes)}",
|
||||
"",
|
||||
"| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |",
|
||||
"|---|---|---|---|---|---|",
|
||||
"| 人物 | 种族 | 性别 | 生卒 | 身份 | 本卷出场 | 角色 |",
|
||||
"|---|---|---|---|---|---|---|",
|
||||
]
|
||||
lines.extend(row.as_row() for row in rows)
|
||||
|
||||
|
||||
+392
@@ -0,0 +1,392 @@
|
||||
"""命令行入口。
|
||||
|
||||
人物主线流程:
|
||||
|
||||
scripts/dfannals status 查看当前进度
|
||||
scripts/dfannals index 解析 exports → 年表 + 人物索引
|
||||
scripts/dfannals threads 挖掘人物线索候选(含信号明细)
|
||||
scripts/dfannals episodes --thread 1 查看某条线索切出的剧集
|
||||
scripts/dfannals cast --thread 1 生成人物表(表格 + 小传)
|
||||
scripts/dfannals episode --index 1 写出某一集(自评闸门 + 文笔层扩写)
|
||||
scripts/dfannals expand --index 1 只跑文笔层:把已有的一集大量扩写(A/B 对照)
|
||||
|
||||
用 scripts/dfannals 启动,不要直接 python3 -m:会话 PATH 上的 python3 可能是
|
||||
编辑器工具链的 venv,里面没有 defusedxml。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import (
|
||||
cast,
|
||||
chronicle,
|
||||
deslop,
|
||||
episodes,
|
||||
factcheck,
|
||||
legends,
|
||||
publish,
|
||||
threads,
|
||||
timeline,
|
||||
)
|
||||
from dfannals import expand as expand_mod
|
||||
from dfannals import volume as volume_mod
|
||||
from dfannals.config import (
|
||||
DATA_DIR,
|
||||
EXPAND_TARGET_CHARS,
|
||||
EXPORT_DIR,
|
||||
GITEA_WEB,
|
||||
LITERARY_MODEL,
|
||||
chapter_dir,
|
||||
ensure_dirs,
|
||||
volume_dir,
|
||||
)
|
||||
|
||||
|
||||
def find_exports(export_dir: Path) -> list[Path]:
|
||||
"""找出 legends 导出文件。
|
||||
|
||||
原生 legends.xml 与 DFHack 的 legends_plus.xml 都要;
|
||||
预处理缓存放在子目录,所以这里不会重复收到同一份数据。
|
||||
"""
|
||||
if not export_dir.is_dir():
|
||||
return []
|
||||
files = sorted(export_dir.glob("*.xml"))
|
||||
base = [f for f in files if "legends_plus" not in f.name]
|
||||
plus = [f for f in files if "legends_plus" in f.name]
|
||||
return base + plus
|
||||
|
||||
|
||||
def load_world(export_dir: Path) -> legends.World:
|
||||
paths = find_exports(export_dir)
|
||||
if not paths:
|
||||
raise SystemExit(
|
||||
f"在 {export_dir} 里没找到 legends XML。\n"
|
||||
"请先导出(游戏内 Export XML + DFHack 的 exportlegends),再复制进这个目录。"
|
||||
)
|
||||
return legends.load(paths)
|
||||
|
||||
|
||||
def _current_volume(state: chronicle.State, world: legends.World) -> int:
|
||||
"""世界换了就开新卷。"""
|
||||
if state.world != world.name:
|
||||
if state.world:
|
||||
state.volume += 1
|
||||
state.done = []
|
||||
state.world = world.name
|
||||
return state.volume
|
||||
|
||||
|
||||
def _volume_for(state: chronicle.State, world: legends.World) -> int:
|
||||
return state.volume if state.world == world.name else 1
|
||||
|
||||
|
||||
def _load_candidates(args: argparse.Namespace) -> tuple[legends.World, list[threads.Thread]]:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
found = threads.find_threads(
|
||||
world, min_interactions=args.min_interactions, min_span=args.min_span
|
||||
)
|
||||
if not found:
|
||||
print("没有候选线索")
|
||||
raise SystemExit(1)
|
||||
return world, found
|
||||
|
||||
|
||||
def cmd_status(args: argparse.Namespace) -> int:
|
||||
state = chronicle.State.load()
|
||||
if not state.world:
|
||||
print("尚未开始。先运行 index。")
|
||||
return 0
|
||||
vdir = volume_dir(state.volume)
|
||||
print(f"世界:{state.world}")
|
||||
print(f"卷号:第 {state.volume} 卷")
|
||||
print(f"已完成集数:{len(state.done)}")
|
||||
for key in state.done:
|
||||
print(f" · {key}")
|
||||
print(f"成品目录:{vdir}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_index(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
state = chronicle.State.load()
|
||||
volume = _current_volume(state, world)
|
||||
|
||||
vdir = volume_dir(volume)
|
||||
timeline.write_outputs(world, vdir)
|
||||
state.save()
|
||||
|
||||
print(legends.describe(world))
|
||||
print()
|
||||
print(f"年表与人物索引已写入 {vdir}")
|
||||
print("下一步:threads 看候选线索,episodes 看剧集规划。")
|
||||
|
||||
if not args.no_push:
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 {volume} 卷索引:{world.name}", [vdir / "timeline.md", vdir / "figures.md"]
|
||||
)
|
||||
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_threads(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
report = threads.render_report(world, found, per_thread=args.samples)
|
||||
print(report)
|
||||
path = threads.save_json(world, found, DATA_DIR / "threads.json")
|
||||
print(f"共 {len(found)} 条候选;明细已写入 {path}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_episodes(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
if not 1 <= args.thread <= len(found):
|
||||
print(f"候选只有 {len(found)} 条,--thread 超出范围")
|
||||
return 1
|
||||
thread = found[args.thread - 1]
|
||||
_, planned = volume_mod.extended_plan(world, thread)
|
||||
print(episodes.render_plan(world, thread, planned))
|
||||
print()
|
||||
print(episodes.budget_report(planned))
|
||||
if args.samples:
|
||||
print("\n=== 前两集事件样例 ===")
|
||||
for ep in planned[:2]:
|
||||
print(f" 第 {ep.index} 集({ep.span})")
|
||||
for e in ep.events[: args.samples]:
|
||||
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_cast(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
if not 1 <= args.thread <= len(found):
|
||||
print(f"候选只有 {len(found)} 条,--thread 超出范围")
|
||||
return 1
|
||||
|
||||
thread = found[args.thread - 1]
|
||||
_, planned = volume_mod.extended_plan(world, thread)
|
||||
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=not args.no_bios)
|
||||
if args.dry_run:
|
||||
print(text[:2000])
|
||||
return 0
|
||||
|
||||
state = chronicle.State.load()
|
||||
volume = _volume_for(state, world)
|
||||
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
|
||||
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
|
||||
print(factcheck.report(suspicions) if suspicions
|
||||
else "专名校验:小传里的专名全部能在史料中找到。")
|
||||
return 0
|
||||
|
||||
|
||||
def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
|
||||
"""把刚写好的骨架稿大量扩写成成稿,覆盖本集文件;骨架另存 `.skeleton.md`。
|
||||
|
||||
骨架仍然保留:它过了自评闸门、字数合规,是溯因和对照的依据。
|
||||
"""
|
||||
ext, planned = volume_mod.extended_plan(world, produced.selection.thread)
|
||||
directory = chapter_dir(volume)
|
||||
skeleton = produced.path.with_name(produced.path.stem + ".skeleton.md")
|
||||
skeleton.write_text(produced.text.strip() + "\n", encoding="utf-8")
|
||||
|
||||
print(f"--- 文笔层扩写中({args.model},目标 {args.target} 字)---")
|
||||
result = expand_mod.expand_episode(
|
||||
world, produced.selection.thread, produced.episode, len(planned), produced.text,
|
||||
prev_tail=volume_mod.prev_tail_of(directory, planned, produced.episode.index),
|
||||
model=args.model, target=args.target, elided=ext.elided,
|
||||
)
|
||||
text = result.text
|
||||
for note in result.notes:
|
||||
print(" " + note)
|
||||
|
||||
suspects = factcheck.check(text, world)
|
||||
if suspects:
|
||||
print("→ 自动修复可疑专名(只改含它们的段落)")
|
||||
fixed = expand_mod.repair_names(world, produced.selection.thread, produced.episode,
|
||||
text, suspects)
|
||||
text = fixed.text
|
||||
for note in fixed.notes:
|
||||
print(" " + note)
|
||||
|
||||
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
|
||||
remaining = factcheck.check(text, world)
|
||||
print(expand_mod.compare(produced.text, text))
|
||||
print(f"成稿专名校验:可疑 {len(remaining)} 处|骨架稿另存 {skeleton.name}")
|
||||
if remaining:
|
||||
print(factcheck.report(remaining, top=5))
|
||||
return text
|
||||
|
||||
|
||||
def cmd_episode(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
state = chronicle.State.load()
|
||||
volume = _volume_for(state, world)
|
||||
|
||||
out_dir = None if args.dry_run else chapter_dir(volume)
|
||||
produced = volume_mod.produce_episode(
|
||||
world, found, episode_index=args.index, out_dir=out_dir
|
||||
)
|
||||
|
||||
print("选择依据:" + produced.selection.reason)
|
||||
print(f"骨架稿字数:{chronicle.cjk_length(produced.text)}")
|
||||
if args.dry_run:
|
||||
print("\n--- 试运行,不落盘 ---\n")
|
||||
print(produced.text[:1600])
|
||||
return 0
|
||||
|
||||
if args.expand:
|
||||
_expand_in_place(volume, world, produced, args)
|
||||
|
||||
state.mark_done(produced.episode.key)
|
||||
state.save()
|
||||
print(f"已写入 {produced.path}")
|
||||
if not args.no_push:
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 {volume} 卷第 {produced.episode.index} 集(人物主线):"
|
||||
f"{produced.episode.span}"
|
||||
)
|
||||
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_expand(args: argparse.Namespace) -> int:
|
||||
"""文笔层:拿已写好的那一集当骨架,大量扩写成场景化长文。
|
||||
|
||||
故意不覆盖原稿:产物写 `*.literary.md`,两稿并存才能比对。也不推送。
|
||||
"""
|
||||
world, found = _load_candidates(args)
|
||||
state = chronicle.State.load()
|
||||
volume = _volume_for(state, world)
|
||||
|
||||
selection = volume_mod.select_thread(world, found, rank=args.thread)
|
||||
ext, planned = volume_mod.extended_plan(world, selection.thread)
|
||||
if args.index > len(planned):
|
||||
print(f"第 {args.thread} 条线索只有 {len(planned)} 集,没有第 {args.index} 集")
|
||||
return 1
|
||||
|
||||
episode = planned[args.index - 1]
|
||||
directory = chapter_dir(volume)
|
||||
draft_path = directory / f"{episode.key}.md"
|
||||
if not draft_path.is_file():
|
||||
print(f"原稿不存在:{draft_path}")
|
||||
print(f"先跑:scripts/dfannals episode --index {args.index}")
|
||||
return 1
|
||||
|
||||
draft = draft_path.read_text(encoding="utf-8")
|
||||
print(f"线索:{'、'.join(world.figure_name(m) for m in selection.thread.members)}")
|
||||
print(f"本集:{episode.span}|原稿 {draft_path.name}({chronicle.cjk_length(draft)} 字)")
|
||||
print()
|
||||
print(deslop.report(draft))
|
||||
print()
|
||||
print(f"--- 文笔层扩写中(模型 {args.model},目标 {args.target} 字)---")
|
||||
|
||||
result = expand_mod.expand_episode(
|
||||
world, selection.thread, episode, len(planned), draft,
|
||||
prev_tail=volume_mod.prev_tail_of(directory, planned, args.index),
|
||||
model=args.model, target=args.target, elided=ext.elided,
|
||||
)
|
||||
text = result.text
|
||||
for note in result.notes:
|
||||
print(" " + note)
|
||||
|
||||
print()
|
||||
print(expand_mod.full_report(draft, text))
|
||||
print()
|
||||
|
||||
before = factcheck.check(draft, world)
|
||||
suspects = factcheck.check(text, world)
|
||||
print(f"专名校验:原稿 {len(before)} 处|扩写稿 {len(suspects)} 处")
|
||||
if suspects:
|
||||
print(factcheck.report(suspects, top=5))
|
||||
print("→ 自动修复可疑专名(只改含它们的段落)")
|
||||
fixed = expand_mod.repair_names(world, selection.thread, episode, text, suspects)
|
||||
for note in fixed.notes:
|
||||
print(" " + note)
|
||||
text = fixed.text
|
||||
left = factcheck.check(text, world)
|
||||
print(f" 复检:剩余可疑专名 {len(left)} 处")
|
||||
if left:
|
||||
print(factcheck.report(left, top=5))
|
||||
|
||||
if args.dry_run:
|
||||
print("\n--- 试运行,不落盘 ---\n")
|
||||
print(text)
|
||||
return 0
|
||||
|
||||
path = expand_mod.literary_path(directory, episode)
|
||||
path.write_text(text.strip() + "\n", encoding="utf-8")
|
||||
print(f"\n已写入 {path}")
|
||||
print("原稿未改动、未推送(A/B 对照用)")
|
||||
return 0
|
||||
|
||||
|
||||
def _add_pipeline_args(p: argparse.ArgumentParser) -> None:
|
||||
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
|
||||
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
|
||||
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
|
||||
help="线索最小跨度年数(默认 30)")
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞人物主线连载管道")
|
||||
sub = parser.add_subparsers(dest="cmd", required=True)
|
||||
|
||||
p = sub.add_parser("status", help="查看当前进度")
|
||||
p.set_defaults(func=cmd_status)
|
||||
|
||||
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
|
||||
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
|
||||
p.set_defaults(func=cmd_index)
|
||||
|
||||
p = sub.add_parser("threads", help="挖掘人物线索候选")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
|
||||
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
|
||||
p.set_defaults(func=cmd_threads)
|
||||
|
||||
p = sub.add_parser("episodes", help="按人物线索切出剧集")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
|
||||
p.set_defaults(func=cmd_episodes)
|
||||
|
||||
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
|
||||
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
|
||||
p.set_defaults(func=cmd_cast)
|
||||
|
||||
p = sub.add_parser("episode", help="产出某一集(自评闸门 + 文笔层扩写)")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
|
||||
p.add_argument("--no-expand", dest="expand", action="store_false",
|
||||
help="只出骨架稿,不做文笔层扩写")
|
||||
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS,
|
||||
help=f"扩写目标字数(默认 {EXPAND_TARGET_CHARS})")
|
||||
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
|
||||
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
|
||||
p.set_defaults(func=cmd_episode, expand=True)
|
||||
|
||||
p = sub.add_parser("expand", help="文笔层:把某一集大量扩写成场景化长文(不覆盖原稿)")
|
||||
_add_pipeline_args(p)
|
||||
p.add_argument("--index", type=int, default=1, help="扩写哪一集(默认 1)")
|
||||
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
|
||||
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS, help="目标字数")
|
||||
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘")
|
||||
p.set_defaults(func=cmd_expand)
|
||||
|
||||
args = parser.parse_args(argv)
|
||||
ensure_dirs()
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
|
||||
@@ -44,6 +44,9 @@ PI_AUTH_JSON = HOME / ".pi/agent/auth.json"
|
||||
PI_MODELS_JSON = HOME / ".pi/agent/models.json"
|
||||
PROVIDER = "workbuddy"
|
||||
MODEL = "cn:deepseek-v4-pro"
|
||||
# 文笔层(扩写)单独选模型:创作向模型写叙事更贴地,
|
||||
# 而推理向模型(deepseek-v4-pro)天然偏总结陈词,容易写成报告。
|
||||
LITERARY_MODEL = "cn:minimax-m3"
|
||||
# 该模型是推理模型:思考与正文共用 max_tokens,必须留足
|
||||
MAX_TOKENS_CHAPTER = 16000
|
||||
MAX_TOKENS_UTIL = 4000
|
||||
@@ -66,6 +69,16 @@ WORLD_HISTORY_YEARS = 250
|
||||
CHAPTER_MIN_CHARS = 800
|
||||
CHAPTER_MAX_CHARS = 1500 # 严格按访谈定的 800–1500,不留“余量”
|
||||
|
||||
# ---------------------------------------------------------------- 文笔层(扩写)
|
||||
# 用户口径:要的是「大量扩写」——原稿是骨架,文笔层把它逐事件展开成现场。
|
||||
EXPAND_TARGET_CHARS = 8000 # 目标字数(用户选定:每集约 8000 字)
|
||||
EXPAND_FLOOR = 6000 # 低于此判太短,继续展开
|
||||
EXPAND_CEILING = 11000 # 高于此判太长,压缩
|
||||
# 一次调用写 8000 字很容易跑偏、漏事件,所以按场景分块展开,每块目标这么多字
|
||||
EXPAND_CHUNK_TARGET = 2600
|
||||
# 长文 + 可能的思考 tokens,预算要给足
|
||||
MAX_TOKENS_EXPAND = 32000
|
||||
|
||||
|
||||
@dataclass
|
||||
class Gateway:
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
"""AI 味机械诊断:把「文笔是否自然」变成可度量的指标。
|
||||
|
||||
规则来自 oh-story-claudecode 的 story-deslop 与 banned-words(MIT),做了两处适配:
|
||||
|
||||
1. **分层**。原文把「仿佛、一丝、坚定、冰冷」放同一张一级表,但「坚定」「冰冷」在
|
||||
史书叙事里是正常词(矮人要塞的"冰冷"可能是字面温度),一律判罪会刷满假阳性。
|
||||
因此分成硬伤词(几乎只在 AI 文本里成串出现)与密度词(真人也用,只看密度)。
|
||||
2. **按密度而非布尔判定**。AI 味是密度问题:出现一次是风格,出现二十次是模板。
|
||||
输出「每千字命中数」,供闸门与扩写模型参考。
|
||||
|
||||
语义类判断(比喻是否套话、是否把一件事掰成三段写)机械检查假阳性太高,
|
||||
交给扩写模型和自评闸门,这里不猜。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from dfannals.chronicle import cjk_length
|
||||
|
||||
# 硬伤词:出现在正常中文里的概率很低,命中一处就该改
|
||||
HARD_WORDS: dict[str, tuple[str, ...]] = {
|
||||
"情态": ("仿佛", "犹如", "宛若", "如同", "一丝", "一抹", "些许", "几分", "隐约",
|
||||
"毫无征兆", "几不可闻", "微不可察"),
|
||||
"动作": ("深吸一口气", "不禁"),
|
||||
"表情": ("眼中闪过", "嘴角勾起", "眉头微皱", "眉眼低垂", "瞳孔微缩", "瞳孔收缩",
|
||||
"瞳孔一缩", "指节泛白", "眼神锐利", "目光锐利"),
|
||||
"心理": ("心中一动", "心头一震", "心下了然", "心中暗道", "心底泛起", "不由得", "心中一凛"),
|
||||
"判断": ("不容置疑", "不容置喙", "不易察觉", "显而易见", "毫无疑问", "不可否认", "前所未有"),
|
||||
"过渡": ("不由自主", "情不自禁", "自然而然", "话锋一转", "闪烁着光芒"),
|
||||
}
|
||||
|
||||
# 密度词:真人写作也用,只有成串出现才算模板腔
|
||||
SOFT_WORDS: dict[str, tuple[str, ...]] = {
|
||||
"形容": ("坚定", "狡黠", "深邃", "凛冽", "冰冷"),
|
||||
"语境敏感": ("突然", "陡然", "骤然", "猛然", "好像", "似乎", "瞬间", "猛地", "死死地"),
|
||||
"弱化副词": ("缓缓", "微微", "轻轻", "淡淡"),
|
||||
}
|
||||
|
||||
# 句式与标点:每条是(规则名,说明,正则)
|
||||
PATTERNS: tuple[tuple[str, str, str], ...] = (
|
||||
("不是A而是B", "最毒的 AI 句式,直接写 B", r"不是[^,。;!?\n]{1,24}[,,]\s*(?:而)?是"),
|
||||
("万能动宾", "「,带着……」万能状语", r"[,,]\s*带着[^,。;!?\n]{0,20}"),
|
||||
("声音描写", "「声音不大,却带着……」", r"声音(?:不大|很轻|很平|平静)[^。\n]{0,14}(?:却|但)?带着"),
|
||||
("认知直述", "告诉而非展示,改为动作或台词", r"(?:他|她|他们|她们)(?:这才|终于)?(?:明白|意识到|感到|知道)"),
|
||||
("章末预告", "「他不知道的是……」空泛预告", r"(?:他|她)不知道的是"),
|
||||
("收束腔", "「这一刻/才刚刚开始」式总结", r"这一刻[,,]|从这一刻开始|才刚刚开始|原来[,,]"),
|
||||
("抽象命运", "命运+齿轮/棋局/獠牙", r"命运[^。\n]{0,8}(?:齿轮|棋局|獠牙|改写|安排)"),
|
||||
("心理模板", "「心中一震/心头一凛」式心理惊吓", r"心(?:中|头|底)(?:一|猛|顿)(?:震|凛|颤)"),
|
||||
("过渡模板", "「取而代之的是」", r"取而代之"),
|
||||
("告诉而非展示", "「显得有些X」「散发着一股X气息」", r"显得(?:有些|十分|非常)?[\u4e00-\u9fff]{1,4}|散发(?:着|出)?[^。\n]{0,8}(?:气息|气场)"),
|
||||
("公式化对话标签", "「好的,他说道」", r"[,,](?:他|她)(?:说道|开口道|沉声道|低声道)"),
|
||||
("英文引号", "模型常直接吐 ASCII 直角引号,中文排版用“”", r'"'),
|
||||
("破折号", "正文禁用,改用句号、逗号或动作断句", r"——|──|(?<!-)--(?!-)"),
|
||||
("省略号", "正文禁用,用动作或短句表达停顿", r"……|\.\.\."),
|
||||
)
|
||||
|
||||
SAMPLE_WIDTH = 14
|
||||
SAMPLES_PER_RULE = 2
|
||||
|
||||
|
||||
@dataclass
|
||||
class Hit:
|
||||
"""一条诊断:命中哪个规则、几次、原文长什么样。"""
|
||||
|
||||
rule: str
|
||||
category: str
|
||||
level: str # hard = 硬伤词+句式+标点;soft = 只看密度的词
|
||||
count: int
|
||||
samples: list[str] = field(default_factory=list)
|
||||
|
||||
def line(self) -> str:
|
||||
ex = ";".join(self.samples)
|
||||
return f"{self.category:<6} {self.rule:<8} ×{self.count:<3} {ex}"
|
||||
|
||||
|
||||
def _context(text: str, start: int, end: int) -> str:
|
||||
left = max(0, start - SAMPLE_WIDTH)
|
||||
right = min(len(text), end + SAMPLE_WIDTH)
|
||||
return text[left:right].replace("\n", " ").strip()
|
||||
|
||||
|
||||
def _scan_word(text: str, word: str) -> tuple[int, list[str]]:
|
||||
count, samples, pos = 0, [], text.find(word)
|
||||
while pos != -1:
|
||||
count += 1
|
||||
if len(samples) < SAMPLES_PER_RULE:
|
||||
samples.append(_context(text, pos, pos + len(word)))
|
||||
pos = text.find(word, pos + len(word))
|
||||
return count, samples
|
||||
|
||||
|
||||
def check(text: str) -> list[Hit]:
|
||||
"""扫一遍文本,返回全部诊断(硬伤在前,同类按命中数降序)。"""
|
||||
hits: list[Hit] = []
|
||||
|
||||
for category, words in HARD_WORDS.items():
|
||||
for word in words:
|
||||
count, samples = _scan_word(text, word)
|
||||
if count:
|
||||
hits.append(Hit(word, category, "hard", count, samples))
|
||||
|
||||
for rule, _note, pattern in PATTERNS:
|
||||
rx = re.compile(pattern)
|
||||
found = list(rx.finditer(text))
|
||||
if found:
|
||||
samples = [_context(text, m.start(), m.end()) for m in found[:SAMPLES_PER_RULE]]
|
||||
hits.append(Hit(rule, "句式" if not rule.startswith(("破折号", "省略号", "英文引号")) else "标点",
|
||||
"hard", len(found), samples))
|
||||
|
||||
for category, words in SOFT_WORDS.items():
|
||||
for word in words:
|
||||
count, samples = _scan_word(text, word)
|
||||
if count:
|
||||
hits.append(Hit(word, category, "soft", count, samples))
|
||||
|
||||
hits.sort(key=lambda h: (h.level != "hard", -h.count, h.rule))
|
||||
return hits
|
||||
|
||||
|
||||
def hard_hits(hits: list[Hit]) -> list[Hit]:
|
||||
return [h for h in hits if h.level == "hard"]
|
||||
|
||||
|
||||
def per_k(text: str) -> float:
|
||||
"""每千字硬伤命中数。空文本返回 0,避免除零。"""
|
||||
length = cjk_length(text) or 1
|
||||
return sum(h.count for h in hard_hits(check(text))) * 1000 / length
|
||||
|
||||
|
||||
def report(text: str, hits: list[Hit] | None = None, top: int = 14) -> str:
|
||||
hits = check(text) if hits is None else hits
|
||||
hard = hard_hits(hits)
|
||||
soft = [h for h in hits if h.level == "soft"]
|
||||
total_hard = sum(h.count for h in hard)
|
||||
total_soft = sum(h.count for h in soft)
|
||||
length = cjk_length(text)
|
||||
density = total_hard * 1000 / (length or 1)
|
||||
|
||||
lines = [
|
||||
f"AI 味诊断:{length} 字|硬伤 {total_hard} 处({density:.1f}/千字)|密度词 {total_soft} 处",
|
||||
]
|
||||
if hard:
|
||||
lines.append("")
|
||||
lines.append("硬伤(命中即改):")
|
||||
lines += [" " + h.line() for h in hard[:top]]
|
||||
if soft:
|
||||
lines.append("")
|
||||
lines.append("密度词(成串才处理):")
|
||||
lines += [" " + h.line() for h in soft[:top]]
|
||||
if not hard and not soft:
|
||||
lines.append(" 没有命中:用词干净。")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def digest(text: str, top: int = 10) -> str:
|
||||
"""给扩写模型看的简短诊断:要求它逐条消灭这些。"""
|
||||
hits = [h for h in check(text) if h.level == "hard"][:top]
|
||||
if not hits:
|
||||
return "(机械检查没有发现 AI 味硬伤;重点放在把概述换成场景。)"
|
||||
return "\n".join(f"- {h.category}『{h.rule}』×{h.count}|例:{h.samples[0] if h.samples else ''}"
|
||||
for h in hits)
|
||||
@@ -0,0 +1,397 @@
|
||||
"""文笔层:把已过闸门的编年稿大量扩写成有现场的长文。
|
||||
|
||||
与 volume.write_episode 的分工:
|
||||
- volume 从史料素材写出「骨架正确」的一集(事实优先,1500 字内);
|
||||
- expand 拿那一稿逐事件展开(场景优先,目标 8000 字)。
|
||||
|
||||
**为什么要分块**:一次调用来写 8000 字,模型会跑偏、提前收尾、漏掉后半段事件。
|
||||
所以按「年份场景」把原稿切块,每块目标约 2600 字,逐块展开后用上一块结尾承接,
|
||||
再把各块拼接。每块都只喂它自己那一段史料,避免不同块写出重复内容。
|
||||
|
||||
这是 A/B 层:产物写成 `<key>.literary.md`,永不覆盖原稿。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import re
|
||||
from collections import Counter
|
||||
from dataclasses import replace
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import deslop, volume
|
||||
from dfannals import episodes as ep_mod
|
||||
from dfannals.chronicle import cjk_length
|
||||
from dfannals.config import (
|
||||
EXPAND_CEILING,
|
||||
EXPAND_CHUNK_TARGET,
|
||||
EXPAND_FLOOR,
|
||||
EXPAND_TARGET_CHARS,
|
||||
LITERARY_MODEL,
|
||||
MAX_TOKENS_EXPAND,
|
||||
PROMPT_DIR,
|
||||
)
|
||||
from dfannals.factcheck import Suspicion
|
||||
from dfannals.legends import World
|
||||
from dfannals.llm import chat
|
||||
from dfannals.normalize import Normalized, normalize
|
||||
from dfannals.threads import Thread
|
||||
|
||||
SPEC_FILE = PROMPT_DIR / "literary-expander.md"
|
||||
TAIL_CHARS = 300 # 上一块结尾取多少字做承接
|
||||
YEAR_RE = re.compile(r"^\s*\d{3,4}\s*年")
|
||||
ANY_YEAR_RE = re.compile(r"(\d{3,4})\s*年")
|
||||
|
||||
|
||||
def scene_groups(draft: str) -> list[list[str]]:
|
||||
"""按「年份开头的段落」切场景组;紧随其后的非年份段落归入同一组。"""
|
||||
groups: list[list[str]] = []
|
||||
for para in (p for p in draft.split("\n") if p.strip()):
|
||||
if YEAR_RE.match(para) or not groups:
|
||||
groups.append([para])
|
||||
else:
|
||||
groups[-1].append(para)
|
||||
return groups
|
||||
|
||||
|
||||
def plan_chunks(draft: str, per_chunk_chars: int) -> list[str]:
|
||||
"""把原稿装进若干块,尽量在场景边界切割。"""
|
||||
groups = scene_groups(draft)
|
||||
chunks: list[list[str]] = []
|
||||
current: list[str] = []
|
||||
size = 0
|
||||
for group in groups:
|
||||
glen = sum(cjk_length(line) for line in group)
|
||||
if current and size + glen > per_chunk_chars:
|
||||
chunks.append(current)
|
||||
current, size = [], 0
|
||||
current += group
|
||||
size += glen
|
||||
if current:
|
||||
chunks.append(current)
|
||||
return ["\n".join(c).strip() for c in chunks if "".join(c).strip()]
|
||||
|
||||
|
||||
def _chunk_episode(episode: ep_mod.Episode, chunk_text: str) -> ep_mod.Episode:
|
||||
"""把这一块覆盖的史料事件挑出来,作为该块的素材。"""
|
||||
years = [int(m.group(1)) for m in ANY_YEAR_RE.finditer(chunk_text)]
|
||||
if not years:
|
||||
return episode
|
||||
low, high = min(years), max(years)
|
||||
events = [e for e in episode.events if low <= e.year <= high] or episode.events
|
||||
return replace(episode, events=events, start_year=low, end_year=high)
|
||||
|
||||
|
||||
def _title_form(value: str) -> str:
|
||||
"""YETI → Yeti、EAGLE_MAN → Eagle_man:正文里模型就是这么写的。"""
|
||||
return "_".join(part[:1].upper() + part[1:].lower() for part in value.split("_"))
|
||||
|
||||
|
||||
def allowed_names(world: World, thread: Thread, episode: ep_mod.Episode) -> list[str]:
|
||||
"""本集史料里真实出现过的专名。
|
||||
|
||||
这份名单既发给模型(事前防幻觉),也用于修复可疑专名(事后)。
|
||||
实测:不给名单时模型造出了 "Farewell" 这个史料里不存在的地名。
|
||||
"""
|
||||
names: set[str] = set()
|
||||
heros, others = ep_mod.cast_of(world, thread, episode)
|
||||
for fid in [*heros, *others, *thread.members]:
|
||||
name = world.figure_name(fid)
|
||||
if name:
|
||||
names.add(name)
|
||||
figure = world.figures.get(fid)
|
||||
if figure and figure.race:
|
||||
names.add(figure.race)
|
||||
names.add(_title_form(figure.race))
|
||||
for event in episode.events:
|
||||
for ref in world.render_refs(event):
|
||||
value = ref.partition("=")[2].strip()
|
||||
if value:
|
||||
names.add(value)
|
||||
return sorted(n for n in names if n)
|
||||
|
||||
|
||||
def chunk_messages(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
total_episodes: int,
|
||||
chunk_text: str,
|
||||
chunk_target: int,
|
||||
index: int,
|
||||
count: int,
|
||||
prev_tail: str = "",
|
||||
elided: Counter | None = None,
|
||||
) -> list[dict]:
|
||||
spec = SPEC_FILE.read_text(encoding="utf-8")
|
||||
material = volume.render_material(world, thread, episode, total_episodes, "", elided)
|
||||
user = [
|
||||
material,
|
||||
"",
|
||||
f"### 这一块原稿(第 {index}/{count} 块,只写这一块的年代范围)",
|
||||
"",
|
||||
chunk_text,
|
||||
"",
|
||||
"### 这一块的 AI 味诊断(这些必须改掉)",
|
||||
"",
|
||||
deslop.digest(chunk_text),
|
||||
"",
|
||||
]
|
||||
if prev_tail:
|
||||
user += ["### 上一块写到哪儿(用于承接状态,不要复述)", "", prev_tail, ""]
|
||||
names = allowed_names(world, thread, episode)
|
||||
if names:
|
||||
user += [
|
||||
"### 允许使用的专名(史料里真实存在,逗号分隔;除此之外的人名地名一律不许写)",
|
||||
"",
|
||||
"、".join(names),
|
||||
"",
|
||||
"(上面提到的种族名可以当普通名词用,不要写成新角色。)",
|
||||
"",
|
||||
]
|
||||
user += [
|
||||
(
|
||||
f"### 要求:把这一块逐事件展开到 {chunk_target} 字左右,只输出正文。"
|
||||
"不要写标题,不要重复上一块已经写过的内容,也不要跳到下一块的年代。"
|
||||
),
|
||||
]
|
||||
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
|
||||
|
||||
|
||||
def _fit_length(
|
||||
messages: list[dict],
|
||||
text: str,
|
||||
*,
|
||||
floor: int,
|
||||
ceiling: int,
|
||||
target: int,
|
||||
gateway=None,
|
||||
model: str = LITERARY_MODEL,
|
||||
max_repairs: int = 1,
|
||||
) -> str:
|
||||
"""把长度拉进区间:太短继续展开,太长压缩(都不许动史实)。"""
|
||||
for _ in range(max_repairs):
|
||||
length = cjk_length(text)
|
||||
if floor <= length <= ceiling:
|
||||
break
|
||||
ask = (
|
||||
f"只有 {length} 字,还差很多。继续展开:把这一段里还没落到现场的事件逐件写成场景,"
|
||||
f"补对白、动作、视角人物此刻的感知与算计,写到 {target} 字左右。"
|
||||
"不许新增史料里没有的事件或专名,也不要用形容词灌水。"
|
||||
if length < floor
|
||||
else f"有 {length} 字,超了。删掉重复交代和没有功能的描写,压到 {ceiling} 字以内,"
|
||||
"冲突与转折的现场感要保留。"
|
||||
)
|
||||
text = chat(messages + [{"role": "assistant", "content": text},
|
||||
{"role": "user", "content": ask}],
|
||||
gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
|
||||
return text.strip()
|
||||
|
||||
|
||||
def build_messages(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
total_episodes: int,
|
||||
draft: str,
|
||||
prev_tail: str = "",
|
||||
elided: Counter | None = None,
|
||||
target: int = EXPAND_TARGET_CHARS,
|
||||
) -> list[dict]:
|
||||
"""整集一次写完时用(目标不长时):素材 + 原稿 + 诊断 + 字数目标。"""
|
||||
spec = SPEC_FILE.read_text(encoding="utf-8")
|
||||
material = volume.render_material(world, thread, episode, total_episodes, prev_tail, elided)
|
||||
user = [
|
||||
material,
|
||||
"",
|
||||
"### 原稿(事实与事件顺序按它,写法不要学它)",
|
||||
"",
|
||||
draft.strip(),
|
||||
"",
|
||||
"### 原稿的 AI 味诊断(这些必须改掉)",
|
||||
"",
|
||||
deslop.digest(draft),
|
||||
"",
|
||||
(
|
||||
f"### 现在开始:按原稿的事件顺序逐件展开,写到 {target} 字左右"
|
||||
f"({EXPAND_FLOOR}–{EXPAND_CEILING} 字以内),只输出正文。"
|
||||
),
|
||||
]
|
||||
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
|
||||
|
||||
|
||||
def expand_episode(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
total_episodes: int,
|
||||
draft: str,
|
||||
prev_tail: str = "",
|
||||
*,
|
||||
gateway=None,
|
||||
model: str = LITERARY_MODEL,
|
||||
target: int = EXPAND_TARGET_CHARS,
|
||||
elided: Counter | None = None,
|
||||
max_repairs: int = 1,
|
||||
) -> Normalized:
|
||||
"""大量扩写:分块展开 → 拼接 → 规范化标点。
|
||||
|
||||
返回 Normalized,notes 里带上每块的实际字数,便于发现哪一块没写够。
|
||||
"""
|
||||
# 标题不进正文块:否则模型会把 "# 守门人" 当正文写进去。最后再拼回开头。
|
||||
lines = draft.split("\n")
|
||||
headings = [ln for ln in lines if ln.strip().startswith("#")]
|
||||
body = "\n".join(ln for ln in lines if not ln.strip().startswith("#")).strip()
|
||||
|
||||
length = cjk_length(body) or 1
|
||||
blocks = max(1, math.ceil(target / EXPAND_CHUNK_TARGET))
|
||||
chunks = plan_chunks(body, max(1, round(length / blocks)))
|
||||
chunk_target = max(700, target // len(chunks))
|
||||
|
||||
parts: list[str] = []
|
||||
notes: list[str] = []
|
||||
for index, chunk in enumerate(chunks, 1):
|
||||
prev_tail_text = parts[-1][-TAIL_CHARS:] if parts else prev_tail[-TAIL_CHARS:]
|
||||
sub_episode = _chunk_episode(episode, chunk)
|
||||
messages = chunk_messages(world, thread, sub_episode, total_episodes, chunk,
|
||||
chunk_target, index, len(chunks), prev_tail_text, elided)
|
||||
text = chat(messages, gateway=gateway, model=model,
|
||||
max_tokens=MAX_TOKENS_EXPAND).content
|
||||
text = _fit_length(messages, text,
|
||||
floor=int(chunk_target * 0.75), ceiling=int(chunk_target * 1.35),
|
||||
target=chunk_target, gateway=gateway, model=model,
|
||||
max_repairs=max_repairs)
|
||||
parts.append(text)
|
||||
notes.append(f"第 {index}/{len(chunks)} 块:{cjk_length(text)} 字(目标 {chunk_target})")
|
||||
|
||||
stitched = "\n\n".join(parts)
|
||||
total_now = cjk_length(stitched)
|
||||
|
||||
# 拼接后总量仍不足,再整篇催一次(只此一次,避免无限加长)
|
||||
if total_now < EXPAND_FLOOR:
|
||||
messages = build_messages(world, thread, episode, total_episodes, body, prev_tail,
|
||||
elided, target)
|
||||
stitched = _fit_length(messages, stitched, floor=EXPAND_FLOOR, ceiling=EXPAND_CEILING,
|
||||
target=target, gateway=gateway, model=model, max_repairs=1)
|
||||
|
||||
result = normalize(stitched.strip())
|
||||
result.notes = notes + result.notes
|
||||
if headings:
|
||||
result.text = "\n\n".join([*headings, result.text])
|
||||
return result
|
||||
|
||||
|
||||
def literary_path(directory: Path, episode: ep_mod.Episode) -> Path:
|
||||
"""扩写稿的落盘位置:与原稿并列,便于 A/B 对照。"""
|
||||
return directory / f"{episode.key}.literary.md"
|
||||
|
||||
|
||||
NAME_FIX_SPEC = """你在做一件很窄的事:正文里出现了史料中不存在的专名(模型凭空造的人名、地名、称号)。
|
||||
|
||||
把包含这些专名的句子改掉:
|
||||
- 优先换成「允许使用的专名」里真实存在、语义接近的名字;
|
||||
- 没有合适的名字,就把句子改成不指名(例如把"把 Farewell 挡回去"改成"把下一个要拦的人挡回去")。
|
||||
|
||||
不许增删情节,不许改动没被点到的句子,不许把段落合并或拆分。
|
||||
只输出改好的段落,每段以 [序号] 开头,序号与段数必须和输入一致。不写任何解释。
|
||||
"""
|
||||
|
||||
|
||||
_NUMBERED_RE = re.compile(r"^\s*\[(\d+)\]\s*(.+?)\s*$")
|
||||
|
||||
|
||||
def _parse_numbered(raw: str) -> dict[int, str]:
|
||||
"""把模型返回的 `[n] 段落` 解析成 {序号: 段落},容忍跨行。"""
|
||||
out: dict[int, str] = {}
|
||||
current: int | None = None
|
||||
buffer: list[str] = []
|
||||
for line in raw.split("\n"):
|
||||
match = _NUMBERED_RE.match(line)
|
||||
if match:
|
||||
if current is not None:
|
||||
out[current] = "\n".join(buffer).strip()
|
||||
current = int(match.group(1))
|
||||
buffer = [match.group(2)]
|
||||
elif current is not None and line.strip():
|
||||
buffer.append(line.strip())
|
||||
if current is not None:
|
||||
out[current] = "\n".join(buffer).strip()
|
||||
return {k: v for k, v in out.items() if v}
|
||||
|
||||
|
||||
def repair_names(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
text: str,
|
||||
suspects: list[Suspicion],
|
||||
*,
|
||||
gateway=None,
|
||||
model: str = LITERARY_MODEL,
|
||||
) -> Normalized:
|
||||
"""只重写含可疑专名的那几段,其他段落一字不动。
|
||||
|
||||
整篇重写代价高(几千字)且模型会顺手改别的地方;段级手术安全得多——
|
||||
段数对不上就不写回,宁可不改也不把全文换掉。
|
||||
"""
|
||||
bad = {s.name for s in suspects}
|
||||
paragraphs = text.split("\n")
|
||||
targets = [i for i, para in enumerate(paragraphs) if any(name in para for name in bad)]
|
||||
if not targets:
|
||||
return normalize(text)
|
||||
|
||||
numbered = "\n".join(f"[{n}] {paragraphs[i]}" for n, i in enumerate(targets, 1))
|
||||
messages = [
|
||||
{"role": "system", "content": NAME_FIX_SPEC},
|
||||
{"role": "user", "content": "\n".join([
|
||||
"### 允许使用的专名", "", "、".join(allowed_names(world, thread, episode)), "",
|
||||
"### 史料里找不到的专名", "",
|
||||
*[f"- {s.name}(出现 {s.count} 次)|例:…{s.context}…" for s in suspects], "",
|
||||
f"### 需要修的段落(共 {len(targets)} 段)", "", numbered, "",
|
||||
f"### 要求:只输出改好的 {len(targets)} 段,每段以 [序号] 开头。",
|
||||
])},
|
||||
]
|
||||
reply = chat(messages, gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
|
||||
fixed = _parse_numbered(reply)
|
||||
|
||||
applied = 0
|
||||
for n, index in enumerate(targets, 1):
|
||||
if n in fixed:
|
||||
paragraphs[index] = fixed[n]
|
||||
applied += 1
|
||||
|
||||
result = normalize("\n".join(paragraphs))
|
||||
result.notes = [f"可疑专名修复:改写 {applied}/{len(targets)} 段"] + result.notes
|
||||
return result
|
||||
|
||||
|
||||
def _hits(text: str) -> tuple[int, float, int]:
|
||||
"""返回(硬伤数、每千字硬伤密度、字数)。"""
|
||||
hard = [h for h in deslop.check(text) if h.level == "hard"]
|
||||
count = sum(h.count for h in hard)
|
||||
length = cjk_length(text)
|
||||
return count, count * 1000 / (length or 1), length
|
||||
|
||||
|
||||
def compare(draft: str, literary: str) -> str:
|
||||
"""A/B 对照:字数与 AI 味硬伤密度。"""
|
||||
rows = ["A/B 对照:"]
|
||||
for label, text in (("原稿", draft), ("扩写稿", literary)):
|
||||
count, density, length = _hits(text)
|
||||
rows.append(f" {label:<8}{length:>6} 字|硬伤 {count:>4} 处({density:.1f}/千字)")
|
||||
before = _hits(draft)[2]
|
||||
if before:
|
||||
rows.append(f" 篇幅倍数:{_hits(literary)[2] / before:.1f}×")
|
||||
return "\n".join(rows)
|
||||
|
||||
|
||||
def full_report(draft: str, literary: str) -> str:
|
||||
"""给 CLI 用:A/B 对照 + 两稿各自的诊断。"""
|
||||
return "\n".join([
|
||||
compare(draft, literary),
|
||||
"",
|
||||
"— 原稿诊断 —",
|
||||
deslop.report(draft),
|
||||
"",
|
||||
"— 扩写稿诊断 —",
|
||||
deslop.report(literary),
|
||||
])
|
||||
@@ -97,14 +97,19 @@ def _is_candidate(name: str, known: set[str]) -> bool:
|
||||
def check(text: str, world: World) -> list[Suspicion]:
|
||||
"""返回可疑专名列表(按出现次数降序)。"""
|
||||
known = known_names(world)
|
||||
# 大小写不该决定是不是幻觉:史料里是 YETI/EAGLE_MAN,正文写成 Yeti/Eagle_man
|
||||
# 是同一件事(模型从种族字段学来的),不能报成凭空编造。
|
||||
folded = {k.casefold() for k in known}
|
||||
found: dict[str, list[str]] = {}
|
||||
|
||||
for match in NAME_RE.finditer(text):
|
||||
raw = match.group(0)
|
||||
if raw.casefold() in folded:
|
||||
continue
|
||||
# 逐级回退:整串不认,就试着拆成更短的已知名字,减少误报
|
||||
if not _is_candidate(raw, known):
|
||||
continue
|
||||
if any(part in known for part in raw.split()):
|
||||
if any(part.casefold() in folded for part in raw.split()):
|
||||
# 名字里有一部分是史料已知的(例如 "Urist the Bold"),不当作凭空编造
|
||||
continue
|
||||
start = max(0, match.start() - 20)
|
||||
|
||||
@@ -45,12 +45,23 @@ class Figure:
|
||||
name: str = ""
|
||||
race: str = ""
|
||||
sex: int | None = None
|
||||
caste: str = "" # 史料里的性别字段(v53 导出用 <caste>MALE/FEMALE</caste>)
|
||||
birth_year: int | None = None
|
||||
death_year: int | None = None
|
||||
profession: str = ""
|
||||
entities: list[int] = field(default_factory=list)
|
||||
sites: list[int] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def gender(self) -> str:
|
||||
"""史料记载的性别,用于避免正文里默认称呼"他"。"""
|
||||
c = (self.caste or "").upper()
|
||||
if c == "FEMALE":
|
||||
return "女"
|
||||
if c == "MALE":
|
||||
return "男"
|
||||
return ""
|
||||
|
||||
@property
|
||||
def alive_span(self) -> str:
|
||||
# DF 用负数占位表示“无此项”(未记录/未死),不要当成真实年份显示
|
||||
@@ -333,6 +344,7 @@ def _add_figure(world: World, elem) -> None:
|
||||
fig = world.figures.setdefault(fid, Figure(id=fid))
|
||||
fig.name = fig.name or _first(d, "name")
|
||||
fig.race = fig.race or _first(d, "race")
|
||||
fig.caste = fig.caste or _first(d, "caste")
|
||||
fig.profession = fig.profession or _first(d, "profession")
|
||||
if fig.sex is None:
|
||||
fig.sex = _int_or_none(_first(d, "sex"))
|
||||
|
||||
+6
-3
@@ -48,7 +48,7 @@ def chat(
|
||||
max_tokens: int = MAX_TOKENS_CHAPTER,
|
||||
temperature: float = 1.0,
|
||||
timeout: int = 900,
|
||||
retries: int = 3,
|
||||
retries: int = 5,
|
||||
) -> Reply:
|
||||
gw = gateway or load_gateway()
|
||||
conn_cls, host, path = _endpoint(gw.base_url)
|
||||
@@ -75,7 +75,8 @@ def chat(
|
||||
raw = resp.read()
|
||||
if resp.status != 200:
|
||||
last = f"HTTP {resp.status}: {raw[:400].decode('utf-8', 'replace')}"
|
||||
if resp.status in (400, 401, 403, 404):
|
||||
# 4xx(除 429 限流)是请求本身的问题,重试无意义
|
||||
if resp.status < 500 and resp.status != 429:
|
||||
break
|
||||
else:
|
||||
data = json.loads(raw)
|
||||
@@ -101,6 +102,8 @@ def chat(
|
||||
finally:
|
||||
conn.close()
|
||||
if attempt < retries:
|
||||
time.sleep(3 * attempt)
|
||||
# 实测网关偶发 502 upstream_unavailable(尤其在大请求上),
|
||||
# 所以 5xx/429/超时都退避重试,上限 60 秒。
|
||||
time.sleep(min(60, 5 * 2 ** (attempt - 1)))
|
||||
|
||||
raise GatewayError(f"网关调用失败({gw.base_url},model={model}):{last}")
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
"""标点规范化:把模型爱用的半角标点改回中文排版。
|
||||
|
||||
模型(尤其是 `"`)经常直接吐 ASCII 引号。这既不符合中文排版,
|
||||
也会让「到底有没有对白」这种统计失真——我第一次统计扩写稿对白时就被骗了
|
||||
(算出 0 处对白,实际有 32 组,只是引号是英文的)。
|
||||
|
||||
思路借自 oh-story-claudecode 的 normalize-punctuation.js(MIT),只保留本书需要的两条:
|
||||
引号配对、CJK 之间的半角标点转全角。**破折号与省略号不转换**——本书正文禁用它们,
|
||||
交给诊断器报警比偷偷替换更好。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
_CJK = r"[\u4e00-\u9fff]"
|
||||
_HALF2FULL = {",": ",", ".": "。", "!": "!", "?": "?", ":": ":", ";": ";"}
|
||||
_CJK_PUNCT = re.compile(f"({_CJK})([,.\u0021?:;])({_CJK})")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Normalized:
|
||||
text: str
|
||||
notes: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def changed(self) -> bool:
|
||||
return bool(self.notes)
|
||||
|
||||
|
||||
def fix_quotes(text: str) -> tuple[str, int]:
|
||||
"""ASCII 直引号 → 中文弯引号“”。
|
||||
|
||||
按出现顺序交替开合。史书正文里不会出现嵌套引号(对白里再说引语时
|
||||
我们用「」),所以交替法够用;真出现奇数个也会被 deslop 诊断出来。
|
||||
"""
|
||||
if '"' not in text:
|
||||
return text, 0
|
||||
out: list[str] = []
|
||||
opening = True
|
||||
for ch in text:
|
||||
if ch == '"':
|
||||
out.append("“" if opening else "”")
|
||||
opening = not opening
|
||||
else:
|
||||
out.append(ch)
|
||||
return "".join(out), text.count('"')
|
||||
|
||||
|
||||
def fix_cjk_punct(text: str) -> tuple[str, int]:
|
||||
"""夹在两个汉字之间的半角标点 → 全角。不动英文专名里的标点。"""
|
||||
total = 0
|
||||
while True:
|
||||
text, n = _CJK_PUNCT.subn(
|
||||
lambda m: m.group(1) + _HALF2FULL[m.group(2)] + m.group(3), text
|
||||
)
|
||||
total += n
|
||||
if not n:
|
||||
return text, total
|
||||
|
||||
|
||||
def _drop_separator_lines(text: str) -> tuple[str, int]:
|
||||
"""删掉模型自作的分隔线(`---`、`***`)。原稿不用它们,空行已经够。"""
|
||||
kept, dropped = [], 0
|
||||
for line in text.split("\n"):
|
||||
stripped = line.strip()
|
||||
if stripped and set(stripped) <= {"-", "*", "=", "—", "·"}:
|
||||
dropped += 1
|
||||
continue
|
||||
kept.append(line)
|
||||
return "\n".join(kept), dropped
|
||||
|
||||
|
||||
_HEADING_RE = re.compile(r"^\s*#{1,6}\s*\S*\s*$")
|
||||
_NUMERAL_RE = re.compile(r"^\s*(?:[一二三四五六七八九十]{1,3}|\d{1,4})\s*$")
|
||||
|
||||
|
||||
def _drop_stray_markings(text: str) -> tuple[str, int]:
|
||||
"""删掉模型自作的小节标记:`## 一`、孤立的「七」。
|
||||
|
||||
分块扩写实测:模型每写一块就自制一个编号小节(## 一 … ## 六),
|
||||
它们不是故事内容,留在正文里就是脏标记。
|
||||
正文第一行标题(形如 `# 守门人`)要保留——那是原稿的标题。
|
||||
"""
|
||||
lines = text.split("\n")
|
||||
first_content = next((i for i, line in enumerate(lines) if line.strip()), None)
|
||||
kept, dropped = [], 0
|
||||
for index, line in enumerate(lines):
|
||||
if index != first_content and line.strip() and (
|
||||
_HEADING_RE.match(line) or _NUMERAL_RE.match(line)
|
||||
):
|
||||
dropped += 1
|
||||
continue
|
||||
kept.append(line)
|
||||
return "\n".join(kept), dropped
|
||||
|
||||
|
||||
def _collapse_duplicate_paragraphs(text: str) -> tuple[str, int]:
|
||||
"""相邻的完全重复段落只留一段(模型跨块写作时会重写上一块的结尾)。"""
|
||||
kept, dropped, prev = [], 0, None
|
||||
for line in text.split("\n"):
|
||||
stripped = line.strip()
|
||||
if stripped and stripped == prev:
|
||||
dropped += 1
|
||||
continue
|
||||
kept.append(line)
|
||||
if stripped:
|
||||
prev = stripped
|
||||
return "\n".join(kept), dropped
|
||||
|
||||
|
||||
def _collapse_duplicate_sentences(text: str) -> tuple[str, int]:
|
||||
"""段内连续重复的句子只留一句(模型爱把同一句复制两遍再往下写)。"""
|
||||
total = 0
|
||||
out: list[str] = []
|
||||
for para in text.split("\n"):
|
||||
if not para.strip():
|
||||
out.append(para)
|
||||
continue
|
||||
pieces = re.split(r"(?<=[。!?])", para)
|
||||
kept: list[str] = []
|
||||
prev = None
|
||||
for piece in pieces:
|
||||
key = piece.strip()
|
||||
if key and key == prev:
|
||||
total += 1
|
||||
continue
|
||||
kept.append(piece)
|
||||
if key:
|
||||
prev = key
|
||||
out.append("".join(kept))
|
||||
return "\n".join(out), total
|
||||
|
||||
|
||||
def _tidy_blank_lines(text: str) -> str:
|
||||
"""压缩多余空行、去掉行尾空白。"""
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
lines = [line.rstrip() for line in text.split("\n")]
|
||||
return "\n".join(lines).strip()
|
||||
|
||||
|
||||
def normalize(text: str) -> Normalized:
|
||||
"""返回规范化后的文本与改动说明(不改字词,只改标点与结构性冗余)。"""
|
||||
notes: list[str] = []
|
||||
|
||||
text, separators = _drop_separator_lines(text)
|
||||
if separators:
|
||||
notes.append(f"删除模型自作的分隔线 {separators} 行")
|
||||
text, markings = _drop_stray_markings(text)
|
||||
if markings:
|
||||
notes.append(f"删除模型自作的小节标记 {markings} 处")
|
||||
text, dup_paras = _collapse_duplicate_paragraphs(text)
|
||||
if dup_paras:
|
||||
notes.append(f"合并重复段落 {dup_paras} 处")
|
||||
text, dup_sents = _collapse_duplicate_sentences(text)
|
||||
if dup_sents:
|
||||
notes.append(f"合并段内重复句子 {dup_sents} 处")
|
||||
|
||||
text, quotes = fix_quotes(text)
|
||||
if quotes:
|
||||
notes.append(f"英文引号 {quotes} 个 → 中文引号")
|
||||
if quotes % 2:
|
||||
notes.append("警告:引号个数为奇数,可能有一处未闭合")
|
||||
|
||||
text, punct = fix_cjk_punct(text)
|
||||
if punct:
|
||||
notes.append(f"汉字间半角标点 {punct} 处 → 全角")
|
||||
|
||||
return Normalized(text=_tidy_blank_lines(text), notes=notes)
|
||||
+23
-1
@@ -65,8 +65,30 @@ def ensure_repo() -> None:
|
||||
_git("config", "user.email", "df-annals@hajim1.art")
|
||||
|
||||
|
||||
def _tracked_empty_sources() -> list[str]:
|
||||
"""已跟踪的源码/文档里,哪些是 0 字节。"""
|
||||
suffixes = (".py", ".sh", ".lua", ".md", ".toml", ".json")
|
||||
out: list[str] = []
|
||||
for rel in _git("ls-files").stdout.split():
|
||||
if not rel.endswith(suffixes):
|
||||
continue
|
||||
path = PROJECT_DIR / rel
|
||||
if path.is_file() and path.stat().st_size == 0:
|
||||
out.append(rel)
|
||||
return out
|
||||
|
||||
|
||||
def commit_and_push(message: str, paths: list[Path] | None = None) -> str:
|
||||
"""提交并推送。返回提交哈希;无改动时返回空串。"""
|
||||
"""提交并推送。返回提交哈希;无改动时返回空串。
|
||||
|
||||
提交前硬拦空文件:实测发生过一次 cli.py 被提交为 0 字节的事故——
|
||||
而 cli.py 是唯一入口,于是远端仓库里的管道完全不可运行,且失败是静默的。
|
||||
这种错误不能靠人复核,必须在提交口拦住。
|
||||
"""
|
||||
empty = _tracked_empty_sources()
|
||||
if empty:
|
||||
raise RuntimeError("拒绝提交:以下已跟踪文件是空的 → " + "、".join(empty))
|
||||
|
||||
if paths:
|
||||
for p in paths:
|
||||
_git("add", "--", str(p.relative_to(PROJECT_DIR)) if p.is_absolute() else str(p))
|
||||
|
||||
+5
-4
@@ -155,8 +155,8 @@ def render_material(
|
||||
for fid in heros:
|
||||
row = rows.get(fid)
|
||||
if row:
|
||||
lines.append(f"- {row.name}|{row.race}|{row.span}|身份:{row.identity or '不详'}"
|
||||
f"|本集出场 {row.appearances} 次")
|
||||
lines.append(f"- {row.name}|{row.race}|性别:{row.gender or '史料未载'}"
|
||||
f"|{row.span}|身份:{row.identity or '不详'}|本集出场 {row.appearances} 次")
|
||||
else:
|
||||
lines.append(f"- {world.figure_name(fid)}")
|
||||
|
||||
@@ -173,8 +173,8 @@ def render_material(
|
||||
for fid in others[:6]:
|
||||
row = rows.get(fid)
|
||||
if row:
|
||||
lines.append(f"- {row.name}|{row.race}|身份:{row.identity or '不详'}"
|
||||
f"|本集出场 {row.appearances} 次")
|
||||
lines.append(f"- {row.name}|{row.race}|性别:{row.gender or '史料未载'}"
|
||||
f"|身份:{row.identity or '不详'}|本集出场 {row.appearances} 次")
|
||||
else:
|
||||
lines.append(f"- {world.figure_name(fid)}")
|
||||
|
||||
@@ -196,6 +196,7 @@ def render_material(
|
||||
"",
|
||||
f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,专名保留英文。",
|
||||
"- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。",
|
||||
"- 性别按上面卡片写,不要默认“他”;史料未载性别时用名字或身份称呼。",
|
||||
"- 史料之外的世界大事不要写;只有影响到主角时才提一句。",
|
||||
]
|
||||
return "\n".join(lines)
|
||||
|
||||
Reference in New Issue
Block a user