文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)

- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
Chen Yi
2026-10-05 22:27:02 +08:00
parent 08a032120a
commit 8630dcad55
23 changed files with 1971 additions and 56 deletions
+9 -6
View File
@@ -54,14 +54,15 @@ class CastRow:
fid: int
name: str
race: str
gender: str # 男/女/空(史料未载)
span: str
role: str # 主角 / 对手 / 配角
appearances: int
identity: str
def as_row(self) -> str:
return (f"| {self.name} | {self.race or '—'} | {self.span} | {self.identity or '—'} | "
f"{self.appearances} | {self.role} |")
return (f"| {self.name} | {self.race or '—'} | {self.gender or '—'} | {self.span} | "
f"{self.identity or '—'} | {self.appearances} | {self.role} |")
def _identity_of(world: World, fid: int, events: list[Event]) -> str:
@@ -107,6 +108,7 @@ def collect_rows(world: World, thread: Thread, episodes: list[Episode], max_othe
fid=fid,
name=world.figure_name(fid),
race=fig.race if fig else "",
gender=fig.gender if fig else "",
span=fig.alive_span if fig else "生卒不详",
role=role,
appearances=appearances.get(fid, 0),
@@ -136,8 +138,9 @@ def _bio_prompt(world: World, rows: list[CastRow], events: list[Event]) -> str:
for row in rows[:BIO_LIMIT]:
facts = [
(
f"{row.name}({row.race or '种族不详'},{row.span},本卷出场 {row.appearances} 次,"
f"身份:{row.identity or '不详'},在故事里是{row.role})"
f"{row.name}({row.race or '种族不详'},{row.gender or '性别史料未载'},{row.span},"
f"本卷出场 {row.appearances} 次,身份:{row.identity or '不详'},"
f"在故事里是{row.role})"
)
]
for e in events:
@@ -197,8 +200,8 @@ def build_cast(
f"本卷主角线索:{'、'.join(world.figure_name(m) for m in thread.members)}",
f"线索跨度:{_span_of(thread, episodes)}",
"",
"| 人物 | 种族 | 生卒 | 身份 | 本卷出场 | 角色 |",
"|---|---|---|---|---|---|",
"| 人物 | 种族 | 性别 | 生卒 | 身份 | 本卷出场 | 角色 |",
"|---|---|---|---|---|---|---|",
]
lines.extend(row.as_row() for row in rows)
+392
View File
@@ -0,0 +1,392 @@
"""命令行入口。
人物主线流程:
scripts/dfannals status 查看当前进度
scripts/dfannals index 解析 exports → 年表 + 人物索引
scripts/dfannals threads 挖掘人物线索候选(含信号明细)
scripts/dfannals episodes --thread 1 查看某条线索切出的剧集
scripts/dfannals cast --thread 1 生成人物表(表格 + 小传)
scripts/dfannals episode --index 1 写出某一集(自评闸门 + 文笔层扩写)
scripts/dfannals expand --index 1 只跑文笔层:把已有的一集大量扩写(A/B 对照)
用 scripts/dfannals 启动,不要直接 python3 -m:会话 PATH 上的 python3 可能是
编辑器工具链的 venv,里面没有 defusedxml。
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from dfannals import (
cast,
chronicle,
deslop,
episodes,
factcheck,
legends,
publish,
threads,
timeline,
)
from dfannals import expand as expand_mod
from dfannals import volume as volume_mod
from dfannals.config import (
DATA_DIR,
EXPAND_TARGET_CHARS,
EXPORT_DIR,
GITEA_WEB,
LITERARY_MODEL,
chapter_dir,
ensure_dirs,
volume_dir,
)
def find_exports(export_dir: Path) -> list[Path]:
"""找出 legends 导出文件。
原生 legends.xml 与 DFHack 的 legends_plus.xml 都要;
预处理缓存放在子目录,所以这里不会重复收到同一份数据。
"""
if not export_dir.is_dir():
return []
files = sorted(export_dir.glob("*.xml"))
base = [f for f in files if "legends_plus" not in f.name]
plus = [f for f in files if "legends_plus" in f.name]
return base + plus
def load_world(export_dir: Path) -> legends.World:
paths = find_exports(export_dir)
if not paths:
raise SystemExit(
f"在 {export_dir} 里没找到 legends XML。\n"
"请先导出(游戏内 Export XML + DFHack 的 exportlegends),再复制进这个目录。"
)
return legends.load(paths)
def _current_volume(state: chronicle.State, world: legends.World) -> int:
"""世界换了就开新卷。"""
if state.world != world.name:
if state.world:
state.volume += 1
state.done = []
state.world = world.name
return state.volume
def _volume_for(state: chronicle.State, world: legends.World) -> int:
return state.volume if state.world == world.name else 1
def _load_candidates(args: argparse.Namespace) -> tuple[legends.World, list[threads.Thread]]:
ensure_dirs()
world = load_world(EXPORT_DIR)
found = threads.find_threads(
world, min_interactions=args.min_interactions, min_span=args.min_span
)
if not found:
print("没有候选线索")
raise SystemExit(1)
return world, found
def cmd_status(args: argparse.Namespace) -> int:
state = chronicle.State.load()
if not state.world:
print("尚未开始。先运行 index。")
return 0
vdir = volume_dir(state.volume)
print(f"世界:{state.world}")
print(f"卷号:第 {state.volume} 卷")
print(f"已完成集数:{len(state.done)}")
for key in state.done:
print(f" · {key}")
print(f"成品目录:{vdir}")
return 0
def cmd_index(args: argparse.Namespace) -> int:
ensure_dirs()
world = load_world(EXPORT_DIR)
state = chronicle.State.load()
volume = _current_volume(state, world)
vdir = volume_dir(volume)
timeline.write_outputs(world, vdir)
state.save()
print(legends.describe(world))
print()
print(f"年表与人物索引已写入 {vdir}")
print("下一步:threads 看候选线索,episodes 看剧集规划。")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷索引:{world.name}", [vdir / "timeline.md", vdir / "figures.md"]
)
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_threads(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
report = threads.render_report(world, found, per_thread=args.samples)
print(report)
path = threads.save_json(world, found, DATA_DIR / "threads.json")
print(f"共 {len(found)} 条候选;明细已写入 {path}")
return 0
def cmd_episodes(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
print(episodes.render_plan(world, thread, planned))
print()
print(episodes.budget_report(planned))
if args.samples:
print("\n=== 前两集事件样例 ===")
for ep in planned[:2]:
print(f" 第 {ep.index} 集({ep.span})")
for e in ep.events[: args.samples]:
print(f" · {e.year}年 {e.type} " + " · ".join(world.render_refs(e)))
return 0
def cmd_cast(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
if not 1 <= args.thread <= len(found):
print(f"候选只有 {len(found)} 条,--thread 超出范围")
return 1
thread = found[args.thread - 1]
_, planned = volume_mod.extended_plan(world, thread)
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=not args.no_bios)
if args.dry_run:
print(text[:2000])
return 0
state = chronicle.State.load()
volume = _volume_for(state, world)
path = cast.write_cast(volume_dir(volume) / "cast.md", text)
print(f"人物表已写入 {path}\n涵盖 {len(rows)} 人:{', '.join(r.name for r in rows)}")
print(factcheck.report(suspicions) if suspicions
else "专名校验:小传里的专名全部能在史料中找到。")
return 0
def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
"""把刚写好的骨架稿大量扩写成成稿,覆盖本集文件;骨架另存 `.skeleton.md`。
骨架仍然保留:它过了自评闸门、字数合规,是溯因和对照的依据。
"""
ext, planned = volume_mod.extended_plan(world, produced.selection.thread)
directory = chapter_dir(volume)
skeleton = produced.path.with_name(produced.path.stem + ".skeleton.md")
skeleton.write_text(produced.text.strip() + "\n", encoding="utf-8")
print(f"--- 文笔层扩写中({args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, produced.selection.thread, produced.episode, len(planned), produced.text,
prev_tail=volume_mod.prev_tail_of(directory, planned, produced.episode.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
suspects = factcheck.check(text, world)
if suspects:
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, produced.selection.thread, produced.episode,
text, suspects)
text = fixed.text
for note in fixed.notes:
print(" " + note)
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
remaining = factcheck.check(text, world)
print(expand_mod.compare(produced.text, text))
print(f"成稿专名校验:可疑 {len(remaining)} 处|骨架稿另存 {skeleton.name}")
if remaining:
print(factcheck.report(remaining, top=5))
return text
def cmd_episode(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
out_dir = None if args.dry_run else chapter_dir(volume)
produced = volume_mod.produce_episode(
world, found, episode_index=args.index, out_dir=out_dir
)
print("选择依据:" + produced.selection.reason)
print(f"骨架稿字数:{chronicle.cjk_length(produced.text)}")
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(produced.text[:1600])
return 0
if args.expand:
_expand_in_place(volume, world, produced, args)
state.mark_done(produced.episode.key)
state.save()
print(f"已写入 {produced.path}")
if not args.no_push:
publish.ensure_repo()
head = publish.commit_and_push(
f"第 {volume} 卷第 {produced.episode.index} 集(人物主线):"
f"{produced.episode.span}"
)
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
return 0
def cmd_expand(args: argparse.Namespace) -> int:
"""文笔层:拿已写好的那一集当骨架,大量扩写成场景化长文。
故意不覆盖原稿:产物写 `*.literary.md`,两稿并存才能比对。也不推送。
"""
world, found = _load_candidates(args)
state = chronicle.State.load()
volume = _volume_for(state, world)
selection = volume_mod.select_thread(world, found, rank=args.thread)
ext, planned = volume_mod.extended_plan(world, selection.thread)
if args.index > len(planned):
print(f"第 {args.thread} 条线索只有 {len(planned)} 集,没有第 {args.index} 集")
return 1
episode = planned[args.index - 1]
directory = chapter_dir(volume)
draft_path = directory / f"{episode.key}.md"
if not draft_path.is_file():
print(f"原稿不存在:{draft_path}")
print(f"先跑:scripts/dfannals episode --index {args.index}")
return 1
draft = draft_path.read_text(encoding="utf-8")
print(f"线索:{'、'.join(world.figure_name(m) for m in selection.thread.members)}")
print(f"本集:{episode.span}|原稿 {draft_path.name}({chronicle.cjk_length(draft)} 字)")
print()
print(deslop.report(draft))
print()
print(f"--- 文笔层扩写中(模型 {args.model},目标 {args.target} 字)---")
result = expand_mod.expand_episode(
world, selection.thread, episode, len(planned), draft,
prev_tail=volume_mod.prev_tail_of(directory, planned, args.index),
model=args.model, target=args.target, elided=ext.elided,
)
text = result.text
for note in result.notes:
print(" " + note)
print()
print(expand_mod.full_report(draft, text))
print()
before = factcheck.check(draft, world)
suspects = factcheck.check(text, world)
print(f"专名校验:原稿 {len(before)} 处|扩写稿 {len(suspects)} 处")
if suspects:
print(factcheck.report(suspects, top=5))
print("→ 自动修复可疑专名(只改含它们的段落)")
fixed = expand_mod.repair_names(world, selection.thread, episode, text, suspects)
for note in fixed.notes:
print(" " + note)
text = fixed.text
left = factcheck.check(text, world)
print(f" 复检:剩余可疑专名 {len(left)} 处")
if left:
print(factcheck.report(left, top=5))
if args.dry_run:
print("\n--- 试运行,不落盘 ---\n")
print(text)
return 0
path = expand_mod.literary_path(directory, episode)
path.write_text(text.strip() + "\n", encoding="utf-8")
print(f"\n已写入 {path}")
print("原稿未改动、未推送(A/B 对照用)")
return 0
def _add_pipeline_args(p: argparse.ArgumentParser) -> None:
p.add_argument("--min-interactions", type=int, default=threads.DEFAULT_MIN_INTERACTIONS,
help="强边阈值:一对人物至少反复互动多少次(默认 5)")
p.add_argument("--min-span", type=int, default=threads.DEFAULT_MIN_SPAN,
help="线索最小跨度年数(默认 30)")
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞人物主线连载管道")
sub = parser.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("status", help="查看当前进度")
p.set_defaults(func=cmd_status)
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
p.set_defaults(func=cmd_index)
p = sub.add_parser("threads", help="挖掘人物线索候选")
_add_pipeline_args(p)
p.add_argument("--top", type=int, default=0, help="只显示前 N 条(0 为全部)")
p.add_argument("--samples", type=int, default=3, help="每条线索展示几条事件样例")
p.set_defaults(func=cmd_threads)
p = sub.add_parser("episodes", help="按人物线索切出剧集")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--samples", type=int, default=0, help="额外打印每集前 N 条事件")
p.set_defaults(func=cmd_episodes)
p = sub.add_parser("cast", help="生成每卷人物表(表格 + 小传)")
_add_pipeline_args(p)
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--no-bios", action="store_true", help="只出表格,不让模型写小传")
p.add_argument("--dry-run", action="store_true", help="只打印不落盘")
p.set_defaults(func=cmd_cast)
p = sub.add_parser("episode", help="产出某一集(自评闸门 + 文笔层扩写)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="写该线索的第几集(默认 1)")
p.add_argument("--no-expand", dest="expand", action="store_false",
help="只出骨架稿,不做文笔层扩写")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS,
help=f"扩写目标字数(默认 {EXPAND_TARGET_CHARS})")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
p.add_argument("--no-push", action="store_true", help="落盘但不推送")
p.set_defaults(func=cmd_episode, expand=True)
p = sub.add_parser("expand", help="文笔层:把某一集大量扩写成场景化长文(不覆盖原稿)")
_add_pipeline_args(p)
p.add_argument("--index", type=int, default=1, help="扩写哪一集(默认 1)")
p.add_argument("--thread", type=int, default=1, help="用第几条候选线索(默认 1)")
p.add_argument("--target", type=int, default=EXPAND_TARGET_CHARS, help="目标字数")
p.add_argument("--model", default=LITERARY_MODEL, help="扩写模型")
p.add_argument("--dry-run", action="store_true", help="只生成不落盘")
p.set_defaults(func=cmd_expand)
args = parser.parse_args(argv)
ensure_dirs()
return args.func(args)
if __name__ == "__main__":
sys.exit(main())
+13
View File
@@ -44,6 +44,9 @@ PI_AUTH_JSON = HOME / ".pi/agent/auth.json"
PI_MODELS_JSON = HOME / ".pi/agent/models.json"
PROVIDER = "workbuddy"
MODEL = "cn:deepseek-v4-pro"
# 文笔层(扩写)单独选模型:创作向模型写叙事更贴地,
# 而推理向模型(deepseek-v4-pro)天然偏总结陈词,容易写成报告。
LITERARY_MODEL = "cn:minimax-m3"
# 该模型是推理模型:思考与正文共用 max_tokens,必须留足
MAX_TOKENS_CHAPTER = 16000
MAX_TOKENS_UTIL = 4000
@@ -66,6 +69,16 @@ WORLD_HISTORY_YEARS = 250
CHAPTER_MIN_CHARS = 800
CHAPTER_MAX_CHARS = 1500 # 严格按访谈定的 800–1500,不留“余量”
# ---------------------------------------------------------------- 文笔层(扩写)
# 用户口径:要的是「大量扩写」——原稿是骨架,文笔层把它逐事件展开成现场。
EXPAND_TARGET_CHARS = 8000 # 目标字数(用户选定:每集约 8000 字)
EXPAND_FLOOR = 6000 # 低于此判太短,继续展开
EXPAND_CEILING = 11000 # 高于此判太长,压缩
# 一次调用写 8000 字很容易跑偏、漏事件,所以按场景分块展开,每块目标这么多字
EXPAND_CHUNK_TARGET = 2600
# 长文 + 可能的思考 tokens,预算要给足
MAX_TOKENS_EXPAND = 32000
@dataclass
class Gateway:
+162
View File
@@ -0,0 +1,162 @@
"""AI 味机械诊断:把「文笔是否自然」变成可度量的指标。
规则来自 oh-story-claudecode 的 story-deslop 与 banned-words(MIT),做了两处适配:
1. **分层**。原文把「仿佛、一丝、坚定、冰冷」放同一张一级表,但「坚定」「冰冷」在
史书叙事里是正常词(矮人要塞的"冰冷"可能是字面温度),一律判罪会刷满假阳性。
因此分成硬伤词(几乎只在 AI 文本里成串出现)与密度词(真人也用,只看密度)。
2. **按密度而非布尔判定**。AI 味是密度问题:出现一次是风格,出现二十次是模板。
输出「每千字命中数」,供闸门与扩写模型参考。
语义类判断(比喻是否套话、是否把一件事掰成三段写)机械检查假阳性太高,
交给扩写模型和自评闸门,这里不猜。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from dfannals.chronicle import cjk_length
# 硬伤词:出现在正常中文里的概率很低,命中一处就该改
HARD_WORDS: dict[str, tuple[str, ...]] = {
"情态": ("仿佛", "犹如", "宛若", "如同", "一丝", "一抹", "些许", "几分", "隐约",
"毫无征兆", "几不可闻", "微不可察"),
"动作": ("深吸一口气", "不禁"),
"表情": ("眼中闪过", "嘴角勾起", "眉头微皱", "眉眼低垂", "瞳孔微缩", "瞳孔收缩",
"瞳孔一缩", "指节泛白", "眼神锐利", "目光锐利"),
"心理": ("心中一动", "心头一震", "心下了然", "心中暗道", "心底泛起", "不由得", "心中一凛"),
"判断": ("不容置疑", "不容置喙", "不易察觉", "显而易见", "毫无疑问", "不可否认", "前所未有"),
"过渡": ("不由自主", "情不自禁", "自然而然", "话锋一转", "闪烁着光芒"),
}
# 密度词:真人写作也用,只有成串出现才算模板腔
SOFT_WORDS: dict[str, tuple[str, ...]] = {
"形容": ("坚定", "狡黠", "深邃", "凛冽", "冰冷"),
"语境敏感": ("突然", "陡然", "骤然", "猛然", "好像", "似乎", "瞬间", "猛地", "死死地"),
"弱化副词": ("缓缓", "微微", "轻轻", "淡淡"),
}
# 句式与标点:每条是(规则名,说明,正则)
PATTERNS: tuple[tuple[str, str, str], ...] = (
("不是A而是B", "最毒的 AI 句式,直接写 B", r"不是[^,。;!?\n]{1,24}[,,]\s*(?:而)?是"),
("万能动宾", "「,带着……」万能状语", r"[,,]\s*带着[^,。;!?\n]{0,20}"),
("声音描写", "「声音不大,却带着……」", r"声音(?:不大|很轻|很平|平静)[^。\n]{0,14}(?:却|但)?带着"),
("认知直述", "告诉而非展示,改为动作或台词", r"(?:他|她|他们|她们)(?:这才|终于)?(?:明白|意识到|感到|知道)"),
("章末预告", "「他不知道的是……」空泛预告", r"(?:他|她)不知道的是"),
("收束腔", "「这一刻/才刚刚开始」式总结", r"这一刻[,,]|从这一刻开始|才刚刚开始|原来[,,]"),
("抽象命运", "命运+齿轮/棋局/獠牙", r"命运[^。\n]{0,8}(?:齿轮|棋局|獠牙|改写|安排)"),
("心理模板", "「心中一震/心头一凛」式心理惊吓", r"心(?:中|头|底)(?:一|猛|顿)(?:震|凛|颤)"),
("过渡模板", "「取而代之的是」", r"取而代之"),
("告诉而非展示", "「显得有些X」「散发着一股X气息」", r"显得(?:有些|十分|非常)?[\u4e00-\u9fff]{1,4}|散发(?:着|出)?[^。\n]{0,8}(?:气息|气场)"),
("公式化对话标签", "「好的,他说道」", r"[,,](?:他|她)(?:说道|开口道|沉声道|低声道)"),
("英文引号", "模型常直接吐 ASCII 直角引号,中文排版用“”", r'"'),
("破折号", "正文禁用,改用句号、逗号或动作断句", r"——|──|(?<!-)--(?!-)"),
("省略号", "正文禁用,用动作或短句表达停顿", r"……|\.\.\."),
)
SAMPLE_WIDTH = 14
SAMPLES_PER_RULE = 2
@dataclass
class Hit:
"""一条诊断:命中哪个规则、几次、原文长什么样。"""
rule: str
category: str
level: str # hard = 硬伤词+句式+标点;soft = 只看密度的词
count: int
samples: list[str] = field(default_factory=list)
def line(self) -> str:
ex = ";".join(self.samples)
return f"{self.category:<6} {self.rule:<8} ×{self.count:<3} {ex}"
def _context(text: str, start: int, end: int) -> str:
left = max(0, start - SAMPLE_WIDTH)
right = min(len(text), end + SAMPLE_WIDTH)
return text[left:right].replace("\n", " ").strip()
def _scan_word(text: str, word: str) -> tuple[int, list[str]]:
count, samples, pos = 0, [], text.find(word)
while pos != -1:
count += 1
if len(samples) < SAMPLES_PER_RULE:
samples.append(_context(text, pos, pos + len(word)))
pos = text.find(word, pos + len(word))
return count, samples
def check(text: str) -> list[Hit]:
"""扫一遍文本,返回全部诊断(硬伤在前,同类按命中数降序)。"""
hits: list[Hit] = []
for category, words in HARD_WORDS.items():
for word in words:
count, samples = _scan_word(text, word)
if count:
hits.append(Hit(word, category, "hard", count, samples))
for rule, _note, pattern in PATTERNS:
rx = re.compile(pattern)
found = list(rx.finditer(text))
if found:
samples = [_context(text, m.start(), m.end()) for m in found[:SAMPLES_PER_RULE]]
hits.append(Hit(rule, "句式" if not rule.startswith(("破折号", "省略号", "英文引号")) else "标点",
"hard", len(found), samples))
for category, words in SOFT_WORDS.items():
for word in words:
count, samples = _scan_word(text, word)
if count:
hits.append(Hit(word, category, "soft", count, samples))
hits.sort(key=lambda h: (h.level != "hard", -h.count, h.rule))
return hits
def hard_hits(hits: list[Hit]) -> list[Hit]:
return [h for h in hits if h.level == "hard"]
def per_k(text: str) -> float:
"""每千字硬伤命中数。空文本返回 0,避免除零。"""
length = cjk_length(text) or 1
return sum(h.count for h in hard_hits(check(text))) * 1000 / length
def report(text: str, hits: list[Hit] | None = None, top: int = 14) -> str:
hits = check(text) if hits is None else hits
hard = hard_hits(hits)
soft = [h for h in hits if h.level == "soft"]
total_hard = sum(h.count for h in hard)
total_soft = sum(h.count for h in soft)
length = cjk_length(text)
density = total_hard * 1000 / (length or 1)
lines = [
f"AI 味诊断:{length} 字|硬伤 {total_hard} 处({density:.1f}/千字)|密度词 {total_soft} 处",
]
if hard:
lines.append("")
lines.append("硬伤(命中即改):")
lines += [" " + h.line() for h in hard[:top]]
if soft:
lines.append("")
lines.append("密度词(成串才处理):")
lines += [" " + h.line() for h in soft[:top]]
if not hard and not soft:
lines.append(" 没有命中:用词干净。")
return "\n".join(lines)
def digest(text: str, top: int = 10) -> str:
"""给扩写模型看的简短诊断:要求它逐条消灭这些。"""
hits = [h for h in check(text) if h.level == "hard"][:top]
if not hits:
return "(机械检查没有发现 AI 味硬伤;重点放在把概述换成场景。)"
return "\n".join(f"- {h.category}『{h.rule}』×{h.count}|例:{h.samples[0] if h.samples else ''}"
for h in hits)
+397
View File
@@ -0,0 +1,397 @@
"""文笔层:把已过闸门的编年稿大量扩写成有现场的长文。
与 volume.write_episode 的分工:
- volume 从史料素材写出「骨架正确」的一集(事实优先,1500 字内);
- expand 拿那一稿逐事件展开(场景优先,目标 8000 字)。
**为什么要分块**:一次调用来写 8000 字,模型会跑偏、提前收尾、漏掉后半段事件。
所以按「年份场景」把原稿切块,每块目标约 2600 字,逐块展开后用上一块结尾承接,
再把各块拼接。每块都只喂它自己那一段史料,避免不同块写出重复内容。
这是 A/B 层:产物写成 `<key>.literary.md`,永不覆盖原稿。
"""
from __future__ import annotations
import math
import re
from collections import Counter
from dataclasses import replace
from pathlib import Path
from dfannals import deslop, volume
from dfannals import episodes as ep_mod
from dfannals.chronicle import cjk_length
from dfannals.config import (
EXPAND_CEILING,
EXPAND_CHUNK_TARGET,
EXPAND_FLOOR,
EXPAND_TARGET_CHARS,
LITERARY_MODEL,
MAX_TOKENS_EXPAND,
PROMPT_DIR,
)
from dfannals.factcheck import Suspicion
from dfannals.legends import World
from dfannals.llm import chat
from dfannals.normalize import Normalized, normalize
from dfannals.threads import Thread
SPEC_FILE = PROMPT_DIR / "literary-expander.md"
TAIL_CHARS = 300 # 上一块结尾取多少字做承接
YEAR_RE = re.compile(r"^\s*\d{3,4}\s*年")
ANY_YEAR_RE = re.compile(r"(\d{3,4})\s*年")
def scene_groups(draft: str) -> list[list[str]]:
"""按「年份开头的段落」切场景组;紧随其后的非年份段落归入同一组。"""
groups: list[list[str]] = []
for para in (p for p in draft.split("\n") if p.strip()):
if YEAR_RE.match(para) or not groups:
groups.append([para])
else:
groups[-1].append(para)
return groups
def plan_chunks(draft: str, per_chunk_chars: int) -> list[str]:
"""把原稿装进若干块,尽量在场景边界切割。"""
groups = scene_groups(draft)
chunks: list[list[str]] = []
current: list[str] = []
size = 0
for group in groups:
glen = sum(cjk_length(line) for line in group)
if current and size + glen > per_chunk_chars:
chunks.append(current)
current, size = [], 0
current += group
size += glen
if current:
chunks.append(current)
return ["\n".join(c).strip() for c in chunks if "".join(c).strip()]
def _chunk_episode(episode: ep_mod.Episode, chunk_text: str) -> ep_mod.Episode:
"""把这一块覆盖的史料事件挑出来,作为该块的素材。"""
years = [int(m.group(1)) for m in ANY_YEAR_RE.finditer(chunk_text)]
if not years:
return episode
low, high = min(years), max(years)
events = [e for e in episode.events if low <= e.year <= high] or episode.events
return replace(episode, events=events, start_year=low, end_year=high)
def _title_form(value: str) -> str:
"""YETI → Yeti、EAGLE_MAN → Eagle_man:正文里模型就是这么写的。"""
return "_".join(part[:1].upper() + part[1:].lower() for part in value.split("_"))
def allowed_names(world: World, thread: Thread, episode: ep_mod.Episode) -> list[str]:
"""本集史料里真实出现过的专名。
这份名单既发给模型(事前防幻觉),也用于修复可疑专名(事后)。
实测:不给名单时模型造出了 "Farewell" 这个史料里不存在的地名。
"""
names: set[str] = set()
heros, others = ep_mod.cast_of(world, thread, episode)
for fid in [*heros, *others, *thread.members]:
name = world.figure_name(fid)
if name:
names.add(name)
figure = world.figures.get(fid)
if figure and figure.race:
names.add(figure.race)
names.add(_title_form(figure.race))
for event in episode.events:
for ref in world.render_refs(event):
value = ref.partition("=")[2].strip()
if value:
names.add(value)
return sorted(n for n in names if n)
def chunk_messages(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
chunk_text: str,
chunk_target: int,
index: int,
count: int,
prev_tail: str = "",
elided: Counter | None = None,
) -> list[dict]:
spec = SPEC_FILE.read_text(encoding="utf-8")
material = volume.render_material(world, thread, episode, total_episodes, "", elided)
user = [
material,
"",
f"### 这一块原稿(第 {index}/{count} 块,只写这一块的年代范围)",
"",
chunk_text,
"",
"### 这一块的 AI 味诊断(这些必须改掉)",
"",
deslop.digest(chunk_text),
"",
]
if prev_tail:
user += ["### 上一块写到哪儿(用于承接状态,不要复述)", "", prev_tail, ""]
names = allowed_names(world, thread, episode)
if names:
user += [
"### 允许使用的专名(史料里真实存在,逗号分隔;除此之外的人名地名一律不许写)",
"",
"、".join(names),
"",
"(上面提到的种族名可以当普通名词用,不要写成新角色。)",
"",
]
user += [
(
f"### 要求:把这一块逐事件展开到 {chunk_target} 字左右,只输出正文。"
"不要写标题,不要重复上一块已经写过的内容,也不要跳到下一块的年代。"
),
]
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
def _fit_length(
messages: list[dict],
text: str,
*,
floor: int,
ceiling: int,
target: int,
gateway=None,
model: str = LITERARY_MODEL,
max_repairs: int = 1,
) -> str:
"""把长度拉进区间:太短继续展开,太长压缩(都不许动史实)。"""
for _ in range(max_repairs):
length = cjk_length(text)
if floor <= length <= ceiling:
break
ask = (
f"只有 {length} 字,还差很多。继续展开:把这一段里还没落到现场的事件逐件写成场景,"
f"补对白、动作、视角人物此刻的感知与算计,写到 {target} 字左右。"
"不许新增史料里没有的事件或专名,也不要用形容词灌水。"
if length < floor
else f"有 {length} 字,超了。删掉重复交代和没有功能的描写,压到 {ceiling} 字以内,"
"冲突与转折的现场感要保留。"
)
text = chat(messages + [{"role": "assistant", "content": text},
{"role": "user", "content": ask}],
gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
return text.strip()
def build_messages(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
draft: str,
prev_tail: str = "",
elided: Counter | None = None,
target: int = EXPAND_TARGET_CHARS,
) -> list[dict]:
"""整集一次写完时用(目标不长时):素材 + 原稿 + 诊断 + 字数目标。"""
spec = SPEC_FILE.read_text(encoding="utf-8")
material = volume.render_material(world, thread, episode, total_episodes, prev_tail, elided)
user = [
material,
"",
"### 原稿(事实与事件顺序按它,写法不要学它)",
"",
draft.strip(),
"",
"### 原稿的 AI 味诊断(这些必须改掉)",
"",
deslop.digest(draft),
"",
(
f"### 现在开始:按原稿的事件顺序逐件展开,写到 {target} 字左右"
f"({EXPAND_FLOOR}–{EXPAND_CEILING} 字以内),只输出正文。"
),
]
return [{"role": "system", "content": spec}, {"role": "user", "content": "\n".join(user)}]
def expand_episode(
world: World,
thread: Thread,
episode: ep_mod.Episode,
total_episodes: int,
draft: str,
prev_tail: str = "",
*,
gateway=None,
model: str = LITERARY_MODEL,
target: int = EXPAND_TARGET_CHARS,
elided: Counter | None = None,
max_repairs: int = 1,
) -> Normalized:
"""大量扩写:分块展开 → 拼接 → 规范化标点。
返回 Normalized,notes 里带上每块的实际字数,便于发现哪一块没写够。
"""
# 标题不进正文块:否则模型会把 "# 守门人" 当正文写进去。最后再拼回开头。
lines = draft.split("\n")
headings = [ln for ln in lines if ln.strip().startswith("#")]
body = "\n".join(ln for ln in lines if not ln.strip().startswith("#")).strip()
length = cjk_length(body) or 1
blocks = max(1, math.ceil(target / EXPAND_CHUNK_TARGET))
chunks = plan_chunks(body, max(1, round(length / blocks)))
chunk_target = max(700, target // len(chunks))
parts: list[str] = []
notes: list[str] = []
for index, chunk in enumerate(chunks, 1):
prev_tail_text = parts[-1][-TAIL_CHARS:] if parts else prev_tail[-TAIL_CHARS:]
sub_episode = _chunk_episode(episode, chunk)
messages = chunk_messages(world, thread, sub_episode, total_episodes, chunk,
chunk_target, index, len(chunks), prev_tail_text, elided)
text = chat(messages, gateway=gateway, model=model,
max_tokens=MAX_TOKENS_EXPAND).content
text = _fit_length(messages, text,
floor=int(chunk_target * 0.75), ceiling=int(chunk_target * 1.35),
target=chunk_target, gateway=gateway, model=model,
max_repairs=max_repairs)
parts.append(text)
notes.append(f"第 {index}/{len(chunks)} 块:{cjk_length(text)} 字(目标 {chunk_target})")
stitched = "\n\n".join(parts)
total_now = cjk_length(stitched)
# 拼接后总量仍不足,再整篇催一次(只此一次,避免无限加长)
if total_now < EXPAND_FLOOR:
messages = build_messages(world, thread, episode, total_episodes, body, prev_tail,
elided, target)
stitched = _fit_length(messages, stitched, floor=EXPAND_FLOOR, ceiling=EXPAND_CEILING,
target=target, gateway=gateway, model=model, max_repairs=1)
result = normalize(stitched.strip())
result.notes = notes + result.notes
if headings:
result.text = "\n\n".join([*headings, result.text])
return result
def literary_path(directory: Path, episode: ep_mod.Episode) -> Path:
"""扩写稿的落盘位置:与原稿并列,便于 A/B 对照。"""
return directory / f"{episode.key}.literary.md"
NAME_FIX_SPEC = """你在做一件很窄的事:正文里出现了史料中不存在的专名(模型凭空造的人名、地名、称号)。
把包含这些专名的句子改掉:
- 优先换成「允许使用的专名」里真实存在、语义接近的名字;
- 没有合适的名字,就把句子改成不指名(例如把"把 Farewell 挡回去"改成"把下一个要拦的人挡回去")。
不许增删情节,不许改动没被点到的句子,不许把段落合并或拆分。
只输出改好的段落,每段以 [序号] 开头,序号与段数必须和输入一致。不写任何解释。
"""
_NUMBERED_RE = re.compile(r"^\s*\[(\d+)\]\s*(.+?)\s*$")
def _parse_numbered(raw: str) -> dict[int, str]:
"""把模型返回的 `[n] 段落` 解析成 {序号: 段落},容忍跨行。"""
out: dict[int, str] = {}
current: int | None = None
buffer: list[str] = []
for line in raw.split("\n"):
match = _NUMBERED_RE.match(line)
if match:
if current is not None:
out[current] = "\n".join(buffer).strip()
current = int(match.group(1))
buffer = [match.group(2)]
elif current is not None and line.strip():
buffer.append(line.strip())
if current is not None:
out[current] = "\n".join(buffer).strip()
return {k: v for k, v in out.items() if v}
def repair_names(
world: World,
thread: Thread,
episode: ep_mod.Episode,
text: str,
suspects: list[Suspicion],
*,
gateway=None,
model: str = LITERARY_MODEL,
) -> Normalized:
"""只重写含可疑专名的那几段,其他段落一字不动。
整篇重写代价高(几千字)且模型会顺手改别的地方;段级手术安全得多——
段数对不上就不写回,宁可不改也不把全文换掉。
"""
bad = {s.name for s in suspects}
paragraphs = text.split("\n")
targets = [i for i, para in enumerate(paragraphs) if any(name in para for name in bad)]
if not targets:
return normalize(text)
numbered = "\n".join(f"[{n}] {paragraphs[i]}" for n, i in enumerate(targets, 1))
messages = [
{"role": "system", "content": NAME_FIX_SPEC},
{"role": "user", "content": "\n".join([
"### 允许使用的专名", "", "、".join(allowed_names(world, thread, episode)), "",
"### 史料里找不到的专名", "",
*[f"- {s.name}(出现 {s.count} 次)|例:…{s.context}…" for s in suspects], "",
f"### 需要修的段落(共 {len(targets)} 段)", "", numbered, "",
f"### 要求:只输出改好的 {len(targets)} 段,每段以 [序号] 开头。",
])},
]
reply = chat(messages, gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
fixed = _parse_numbered(reply)
applied = 0
for n, index in enumerate(targets, 1):
if n in fixed:
paragraphs[index] = fixed[n]
applied += 1
result = normalize("\n".join(paragraphs))
result.notes = [f"可疑专名修复:改写 {applied}/{len(targets)} 段"] + result.notes
return result
def _hits(text: str) -> tuple[int, float, int]:
"""返回(硬伤数、每千字硬伤密度、字数)。"""
hard = [h for h in deslop.check(text) if h.level == "hard"]
count = sum(h.count for h in hard)
length = cjk_length(text)
return count, count * 1000 / (length or 1), length
def compare(draft: str, literary: str) -> str:
"""A/B 对照:字数与 AI 味硬伤密度。"""
rows = ["A/B 对照:"]
for label, text in (("原稿", draft), ("扩写稿", literary)):
count, density, length = _hits(text)
rows.append(f" {label:<8}{length:>6} 字|硬伤 {count:>4} 处({density:.1f}/千字)")
before = _hits(draft)[2]
if before:
rows.append(f" 篇幅倍数:{_hits(literary)[2] / before:.1f}×")
return "\n".join(rows)
def full_report(draft: str, literary: str) -> str:
"""给 CLI 用:A/B 对照 + 两稿各自的诊断。"""
return "\n".join([
compare(draft, literary),
"",
"— 原稿诊断 —",
deslop.report(draft),
"",
"— 扩写稿诊断 —",
deslop.report(literary),
])
+6 -1
View File
@@ -97,14 +97,19 @@ def _is_candidate(name: str, known: set[str]) -> bool:
def check(text: str, world: World) -> list[Suspicion]:
"""返回可疑专名列表(按出现次数降序)。"""
known = known_names(world)
# 大小写不该决定是不是幻觉:史料里是 YETI/EAGLE_MAN,正文写成 Yeti/Eagle_man
# 是同一件事(模型从种族字段学来的),不能报成凭空编造。
folded = {k.casefold() for k in known}
found: dict[str, list[str]] = {}
for match in NAME_RE.finditer(text):
raw = match.group(0)
if raw.casefold() in folded:
continue
# 逐级回退:整串不认,就试着拆成更短的已知名字,减少误报
if not _is_candidate(raw, known):
continue
if any(part in known for part in raw.split()):
if any(part.casefold() in folded for part in raw.split()):
# 名字里有一部分是史料已知的(例如 "Urist the Bold"),不当作凭空编造
continue
start = max(0, match.start() - 20)
+12
View File
@@ -45,12 +45,23 @@ class Figure:
name: str = ""
race: str = ""
sex: int | None = None
caste: str = "" # 史料里的性别字段(v53 导出用 <caste>MALE/FEMALE</caste>)
birth_year: int | None = None
death_year: int | None = None
profession: str = ""
entities: list[int] = field(default_factory=list)
sites: list[int] = field(default_factory=list)
@property
def gender(self) -> str:
"""史料记载的性别,用于避免正文里默认称呼"他"。"""
c = (self.caste or "").upper()
if c == "FEMALE":
return "女"
if c == "MALE":
return "男"
return ""
@property
def alive_span(self) -> str:
# DF 用负数占位表示“无此项”(未记录/未死),不要当成真实年份显示
@@ -333,6 +344,7 @@ def _add_figure(world: World, elem) -> None:
fig = world.figures.setdefault(fid, Figure(id=fid))
fig.name = fig.name or _first(d, "name")
fig.race = fig.race or _first(d, "race")
fig.caste = fig.caste or _first(d, "caste")
fig.profession = fig.profession or _first(d, "profession")
if fig.sex is None:
fig.sex = _int_or_none(_first(d, "sex"))
+6 -3
View File
@@ -48,7 +48,7 @@ def chat(
max_tokens: int = MAX_TOKENS_CHAPTER,
temperature: float = 1.0,
timeout: int = 900,
retries: int = 3,
retries: int = 5,
) -> Reply:
gw = gateway or load_gateway()
conn_cls, host, path = _endpoint(gw.base_url)
@@ -75,7 +75,8 @@ def chat(
raw = resp.read()
if resp.status != 200:
last = f"HTTP {resp.status}: {raw[:400].decode('utf-8', 'replace')}"
if resp.status in (400, 401, 403, 404):
# 4xx(除 429 限流)是请求本身的问题,重试无意义
if resp.status < 500 and resp.status != 429:
break
else:
data = json.loads(raw)
@@ -101,6 +102,8 @@ def chat(
finally:
conn.close()
if attempt < retries:
time.sleep(3 * attempt)
# 实测网关偶发 502 upstream_unavailable(尤其在大请求上),
# 所以 5xx/429/超时都退避重试,上限 60 秒。
time.sleep(min(60, 5 * 2 ** (attempt - 1)))
raise GatewayError(f"网关调用失败({gw.base_url},model={model}):{last}")
+169
View File
@@ -0,0 +1,169 @@
"""标点规范化:把模型爱用的半角标点改回中文排版。
模型(尤其是 `"`)经常直接吐 ASCII 引号。这既不符合中文排版,
也会让「到底有没有对白」这种统计失真——我第一次统计扩写稿对白时就被骗了
(算出 0 处对白,实际有 32 组,只是引号是英文的)。
思路借自 oh-story-claudecode 的 normalize-punctuation.js(MIT),只保留本书需要的两条:
引号配对、CJK 之间的半角标点转全角。**破折号与省略号不转换**——本书正文禁用它们,
交给诊断器报警比偷偷替换更好。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
_CJK = r"[\u4e00-\u9fff]"
_HALF2FULL = {",": ",", ".": "。", "!": "!", "?": "?", ":": ":", ";": ";"}
_CJK_PUNCT = re.compile(f"({_CJK})([,.\u0021?:;])({_CJK})")
@dataclass
class Normalized:
text: str
notes: list[str] = field(default_factory=list)
@property
def changed(self) -> bool:
return bool(self.notes)
def fix_quotes(text: str) -> tuple[str, int]:
"""ASCII 直引号 → 中文弯引号“”。
按出现顺序交替开合。史书正文里不会出现嵌套引号(对白里再说引语时
我们用「」),所以交替法够用;真出现奇数个也会被 deslop 诊断出来。
"""
if '"' not in text:
return text, 0
out: list[str] = []
opening = True
for ch in text:
if ch == '"':
out.append("“" if opening else "”")
opening = not opening
else:
out.append(ch)
return "".join(out), text.count('"')
def fix_cjk_punct(text: str) -> tuple[str, int]:
"""夹在两个汉字之间的半角标点 → 全角。不动英文专名里的标点。"""
total = 0
while True:
text, n = _CJK_PUNCT.subn(
lambda m: m.group(1) + _HALF2FULL[m.group(2)] + m.group(3), text
)
total += n
if not n:
return text, total
def _drop_separator_lines(text: str) -> tuple[str, int]:
"""删掉模型自作的分隔线(`---`、`***`)。原稿不用它们,空行已经够。"""
kept, dropped = [], 0
for line in text.split("\n"):
stripped = line.strip()
if stripped and set(stripped) <= {"-", "*", "=", "—", "·"}:
dropped += 1
continue
kept.append(line)
return "\n".join(kept), dropped
_HEADING_RE = re.compile(r"^\s*#{1,6}\s*\S*\s*$")
_NUMERAL_RE = re.compile(r"^\s*(?:[一二三四五六七八九十]{1,3}|\d{1,4})\s*$")
def _drop_stray_markings(text: str) -> tuple[str, int]:
"""删掉模型自作的小节标记:`## 一`、孤立的「七」。
分块扩写实测:模型每写一块就自制一个编号小节(## 一 … ## 六),
它们不是故事内容,留在正文里就是脏标记。
正文第一行标题(形如 `# 守门人`)要保留——那是原稿的标题。
"""
lines = text.split("\n")
first_content = next((i for i, line in enumerate(lines) if line.strip()), None)
kept, dropped = [], 0
for index, line in enumerate(lines):
if index != first_content and line.strip() and (
_HEADING_RE.match(line) or _NUMERAL_RE.match(line)
):
dropped += 1
continue
kept.append(line)
return "\n".join(kept), dropped
def _collapse_duplicate_paragraphs(text: str) -> tuple[str, int]:
"""相邻的完全重复段落只留一段(模型跨块写作时会重写上一块的结尾)。"""
kept, dropped, prev = [], 0, None
for line in text.split("\n"):
stripped = line.strip()
if stripped and stripped == prev:
dropped += 1
continue
kept.append(line)
if stripped:
prev = stripped
return "\n".join(kept), dropped
def _collapse_duplicate_sentences(text: str) -> tuple[str, int]:
"""段内连续重复的句子只留一句(模型爱把同一句复制两遍再往下写)。"""
total = 0
out: list[str] = []
for para in text.split("\n"):
if not para.strip():
out.append(para)
continue
pieces = re.split(r"(?<=[。!?])", para)
kept: list[str] = []
prev = None
for piece in pieces:
key = piece.strip()
if key and key == prev:
total += 1
continue
kept.append(piece)
if key:
prev = key
out.append("".join(kept))
return "\n".join(out), total
def _tidy_blank_lines(text: str) -> str:
"""压缩多余空行、去掉行尾空白。"""
text = re.sub(r"\n{3,}", "\n\n", text)
lines = [line.rstrip() for line in text.split("\n")]
return "\n".join(lines).strip()
def normalize(text: str) -> Normalized:
"""返回规范化后的文本与改动说明(不改字词,只改标点与结构性冗余)。"""
notes: list[str] = []
text, separators = _drop_separator_lines(text)
if separators:
notes.append(f"删除模型自作的分隔线 {separators} 行")
text, markings = _drop_stray_markings(text)
if markings:
notes.append(f"删除模型自作的小节标记 {markings} 处")
text, dup_paras = _collapse_duplicate_paragraphs(text)
if dup_paras:
notes.append(f"合并重复段落 {dup_paras} 处")
text, dup_sents = _collapse_duplicate_sentences(text)
if dup_sents:
notes.append(f"合并段内重复句子 {dup_sents} 处")
text, quotes = fix_quotes(text)
if quotes:
notes.append(f"英文引号 {quotes} 个 → 中文引号")
if quotes % 2:
notes.append("警告:引号个数为奇数,可能有一处未闭合")
text, punct = fix_cjk_punct(text)
if punct:
notes.append(f"汉字间半角标点 {punct} 处 → 全角")
return Normalized(text=_tidy_blank_lines(text), notes=notes)
+23 -1
View File
@@ -65,8 +65,30 @@ def ensure_repo() -> None:
_git("config", "user.email", "df-annals@hajim1.art")
def _tracked_empty_sources() -> list[str]:
"""已跟踪的源码/文档里,哪些是 0 字节。"""
suffixes = (".py", ".sh", ".lua", ".md", ".toml", ".json")
out: list[str] = []
for rel in _git("ls-files").stdout.split():
if not rel.endswith(suffixes):
continue
path = PROJECT_DIR / rel
if path.is_file() and path.stat().st_size == 0:
out.append(rel)
return out
def commit_and_push(message: str, paths: list[Path] | None = None) -> str:
"""提交并推送。返回提交哈希;无改动时返回空串。"""
"""提交并推送。返回提交哈希;无改动时返回空串。
提交前硬拦空文件:实测发生过一次 cli.py 被提交为 0 字节的事故——
而 cli.py 是唯一入口,于是远端仓库里的管道完全不可运行,且失败是静默的。
这种错误不能靠人复核,必须在提交口拦住。
"""
empty = _tracked_empty_sources()
if empty:
raise RuntimeError("拒绝提交:以下已跟踪文件是空的 → " + "、".join(empty))
if paths:
for p in paths:
_git("add", "--", str(p.relative_to(PROJECT_DIR)) if p.is_absolute() else str(p))
+5 -4
View File
@@ -155,8 +155,8 @@ def render_material(
for fid in heros:
row = rows.get(fid)
if row:
lines.append(f"- {row.name}|{row.race}|{row.span}|身份:{row.identity or '不详'}"
f"|本集出场 {row.appearances} 次")
lines.append(f"- {row.name}|{row.race}|性别:{row.gender or '史料未载'}"
f"|{row.span}|身份:{row.identity or '不详'}|本集出场 {row.appearances} 次")
else:
lines.append(f"- {world.figure_name(fid)}")
@@ -173,8 +173,8 @@ def render_material(
for fid in others[:6]:
row = rows.get(fid)
if row:
lines.append(f"- {row.name}|{row.race}|身份:{row.identity or '不详'}"
f"|本集出场 {row.appearances} 次")
lines.append(f"- {row.name}|{row.race}|性别:{row.gender or '史料未载'}"
f"|身份:{row.identity or '不详'}|本集出场 {row.appearances} 次")
else:
lines.append(f"- {world.figure_name(fid)}")
@@ -196,6 +196,7 @@ def render_material(
"",
f"- 本集 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,中文正文,专名保留英文。",
"- 贴着主角写:他们的目标、算计、得失做主语;对手要是个具体的人。",
"- 性别按上面卡片写,不要默认“他”;史料未载性别时用名字或身份称呼。",
"- 史料之外的世界大事不要写;只有影响到主角时才提一句。",
]
return "\n".join(lines)