第 1 卷三集全部转为成稿(分块扩写 + AI 腔定点返修),文档同步

- 每集两稿:NNN-….md 为成稿(9492/11260/11869 字),NNN-….skeleton.md 为过闸门的骨架稿
- 新增 expand.repair_style:只重写含 AI 腔标记的段落(本例出现了 14 处破折号),
  三集硬伤密度 1.2/1.0/1.5 → 0.1/0.3/0.0(每千字)
- episode 命令接入定点返修;expand 命令保留 A/B 对照用法
- README 同步文笔层、本土化、子命令、以及本轮新踩的坑(\b 盲点、枚举大小写、自制小节标记)
- 第 3 集骨架连换三条线索仍未过闸门(真实拒绝),其成稿基于既有骨架扩写
This commit is contained in:
Chen Yi
2026-10-05 23:14:24 +08:00
parent 16b476167e
commit 70ce758e8d
11 changed files with 1470 additions and 458 deletions
+16
View File
@@ -212,6 +212,8 @@ def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
for note in fixed.notes:
print(" " + note)
text = _polish(text, world, produced.selection.thread, produced.episode, args.model)
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
remaining = factcheck.check(text, world)
print(expand_mod.compare(produced.text, text))
@@ -221,6 +223,20 @@ def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
return text
def _polish(text: str, world, thread, episode, model: str, max_rounds: int = 2) -> str:
"""定点返修 AI 腔标记(破折号、「不是A而是B」等),只改命中的段落。"""
for round_no in range(1, max_rounds + 1):
if not expand_mod.style_targets(text):
return text
result = expand_mod.repair_style(world, thread, episode, text, model=model)
for note in result.notes:
print(f" (第 {round_no} 轮){note}")
if result.text == text:
return text
text = result.text
return text
def cmd_episode(args: argparse.Namespace) -> int:
world, found = _load_candidates(args)
state = chronicle.State.load()
+75 -8
View File
@@ -320,6 +320,79 @@ def _parse_numbered(raw: str) -> dict[int, str]:
return {k: v for k, v in out.items() if v}
STYLE_FIX_SPEC = """你在做一件很窄的事:清掉正文里的 AI 腔标记,其他内容一律不动。
要清掉的:
- 破折号「——」与省略号「……」:改成句号、逗号、短句或动作断句;
- 「不是A,而是B」句式:直接写 B;
- 「,带着……」这类万能状语;
- 「他/他知道/明白/意识到/感到」这类直述认知:改成动作或台词;
- 仿佛/犹如/一丝/一抹/心中一震/眼中闪过/深吸一口气 等禁用词:换成具体描写。
不许增删情节,不许改动没被点到的句子,不许合并或拆分段落。
只输出改好的段落,每段以 [序号] 开头,序号与段数必须和输入一致。不写任何解释。
"""
def _apply_fixes(text: str, targets: list[int], reply: str) -> tuple[str, int]:
"""把模型返回的 `[n] 段落` 写回原位;段号对不上就跳过,宁可不改。"""
paragraphs = text.split("\n")
fixed = _parse_numbered(reply)
applied = 0
for n, index in enumerate(targets, 1):
if n in fixed:
paragraphs[index] = fixed[n]
applied += 1
return "\n".join(paragraphs), applied
def style_targets(text: str, limit: int = 8) -> list[int]:
"""挑出含 AI 腔硬伤的段落(最多 limit 段)。"""
targets: list[int] = []
for index, para in enumerate(text.split("\n")):
if not para.strip():
continue
if any(hit.level == "hard" for hit in deslop.check(para)):
targets.append(index)
return targets[:limit]
def repair_style(
world: World,
thread: Thread,
episode: ep_mod.Episode,
text: str,
*,
gateway=None,
model: str = LITERARY_MODEL,
limit: int = 8,
) -> Normalized:
"""定点清掉 AI 腔标记,只重写命中的段落。
实测:一次扩写能冒出 14 处破折号(提示词里明确禁用也拦不住)。
确定性地把「——」一律换成逗号会读着别扭,所以交回模型改,但只改命中的段。
"""
targets = style_targets(text, limit)
if not targets:
return normalize(text)
paragraphs = text.split("\n")
numbered = "\n".join(f"[{n}] {paragraphs[i]}" for n, i in enumerate(targets, 1))
messages = [
{"role": "system", "content": STYLE_FIX_SPEC},
{"role": "user", "content": "\n".join([
"### 允许使用的专名", "", "、".join(allowed_names(world, thread, episode)), "",
"### 全文的 AI 腔诊断", "", deslop.digest(text, top=12), "",
f"### 需要改的段落(共 {len(targets)} 段)", "", numbered, "",
f"### 要求:只输出改好的 {len(targets)} 段,每段以 [序号] 开头。",
])},
]
reply = chat(messages, gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
fixed_text, applied = _apply_fixes(text, targets, reply)
result = normalize(fixed_text)
result.notes = [f"AI 腔定点返修:改写 {applied}/{len(targets)} 段"] + result.notes
return result
def repair_names(
world: World,
thread: Thread,
@@ -353,15 +426,9 @@ def repair_names(
])},
]
reply = chat(messages, gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
fixed = _parse_numbered(reply)
fixed_text, applied = _apply_fixes(text, targets, reply)
applied = 0
for n, index in enumerate(targets, 1):
if n in fixed:
paragraphs[index] = fixed[n]
applied += 1
result = normalize("\n".join(paragraphs))
result = normalize(fixed_text)
result.notes = [f"可疑专名修复:改写 {applied}/{len(targets)} 段"] + result.notes
return result