第 1 卷三集全部转为成稿(分块扩写 + AI 腔定点返修),文档同步
- 每集两稿:NNN-….md 为成稿(9492/11260/11869 字),NNN-….skeleton.md 为过闸门的骨架稿 - 新增 expand.repair_style:只重写含 AI 腔标记的段落(本例出现了 14 处破折号), 三集硬伤密度 1.2/1.0/1.5 → 0.1/0.3/0.0(每千字) - episode 命令接入定点返修;expand 命令保留 A/B 对照用法 - README 同步文笔层、本土化、子命令、以及本轮新踩的坑(\b 盲点、枚举大小写、自制小节标记) - 第 3 集骨架连换三条线索仍未过闸门(真实拒绝),其成稿基于既有骨架扩写
This commit is contained in:
@@ -212,6 +212,8 @@ def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
|
||||
for note in fixed.notes:
|
||||
print(" " + note)
|
||||
|
||||
text = _polish(text, world, produced.selection.thread, produced.episode, args.model)
|
||||
|
||||
produced.path.write_text(text.strip() + "\n", encoding="utf-8")
|
||||
remaining = factcheck.check(text, world)
|
||||
print(expand_mod.compare(produced.text, text))
|
||||
@@ -221,6 +223,20 @@ def _expand_in_place(volume: int, world: legends.World, produced, args) -> str:
|
||||
return text
|
||||
|
||||
|
||||
def _polish(text: str, world, thread, episode, model: str, max_rounds: int = 2) -> str:
|
||||
"""定点返修 AI 腔标记(破折号、「不是A而是B」等),只改命中的段落。"""
|
||||
for round_no in range(1, max_rounds + 1):
|
||||
if not expand_mod.style_targets(text):
|
||||
return text
|
||||
result = expand_mod.repair_style(world, thread, episode, text, model=model)
|
||||
for note in result.notes:
|
||||
print(f" (第 {round_no} 轮){note}")
|
||||
if result.text == text:
|
||||
return text
|
||||
text = result.text
|
||||
return text
|
||||
|
||||
|
||||
def cmd_episode(args: argparse.Namespace) -> int:
|
||||
world, found = _load_candidates(args)
|
||||
state = chronicle.State.load()
|
||||
|
||||
+75
-8
@@ -320,6 +320,79 @@ def _parse_numbered(raw: str) -> dict[int, str]:
|
||||
return {k: v for k, v in out.items() if v}
|
||||
|
||||
|
||||
STYLE_FIX_SPEC = """你在做一件很窄的事:清掉正文里的 AI 腔标记,其他内容一律不动。
|
||||
|
||||
要清掉的:
|
||||
- 破折号「——」与省略号「……」:改成句号、逗号、短句或动作断句;
|
||||
- 「不是A,而是B」句式:直接写 B;
|
||||
- 「,带着……」这类万能状语;
|
||||
- 「他/他知道/明白/意识到/感到」这类直述认知:改成动作或台词;
|
||||
- 仿佛/犹如/一丝/一抹/心中一震/眼中闪过/深吸一口气 等禁用词:换成具体描写。
|
||||
|
||||
不许增删情节,不许改动没被点到的句子,不许合并或拆分段落。
|
||||
只输出改好的段落,每段以 [序号] 开头,序号与段数必须和输入一致。不写任何解释。
|
||||
"""
|
||||
|
||||
|
||||
def _apply_fixes(text: str, targets: list[int], reply: str) -> tuple[str, int]:
|
||||
"""把模型返回的 `[n] 段落` 写回原位;段号对不上就跳过,宁可不改。"""
|
||||
paragraphs = text.split("\n")
|
||||
fixed = _parse_numbered(reply)
|
||||
applied = 0
|
||||
for n, index in enumerate(targets, 1):
|
||||
if n in fixed:
|
||||
paragraphs[index] = fixed[n]
|
||||
applied += 1
|
||||
return "\n".join(paragraphs), applied
|
||||
|
||||
|
||||
def style_targets(text: str, limit: int = 8) -> list[int]:
|
||||
"""挑出含 AI 腔硬伤的段落(最多 limit 段)。"""
|
||||
targets: list[int] = []
|
||||
for index, para in enumerate(text.split("\n")):
|
||||
if not para.strip():
|
||||
continue
|
||||
if any(hit.level == "hard" for hit in deslop.check(para)):
|
||||
targets.append(index)
|
||||
return targets[:limit]
|
||||
|
||||
|
||||
def repair_style(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
episode: ep_mod.Episode,
|
||||
text: str,
|
||||
*,
|
||||
gateway=None,
|
||||
model: str = LITERARY_MODEL,
|
||||
limit: int = 8,
|
||||
) -> Normalized:
|
||||
"""定点清掉 AI 腔标记,只重写命中的段落。
|
||||
|
||||
实测:一次扩写能冒出 14 处破折号(提示词里明确禁用也拦不住)。
|
||||
确定性地把「——」一律换成逗号会读着别扭,所以交回模型改,但只改命中的段。
|
||||
"""
|
||||
targets = style_targets(text, limit)
|
||||
if not targets:
|
||||
return normalize(text)
|
||||
paragraphs = text.split("\n")
|
||||
numbered = "\n".join(f"[{n}] {paragraphs[i]}" for n, i in enumerate(targets, 1))
|
||||
messages = [
|
||||
{"role": "system", "content": STYLE_FIX_SPEC},
|
||||
{"role": "user", "content": "\n".join([
|
||||
"### 允许使用的专名", "", "、".join(allowed_names(world, thread, episode)), "",
|
||||
"### 全文的 AI 腔诊断", "", deslop.digest(text, top=12), "",
|
||||
f"### 需要改的段落(共 {len(targets)} 段)", "", numbered, "",
|
||||
f"### 要求:只输出改好的 {len(targets)} 段,每段以 [序号] 开头。",
|
||||
])},
|
||||
]
|
||||
reply = chat(messages, gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
|
||||
fixed_text, applied = _apply_fixes(text, targets, reply)
|
||||
result = normalize(fixed_text)
|
||||
result.notes = [f"AI 腔定点返修:改写 {applied}/{len(targets)} 段"] + result.notes
|
||||
return result
|
||||
|
||||
|
||||
def repair_names(
|
||||
world: World,
|
||||
thread: Thread,
|
||||
@@ -353,15 +426,9 @@ def repair_names(
|
||||
])},
|
||||
]
|
||||
reply = chat(messages, gateway=gateway, model=model, max_tokens=MAX_TOKENS_EXPAND).content
|
||||
fixed = _parse_numbered(reply)
|
||||
fixed_text, applied = _apply_fixes(text, targets, reply)
|
||||
|
||||
applied = 0
|
||||
for n, index in enumerate(targets, 1):
|
||||
if n in fixed:
|
||||
paragraphs[index] = fixed[n]
|
||||
applied += 1
|
||||
|
||||
result = normalize("\n".join(paragraphs))
|
||||
result = normalize(fixed_text)
|
||||
result.notes = [f"可疑专名修复:改写 {applied}/{len(targets)} 段"] + result.notes
|
||||
return result
|
||||
|
||||
|
||||
Reference in New Issue
Block a user