文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)

- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
This commit is contained in:
Chen Yi
2026-10-05 22:27:02 +08:00
parent 08a032120a
commit 8630dcad55
23 changed files with 1971 additions and 56 deletions
+1 -1
View File
@@ -10,7 +10,7 @@
<historical_figures>
<historical_figure><id>10</id><name>Urist McMiner</name><race>DWARF</race><appeared>12</appeared><birth_year>12</birth_year><death_year>88</death_year><profession>MINER</profession></historical_figure>
<historical_figure><id>11</id><name>Kogan Deathspear</name><race>GOBLIN</race><appeared>4</appeared><birth_year>4</birth_year><profession>AXEMAN</profession></historical_figure>
<historical_figure><id>12</id><name>liri fogbalded</name><race>ELF</race><appeared>20</appeared><birth_year>20</birth_year></historical_figure>
<historical_figure><id>12</id><name>liri fogbalded</name><race>ELF</race><caste>FEMALE</caste><appeared>20</appeared><birth_year>20</birth_year></historical_figure>
</historical_figures>
<entities>
<entity><id>100</id><name>The Steel Confederacy</name><type>Civilization</type><race>DWARF</race></entity>
+90
View File
@@ -0,0 +1,90 @@
"""AI 味诊断器的回归测试:能抓住植入的 AI 腔,也不冤枉正常句子。"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import deslop
# 故意堆满 AI 腔:禁用词、万能动宾、收束腔、破折号、省略号、认知直述
AI_TEXT = (
"他心中一震,眼中闪过一丝惊讶,仿佛看见了命运的獠牙——这一刻,他终于明白,"
"一切都已经太晚了。他深吸一口气,嘴角勾起一抹冷笑,带着不容置疑的语气说:"
"“走吧。”他知道,属于他的反击才刚刚开始……"
)
CLEAN_TEXT = (
"Ral把绳子绕了两圈,打了个结。她没说话,把账本推过去,指着第三行。"
"Guspu看了很久,把账本合上,还给她。窗外的风把灯吹得晃,两人谁都没去扶。"
)
def test_hard_hits_are_caught() -> None:
hits = deslop.hard_hits(deslop.check(AI_TEXT))
rules = {h.rule for h in hits}
for expected in ("心理模板", "眼中闪过", "一丝", "仿佛", "深吸一口气", "嘴角勾起",
"不容置疑", "破折号", "省略号"):
assert expected in rules, f"漏检: {expected}(实际命中 {sorted(rules)})"
def test_cognitive_and_closure_patterns() -> None:
hits = deslop.hard_hits(deslop.check(AI_TEXT))
rules = {h.rule for h in hits}
assert "认知直述" in rules, "「他知道」应被抓住"
assert "收束腔" in rules, "「这一刻」「才刚刚开始」应被抓住"
assert "万能动宾" in rules, "「,带着……」应被抓住"
assert "抽象命运" in rules, "「命运的獠牙」应被抓住"
def test_clean_text_has_no_hard_hits() -> None:
hits = deslop.hard_hits(deslop.check(CLEAN_TEXT))
assert not hits, f"正常句子被误判: {[h.line() for h in hits]}"
def test_not_a_but_b_pattern() -> None:
hits = deslop.check("他不是冷漠,而是绝望。")
assert any(h.rule == "不是A而是B" for h in deslop.hard_hits(hits))
def test_density_separates_ai_from_clean() -> None:
ai, clean = deslop.per_k(AI_TEXT), deslop.per_k(CLEAN_TEXT)
assert ai > clean, f"AI 稿密度应高于正常稿:{ai:.1f} vs {clean:.1f}"
assert clean == 0.0
def test_report_and_digest_render() -> None:
rep = deslop.report(AI_TEXT)
assert "硬伤" in rep and "字" in rep
dig = deslop.digest(AI_TEXT)
assert dig.startswith("- "), dig[:80]
assert "机械检查没有发现" in deslop.digest(CLEAN_TEXT)
def test_digest_lists_each_rule_once() -> None:
dig = deslop.digest(AI_TEXT)
rules = [line.split("『")[1].split("』")[0] for line in dig.splitlines()]
assert len(rules) == len(set(rules)), f"同一规则重复列出: {rules}"
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
except Exception as exc: # noqa: BLE001
failed += 1
print(f" ✗ {fn.__name__}: {type(exc).__name__}: {exc}")
else:
print(f" ✓ {fn.__name__}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())
+29
View File
@@ -75,6 +75,35 @@ def test_figure_ids_excludes_placeholder_and_dedups() -> None:
assert len(battle.figure_ids()) == len(set(battle.figure_ids()))
def test_caste_parsed_into_gender() -> None:
"""回归:v53 导出用 <caste> 而非 <sex>,漏读会把女性主角写成"他"。"""
world = load_world()
assert world.figures[12].caste == "FEMALE"
assert world.figures[12].gender == "女"
assert world.figures[10].gender == "", "未载性别的应返回空串,不能猜"
def test_cast_table_shows_gender() -> None:
"""人物表必须带性别列,否则写作者只能猜。"""
from collections import Counter
from dfannals import cast
from dfannals import episodes as ep_mod
from dfannals.threads import Thread
world = load_world()
thread = Thread(
members=[10, 11, 12], interactions=5, turns=1, span=58, first_year=30, last_year=88,
types=Counter(), races=Counter(), non_person_races=[], events=world.events,
)
planned = ep_mod.plan_episodes(thread)
text, rows, suspicions = cast.build_cast(world, thread, planned, with_bios=False)
assert "| 人物 | 种族 | 性别 |" in text, text.splitlines()[4] if len(text.splitlines()) > 4 else text
assert not suspicions
elf = next(r for r in rows if r.fid == 12)
assert elf.gender == "女"
def test_fake_names_are_caught() -> None:
"""反例:凭空编造的人名地名必须被抓出来。"""
world = load_world()
+86
View File
@@ -0,0 +1,86 @@
"""标点与结构清理的回归测试。"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dfannals import normalize
def test_ascii_quotes_become_chinese() -> None:
res = normalize.normalize('他说:"好。" 她说:"不。"')
assert res.text == "他说:“好。” 她说:“不。”", res.text
assert any("英文引号" in n for n in res.notes)
def test_halfwidth_punct_between_cjk() -> None:
res = normalize.normalize("他说,好。她走,停。")
assert "," in res.text and "," not in res.text, res.text
def test_english_punctuation_untouched() -> None:
src = "Mon Sagus, The Plane of Dawn"
assert normalize.normalize(src).text == src
def test_separator_lines_dropped() -> None:
res = normalize.normalize("甲段。\n\n---\n\n乙段。")
assert "---" not in res.text, res.text
assert any("分隔线" in n for n in res.notes)
def test_stray_section_markings_dropped_but_title_kept() -> None:
res = normalize.normalize("# 守门人\n\n甲段。\n\n## 一\n\n乙段。\n\n七\n\n丙段。")
assert res.text.startswith("# 守门人"), res.text
assert "## 一" not in res.text, res.text
assert "\n七\n" not in res.text, res.text
assert any("小节标记" in n for n in res.notes)
def test_duplicate_paragraph_merged() -> None:
res = normalize.normalize("同一段。\n同一段。\n不同段。")
assert res.text.count("同一段。") == 1, res.text
assert any("重复段落" in n for n in res.notes)
def test_duplicate_sentences_merged() -> None:
res = normalize.normalize("他走了。他走了。她留下。")
assert res.text.count("他走了。") == 1, res.text
assert any("重复句子" in n for n in res.notes)
def test_blank_lines_tidied() -> None:
res = normalize.normalize("甲。\n\n\n\n乙。")
assert "\n\n\n" not in res.text
assert res.text == "甲。\n\n乙。"
def test_clean_text_untouched() -> None:
src = "Ral把绳子绕了两圈,打了个结。她没说话。"
res = normalize.normalize(src)
assert res.text == src
assert not res.changed, res.notes
def _main() -> int:
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
failed = 0
for fn in tests:
try:
fn()
except AssertionError as exc:
failed += 1
print(f" ✗ {fn.__name__}: {exc}")
except Exception as exc: # noqa: BLE001
failed += 1
print(f" ✗ {fn.__name__}: {type(exc).__name__}: {exc}")
else:
print(f" ✓ {fn.__name__}")
print(f"\n{len(tests) - failed}/{len(tests)} 通过")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(_main())