第 1 卷第 1 章:1–9 年
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""矮人要塞编年史生成管道。"""
|
||||
@@ -0,0 +1,159 @@
|
||||
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
|
||||
# 说明:LSP 的 Python 环境看不到本项目包(未安装进它的 site-packages),
|
||||
# 会把「包内互相 import」误报为缺失模块。运行时导入已由执行验证通过。
|
||||
|
||||
"""组织 prompt、调用模型、控制篇幅、跑事实校验,产出一章成品。"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import factcheck
|
||||
from dfannals import slice as chapter_slice
|
||||
from dfannals.config import (
|
||||
CHAPTER_MAX_CHARS,
|
||||
CHAPTER_MIN_CHARS,
|
||||
DATA_DIR,
|
||||
PROMPT_DIR,
|
||||
Gateway,
|
||||
)
|
||||
from dfannals.legends import World
|
||||
from dfannals.llm import Reply, chat
|
||||
|
||||
STATE_FILE = DATA_DIR / "state.json"
|
||||
SPEC_FILE = PROMPT_DIR / "chronicler.md"
|
||||
|
||||
TAIL_CHARS = 600 # 交给下一章衔接的上一章结尾长度
|
||||
REPAIR_ATTEMPTS = 2
|
||||
|
||||
|
||||
@dataclass
|
||||
class State:
|
||||
world: str = ""
|
||||
volume: int = 1
|
||||
done: list[str] = None # type: ignore[assignment]
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.done is None:
|
||||
self.done = []
|
||||
|
||||
@classmethod
|
||||
def load(cls, path: Path = STATE_FILE) -> State:
|
||||
if not path.is_file():
|
||||
return cls()
|
||||
try:
|
||||
data = json.loads(path.read_text())
|
||||
except json.JSONDecodeError:
|
||||
return cls()
|
||||
return cls(world=data.get("world", ""), volume=data.get("volume", 1), done=data.get("done", []))
|
||||
|
||||
def save(self, path: Path = STATE_FILE) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps({"world": self.world, "volume": self.volume, "done": self.done},
|
||||
ensure_ascii=False, indent=2))
|
||||
|
||||
def mark_done(self, key: str) -> None:
|
||||
if key not in self.done:
|
||||
self.done.append(key)
|
||||
|
||||
|
||||
def cjk_length(text: str) -> int:
|
||||
"""按中文习惯计字数:CJK 字符与拉丁单词都算 1。"""
|
||||
without_code = re.sub(r"```.*?```", "", text, flags=re.S)
|
||||
without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.M)
|
||||
n = 0
|
||||
for ch in without_meta:
|
||||
if (
|
||||
unicodedata.east_asian_width(ch) in ("W", "F")
|
||||
or ch.isalnum()
|
||||
or ch in ",。!?;:、"
|
||||
):
|
||||
n += 1
|
||||
return n
|
||||
|
||||
|
||||
def previous_tail(chapter_dir: Path, chapters: list[chapter_slice.Chapter], key: str) -> str:
|
||||
"""取上一章结尾,作为本章的衔接线索。"""
|
||||
idx = next((i for i, c in enumerate(chapters) if c.key == key), None)
|
||||
if idx is None or idx == 0:
|
||||
return ""
|
||||
prev = chapters[idx - 1]
|
||||
path = chapter_dir / f"{prev.key}.md"
|
||||
if not path.is_file():
|
||||
return ""
|
||||
text = path.read_text(encoding="utf-8").strip()
|
||||
return text[-TAIL_CHARS:]
|
||||
|
||||
|
||||
def build_messages(world: World, chapter: chapter_slice.Chapter, tail: str) -> list[dict]:
|
||||
spec = SPEC_FILE.read_text(encoding="utf-8")
|
||||
payload = chapter_slice.material(world, chapter)
|
||||
user = [
|
||||
f"世界:{world.name}" + (f"({world.altname})" if world.altname else ""),
|
||||
f"本书体例:第 {chapter.index} 章,覆盖 {chapter.span}",
|
||||
"",
|
||||
"以下是本章的史料素材(来自 legends 导出,事件名称保留游戏原文):",
|
||||
"",
|
||||
payload,
|
||||
]
|
||||
if tail:
|
||||
user += [
|
||||
"",
|
||||
"### 上一章结尾(用于衔接,不要复述,只承接状态)",
|
||||
"",
|
||||
tail,
|
||||
]
|
||||
user += [
|
||||
"",
|
||||
f"现在请写出本章正文,{CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字。",
|
||||
]
|
||||
return [
|
||||
{"role": "system", "content": spec},
|
||||
{"role": "user", "content": "\n".join(user)},
|
||||
]
|
||||
|
||||
|
||||
def _repair(messages: list[dict], draft: str, length: int, gateway: Gateway | None) -> str:
|
||||
target = (
|
||||
f"内容太短({length} 字),请扩写到 {CHAPTER_MIN_CHARS}–{CHAPTER_MAX_CHARS} 字,补充细节与对白,不要注水。"
|
||||
if length < CHAPTER_MIN_CHARS
|
||||
else f"内容太长({length} 字),请压缩到 {CHAPTER_MAX_CHARS} 字以内,删掉次要枝节,保留主干与最有趣的段落。"
|
||||
)
|
||||
reply: Reply = chat(
|
||||
messages + [{"role": "assistant", "content": draft}, {"role": "user", "content": target}],
|
||||
gateway=gateway,
|
||||
)
|
||||
return reply.content
|
||||
|
||||
|
||||
def generate(
|
||||
world: World,
|
||||
chapter: chapter_slice.Chapter,
|
||||
*,
|
||||
tail: str = "",
|
||||
gateway: Gateway | None = None,
|
||||
max_repairs: int = REPAIR_ATTEMPTS,
|
||||
) -> tuple[str, list[factcheck.Suspicion], Reply | None]:
|
||||
"""生成一章:调用模型 → 篇幅修正 → 事实校验 → 标注。"""
|
||||
messages = build_messages(world, chapter, tail)
|
||||
reply = chat(messages, gateway=gateway)
|
||||
body = reply.content
|
||||
|
||||
for _ in range(max_repairs):
|
||||
length = cjk_length(body)
|
||||
if CHAPTER_MIN_CHARS <= length <= CHAPTER_MAX_CHARS:
|
||||
break
|
||||
body = _repair(messages, body, length, gateway)
|
||||
|
||||
suspicions = factcheck.check(body, world)
|
||||
return factcheck.annotate(body, suspicions), suspicions, reply
|
||||
|
||||
|
||||
def write_chapter(chapter_dir: Path, chapter: chapter_slice.Chapter, body: str) -> Path:
|
||||
chapter_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = chapter_dir / f"{chapter.key}.md"
|
||||
path.write_text(body.strip() + "\n", encoding="utf-8")
|
||||
return path
|
||||
+181
@@ -0,0 +1,181 @@
|
||||
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
|
||||
# 说明:LSP 的 Python 环境看不到本项目包(未安装进它的 site-packages),
|
||||
# 会把「包内互相 import」误报为缺失模块。运行时导入已由执行验证通过。
|
||||
|
||||
"""命令行入口。
|
||||
|
||||
python3 -m dfannals.cli status 查看当前进度
|
||||
python3 -m dfannals.cli index 解析 exports → 年表 + 人物索引
|
||||
python3 -m dfannals.cli plan 显示切章方案(不调用模型)
|
||||
python3 -m dfannals.cli next 生成下一章 → 校验 → 推送
|
||||
python3 -m dfannals.cli run 一条命令跑完:索引 + 下一章 + 推送
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import chronicle, factcheck, legends, publish, timeline
|
||||
from dfannals import slice as chapter_slice
|
||||
from dfannals.config import EXPORT_DIR, GITEA_WEB, chapter_dir, ensure_dirs, volume_dir
|
||||
|
||||
|
||||
def find_exports(export_dir: Path) -> list[Path]:
|
||||
"""找出 legends 导出文件。原生 legends.xml 与 DFHack 的 legends_plus.xml 都要。"""
|
||||
if not export_dir.is_dir():
|
||||
return []
|
||||
files = sorted(export_dir.glob("*.xml"))
|
||||
base = [f for f in files if not f.name.endswith("-legends_plus.xml") and "legends_plus" not in f.name]
|
||||
plus = [f for f in files if "legends_plus" in f.name]
|
||||
return base + plus
|
||||
|
||||
|
||||
def load_world(export_dir: Path) -> legends.World:
|
||||
paths = find_exports(export_dir)
|
||||
if not paths:
|
||||
raise SystemExit(
|
||||
f"在 {export_dir} 里没找到 legends XML。\n"
|
||||
"请先在游戏里导出(DF 的 Export XML + DFHack 的 exportlegends),"
|
||||
"再把文件复制进这个目录。"
|
||||
)
|
||||
return legends.load(paths)
|
||||
|
||||
|
||||
def _current_volume(state: chronicle.State, world: legends.World) -> int:
|
||||
"""世界换了就开新卷。"""
|
||||
if state.world != world.name:
|
||||
if state.world:
|
||||
state.volume += 1
|
||||
state.done = []
|
||||
state.world = world.name
|
||||
return state.volume
|
||||
|
||||
|
||||
def cmd_status(args: argparse.Namespace) -> int:
|
||||
state = chronicle.State.load()
|
||||
if not state.world:
|
||||
print("尚未开始。先运行 index。")
|
||||
return 0
|
||||
vdir = volume_dir(state.volume)
|
||||
print(f"世界:{state.world}")
|
||||
print(f"卷号:第 {state.volume} 卷")
|
||||
print(f"已完成章节:{len(state.done)}")
|
||||
for key in state.done:
|
||||
print(f" · {key}")
|
||||
print(f"成品目录:{vdir}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_index(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
state = chronicle.State.load()
|
||||
volume = _current_volume(state, world)
|
||||
chapters = chapter_slice.plan(world)
|
||||
|
||||
vdir = volume_dir(volume)
|
||||
timeline.write_outputs(world, vdir)
|
||||
state.save()
|
||||
|
||||
print(legends.describe(world))
|
||||
print()
|
||||
print(chapter_slice.overview(world, chapters))
|
||||
print()
|
||||
print(f"年表与人物索引已写入 {vdir}")
|
||||
|
||||
if not args.no_push:
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 {volume} 卷索引:{world.name}({len(chapters)} 章规划)",
|
||||
[vdir / "timeline.md", vdir / "figures.md"],
|
||||
)
|
||||
print(f"已推送索引:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_plan(args: argparse.Namespace) -> int:
|
||||
world = load_world(EXPORT_DIR)
|
||||
chapters = chapter_slice.plan(world)
|
||||
print(chapter_slice.overview(world, chapters))
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_next(args: argparse.Namespace) -> int:
|
||||
ensure_dirs()
|
||||
world = load_world(EXPORT_DIR)
|
||||
state = chronicle.State.load()
|
||||
volume = _current_volume(state, world)
|
||||
chapters = chapter_slice.plan(world)
|
||||
cdir = chapter_dir(volume)
|
||||
|
||||
todo = [c for c in chapters if c.key not in state.done]
|
||||
if not todo:
|
||||
print("本卷史料已全部写完。需要新世界才能开新卷。")
|
||||
return 0
|
||||
|
||||
chapter = todo[0]
|
||||
print(f"正在写第 {chapter.index} 章:{chapter.span}({len(chapter.events)} 条事件)")
|
||||
tail = chronicle.previous_tail(cdir, chapters, chapter.key)
|
||||
body, suspicions, reply = chronicle.generate(world, chapter, tail=tail)
|
||||
length = chronicle.cjk_length(body)
|
||||
print(f"生成完成:{length} 字" + (f" | 总 token {reply.total_tokens}" if reply else ""))
|
||||
|
||||
print(factcheck.report(suspicions))
|
||||
|
||||
if args.dry_run:
|
||||
print("\n--- 试运行,不写文件 ---\n")
|
||||
print(body[:1200])
|
||||
return 0
|
||||
|
||||
path = chronicle.write_chapter(cdir, chapter, body)
|
||||
state.mark_done(chapter.key)
|
||||
state.save()
|
||||
print(f"已写入 {path}")
|
||||
|
||||
publish.ensure_repo()
|
||||
head = publish.commit_and_push(
|
||||
f"第 {volume} 卷第 {chapter.index} 章:{chapter.span}"
|
||||
+ (f"(专名校验可疑 {len(suspicions)} 处)" if suspicions else "")
|
||||
)
|
||||
print(f"已推送:{head or '(无改动)'} → {GITEA_WEB}")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_run(args: argparse.Namespace) -> int:
|
||||
rc = cmd_index(args)
|
||||
if rc:
|
||||
return rc
|
||||
return cmd_next(args)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(prog="dfannals", description="矮人要塞编年史生成管道")
|
||||
sub = parser.add_subparsers(dest="cmd", required=True)
|
||||
|
||||
p = sub.add_parser("status", help="查看当前进度")
|
||||
p.set_defaults(func=cmd_status)
|
||||
|
||||
p = sub.add_parser("index", help="解析 exports → 年表 + 人物索引")
|
||||
p.add_argument("--no-push", action="store_true", help="只写本地,不推送")
|
||||
p.set_defaults(func=cmd_index)
|
||||
|
||||
p = sub.add_parser("plan", help="显示切章方案")
|
||||
p.set_defaults(func=cmd_plan)
|
||||
|
||||
p = sub.add_parser("next", help="生成下一章")
|
||||
p.add_argument("--dry-run", action="store_true", help="只生成不落盘、不推送")
|
||||
p.set_defaults(func=cmd_next)
|
||||
|
||||
p = sub.add_parser("run", help="一条命令:索引 + 下一章 + 推送")
|
||||
p.add_argument("--no-push", action="store_true")
|
||||
p.add_argument("--dry-run", action="store_true")
|
||||
p.set_defaults(func=cmd_run)
|
||||
|
||||
args = parser.parse_args(argv)
|
||||
ensure_dirs()
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,119 @@
|
||||
"""项目配置:路径、模型、网关、发布目标。
|
||||
|
||||
原则:API key 不落本项目文件,运行时从 Pi 的 auth.json 读取。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
HOME = Path.home()
|
||||
|
||||
# ---------------------------------------------------------------- 目录
|
||||
PROJECT_DIR = Path(__file__).resolve().parents[1] # /home/chendy/df-annals
|
||||
DATA_DIR = PROJECT_DIR / "data" # 原始数据,不入库
|
||||
EXPORT_DIR = DATA_DIR / "exports" # legends 导出原始文件
|
||||
OUTPUT_DIR = PROJECT_DIR / "output" # 成品,入库
|
||||
CHAPTER_DIR = OUTPUT_DIR / "chapters"
|
||||
PROMPT_DIR = PROJECT_DIR / "prompts"
|
||||
|
||||
|
||||
def volume_dir(volume: int) -> Path:
|
||||
"""每一卷(一个世界)独立的成品目录。"""
|
||||
return OUTPUT_DIR / "volumes" / f"v{volume}"
|
||||
|
||||
|
||||
def chapter_dir(volume: int) -> Path:
|
||||
return volume_dir(volume) / "chapters"
|
||||
|
||||
# ---------------------------------------------------------------- 游戏(Windows 侧)
|
||||
WINDOWS_USER = "23518"
|
||||
WIN_DF_ROOT = Path(f"/mnt/c/Users/{WINDOWS_USER}/DwarfFortress") # 下载/安装根目录
|
||||
WIN_DF_DIR_CANDIDATES = [
|
||||
WIN_DF_ROOT / "df_53_16_win",
|
||||
WIN_DF_ROOT / "Dwarf Fortress",
|
||||
WIN_DF_ROOT,
|
||||
]
|
||||
DF_VERSION = "53.16"
|
||||
DFHACK_VERSION = "53.16-r2"
|
||||
|
||||
# ---------------------------------------------------------------- LLM 网关
|
||||
PI_AUTH_JSON = HOME / ".pi/agent/auth.json"
|
||||
PI_MODELS_JSON = HOME / ".pi/agent/models.json"
|
||||
PROVIDER = "workbuddy"
|
||||
MODEL = "cn:deepseek-v4-pro"
|
||||
# 该模型是推理模型:思考与正文共用 max_tokens,必须留足
|
||||
MAX_TOKENS_CHAPTER = 16000
|
||||
MAX_TOKENS_UTIL = 4000
|
||||
|
||||
# ---------------------------------------------------------------- 发布
|
||||
GITEA_SSH_HOST = "124.222.29.26"
|
||||
GITEA_SSH_PORT = 2222
|
||||
GITEA_USER = "gitadmin"
|
||||
GITEA_REPO = "dwarf-fortress-annals"
|
||||
GITEA_WEB = "http://124.222.29.26:3000"
|
||||
GITEA_REMOTE = f"ssh://git@{GITEA_SSH_HOST}:{GITEA_SSH_PORT}/{GITEA_USER}/{GITEA_REPO}.git"
|
||||
SSH_KEY = HOME / ".ssh/id_ed25519"
|
||||
|
||||
# ---------------------------------------------------------------- 世界与章节
|
||||
WORLD_SIZE = "Medium"
|
||||
WORLD_HISTORY_YEARS = 250
|
||||
CHAPTER_MIN_CHARS = 800
|
||||
CHAPTER_MAX_CHARS = 1600 # 契约上限 1500,留出标点/换行余量
|
||||
|
||||
|
||||
@dataclass
|
||||
class Gateway:
|
||||
base_url: str
|
||||
key: str
|
||||
model: str = MODEL
|
||||
provider: str = PROVIDER
|
||||
|
||||
|
||||
def load_gateway(provider: str | None = None) -> Gateway:
|
||||
"""从 Pi 配置读取网关地址与密钥。密钥只驻留内存,不打印、不落盘。"""
|
||||
provider = provider or PROVIDER
|
||||
|
||||
def _read_json(path: Path, label: str) -> dict:
|
||||
try:
|
||||
return json.loads(path.read_text())
|
||||
except FileNotFoundError:
|
||||
raise SystemExit(f"找不到 Pi 的{label}文件: {path}") from None
|
||||
except json.JSONDecodeError as exc:
|
||||
raise SystemExit(f"Pi 的{label}文件不是合法 JSON: {path} ({exc})") from None
|
||||
|
||||
auth = _read_json(PI_AUTH_JSON, "凭据")
|
||||
models = _read_json(PI_MODELS_JSON, "模型配置")
|
||||
providers = models.get("providers", models)
|
||||
if provider not in providers:
|
||||
raise SystemExit(f"Pi 配置里没有 provider: {provider}")
|
||||
if provider not in auth:
|
||||
raise SystemExit(f"Pi 未授权 provider: {provider}")
|
||||
entry = auth[provider]
|
||||
key = entry.get("key") if isinstance(entry, dict) else entry
|
||||
if not key:
|
||||
raise SystemExit(f"provider {provider} 的凭据为空")
|
||||
return Gateway(base_url=providers[provider]["baseUrl"].rstrip("/"), key=key, provider=provider)
|
||||
|
||||
|
||||
def find_df_dir() -> Path:
|
||||
"""定位已安装的 DF 目录(含 Dwarf Fortress.exe)。"""
|
||||
override = os.environ.get("DF_DIR")
|
||||
if override:
|
||||
p = Path(override)
|
||||
if p.is_dir():
|
||||
return p
|
||||
raise SystemExit(f"DF_DIR 不存在: {p}")
|
||||
for cand in WIN_DF_DIR_CANDIDATES:
|
||||
if (cand / "Dwarf Fortress.exe").is_file():
|
||||
return cand
|
||||
raise SystemExit(
|
||||
"未找到 Dwarf Fortress.exe。候选路径:" + ", ".join(str(c) for c in WIN_DF_DIR_CANDIDATES)
|
||||
)
|
||||
|
||||
|
||||
def ensure_dirs() -> None:
|
||||
for d in (DATA_DIR, EXPORT_DIR, OUTPUT_DIR, CHAPTER_DIR, PROMPT_DIR):
|
||||
d.mkdir(parents=True, exist_ok=True)
|
||||
@@ -0,0 +1,141 @@
|
||||
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
|
||||
# 说明:LSP 的 Python 环境看不到本项目包(未安装进它的 site-packages),
|
||||
# 会把「包内互相 import」误报为缺失模块。运行时导入已由执行验证通过。
|
||||
|
||||
"""专名事实校验:正文里出现的专有名词必须能在史料中找到。
|
||||
|
||||
策略(按访谈决策:自动直推、不阻断,但标注可疑处):
|
||||
- 抽出正文中的拉丁字母专名序列(本项目的约定是专名保留英文原文)
|
||||
- 与史料已知专名集合比对
|
||||
- 未知专名既不阻断推送,也不静默放过,而是连同上下文一起标注出来
|
||||
|
||||
注意:这一步只能拦住"凭空出现的人名地名",拦不住编造的对白与心理描写——
|
||||
后者按用户选择属于允许的自由演绎范围。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
from dfannals.legends import World, pretty
|
||||
|
||||
# 正文里允许出现、但不算专名的常见英文词(游戏术语/通用词)
|
||||
COMMON_WORDS = {
|
||||
"Dwarf", "Dwarves", "Dwarven", "Fortress", "Fort", "Mountain", "Mountains", "Hill",
|
||||
"River", "Lake", "Forest", "Desert", "Ocean", "Cave", "Caves", "Town", "City",
|
||||
"King", "Queen", "Lord", "Lady", "Captain", "General", "Soldier", "Miner", "Smith",
|
||||
"Hammer", "Axe", "Sword", "Shield", "Gold", "Silver", "Iron", "Steel", "Copper",
|
||||
"Bronze", "Wood", "Stone", "Water", "Beer", "Ale", "Wine", "Goblin", "Goblins",
|
||||
"Elf", "Elves", "Human", "Humans", "Kobold", "Kobolds", "Troll", "Trolls",
|
||||
"Beast", "Beasts", "Titan", "Titans", "Dragon", "Dragons", "Demon", "Demons",
|
||||
"God", "Gods", "Temple", "Shrine", "Tomb", "Market", "Tavern", "Library",
|
||||
"Legend", "Legends", "Annals", "Chronicle", "Chapter", "Year", "Years",
|
||||
"North", "South", "East", "West", "Spring", "Summer", "Autumn", "Winter",
|
||||
"The", "A", "An", "And", "Or", "But", "In", "On", "At", "By", "Of", "To", "For",
|
||||
}
|
||||
|
||||
# 拉丁专名序列:首字母大写的词,可含 ' - 与数字
|
||||
NAME_RE = re.compile(r"\b[A-Z][A-Za-z'\-]*(?:\s+[A-Z][A-Za-z'\-]*)*\b")
|
||||
_CJK = re.compile(r"[\u4e00-\u9fff]")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Suspicion:
|
||||
name: str
|
||||
count: int
|
||||
context: str
|
||||
|
||||
|
||||
def known_names(world: World) -> set[str]:
|
||||
"""史料中出现过的全部专名,含单词与整名两种粒度。"""
|
||||
names: set[str] = set()
|
||||
|
||||
def add(value: str) -> None:
|
||||
value = (value or "").strip()
|
||||
if not value:
|
||||
return
|
||||
names.add(value)
|
||||
# 同时收录展示形式:正文里专名会做首字母大写,两边必须一致
|
||||
shown = pretty(value)
|
||||
names.add(shown)
|
||||
for token in re.split(r"[\s,]+", shown):
|
||||
token = token.strip("'\"()[]")
|
||||
if token and token[0].isupper():
|
||||
names.add(token)
|
||||
# 原样小写形式也保留,避免漏报
|
||||
names.add(value.lower())
|
||||
|
||||
# 世界名必须拆词:否则 "The World of Prophecy" 里的 Prophecy 会被当成凭空编造
|
||||
add(world.name)
|
||||
add(world.altname)
|
||||
|
||||
for fig in world.figures.values():
|
||||
add(fig.name)
|
||||
add(fig.race)
|
||||
add(fig.profession)
|
||||
for ent in world.entities.values():
|
||||
add(ent.name)
|
||||
add(ent.type)
|
||||
add(ent.race)
|
||||
for site in world.sites.values():
|
||||
add(site.name)
|
||||
add(site.type)
|
||||
return names
|
||||
|
||||
|
||||
def _is_candidate(name: str, known: set[str]) -> bool:
|
||||
if name in known:
|
||||
return False
|
||||
tokens = name.split()
|
||||
# 全部由常见词组成的串不是专名
|
||||
if all(t in COMMON_WORDS for t in tokens):
|
||||
return False
|
||||
# 全大写的是史料里的枚举值(GOBLIN、MINER、DARK_FORTRESS),不是人名地名
|
||||
return not all(len(t) >= 2 and t.isupper() for t in tokens)
|
||||
|
||||
|
||||
def check(text: str, world: World) -> list[Suspicion]:
|
||||
"""返回可疑专名列表(按出现次数降序)。"""
|
||||
known = known_names(world)
|
||||
found: dict[str, list[str]] = {}
|
||||
|
||||
for match in NAME_RE.finditer(text):
|
||||
raw = match.group(0)
|
||||
# 逐级回退:整串不认,就试着拆成更短的已知名字,减少误报
|
||||
if not _is_candidate(raw, known):
|
||||
continue
|
||||
if any(part in known for part in raw.split()):
|
||||
# 名字里有一部分是史料已知的(例如 "Urist the Bold"),不当作凭空编造
|
||||
continue
|
||||
start = max(0, match.start() - 20)
|
||||
end = min(len(text), match.end() + 20)
|
||||
found.setdefault(raw, []).append(text[start:end].replace("\n", " "))
|
||||
|
||||
return [
|
||||
Suspicion(name=n, count=len(ctxs), context=ctxs[0])
|
||||
for n, ctxs in sorted(found.items(), key=lambda kv: (-len(kv[1]), kv[0]))
|
||||
]
|
||||
|
||||
|
||||
def report(suspicions: list[Suspicion], top: int = 12) -> str:
|
||||
if not suspicions:
|
||||
return "专名校验:未发现史料之外的专有名词。"
|
||||
lines = [f"专名校验:发现 {len(suspicions)} 个可疑专名(不阻断推送,仅标注)"]
|
||||
for s in suspicions[:top]:
|
||||
lines.append(f" · {s.name} ×{s.count} —— 上下文:…{s.context}…")
|
||||
if len(suspicions) > top:
|
||||
lines.append(f" (其余 {len(suspicions) - top} 个见校验报告文件)")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def annotate(chapter_md: str, suspicions: list[Suspicion]) -> str:
|
||||
"""在章节末尾追加可见的校验提示。"""
|
||||
if not suspicions:
|
||||
return chapter_md
|
||||
names = "、".join(f"{s.name}×{s.count}" for s in suspicions[:10])
|
||||
block = (
|
||||
"\n\n---\n\n"
|
||||
f"> ⚠️ **史料核验提示**:以下专名未在 legends 史料中找到,可能是演绎或错误:{names}。\n"
|
||||
"> 其余对白与心理描写属于允许的文学演绎,不在校验范围内。\n"
|
||||
)
|
||||
return chapter_md.rstrip() + block
|
||||
@@ -0,0 +1,347 @@
|
||||
"""解析矮人要塞的 legends 导出。
|
||||
|
||||
数据来源有两份,按 id 合并:
|
||||
1. ``legends.xml`` —— DF 原生 "Export XML" 产物:名字、年份、事件主体
|
||||
2. ``legends_plus.xml`` —— DFHack ``exportlegends`` 产物:补充字段(sex/race 等)
|
||||
|
||||
因为大世界的导出可达数十 MB,这里用 iterparse 流式解析。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from defusedxml.ElementTree import iterparse
|
||||
|
||||
# 事件里指向其他实体的字段 → 指向哪类对象
|
||||
ID_FIELDS: dict[str, str] = {
|
||||
"hfid": "figure",
|
||||
"hist_figure_id": "figure",
|
||||
"slayer_hfid": "figure",
|
||||
"slayer_item_id": "artifact",
|
||||
"target_hfid": "figure",
|
||||
"source_hfid": "figure",
|
||||
"site_id": "site",
|
||||
"site_civ_id": "entity",
|
||||
"entity_id": "entity",
|
||||
"attacker_civ_id": "entity",
|
||||
"defender_civ_id": "entity",
|
||||
"artifact_id": "artifact",
|
||||
"region_id": "region",
|
||||
"feature_layer_id": "region",
|
||||
"deity": "figure",
|
||||
"worshipper_hfid": "figure",
|
||||
"creature_id": "creature",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class Figure:
|
||||
id: int
|
||||
name: str = ""
|
||||
race: str = ""
|
||||
sex: int | None = None
|
||||
birth_year: int | None = None
|
||||
death_year: int | None = None
|
||||
profession: str = ""
|
||||
entities: list[int] = field(default_factory=list)
|
||||
sites: list[int] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def alive_span(self) -> str:
|
||||
# DF 用负数占位表示“无此项”(未记录/未死),不要当成真实年份显示
|
||||
birth = self.birth_year if (self.birth_year or 0) >= 0 else None
|
||||
death = self.death_year if (self.death_year or 0) >= 0 else None
|
||||
if birth is None and death is None:
|
||||
return "生卒不详"
|
||||
b = "?" if birth is None else str(birth)
|
||||
d = "在世/不详" if death is None else str(death)
|
||||
return f"{b}–{d}"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Event:
|
||||
id: int
|
||||
year: int
|
||||
seconds72: int = 0
|
||||
type: str = ""
|
||||
fields: dict[str, list[str]] = field(default_factory=dict)
|
||||
|
||||
def ids(self, tag: str) -> list[int]:
|
||||
out: list[int] = []
|
||||
for raw in self.fields.get(tag, []):
|
||||
try:
|
||||
out.append(int(raw))
|
||||
except ValueError:
|
||||
continue
|
||||
return out
|
||||
|
||||
def text(self, tag: str) -> str:
|
||||
vals = self.fields.get(tag)
|
||||
return vals[0] if vals else ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class Entity:
|
||||
id: int
|
||||
name: str = ""
|
||||
type: str = ""
|
||||
race: str = ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class Site:
|
||||
id: int
|
||||
name: str = ""
|
||||
type: str = ""
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ 展示用工具
|
||||
def pretty(name: str) -> str:
|
||||
"""展示用:DF 的程序化名字全小写,正文里按首字母大写会更像专名。
|
||||
|
||||
只影响渲染,不改变史料本身。
|
||||
"""
|
||||
if not name:
|
||||
return name
|
||||
return " ".join(part[:1].upper() + part[1:] for part in name.split())
|
||||
|
||||
|
||||
@dataclass
|
||||
class World:
|
||||
name: str = ""
|
||||
altname: str = ""
|
||||
figures: dict[int, Figure] = field(default_factory=dict)
|
||||
events: list[Event] = field(default_factory=list)
|
||||
entities: dict[int, Entity] = field(default_factory=dict)
|
||||
sites: dict[int, Site] = field(default_factory=dict)
|
||||
sources: list[str] = field(default_factory=list)
|
||||
|
||||
# ---------------------------------------------------------- 查询
|
||||
# 约定:id 为负表示“无此项”(DF 用 -1 占位),返回空串让调用方跳过。
|
||||
def figure_name(self, fid: int) -> str:
|
||||
if fid < 0:
|
||||
return ""
|
||||
fig = self.figures.get(fid)
|
||||
return pretty(fig.name) if fig and fig.name else f"HF#{fid}"
|
||||
|
||||
def entity_name(self, eid: int) -> str:
|
||||
if eid < 0:
|
||||
return ""
|
||||
ent = self.entities.get(eid)
|
||||
return pretty(ent.name) if ent and ent.name else f"Entity#{eid}"
|
||||
|
||||
def site_name(self, sid: int) -> str:
|
||||
if sid < 0:
|
||||
return ""
|
||||
st = self.sites.get(sid)
|
||||
return pretty(st.name) if st and st.name else f"Site#{sid}"
|
||||
|
||||
def event_years(self) -> tuple[int, int]:
|
||||
years = [e.year for e in self.events if e.year]
|
||||
return (min(years), max(years)) if years else (0, 0)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ 解析工具
|
||||
def _int_or_none(text: str | None) -> int | None:
|
||||
if text is None:
|
||||
return None
|
||||
try:
|
||||
return int(text)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def _children(elem: Any) -> dict[str, list[str]]:
|
||||
"""把元素的子标签收成 {tag: [文本, ...]}。"""
|
||||
out: dict[str, list[str]] = {}
|
||||
for child in elem:
|
||||
val = (child.text or "").strip()
|
||||
out.setdefault(child.tag, []).append(val)
|
||||
return out
|
||||
|
||||
|
||||
def _first(d: dict[str, list[str]], *keys: str) -> str:
|
||||
for k in keys:
|
||||
if d.get(k):
|
||||
return d[k][0]
|
||||
return ""
|
||||
|
||||
|
||||
# XML 1.0 不允许的控制字符(保留 \t \n \r)
|
||||
ILLEGAL_BYTES = bytes(range(0x09)) + b"\x0b\x0c" + bytes(range(0x0e, 0x20))
|
||||
CLEAN_CACHE = Path("data/exports/clean")
|
||||
|
||||
|
||||
def sanitize(src: Path, dst: Path, chunk_size: int = 8 << 20) -> Path:
|
||||
"""把 DF 导出的 legends XML 转成合规的 UTF-8 文件。
|
||||
|
||||
DF 会在名字里嵌入 \\x10/\\x11 等 CP437 控制字节做专名标记,
|
||||
XML 1.0 禁止这些字节,严格解析器会直接报错。
|
||||
这里做两件事:剔除非法控制字节;把 CP437 声明与实际字节统一成 UTF-8。
|
||||
cp437 是单字节编码,因此可以按块转换而不会切断字符。
|
||||
"""
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
with src.open("rb") as fin, dst.open("wb") as fout:
|
||||
first = True
|
||||
while True:
|
||||
buf = fin.read(chunk_size)
|
||||
if not buf:
|
||||
break
|
||||
buf = buf.translate(None, ILLEGAL_BYTES)
|
||||
if first:
|
||||
buf = buf.replace(b"encoding='CP437'", b"encoding='UTF-8'", 1)
|
||||
buf = buf.replace(b'encoding="CP437"', b'encoding="UTF-8"', 1)
|
||||
first = False
|
||||
fout.write(buf.decode("cp437").encode("utf-8"))
|
||||
return dst
|
||||
|
||||
|
||||
def clean_copy(src: Path) -> Path:
|
||||
"""返回可解析的干净副本,带缓存(源文件更新则重建)。
|
||||
|
||||
缓存写到 data/exports/clean/ 子目录,避开 find_exports 的 *.xml 扫描,
|
||||
否则干净副本会被当作第二份史料,导致事件数翻倍。
|
||||
"""
|
||||
dst = CLEAN_CACHE / src.name
|
||||
if dst.is_file() and dst.stat().st_mtime >= src.stat().st_mtime and dst.stat().st_size > 0:
|
||||
return dst
|
||||
return sanitize(src, dst)
|
||||
|
||||
|
||||
def _reject_unsafe_xml(path: Path) -> None:
|
||||
"""拒绝带 DTD / 实体声明的文件。
|
||||
|
||||
DOCTYPE 必须出现在根元素之前,因此只检查文件头部即可拦住
|
||||
XXE 与 billion-laughs 这两类实体展开攻击。
|
||||
"""
|
||||
with path.open("rb") as fh:
|
||||
head = fh.read(65536)
|
||||
for marker in (b"<!DOCTYPE", b"<!ENTITY"):
|
||||
if marker in head:
|
||||
raise ValueError(f"拒绝解析含 DTD/实体声明的文件({marker.decode()}): {path}")
|
||||
|
||||
|
||||
WRAPPERS = ("historical_figures", "historical_events", "entities", "sites")
|
||||
|
||||
|
||||
def parse_file(path: Path, world: World) -> None:
|
||||
"""流式解析一个 legends XML,合并进 world。
|
||||
|
||||
用元素深度而不是包裹标签的进出状态来控制语义:
|
||||
只有 depth == 2(即 <df_world> 的直接子元素)才会被当作世界级字段,
|
||||
这样历史人物、地区里的 <name> 永远不会污染世界名。
|
||||
"""
|
||||
_reject_unsafe_xml(path)
|
||||
world.sources.append(str(path))
|
||||
depth = 0
|
||||
parser = iterparse(str(path), events=("start", "end"))
|
||||
for event, elem in parser:
|
||||
if event == "start":
|
||||
depth += 1
|
||||
continue
|
||||
|
||||
tag = elem.tag
|
||||
if depth == 2:
|
||||
if tag == "name":
|
||||
world.name = world.name or (elem.text or "").strip()
|
||||
elif tag == "altname":
|
||||
world.altname = world.altname or (elem.text or "").strip()
|
||||
elif tag == "historical_figure":
|
||||
_add_figure(world, elem)
|
||||
elem.clear()
|
||||
elif tag == "historical_event":
|
||||
_add_event(world, elem)
|
||||
elem.clear()
|
||||
elif tag == "entity":
|
||||
_add_entity(world, elem)
|
||||
elem.clear()
|
||||
elif tag == "site":
|
||||
_add_site(world, elem)
|
||||
elem.clear()
|
||||
|
||||
depth -= 1
|
||||
|
||||
|
||||
def _add_figure(world: World, elem) -> None:
|
||||
d = _children(elem)
|
||||
fid = _int_or_none(_first(d, "id"))
|
||||
if fid is None:
|
||||
return
|
||||
fig = world.figures.setdefault(fid, Figure(id=fid))
|
||||
fig.name = fig.name or _first(d, "name")
|
||||
fig.race = fig.race or _first(d, "race")
|
||||
fig.profession = fig.profession or _first(d, "profession")
|
||||
if fig.sex is None:
|
||||
fig.sex = _int_or_none(_first(d, "sex"))
|
||||
if fig.birth_year is None:
|
||||
fig.birth_year = _int_or_none(_first(d, "birth_year"))
|
||||
if fig.death_year is None:
|
||||
fig.death_year = _int_or_none(_first(d, "death_year"))
|
||||
|
||||
|
||||
def _add_event(world: World, elem) -> None:
|
||||
d = _children(elem)
|
||||
eid = _int_or_none(_first(d, "id"))
|
||||
year = _int_or_none(_first(d, "year"))
|
||||
if eid is None or year is None:
|
||||
return
|
||||
world.events.append(
|
||||
Event(
|
||||
id=eid,
|
||||
year=year,
|
||||
seconds72=_int_or_none(_first(d, "seconds72")) or 0,
|
||||
type=_first(d, "type"),
|
||||
fields=d,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _add_entity(world: World, elem) -> None:
|
||||
d = _children(elem)
|
||||
eid = _int_or_none(_first(d, "id"))
|
||||
if eid is None:
|
||||
return
|
||||
ent = world.entities.setdefault(eid, Entity(id=eid))
|
||||
ent.name = ent.name or _first(d, "name")
|
||||
ent.type = ent.type or _first(d, "type")
|
||||
ent.race = ent.race or _first(d, "race")
|
||||
|
||||
|
||||
def _add_site(world: World, elem) -> None:
|
||||
d = _children(elem)
|
||||
sid = _int_or_none(_first(d, "id"))
|
||||
if sid is None:
|
||||
return
|
||||
st = world.sites.setdefault(sid, Site(id=sid))
|
||||
st.name = st.name or _first(d, "name")
|
||||
st.type = st.type or _first(d, "type")
|
||||
|
||||
|
||||
def load(paths: list[Path]) -> World:
|
||||
world = World()
|
||||
for p in paths:
|
||||
if not p.is_file():
|
||||
raise FileNotFoundError(f"缺少 legends 导出文件: {p}")
|
||||
if p.stat().st_size == 0:
|
||||
raise ValueError(f"legends 导出文件为空: {p}")
|
||||
parse_file(clean_copy(p), world)
|
||||
world.events.sort(key=lambda e: (e.year, e.seconds72, e.id))
|
||||
return world
|
||||
|
||||
|
||||
def describe(world: World, top: int = 8) -> str:
|
||||
lo, hi = world.event_years()
|
||||
by_type: dict[str, int] = {}
|
||||
for e in world.events:
|
||||
by_type[e.type] = by_type.get(e.type, 0) + 1
|
||||
hot = sorted(by_type.items(), key=lambda kv: -kv[1])[:top]
|
||||
lines = [
|
||||
f"世界: {world.name}" + (f"({world.altname})" if world.altname else ""),
|
||||
f"历史跨度: {lo}–{hi}({hi - lo} 年)| 事件 {len(world.events)} 条",
|
||||
f"历史人物 {len(world.figures)} | 文明/组织 {len(world.entities)} | 地点 {len(world.sites)}",
|
||||
"高频事件类型: " + ", ".join(f"{t}×{n}" for t, n in hot),
|
||||
f"来源: {', '.join(Path(s).name for s in world.sources)}",
|
||||
]
|
||||
return "\n".join(lines)
|
||||
+106
@@ -0,0 +1,106 @@
|
||||
"""调用用户自建网关(OpenAI 兼容)生成文本。
|
||||
|
||||
要点:
|
||||
- 密钥运行时从 Pi 的 auth.json 读取,只在内存中,永不打印或写入产物。
|
||||
- 目标是推理模型,思考与正文共用 max_tokens,因此默认给足预算。
|
||||
- 传输层用 http.client(方案显式限定 http/https),不依赖第三方库。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import http.client
|
||||
import json
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from dfannals.config import MAX_TOKENS_CHAPTER, MODEL, Gateway, load_gateway
|
||||
|
||||
|
||||
@dataclass
|
||||
class Reply:
|
||||
content: str
|
||||
reasoning_tokens: int
|
||||
total_tokens: int
|
||||
model: str
|
||||
|
||||
|
||||
class GatewayError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def _endpoint(base_url: str) -> tuple[type[http.client.HTTPConnection], str, str]:
|
||||
"""解析网关地址,只接受 http/https。"""
|
||||
parts = urlsplit(base_url)
|
||||
if parts.scheme not in ("http", "https"):
|
||||
raise GatewayError(f"网关地址必须是 http(s):{base_url!r}")
|
||||
if not parts.netloc:
|
||||
raise GatewayError(f"网关地址缺少主机:{base_url!r}")
|
||||
cls = http.client.HTTPSConnection if parts.scheme == "https" else http.client.HTTPConnection
|
||||
path = parts.path.rstrip("/") + "/chat/completions"
|
||||
return cls, parts.netloc, path
|
||||
|
||||
|
||||
def chat(
|
||||
messages: list[dict],
|
||||
*,
|
||||
gateway: Gateway | None = None,
|
||||
model: str = MODEL,
|
||||
max_tokens: int = MAX_TOKENS_CHAPTER,
|
||||
temperature: float = 1.0,
|
||||
timeout: int = 900,
|
||||
retries: int = 3,
|
||||
) -> Reply:
|
||||
gw = gateway or load_gateway()
|
||||
conn_cls, host, path = _endpoint(gw.base_url)
|
||||
body = json.dumps(
|
||||
{
|
||||
"model": model,
|
||||
"messages": messages,
|
||||
"max_tokens": max_tokens,
|
||||
"temperature": temperature,
|
||||
}
|
||||
).encode()
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
# 密钥仅在此处作为请求头使用,不落盘、不打印
|
||||
"Authorization": "Bearer " + gw.key,
|
||||
}
|
||||
|
||||
last: str | None = None
|
||||
for attempt in range(1, retries + 1):
|
||||
conn = conn_cls(host, timeout=timeout)
|
||||
try:
|
||||
conn.request("POST", path, body=body, headers=headers)
|
||||
resp = conn.getresponse()
|
||||
raw = resp.read()
|
||||
if resp.status != 200:
|
||||
last = f"HTTP {resp.status}: {raw[:400].decode('utf-8', 'replace')}"
|
||||
if resp.status in (400, 401, 403, 404):
|
||||
break
|
||||
else:
|
||||
data = json.loads(raw)
|
||||
choice = data["choices"][0]["message"]
|
||||
usage = data.get("usage") or {}
|
||||
content = (choice.get("content") or "").strip()
|
||||
if not content:
|
||||
raise GatewayError(
|
||||
"模型只产出了思考、没有正文;"
|
||||
f"thinking_tokens={usage.get('completion_thinking_tokens')},请加大 max_tokens"
|
||||
)
|
||||
return Reply(
|
||||
content=content,
|
||||
reasoning_tokens=int(usage.get("completion_thinking_tokens") or 0),
|
||||
total_tokens=int(usage.get("total_tokens") or 0),
|
||||
model=data.get("model", model),
|
||||
)
|
||||
except (OSError, http.client.HTTPException, json.JSONDecodeError, KeyError) as exc:
|
||||
last = f"{type(exc).__name__}: {exc}"
|
||||
except GatewayError as exc:
|
||||
last = str(exc)
|
||||
break
|
||||
finally:
|
||||
conn.close()
|
||||
if attempt < retries:
|
||||
time.sleep(3 * attempt)
|
||||
|
||||
raise GatewayError(f"网关调用失败({gw.base_url},model={model}):{last}")
|
||||
@@ -0,0 +1,79 @@
|
||||
"""把成品章节与年表推送到自建 Gitea。
|
||||
|
||||
- 推送凭据复用现有 ~/.ssh/id_ed25519(已验证在 Gitea 注册为 gitadmin 的 chendy-ubuntu-wsl)。
|
||||
- 原始 legends 数据与存档不入库(见 .gitignore)。
|
||||
- 自动直推,不设人工审核环节(按访谈决策)。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals.config import GITEA_REMOTE, PROJECT_DIR, SSH_KEY
|
||||
|
||||
GITIGNORE = """\
|
||||
# 原始数据与存档:不入库
|
||||
data/
|
||||
vendor/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
*.log
|
||||
"""
|
||||
|
||||
|
||||
def _git(*args: str, check: bool = True) -> subprocess.CompletedProcess:
|
||||
env = {
|
||||
"GIT_SSH_COMMAND": f"ssh -i {SSH_KEY} -o IdentitiesOnly=yes -o StrictHostKeyChecking=accept-new",
|
||||
"GIT_TERMINAL_PROMPT": "0",
|
||||
}
|
||||
proc = subprocess.run(
|
||||
["git", *args],
|
||||
cwd=PROJECT_DIR,
|
||||
env={**_base_env(), **env},
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
if check and proc.returncode != 0:
|
||||
raise RuntimeError(f"git {' '.join(args)} 失败:{proc.stderr.strip() or proc.stdout.strip()}")
|
||||
return proc
|
||||
|
||||
|
||||
def _base_env() -> dict:
|
||||
import os
|
||||
|
||||
env = dict(os.environ)
|
||||
env.pop("GIT_DIR", None)
|
||||
return env
|
||||
|
||||
|
||||
def ensure_repo() -> None:
|
||||
"""初始化仓库、写 .gitignore、设置 remote 与提交身份。"""
|
||||
if not (PROJECT_DIR / ".git").is_dir():
|
||||
_git("init", "-b", "main")
|
||||
(PROJECT_DIR / ".gitignore").write_text(GITIGNORE)
|
||||
|
||||
remotes = _git("remote").stdout.split()
|
||||
if "origin" in remotes:
|
||||
_git("remote", "set-url", "origin", GITEA_REMOTE)
|
||||
else:
|
||||
_git("remote", "add", "origin", GITEA_REMOTE)
|
||||
|
||||
if not _git("config", "user.name", check=False).stdout.strip():
|
||||
_git("config", "user.name", "DF Annals Bot")
|
||||
if not _git("config", "user.email", check=False).stdout.strip():
|
||||
_git("config", "user.email", "df-annals@hajim1.art")
|
||||
|
||||
|
||||
def commit_and_push(message: str, paths: list[Path] | None = None) -> str:
|
||||
"""提交并推送。返回提交哈希;无改动时返回空串。"""
|
||||
if paths:
|
||||
for p in paths:
|
||||
_git("add", "--", str(p.relative_to(PROJECT_DIR)) if p.is_absolute() else str(p))
|
||||
else:
|
||||
_git("add", "-A")
|
||||
if not _git("status", "--porcelain").stdout.strip():
|
||||
return ""
|
||||
_git("commit", "-m", message)
|
||||
head = _git("rev-parse", "--short", "HEAD").stdout.strip()
|
||||
_git("push", "-u", "origin", "HEAD:main")
|
||||
return head
|
||||
@@ -0,0 +1,140 @@
|
||||
"""事件显著度评分与小人物过滤。
|
||||
|
||||
真实数据给的教训:40 万条事件里绝大多数是流水账
|
||||
(change hf state 7.3 万、change hf job 5.8 万、written content composed 3.9 万),
|
||||
直接按类型切章会得到 5910 章,毫无叙事价值。
|
||||
|
||||
因此策略是:
|
||||
1. 只保留天然有叙事张力的事件类型(攻城、灭国、神器、死亡、结盟……)
|
||||
2. 再按"参与者有多重要"做二次排序,让编年史跟着大人物走
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter
|
||||
|
||||
from dfannals.legends import Event, World
|
||||
|
||||
# 事件类型 → 基础分。0 分表示不单独叙事(流水账)
|
||||
NARRATIVE: dict[str, int] = {
|
||||
# ---- 天崩地裂级
|
||||
"destroyed site": 100,
|
||||
"razed structure": 100,
|
||||
"entity overthrown": 100,
|
||||
"hf destroyed site": 95,
|
||||
"plundered site": 95,
|
||||
"field battle": 95,
|
||||
"entity dissolved": 90,
|
||||
"holy city declaration": 90,
|
||||
"site taken over": 90,
|
||||
"entity breach feature layer": 90,
|
||||
"entity alliance formed": 85,
|
||||
"peace accepted": 85,
|
||||
"hf enslaved": 85,
|
||||
"hf revived": 80,
|
||||
"entity relocate": 80,
|
||||
"hf ransomed": 80,
|
||||
"peace rejected": 78,
|
||||
# ---- 重大
|
||||
"created site": 75,
|
||||
"reclaim site": 72,
|
||||
"attacked site": 70,
|
||||
"hf attacked site": 70,
|
||||
"entity created": 70,
|
||||
"artifact destroyed": 70,
|
||||
"new site leader": 65,
|
||||
"created world construction": 60,
|
||||
"entity persecuted": 60,
|
||||
"hf convicted": 60,
|
||||
"site dispute": 55,
|
||||
"failed intrigue corruption": 55,
|
||||
"artifact recovered": 60,
|
||||
"artifact possessed": 60,
|
||||
# ---- 值得一记
|
||||
"artifact created": 45,
|
||||
"artifact lost": 45,
|
||||
"creature devoured": 42,
|
||||
"hf abducted": 40,
|
||||
"assume identity": 40,
|
||||
"knowledge discovered": 38,
|
||||
"artifact given": 36,
|
||||
"agreement formed": 36,
|
||||
"hf gains secret goal": 36,
|
||||
"item stolen": 34,
|
||||
"hf performed horrible experiments": 45,
|
||||
"failed frame attempt": 40,
|
||||
"hf profaned structure": 38,
|
||||
"changed creature type": 36,
|
||||
"hf interrogated": 34,
|
||||
"created structure": 30,
|
||||
"hf wounded": 30,
|
||||
"hf learns secret": 34,
|
||||
"changed hf body state": 30,
|
||||
}
|
||||
|
||||
# 死亡单独处理:分数取决于死者本身有多重要
|
||||
DEATH_TYPES = ("hf died",)
|
||||
|
||||
CATEGORIES: list[tuple[tuple[str, ...], str]] = [
|
||||
(("war", "battle", "siege", "raid", "attack", "plunder", "destroy", "sack", "taken over"), "战事"),
|
||||
(("died", "death", "slain", "murder", "assassin", "enslaved", "ransomed", "revived"), "生死"),
|
||||
(("created site", "reclaim", "relocate", "dissolved", "overthrown", "alliance", "entity created"), "兴亡"),
|
||||
(("artifact", "masterpiece", "construction", "structure", "razed"), "器物"),
|
||||
(("peace", "agreement", "tribute", "persecuted", "convicted", "dispute", "intrigue"), "政争"),
|
||||
(("knowledge", "secret", "abducted", "identity", "experiment", "profaned", "devoured"), "异闻"),
|
||||
]
|
||||
|
||||
|
||||
def category(event: Event) -> str:
|
||||
t = event.type.lower()
|
||||
for keys, label in CATEGORIES:
|
||||
if any(k in t for k in keys):
|
||||
return label
|
||||
return "杂记"
|
||||
|
||||
|
||||
def prominence(world: World) -> Counter[int]:
|
||||
"""每个人物被卷入的事件次数——用来衡量他在史料里有多重要。"""
|
||||
counts: Counter[int] = Counter()
|
||||
for e in world.events:
|
||||
seen: set[int] = set()
|
||||
for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity", "worshipper_hfid"):
|
||||
for fid in e.ids(tag):
|
||||
seen.add(fid)
|
||||
counts.update(seen)
|
||||
return counts
|
||||
|
||||
|
||||
def _participants(event: Event) -> list[int]:
|
||||
out: list[int] = []
|
||||
for tag in ("hfid", "slayer_hfid", "target_hfid", "source_hfid", "deity"):
|
||||
out.extend(event.ids(tag))
|
||||
return out
|
||||
|
||||
|
||||
# 人物重要度阈值:参与事件数达到这个量级才算"要角"
|
||||
PROMINENT = 60
|
||||
|
||||
|
||||
def significance(event: Event, prom: Counter[int] | None = None, prominent: int = PROMINENT) -> int:
|
||||
"""事件的故事价值。0 表示不值得单独写。"""
|
||||
t = event.type.lower()
|
||||
base = NARRATIVE.get(t, 0)
|
||||
people = _participants(event)
|
||||
top = max((prom.get(p, 0) for p in people), default=0) if prom else 0
|
||||
|
||||
if t in DEATH_TYPES:
|
||||
# 无名死者不写;要角之死按重要度加分
|
||||
if top < prominent:
|
||||
return 0
|
||||
return 55 + min(45, top // 10)
|
||||
|
||||
if base == 0:
|
||||
return 0
|
||||
|
||||
# 参与者越重要,越值得写;但不要盖过类型本身的分量
|
||||
bonus = min(25, top // 25) if prom else 0
|
||||
return base + bonus
|
||||
|
||||
|
||||
def is_narrative(event: Event, prom: Counter[int] | None = None) -> bool:
|
||||
return significance(event, prom) > 0
|
||||
@@ -0,0 +1,195 @@
|
||||
"""把史料切成章节素材。
|
||||
|
||||
真实数据(403853 条事件 / 250 年)证明:按"重要事件数量"切章会得到几千章。
|
||||
所以改为先按时间窗分桶(每窗 10 年),再在窗内按显著度取前 N 条——
|
||||
这样一章天然覆盖一段年月,既有大事件链,也不会漏掉时代的推进。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from dfannals import legends as lg
|
||||
from dfannals import score
|
||||
from dfannals.legends import Event, World
|
||||
|
||||
YEARS_PER_CHAPTER = 10 # 一个时间窗覆盖的年数
|
||||
EVENTS_PER_CHAPTER = 12 # 每章最多收录的史料条目
|
||||
MIN_EVENTS_PER_CHAPTER = 3 # 少于这个数量就并入上一章(太平年月)
|
||||
MAX_PER_CATEGORY = 4 # 同一类事件每章上限,避免整章都是讣告
|
||||
|
||||
|
||||
@dataclass
|
||||
class Chapter:
|
||||
index: int
|
||||
start_year: int
|
||||
end_year: int
|
||||
events: list[Event] = field(default_factory=list)
|
||||
quiet_years: list[tuple[int, int]] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def key(self) -> str:
|
||||
return f"{self.index:03d}-{self.start_year}-{self.end_year}"
|
||||
|
||||
@property
|
||||
def span(self) -> str:
|
||||
if self.start_year == self.end_year:
|
||||
return f"{self.start_year} 年"
|
||||
return f"{self.start_year}–{self.end_year} 年"
|
||||
|
||||
def categories(self) -> list[str]:
|
||||
seen: list[str] = []
|
||||
for e in self.events:
|
||||
c = score.category(e)
|
||||
if c not in seen:
|
||||
seen.append(c)
|
||||
return seen
|
||||
|
||||
|
||||
def plan(
|
||||
world: World,
|
||||
years_per_chapter: int = YEARS_PER_CHAPTER,
|
||||
per_chapter: int = EVENTS_PER_CHAPTER,
|
||||
min_events: int = MIN_EVENTS_PER_CHAPTER,
|
||||
max_per_category: int = MAX_PER_CATEGORY,
|
||||
) -> list[Chapter]:
|
||||
prom = score.prominence(world)
|
||||
scored = [(score.significance(e, prom), e) for e in world.events]
|
||||
scored = [pair for pair in scored if pair[0] > 0]
|
||||
if not scored:
|
||||
lo, hi = world.event_years()
|
||||
return [Chapter(index=1, start_year=lo, end_year=hi)]
|
||||
|
||||
lo = min(e.year for _, e in scored)
|
||||
hi = max(e.year for _, e in scored)
|
||||
|
||||
buckets: dict[int, list[tuple[int, Event]]] = defaultdict(list)
|
||||
for sig, e in scored:
|
||||
buckets[e.year // years_per_chapter].append((sig, e))
|
||||
|
||||
chapters: list[Chapter] = []
|
||||
for b in range(lo // years_per_chapter, hi // years_per_chapter + 1):
|
||||
window_start = max(b * years_per_chapter, lo)
|
||||
window_end = min(b * years_per_chapter + years_per_chapter - 1, hi)
|
||||
ranked = sorted(buckets.get(b, []), key=lambda p: (-p[0], p[1].year, p[1].id))
|
||||
# 按类别限额挑选:让一章里有战事、兴亡、器物,而不是 12 条死亡讣告
|
||||
picks: list[tuple[int, Event]] = []
|
||||
used: dict[str, int] = {}
|
||||
for sig, ev in ranked:
|
||||
cat = score.category(ev)
|
||||
if used.get(cat, 0) >= max_per_category:
|
||||
continue
|
||||
used[cat] = used.get(cat, 0) + 1
|
||||
picks.append((sig, ev))
|
||||
if len(picks) >= per_chapter:
|
||||
break
|
||||
events = sorted((e for _, e in picks), key=lambda e: (e.year, e.seconds72, e.id))
|
||||
|
||||
if not events:
|
||||
if chapters:
|
||||
chapters[-1].quiet_years.append((window_start, window_end))
|
||||
chapters[-1].end_year = window_end
|
||||
continue
|
||||
|
||||
if len(events) < min_events and chapters:
|
||||
chapters[-1].events.extend(events)
|
||||
chapters[-1].end_year = window_end
|
||||
continue
|
||||
|
||||
chapters.append(
|
||||
Chapter(index=len(chapters) + 1, start_year=window_start, end_year=window_end, events=events)
|
||||
)
|
||||
|
||||
for i, c in enumerate(chapters, 1):
|
||||
c.index = i
|
||||
return chapters
|
||||
|
||||
|
||||
def _figure_line(world: World, fid: int, prom) -> str:
|
||||
fig = world.figures.get(fid)
|
||||
if not fig:
|
||||
return f"- HF#{fid}(史料无记载)"
|
||||
bits = [lg.pretty(fig.name) or f"HF#{fid}", fig.race or "未知种族", fig.alive_span]
|
||||
if fig.profession:
|
||||
bits.append(fig.profession)
|
||||
bits.append(f"史料出现 {prom.get(fid, 0)} 次")
|
||||
return "- " + " | ".join(bits)
|
||||
|
||||
|
||||
def material(world: World, chapter: Chapter, max_figures: int = 40) -> str:
|
||||
"""把一章的史料渲染成给模型看的素材文本。"""
|
||||
prom = score.prominence(world)
|
||||
lines: list[str] = [
|
||||
f"## 本章范围:{chapter.span}",
|
||||
"",
|
||||
f"共 {len(chapter.events)} 条史料,类型:{('、'.join(chapter.categories())) or '无'}。",
|
||||
"",
|
||||
"### 事件流水(按时间排序,专名保留游戏原文)",
|
||||
"",
|
||||
]
|
||||
|
||||
participation: dict[int, int] = {}
|
||||
for e in chapter.events:
|
||||
chunks = [f"{e.year}年", e.type]
|
||||
for tag, kind in lg.ID_FIELDS.items():
|
||||
for i in e.ids(tag):
|
||||
# id 为负是“无此项”,不要渲染成 HF#-1 这类噪音
|
||||
if kind == "figure":
|
||||
name = world.figure_name(i)
|
||||
elif kind == "site":
|
||||
name = world.site_name(i)
|
||||
elif kind == "entity":
|
||||
name = world.entity_name(i)
|
||||
else:
|
||||
name = ""
|
||||
if not name:
|
||||
continue
|
||||
chunks.append(f"{tag}={name}")
|
||||
if kind == "figure":
|
||||
participation[i] = participation.get(i, 0) + 1
|
||||
extra = []
|
||||
for field_name in ("state", "reason", "circumstance"):
|
||||
v = e.text(field_name)
|
||||
if v and v not in ("-1", ""):
|
||||
extra.append(f"{field_name}={v}")
|
||||
if extra:
|
||||
chunks.append("·".join(extra))
|
||||
lines.append("- " + " · ".join(chunks))
|
||||
|
||||
if chapter.quiet_years:
|
||||
spans = "、".join(f"{a}–{b} 年" for a, b in chapter.quiet_years)
|
||||
lines += [
|
||||
"",
|
||||
"### 太平年月(史料无值得立传的事件)",
|
||||
"",
|
||||
f"- {spans}:这段时期只留下琐碎记录,可一笔带过。",
|
||||
]
|
||||
|
||||
lines += ["", "### 本章登场人物(史料中的身份信息)", ""]
|
||||
for fid, _ in sorted(participation.items(), key=lambda kv: -prom.get(kv[0], 0))[:max_figures]:
|
||||
lines.append(_figure_line(world, fid, prom))
|
||||
|
||||
sites = sorted({n for e in chapter.events for i in e.ids("site_id") if (n := world.site_name(i))})
|
||||
if sites:
|
||||
lines += ["", "### 相关地点", "", "- " + "、".join(sites)]
|
||||
|
||||
ents = sorted({n for e in chapter.events for i in e.ids("entity_id") if (n := world.entity_name(i))})
|
||||
if ents:
|
||||
lines += ["", "### 相关势力", "", "- " + "、".join(ents)]
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def overview(world: World, chapters: list[Chapter], limit: int = 10) -> str:
|
||||
lo, hi = world.event_years()
|
||||
lines = [
|
||||
f"世界:{world.name}" + (f"({world.altname})" if world.altname else ""),
|
||||
f"历史:{lo}–{hi} 年 | 史料事件 {len(world.events)} 条 | "
|
||||
f"人物 {len(world.figures)} | 势力 {len(world.entities)} | 地点 {len(world.sites)}",
|
||||
f"规划章节:{len(chapters)} 章(每章约 {YEARS_PER_CHAPTER} 年 / 最多 {EVENTS_PER_CHAPTER} 条史料)",
|
||||
]
|
||||
for c in chapters[:limit]:
|
||||
lines.append(f" 第 {c.index} 章 {c.span}:{len(c.events)} 条({('、'.join(c.categories())) or '无'})")
|
||||
if len(chapters) > limit:
|
||||
lines.append(f" …… 其余 {len(chapters) - limit} 章略")
|
||||
return "\n".join(lines)
|
||||
@@ -0,0 +1,108 @@
|
||||
# pyright: reportMissingImports=false, reportAttributeAccessIssue=false
|
||||
# 说明:LSP 的 Python 环境看不到本项目包(未安装进它的 site-packages),
|
||||
# 会把「包内互相 import」误报为缺失模块。运行时导入已由执行验证通过。
|
||||
|
||||
"""生成迷你年表与人物索引(随成品入库的两种辅助文件)。"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
from dfannals import legends as lg
|
||||
from dfannals import score
|
||||
from dfannals.legends import Event, World
|
||||
|
||||
# 评分标准统一在 dfannals/score.py,年表与切章共用同一套。
|
||||
significance = score.significance
|
||||
category = score.category
|
||||
|
||||
|
||||
def event_line(world: World, e: Event) -> str:
|
||||
"""把一条事件压成一行可读文字。"""
|
||||
chunks: list[str] = [e.type]
|
||||
for tag, kind in lg.ID_FIELDS.items():
|
||||
for i in e.ids(tag):
|
||||
if kind == "figure":
|
||||
name = world.figure_name(i)
|
||||
elif kind == "site":
|
||||
name = world.site_name(i)
|
||||
elif kind == "entity":
|
||||
name = world.entity_name(i)
|
||||
else:
|
||||
name = ""
|
||||
if name:
|
||||
chunks.append(f"{tag}={name}")
|
||||
return f"- **{e.year}** [{category(e)}] " + " · ".join(chunks)
|
||||
|
||||
|
||||
def build_timeline(world: World, min_score: int = 70, per_decade: int = 14) -> str:
|
||||
"""按十年分组的重要事件年表。"""
|
||||
prom = score.prominence(world)
|
||||
lo, hi = world.event_years()
|
||||
important = [e for e in world.events if significance(e, prom) >= min_score]
|
||||
buckets: dict[int, list[Event]] = {}
|
||||
for e in important:
|
||||
buckets.setdefault((e.year // 10) * 10, []).append(e)
|
||||
|
||||
lines = [
|
||||
f"# {world.name} · 年表",
|
||||
"",
|
||||
f"- 历史跨度:{lo}–{hi} 年" if hi else "- 历史跨度:未知",
|
||||
f"- 事件总数:{len(world.events)},其中重要事件 {len(important)}",
|
||||
f"- 历史人物 {len(world.figures)} 位 · 文明与组织 {len(world.entities)} 个 · 地点 {len(world.sites)} 处",
|
||||
"",
|
||||
"> 本表由 legends 导出数据自动生成,只收录叙事权重较高的事件;"
|
||||
"完整数据留在本地,不入库。",
|
||||
"",
|
||||
]
|
||||
for decade in sorted(buckets):
|
||||
evs = sorted(buckets[decade], key=lambda e: (e.year, e.seconds72, e.id))
|
||||
shown = sorted(evs, key=lambda e: -significance(e, prom))[:per_decade]
|
||||
shown.sort(key=lambda e: (e.year, e.seconds72, e.id))
|
||||
lines.append(f"## {decade}s")
|
||||
lines.append("")
|
||||
lines.extend(event_line(world, e) for e in shown)
|
||||
if len(evs) > len(shown):
|
||||
lines.append(f"- …另有 {len(evs) - len(shown)} 条次要事件未列")
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def build_figures(world: World, top: int = 120) -> str:
|
||||
"""人物索引:按参与事件数排序,给出身份与生卒。"""
|
||||
involvement: Counter[int] = Counter()
|
||||
for e in world.events:
|
||||
seen: set[int] = set()
|
||||
for tag, kind in lg.ID_FIELDS.items():
|
||||
if kind != "figure":
|
||||
continue
|
||||
for i in e.ids(tag):
|
||||
seen.add(i)
|
||||
involvement.update(seen)
|
||||
|
||||
lines = [
|
||||
f"# {world.name} · 人物索引",
|
||||
"",
|
||||
f"共 {len(world.figures)} 位历史人物,按在史料中出现的次数排序(前 {top} 位)。",
|
||||
"",
|
||||
"| 人物 | 种族 | 生卒 | 职业 | 事件数 |",
|
||||
"|---|---|---|---|---|",
|
||||
]
|
||||
ranked = sorted(world.figures.values(), key=lambda f: -involvement.get(f.id, 0))
|
||||
for fig in ranked[:top]:
|
||||
name = lg.pretty(fig.name) or f"HF#{fig.id}"
|
||||
prof = fig.profession or "—"
|
||||
lines.append(
|
||||
f"| {name} | {fig.race or '—'} | {fig.alive_span} | {prof} | {involvement.get(fig.id, 0)} |"
|
||||
)
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def write_outputs(world: World, out_dir: Path, min_score: int = 40) -> list[Path]:
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
timeline = out_dir / "timeline.md"
|
||||
figures = out_dir / "figures.md"
|
||||
timeline.write_text(build_timeline(world, min_score=min_score), encoding="utf-8")
|
||||
figures.write_text(build_figures(world), encoding="utf-8")
|
||||
return [timeline, figures]
|
||||
Reference in New Issue
Block a user