- 移除 output/volumes/v1/chapters/001-1-9.md 与 002-10-19.md(仅普通提交,git 历史保留) - 移除被取代的写作规范 prompts/chronicler.md 与旧的按年代切章模块 - README 全面改写为人物主线流程、参数位置与踩坑记录 - scripts/dfannals 启动器:自动挑一个装了 defusedxml 的解释器
71 lines
2.1 KiB
Python
71 lines
2.1 KiB
Python
"""进度状态与中文篇幅统计。
|
|
|
|
(旧版这里还有整套「按年代切章 → 写编年史」的逻辑;那条路径已经被人物主线取代,
|
|
相关代码与提示词一并删除,只留下仍然需要的东西。)
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
import unicodedata
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
|
|
from dfannals.config import DATA_DIR
|
|
|
|
STATE_FILE = DATA_DIR / "state.json"
|
|
|
|
|
|
@dataclass
|
|
class State:
|
|
world: str = ""
|
|
volume: int = 1
|
|
done: list[str] = field(default_factory=list)
|
|
|
|
@classmethod
|
|
def load(cls, path: Path = STATE_FILE) -> State:
|
|
if not path.is_file():
|
|
return cls()
|
|
try:
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
except (json.JSONDecodeError, OSError):
|
|
return cls()
|
|
if not isinstance(data, dict):
|
|
return cls()
|
|
return cls(
|
|
world=data.get("world", ""),
|
|
volume=data.get("volume", 1),
|
|
done=list(data.get("done", [])),
|
|
)
|
|
|
|
def save(self, path: Path = STATE_FILE) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(
|
|
json.dumps({"world": self.world, "volume": self.volume, "done": self.done},
|
|
ensure_ascii=False, indent=2),
|
|
encoding="utf-8",
|
|
)
|
|
|
|
def mark_done(self, key: str) -> None:
|
|
if key not in self.done:
|
|
self.done.append(key)
|
|
|
|
|
|
def cjk_length(text: str) -> int:
|
|
"""按中文习惯计字数:CJK 字符、拉丁字母数字、中文标点各算 1。
|
|
|
|
统计前先去掉代码块与 Markdown 行首标记(标题、引用、列表符),
|
|
使结果接近读者感知的"正文字数"。
|
|
"""
|
|
without_code = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
|
|
without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.MULTILINE)
|
|
n = 0
|
|
for ch in without_meta:
|
|
if (
|
|
unicodedata.east_asian_width(ch) in ("W", "F")
|
|
or ch.isalnum()
|
|
or ch in ",。!?;:、"
|
|
):
|
|
n += 1
|
|
return n
|