"""进度状态与中文篇幅统计。 (旧版这里还有整套「按年代切章 → 写编年史」的逻辑;那条路径已经被人物主线取代, 相关代码与提示词一并删除,只留下仍然需要的东西。) """ from __future__ import annotations import json import re import unicodedata from dataclasses import dataclass, field from pathlib import Path from dfannals.config import DATA_DIR STATE_FILE = DATA_DIR / "state.json" @dataclass class State: world: str = "" volume: int = 1 done: list[str] = field(default_factory=list) @classmethod def load(cls, path: Path = STATE_FILE) -> State: if not path.is_file(): return cls() try: data = json.loads(path.read_text(encoding="utf-8")) except (json.JSONDecodeError, OSError): return cls() if not isinstance(data, dict): return cls() return cls( world=data.get("world", ""), volume=data.get("volume", 1), done=list(data.get("done", [])), ) def save(self, path: Path = STATE_FILE) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text( json.dumps({"world": self.world, "volume": self.volume, "done": self.done}, ensure_ascii=False, indent=2), encoding="utf-8", ) def mark_done(self, key: str) -> None: if key not in self.done: self.done.append(key) def cjk_length(text: str) -> int: """按中文习惯计字数:CJK 字符、拉丁字母数字、中文标点各算 1。 统计前先去掉代码块与 Markdown 行首标记(标题、引用、列表符), 使结果接近读者感知的"正文字数"。 """ without_code = re.sub(r"```.*?```", "", text, flags=re.DOTALL) without_meta = re.sub(r"^\s*[#>|\-*].*$", "", without_code, flags=re.MULTILINE) n = 0 for ch in without_meta: if ( unicodedata.east_asian_width(ch) in ("W", "F") or ch.isalnum() or ch in ",。!?;:、" ): n += 1 return n