Files
Chen Yi 8630dcad55 文笔层:AI 味机检、标点规范化、分块大量扩写(第 1 集成稿 10805 字)
- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行)
- 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性)
- 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告)
- 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记)
- 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写
- 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名
- episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照
- 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试
- 提交前拦截「已跟踪文件为空」,防止上述事故复发
2026-10-05 22:27:02 +08:00

110 lines
3.9 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""调用用户自建网关(OpenAI 兼容)生成文本。
要点:
- 密钥运行时从 Pi 的 auth.json 读取,只在内存中,永不打印或写入产物。
- 目标是推理模型,思考与正文共用 max_tokens,因此默认给足预算。
- 传输层用 http.client(方案显式限定 http/https),不依赖第三方库。
"""
from __future__ import annotations
import http.client
import json
import time
from dataclasses import dataclass
from urllib.parse import urlsplit
from dfannals.config import MAX_TOKENS_CHAPTER, MODEL, Gateway, load_gateway
@dataclass
class Reply:
content: str
reasoning_tokens: int
total_tokens: int
model: str
class GatewayError(RuntimeError):
pass
def _endpoint(base_url: str) -> tuple[type[http.client.HTTPConnection], str, str]:
"""解析网关地址,只接受 http/https。"""
parts = urlsplit(base_url)
if parts.scheme not in ("http", "https"):
raise GatewayError(f"网关地址必须是 http(s):{base_url!r}")
if not parts.netloc:
raise GatewayError(f"网关地址缺少主机:{base_url!r}")
cls = http.client.HTTPSConnection if parts.scheme == "https" else http.client.HTTPConnection
path = parts.path.rstrip("/") + "/chat/completions"
return cls, parts.netloc, path
def chat(
messages: list[dict],
*,
gateway: Gateway | None = None,
model: str = MODEL,
max_tokens: int = MAX_TOKENS_CHAPTER,
temperature: float = 1.0,
timeout: int = 900,
retries: int = 5,
) -> Reply:
gw = gateway or load_gateway()
conn_cls, host, path = _endpoint(gw.base_url)
body = json.dumps(
{
"model": model,
"messages": messages,
"max_tokens": max_tokens,
"temperature": temperature,
}
).encode()
headers = {
"Content-Type": "application/json",
# 密钥仅在此处作为请求头使用,不落盘、不打印
"Authorization": "Bearer " + gw.key,
}
last: str | None = None
for attempt in range(1, retries + 1):
conn = conn_cls(host, timeout=timeout)
try:
conn.request("POST", path, body=body, headers=headers)
resp = conn.getresponse()
raw = resp.read()
if resp.status != 200:
last = f"HTTP {resp.status}: {raw[:400].decode('utf-8', 'replace')}"
# 4xx(除 429 限流)是请求本身的问题,重试无意义
if resp.status < 500 and resp.status != 429:
break
else:
data = json.loads(raw)
choice = data["choices"][0]["message"]
usage = data.get("usage") or {}
content = (choice.get("content") or "").strip()
if not content:
raise GatewayError(
"模型只产出了思考、没有正文;"
f"thinking_tokens={usage.get('completion_thinking_tokens')},请加大 max_tokens"
)
return Reply(
content=content,
reasoning_tokens=int(usage.get("completion_thinking_tokens") or 0),
total_tokens=int(usage.get("total_tokens") or 0),
model=data.get("model", model),
)
except (OSError, http.client.HTTPException, json.JSONDecodeError, KeyError) as exc:
last = f"{type(exc).__name__}: {exc}"
except GatewayError as exc:
last = str(exc)
break
finally:
conn.close()
if attempt < retries:
# 实测网关偶发 502 upstream_unavailable(尤其在大请求上),
# 所以 5xx/429/超时都退避重试,上限 60 秒。
time.sleep(min(60, 5 * 2 ** (attempt - 1)))
raise GatewayError(f"网关调用失败({gw.base_url},model={model}):{last}")