- 修复上次提交把 dfannals/cli.py 写成 0 字节的问题(它是唯一入口,导致管道不可运行) - 性别:解析 <caste>,人物表与写作素材带性别(Ral Fastenhatchets 实为女性) - 新增 dfannals/deslop.py:AI 味机械诊断(硬伤词/句式/标点,按千字密度报告) - 新增 dfannals/normalize.py:标点与结构清理(引号配对、重复段落与句子、模型自加的小节标记) - 新增 dfannals/expand.py 与 prompts/literary-expander.md:按年份场景分块大量扩写 - 专名防幻觉:每块附史料专名白名单,事后按段自动修复可疑专名 - episode 命令并进文笔层(骨架稿另存 .skeleton.md),新增 expand 命令做 A/B 对照 - 新增 notes/switched-threads.md 与 test_deslop / test_normalize 回归测试 - 提交前拦截「已跟踪文件为空」,防止上述事故复发
110 lines
3.9 KiB
Python
110 lines
3.9 KiB
Python
"""调用用户自建网关(OpenAI 兼容)生成文本。
|
||
|
||
要点:
|
||
- 密钥运行时从 Pi 的 auth.json 读取,只在内存中,永不打印或写入产物。
|
||
- 目标是推理模型,思考与正文共用 max_tokens,因此默认给足预算。
|
||
- 传输层用 http.client(方案显式限定 http/https),不依赖第三方库。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import http.client
|
||
import json
|
||
import time
|
||
from dataclasses import dataclass
|
||
from urllib.parse import urlsplit
|
||
|
||
from dfannals.config import MAX_TOKENS_CHAPTER, MODEL, Gateway, load_gateway
|
||
|
||
|
||
@dataclass
|
||
class Reply:
|
||
content: str
|
||
reasoning_tokens: int
|
||
total_tokens: int
|
||
model: str
|
||
|
||
|
||
class GatewayError(RuntimeError):
|
||
pass
|
||
|
||
|
||
def _endpoint(base_url: str) -> tuple[type[http.client.HTTPConnection], str, str]:
|
||
"""解析网关地址,只接受 http/https。"""
|
||
parts = urlsplit(base_url)
|
||
if parts.scheme not in ("http", "https"):
|
||
raise GatewayError(f"网关地址必须是 http(s):{base_url!r}")
|
||
if not parts.netloc:
|
||
raise GatewayError(f"网关地址缺少主机:{base_url!r}")
|
||
cls = http.client.HTTPSConnection if parts.scheme == "https" else http.client.HTTPConnection
|
||
path = parts.path.rstrip("/") + "/chat/completions"
|
||
return cls, parts.netloc, path
|
||
|
||
|
||
def chat(
|
||
messages: list[dict],
|
||
*,
|
||
gateway: Gateway | None = None,
|
||
model: str = MODEL,
|
||
max_tokens: int = MAX_TOKENS_CHAPTER,
|
||
temperature: float = 1.0,
|
||
timeout: int = 900,
|
||
retries: int = 5,
|
||
) -> Reply:
|
||
gw = gateway or load_gateway()
|
||
conn_cls, host, path = _endpoint(gw.base_url)
|
||
body = json.dumps(
|
||
{
|
||
"model": model,
|
||
"messages": messages,
|
||
"max_tokens": max_tokens,
|
||
"temperature": temperature,
|
||
}
|
||
).encode()
|
||
headers = {
|
||
"Content-Type": "application/json",
|
||
# 密钥仅在此处作为请求头使用,不落盘、不打印
|
||
"Authorization": "Bearer " + gw.key,
|
||
}
|
||
|
||
last: str | None = None
|
||
for attempt in range(1, retries + 1):
|
||
conn = conn_cls(host, timeout=timeout)
|
||
try:
|
||
conn.request("POST", path, body=body, headers=headers)
|
||
resp = conn.getresponse()
|
||
raw = resp.read()
|
||
if resp.status != 200:
|
||
last = f"HTTP {resp.status}: {raw[:400].decode('utf-8', 'replace')}"
|
||
# 4xx(除 429 限流)是请求本身的问题,重试无意义
|
||
if resp.status < 500 and resp.status != 429:
|
||
break
|
||
else:
|
||
data = json.loads(raw)
|
||
choice = data["choices"][0]["message"]
|
||
usage = data.get("usage") or {}
|
||
content = (choice.get("content") or "").strip()
|
||
if not content:
|
||
raise GatewayError(
|
||
"模型只产出了思考、没有正文;"
|
||
f"thinking_tokens={usage.get('completion_thinking_tokens')},请加大 max_tokens"
|
||
)
|
||
return Reply(
|
||
content=content,
|
||
reasoning_tokens=int(usage.get("completion_thinking_tokens") or 0),
|
||
total_tokens=int(usage.get("total_tokens") or 0),
|
||
model=data.get("model", model),
|
||
)
|
||
except (OSError, http.client.HTTPException, json.JSONDecodeError, KeyError) as exc:
|
||
last = f"{type(exc).__name__}: {exc}"
|
||
except GatewayError as exc:
|
||
last = str(exc)
|
||
break
|
||
finally:
|
||
conn.close()
|
||
if attempt < retries:
|
||
# 实测网关偶发 502 upstream_unavailable(尤其在大请求上),
|
||
# 所以 5xx/429/超时都退避重试,上限 60 秒。
|
||
time.sleep(min(60, 5 * 2 ** (attempt - 1)))
|
||
|
||
raise GatewayError(f"网关调用失败({gw.base_url},model={model}):{last}")
|