fix(llm): 收编波五件——L4/L5 累积降级链+配平起点避字符串内左括号+响应体单次解析连接复用+400 自愈不占重试预算+usage 上游未给 None 化 (审计 P3-1..5 收编) [vps] [no-doc]

This commit is contained in:
2026-10-05 22:26:17 +08:00
parent ce5b8f59c2
commit 472559380e
5 changed files with 181 additions and 64 deletions
+72 -51
View File
@@ -35,15 +35,16 @@ class LLMClient:
async def chat_json(self, messages: list[dict], *, temperature: float = 0.2,
max_tokens: int = 2000,
want_usage: bool = False) -> dict | tuple[dict, dict]:
want_usage: bool = False) -> dict | tuple[dict, dict | None]:
"""一轮对话→dict;want_usage=True 返 (dict, usage)(P4-3 漏斗 token 计量,
usage 含 prompt_tokens/completion_tokens)。默认调用形态零变化。"""
usage 含 prompt_tokens/completion_tokens;P3-5: 上游没给 usage 时为
None——不再伪装 {} 与「0 token」不可分)。默认调用形态零变化。"""
obj, usage = await self._chat_json(messages, temperature=temperature,
max_tokens=max_tokens)
return (obj, usage) if want_usage else obj
async def _chat_json(self, messages: list[dict], *, temperature: float,
max_tokens: int) -> tuple[dict, dict]:
max_tokens: int) -> tuple[dict, dict | None]:
payload: dict = {
"model": self._cfg.model,
"messages": messages,
@@ -53,58 +54,79 @@ class LLMClient:
}
healed = {"response_format": False, "temperature": False}
attempts = 0
while True:
attempts += 1
try:
status, text, headers = await self._post(payload)
except httpx.TransportError as e:
# 传输级故障(连接拒绝/超时/断连)视作一次失败 attempt,同款退避
if attempts >= _MAX_ATTEMPTS:
raise LLMError(f"LLM 连接失败: {e}") from e
await self._sleep(_BACKOFF_SECONDS[min(attempts - 1, 1)])
continue
if status == 200:
usage = json.loads(text).get("usage") or {}
logger.info("llm tokens prompt=%s completion=%s",
usage.get("prompt_tokens"),
usage.get("completion_tokens"))
# P3-3: 单 AsyncClient 复用整轮(含自愈重试/strict 重试),不再每
# attempt 新建连接池
async with httpx.AsyncClient(timeout=self._cfg.timeout,
transport=self._transport) as client:
while True:
attempts += 1
try:
return robust_json_parse(self._content(text)), usage
except JSONParseError:
return await self._strict_retry(messages, payload)
if status in (401, 403, 404):
raise LLMError(f"LLM 鉴权/端点错误(HTTP {status}):检查 "
f"SANGUO_LLM_API_KEY/SANGUO_LLM_BASE_URL")
if status == 400:
if ("response_format" in text
and not healed["response_format"]):
payload.pop("response_format", None)
healed["response_format"] = True
status, text, headers = await self._post(client, payload)
except httpx.TransportError as e:
# 传输级故障(连接拒绝/超时/断连)视作一次失败 attempt,同款退避
if attempts >= _MAX_ATTEMPTS:
raise LLMError(f"LLM 连接失败: {e}") from e
await self._sleep(_BACKOFF_SECONDS[min(attempts - 1, 1)])
continue
if ("temperature" in text and not healed["temperature"]):
payload.pop("temperature", None)
healed["temperature"] = True
if status == 200:
# P3-3: 响应体只 json.loads 一次,usage/content 同源传递
try:
body = json.loads(text)
except ValueError as e:
raise LLMError(f"LLM 响应非 JSON: {e}") from e
usage = body.get("usage")
if not isinstance(usage, dict):
usage = None # P3-5: 缺失/异形返 None,不伪装 0 token
logger.info("llm tokens prompt=%s completion=%s",
(usage or {}).get("prompt_tokens"),
(usage or {}).get("completion_tokens"))
try:
return robust_json_parse(self._content(body)), usage
except JSONParseError:
return await self._strict_retry(client, messages,
payload)
if status in (401, 403, 404):
raise LLMError(f"LLM 鉴权/端点错误(HTTP {status}):检查 "
f"SANGUO_LLM_API_KEY/SANGUO_LLM_BASE_URL")
if status == 400:
if ("response_format" in text
and not healed["response_format"]):
payload.pop("response_format", None)
healed["response_format"] = True
attempts -= 1 # P3-4: 自愈重发不占 429/5xx 重试预算
continue
if ("temperature" in text and not healed["temperature"]):
payload.pop("temperature", None)
healed["temperature"] = True
attempts -= 1 # P3-4: 同上
continue
raise LLMError(f"LLM 拒绝请求(400): {text[:200]}")
if status == 429 or status >= 500:
if attempts >= _MAX_ATTEMPTS:
raise LLMError(f"LLM 重试耗尽(HTTP {status}): {text[:200]}")
await self._sleep(self._retry_delay(status, headers, attempts))
continue
raise LLMError(f"LLM 拒绝请求(400): {text[:200]}")
if status == 429 or status >= 500:
if attempts >= _MAX_ATTEMPTS:
raise LLMError(f"LLM 重试耗尽(HTTP {status}): {text[:200]}")
await self._sleep(self._retry_delay(status, headers, attempts))
continue
raise LLMError(f"LLM 未预期状态 HTTP {status}: {text[:200]}")
raise LLMError(f"LLM 未预期状态 HTTP {status}: {text[:200]}")
async def _strict_retry(self, messages: list[dict], payload: dict) -> dict:
async def _strict_retry(self, client: httpx.AsyncClient,
messages: list[dict], payload: dict) -> dict:
payload = dict(payload, messages=messages + [
{"role": "user", "content": _STRICT_NOTE}])
try:
status, text, _headers = await self._post(payload)
status, text, _headers = await self._post(client, payload)
except httpx.TransportError as e:
raise LLMError(f"LLM strict 重试连接失败: {e}") from e
if status != 200:
raise LLMError(f"LLM strict 重试失败(HTTP {status}): {text[:200]}")
try:
usage = json.loads(text).get("usage") or {}
return robust_json_parse(self._content(text)), usage
body = json.loads(text) # P3-3: 同响应体单次解析
except ValueError as e:
raise LLMError(f"LLM strict 重试响应非 JSON: {e}") from e
usage = body.get("usage")
if not isinstance(usage, dict):
usage = None # P3-5: 同主链口径
try:
return robust_json_parse(self._content(body)), usage
except JSONParseError as e:
raise LLMError(f"LLM 返回非 JSON(两轮): {e}") from e
@@ -122,20 +144,19 @@ class LLMClient:
pass
return _BACKOFF_SECONDS[min(attempts - 1, 1)]
async def _post(self, payload: dict) -> tuple[int, str, httpx.Headers]:
async def _post(self, client: httpx.AsyncClient,
payload: dict) -> tuple[int, str, httpx.Headers]:
req_headers = {"Authorization": f"Bearer {self._cfg.api_key}",
"Content-Type": "application/json"}
async with httpx.AsyncClient(timeout=self._cfg.timeout,
transport=self._transport) as client:
resp = await client.post(chat_url(self._cfg.base_url),
headers=req_headers, json=payload)
resp = await client.post(chat_url(self._cfg.base_url),
headers=req_headers, json=payload)
return resp.status_code, resp.text, resp.headers
@staticmethod
def _content(text: str) -> str:
def _content(body: dict) -> str:
try:
content = json.loads(text)["choices"][0]["message"]["content"]
except (KeyError, IndexError, TypeError, ValueError) as e:
content = body["choices"][0]["message"]["content"]
except (KeyError, IndexError, TypeError) as e:
raise LLMError(f"LLM 响应缺 choices/message: {e}") from e
if content is None:
# P2-3: 兼容端点空回复 content:null 不罕见——按空回复降级走
+16 -12
View File
@@ -24,13 +24,14 @@ def _loads_dict(text: str) -> dict:
def _balanced_braces(text: str) -> str | None:
"""首个字符串感知的配平 {} 子串(tickflow 三级候选之三)."""
start = text.find("{")
if start < 0:
return None
depth, in_str, esc = 0, False, False
for i in range(start, len(text)):
ch = text[i]
"""首个字符串感知的配平 {} 子串(tickflow 三级候选之三).
P3-2: 起点扫描同样字符串感知——旧 find("{") 会把字符串内的 { 当起点,
配平块从头错位,拖死 L4/L5 全链(假阴性仅多 502)."""
start = -1
depth = 0
in_str = esc = False
for i, ch in enumerate(text):
if in_str:
if esc:
esc = False
@@ -42,8 +43,10 @@ def _balanced_braces(text: str) -> str | None:
if ch == '"':
in_str = True
elif ch == "{":
if start < 0:
start = i
depth += 1
elif ch == "}":
elif ch == "}" and start >= 0:
depth -= 1
if depth == 0:
return text[start:i + 1]
@@ -98,10 +101,11 @@ def robust_json_parse(text: str) -> dict:
balanced = _balanced_braces(text)
if balanced:
candidates.append(balanced)
# L4: 常见 LaTeX 转义残留破坏 json.loads
candidates.append(re.sub(r"\\([()\[\]])", r"\1", balanced))
# L5: 尾逗号清洗(字符串感知) + 尾随垃圾截断(重截一次)
candidates.append(_clean_trailing_commas(balanced))
# P3-1: 累积降级链——L4 修反斜杠、L5 在 L4 输出上再清尾逗号;
# 旧并行候选下「LaTeX 残留+尾逗号」同现时两级各自单独修都不够
l4 = re.sub(r"\\([()\[\]])", r"\1", balanced)
candidates.append(l4)
candidates.append(_clean_trailing_commas(l4))
last_err: Exception = JSONParseError("空文本")
for cand in candidates:
try: