- 剥 think 归一化(py/Java 语义对拍): 剥头部 <think> 块→整体试 JSON→字符串感知花括号 配平提取(伪 JSON 顺延)→全失败原样透传维持原错误路径; Java 用 FAIL_ON_TRAILING_TOKENS 专用 mapper 对拍 json.loads 严格语义; py 9 用例+全套 68 绿, Java 7 用例+模块 57 绿 - 升级门判定: highspeed 结构层 52/52(泄漏 27/52 剥除全救)但策划探针 P1 4/13>基线 2/13 (2 条特有'裸 config 漏三键包裹'契约失败, schemaOk 11/13)+时延无增益→按'任何缺口=留任' 铁律不过门; 待补=prompt 形状强化+M2.7 时延基线 - 防御对 M2.7 当日泄漏回归(batch-002 折损根因)同等生效, 部署后 batch-002b 补产可恢复 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
236 lines
11 KiB
Python
236 lines
11 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
new-api LLM 通道客户端(D3 件)—— spec:HJ-AGENT-LOOP-EXEC-001 §7.1
|
||
|
||
配方完全沿 C1 spike 已验通道(gen_spike.py:28-30/100-110,52/52 通过):
|
||
- BASE = http://100.64.0.8:3000(new-api),模型 MiniMax-M2.7;
|
||
- response_format=json_object、temperature=0.4、调用间隔 0.3s 节流;
|
||
- 重试 ×2(共最多 3 次尝试),指数退避 1s/2s;
|
||
- 密钥只走环境变量 NEWAPI_KEY(严禁写进 repo 文件,§15-3);缺失由 run_batch 启动时校验退出,
|
||
本客户端构造时再设防一道。
|
||
- 剥 think 防御(模型评估矩阵 2026-06-10 前置):MiniMax-M2.7-highspeed / M3 等推理型通道的
|
||
content 概率性带 <think>…</think> 推理块前缀(与 M3 出局同病灶);取 content 后、下游 json
|
||
解析前做归一化 normalize_llm_json_content(剥 think → 整体试 JSON → 提取首个配平 {...} 块;
|
||
全失败原样返回,维持下游原错误路径)。
|
||
|
||
预算闸接线:每次 HTTP 尝试前回调 budget_cb()(编排器传入 BudgetGuard.note_llm_call 绑定),
|
||
超预算由回调抛 BudgetExceeded 中止——计数口径=「每次调用」而非「每次成功」(§7.3-①)。
|
||
"""
|
||
|
||
import json
|
||
import os
|
||
import time
|
||
import urllib.error
|
||
import urllib.request
|
||
|
||
# 通道默认值(与 gen_spike 配方一致;可用环境变量覆盖,便于换環境不改代码)
|
||
DEFAULT_BASE = os.environ.get("NEWAPI_BASE", "http://100.64.0.8:3000")
|
||
DEFAULT_MODEL = os.environ.get("NEWAPI_MODEL", "MiniMax-M2.7")
|
||
# 换模型抽检(§7.3-⑦)所用独立复核模型;未配置时抽检段如实降级并在报告显著标注
|
||
AUDIT_MODEL = os.environ.get("NEWAPI_AUDIT_MODEL", "")
|
||
|
||
# 通道纪律常量(沿 spike:温度 0.4 / 间隔 0.3s / 重试 ×2 / 单次超时 90s)
|
||
TEMPERATURE = 0.4
|
||
CALL_INTERVAL_SECONDS = 0.3
|
||
MAX_RETRIES = 2
|
||
REQUEST_TIMEOUT_SECONDS = 90
|
||
|
||
|
||
class LlmError(Exception):
|
||
"""LLM 通道失败(重试耗尽):编排器按 §11-F1 infra_llm 处置。"""
|
||
|
||
|
||
# 推理块定界符(MiniMax-M2.7-highspeed / M3 等推理型模型概率性输出在 content 头部)
|
||
_THINK_OPEN = "<think>"
|
||
_THINK_CLOSE = "</think>"
|
||
|
||
|
||
def _is_valid_json(text):
|
||
"""整体是否为合法 JSON 文本(空串/纯空白视为非法;与 Java 侧 isStrictJson 对拍)。"""
|
||
if not text or not text.strip():
|
||
return False
|
||
try:
|
||
json.loads(text)
|
||
return True
|
||
except ValueError:
|
||
return False
|
||
|
||
|
||
def _scan_balanced_object_end(text, start):
|
||
"""
|
||
从 text[start]=='{' 起做字符串感知的花括号配平扫描,返回配平闭括号下标;扫不到返回 -1。
|
||
|
||
字符串感知:JSON 字符串字面量内的花括号、转义引号(\\")不参与配平计数,
|
||
防止 {"a": "}"} 这类含括号字符串被截断成非法片段。
|
||
"""
|
||
depth = 0
|
||
in_string = False # 当前是否处于 JSON 字符串字面量内
|
||
escaped = False # 字符串内上一字符是否为反斜杠转义
|
||
for i in range(start, len(text)):
|
||
ch = text[i]
|
||
if in_string:
|
||
if escaped:
|
||
escaped = False
|
||
elif ch == "\\":
|
||
escaped = True
|
||
elif ch == '"':
|
||
in_string = False
|
||
elif ch == '"':
|
||
in_string = True
|
||
elif ch == "{":
|
||
depth += 1
|
||
elif ch == "}":
|
||
depth -= 1
|
||
if depth == 0:
|
||
return i
|
||
return -1
|
||
|
||
|
||
def _extract_first_json_object(text):
|
||
"""
|
||
提取首个「花括号配平且本身可解析为合法 JSON」的 {...} 块;找不到返回 None。
|
||
|
||
起点按出现顺序逐个尝试:某个 { 起点配平失败(think 无闭合截断)或片段解析失败
|
||
(推理文本里的伪 JSON,如 {target: 10}),则顺延到下一个 { 起点继续,保证夹叙夹议
|
||
场景下真正的 JSON 输出不被推理杂文抢先吞掉。
|
||
"""
|
||
search_from = 0
|
||
while True:
|
||
start = text.find("{", search_from)
|
||
if start == -1:
|
||
return None
|
||
end = _scan_balanced_object_end(text, start)
|
||
if end != -1:
|
||
candidate = text[start:end + 1]
|
||
if _is_valid_json(candidate):
|
||
return candidate
|
||
# 该起点配平失败或片段非法:从下一个 { 起点继续
|
||
search_from = start + 1
|
||
|
||
|
||
def normalize_llm_json_content(content):
|
||
"""
|
||
LLM 响应 content 归一化(剥 think 防御)——取 content 后、下游 json 解析前调用。
|
||
|
||
语义三步(与 Java 侧 ExecutorLlmClient.normalizeJsonContent 双侧对拍,必须保持一致):
|
||
① 若以 <think> 开头且存在 </think>:剥除该推理块及前后空白;
|
||
② 剥后(或无 think 时)若整体即合法 JSON:直接采用;否则提取首个花括号配平的 {...} 块再试
|
||
(防 think 无闭合 / 夹叙夹议把 JSON 埋进杂文);
|
||
③ 全失败:原样返回 content,维持下游原错误路径(json 解析失败的归因与防御前完全一致)。
|
||
|
||
:param content: LLM 返回的原始 content 文本
|
||
:return: 归一化后的 JSON 文本;无法归一化时返回原 content
|
||
"""
|
||
if not isinstance(content, str) or not content:
|
||
return content
|
||
text = content.strip()
|
||
# ① 剥除头部 <think>…</think> 推理块(仅处理头部前缀形态,与实测病灶一致)
|
||
if text.startswith(_THINK_OPEN):
|
||
close = text.find(_THINK_CLOSE)
|
||
if close != -1:
|
||
text = text[close + len(_THINK_CLOSE):].strip()
|
||
# ② 整体即合法 JSON:直接采用(纯 JSON 原样、think 剥净后的常规形态)
|
||
if _is_valid_json(text):
|
||
return text
|
||
# ②' 整体非法:提取首个配平且可解析的 {...} 块(think 无闭合 / 前后杂文场景)
|
||
block = _extract_first_json_object(text)
|
||
if block is not None:
|
||
return block
|
||
# ③ 全失败:维持原内容与原错误路径
|
||
return content
|
||
|
||
|
||
class LlmClient(object):
|
||
"""最小 LLM 客户端:纯标准库 urllib,单方法 chat_json。"""
|
||
|
||
def __init__(self, base=None, key=None, model=None, sleeper=time.sleep, opener=None):
|
||
"""
|
||
:param base: new-api 地址(默认环境/常量)
|
||
:param key: API Key(默认读 NEWAPI_KEY 环境变量;空则构造即失败——密钥纪律)
|
||
:param model: 默认模型
|
||
:param sleeper: 可注入睡眠函数(单测桩)
|
||
:param opener: 可注入 urlopen(单测桩)
|
||
"""
|
||
self.base = (base or DEFAULT_BASE).rstrip("/")
|
||
self.key = key if key is not None else os.environ.get("NEWAPI_KEY", "")
|
||
self.model = model or DEFAULT_MODEL
|
||
self._sleep = sleeper
|
||
self._open = opener or urllib.request.urlopen
|
||
if not self.key:
|
||
# 密钥纪律(§15-3):缺 NEWAPI_KEY 立即失败,绝不带空 key 发请求
|
||
raise LlmError("缺少环境变量 NEWAPI_KEY,拒绝初始化 LLM 通道(密钥严禁入 repo,只走环境变量)")
|
||
|
||
def chat_json(self, system, user, model=None, budget_cb=None, log=None):
|
||
"""
|
||
发起一次 JSON-object 约束的对话补全。
|
||
|
||
:param system: 系统提示词(来自 Registry 渲染,prompt 不内嵌代码——§7.1)
|
||
:param user: 用户消息(变量已渲染)
|
||
:param model: 覆盖模型(换模型抽检用)
|
||
:param budget_cb: 每次 HTTP 尝试前回调(预算闸计数;可抛 BudgetExceeded 中止)
|
||
:param log: 中文日志函数(默认 print;外部交互必须可追溯)
|
||
:return: (content: str, attempts: int)
|
||
:raises LlmError: 重试 ×2 后仍失败
|
||
"""
|
||
log = log or (lambda msg: print(msg, flush=True))
|
||
use_model = model or self.model
|
||
# prompt 不内嵌代码(§7.1):编排器把 Registry 渲染后的完整 prompt 文档作为 user 消息,
|
||
# system 传空串即省略——本客户端不携带任何代码内置提示词
|
||
messages = []
|
||
if system:
|
||
messages.append({"role": "system", "content": system})
|
||
messages.append({"role": "user", "content": user})
|
||
body = json.dumps({
|
||
"model": use_model,
|
||
"temperature": TEMPERATURE,
|
||
"response_format": {"type": "json_object"},
|
||
# 显式 max_tokens 固化(模型评估矩阵 2026-06-10 实证):M2.7/M3/deepseek 系均为推理型输出
|
||
# reasoning_content,吃光网关缺省额度会得空 content(C6.1 实测 22 次空重试、leg3 显式给额 5 救 5)。
|
||
"max_tokens": 4096,
|
||
"messages": messages,
|
||
}).encode("utf-8")
|
||
|
||
last_err = None
|
||
attempts = 0
|
||
for attempt in range(1, MAX_RETRIES + 2): # 1 次原始 + 2 次重试
|
||
if budget_cb is not None:
|
||
budget_cb() # 预算闸:每次尝试都计数,超限抛 BudgetExceeded
|
||
attempts = attempt
|
||
req = urllib.request.Request(self.base + "/v1/chat/completions", data=body, method="POST")
|
||
req.add_header("Authorization", "Bearer " + self.key)
|
||
req.add_header("Content-Type", "application/json")
|
||
try:
|
||
with self._open(req, timeout=REQUEST_TIMEOUT_SECONDS) as resp:
|
||
data = json.loads(resp.read().decode("utf-8"))
|
||
content = data.get("choices", [{}])[0].get("message", {}).get("content", "")
|
||
if not content:
|
||
# 空补全按通道异常处理(可重试)
|
||
raise urllib.error.URLError("LLM 返回空 content")
|
||
# 剥 think 防御:推理型通道(M2.7-highspeed/M3 等)content 概率性带 <think> 块/夹杂文本,
|
||
# 在下游 json 解析前归一化;全失败原样透传,维持原错误路径(外部交互可追溯:变更必留日志)
|
||
normalized = normalize_llm_json_content(content)
|
||
if normalized != content:
|
||
log("[llm] 响应含 <think> 推理块/夹杂文本,已归一化提取 JSON(%d→%d 字符,model=%s)"
|
||
% (len(content), len(normalized), use_model))
|
||
content = normalized
|
||
# 通道节流:沿 spike 纪律每次调用后间隔 0.3s(§7.3-⑤)
|
||
self._sleep(CALL_INTERVAL_SECONDS)
|
||
return content, attempts
|
||
except urllib.error.HTTPError as ex:
|
||
# HTTP 层错误:读响应片段入日志便于排障(外部交互可追溯)
|
||
detail = ""
|
||
try:
|
||
detail = ex.read().decode("utf-8", "replace")[:200]
|
||
except Exception: # noqa: BLE001 —— 读错误体失败不掩盖原错误
|
||
pass
|
||
last_err = "http_%s:%s" % (ex.code, detail)
|
||
log("[llm] 第 %d 次尝试 HTTP 错误 %s(model=%s)" % (attempt, last_err, use_model))
|
||
except Exception as ex: # 网络超时/连接拒绝等
|
||
last_err = "err:%s" % ex
|
||
log("[llm] 第 %d 次尝试通道异常 %s(model=%s)" % (attempt, last_err, use_model))
|
||
# 指数退避后重试(1s, 2s)
|
||
if attempt <= MAX_RETRIES:
|
||
self._sleep(1.0 * (2 ** (attempt - 1)))
|
||
raise LlmError("LLM 调用重试 ×%d 后仍失败:%s" % (MAX_RETRIES, last_err))
|