339 lines
13 KiB
Python
339 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
"""统一 Claude CLI 运行时的冻结配置、联合回执和隔离测试。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import json
|
||
import os
|
||
import pathlib
|
||
import subprocess
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
from decimal import Decimal
|
||
|
||
SCRIPT_DIR = pathlib.Path(__file__).resolve().parent
|
||
sys.path.insert(0, str(SCRIPT_DIR))
|
||
|
||
from claude_runtime import ( # noqa: E402
|
||
ClaudeRuntimeError,
|
||
ExecutionProfile,
|
||
build_sandbox_command,
|
||
canonical_json,
|
||
run_claude,
|
||
sha256_json,
|
||
sha256_text,
|
||
verify_execution_profile,
|
||
)
|
||
|
||
|
||
FULL_MODEL_ID = "claude-opus-4-1-20250805"
|
||
CAPABILITY_MODEL_ID = "claude-opus-4-8[1m]"
|
||
OUTPUT_SCHEMA = {
|
||
"type": "object",
|
||
"additionalProperties": False,
|
||
"required": ["value"],
|
||
"properties": {"value": {"type": "string", "minLength": 1}},
|
||
}
|
||
SYSTEM_PROMPT = "只返回符合给定 JSON Schema 的测试对象。"
|
||
|
||
|
||
def execution_profile(
|
||
executable: str = "/usr/bin/true",
|
||
*,
|
||
model_alias: str = "opus",
|
||
resolved_model_id: str = FULL_MODEL_ID,
|
||
) -> ExecutionProfile:
|
||
"""构造不依赖真实模型、但字段完整的冻结测试 profile。"""
|
||
|
||
executable_hash = hashlib.sha256(pathlib.Path(executable).read_bytes()).hexdigest()
|
||
return ExecutionProfile(
|
||
profile_version="claude-offline-v1",
|
||
adapter_role="writer",
|
||
claude_executable_path=executable,
|
||
claude_executable_sha256=executable_hash,
|
||
claude_cli_version="2.1.211",
|
||
model_alias=model_alias,
|
||
resolved_model_id=resolved_model_id,
|
||
effort="high",
|
||
max_budget_usd_per_call=Decimal("1.500000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=10000,
|
||
json_schema_id="test-output-v1",
|
||
json_schema=OUTPUT_SCHEMA,
|
||
json_schema_sha256=sha256_json(OUTPUT_SCHEMA),
|
||
system_prompt_id="test-system-v1",
|
||
system_prompt=SYSTEM_PROMPT,
|
||
system_prompt_sha256=sha256_text(SYSTEM_PROMPT),
|
||
normal_terminal_reasons=("success",),
|
||
)
|
||
|
||
|
||
def success_envelope() -> dict[str, object]:
|
||
"""构造成本、usage、模型身份都能互相核对的成功 envelope。"""
|
||
|
||
return {
|
||
"type": "result",
|
||
"subtype": "success",
|
||
"is_error": False,
|
||
"terminal_reason": "success",
|
||
"stop_reason": "end_turn",
|
||
"api_error_status": None,
|
||
"total_cost_usd": "0.120000",
|
||
"usage": {"input_tokens": 10, "output_tokens": 2},
|
||
"modelUsage": {
|
||
FULL_MODEL_ID: {
|
||
"inputTokens": 10,
|
||
"outputTokens": 2,
|
||
"costUSD": "0.120000",
|
||
}
|
||
},
|
||
"structured_output": {"value": "ok"},
|
||
"result": "此字段不是业务输出",
|
||
}
|
||
|
||
|
||
class FakeRunner:
|
||
"""记录 subprocess 参数并返回预设结果,测试绝不调用真实模型。"""
|
||
|
||
def __init__(self, result: subprocess.CompletedProcess[str] | BaseException):
|
||
self.result = result
|
||
self.calls: list[tuple[list[str], dict[str, object]]] = []
|
||
|
||
def __call__(self, command: list[str], **kwargs: object) -> subprocess.CompletedProcess[str]:
|
||
"""保存调用现场,并按测试场景返回或抛出。"""
|
||
|
||
self.calls.append((command, kwargs))
|
||
if isinstance(self.result, BaseException):
|
||
raise self.result
|
||
return self.result
|
||
|
||
|
||
def runner_for(envelope: dict[str, object], *, returncode: int = 0, stderr: str = "") -> FakeRunner:
|
||
"""把 envelope 包装成 subprocess 完成结果。"""
|
||
|
||
stdout = json.dumps(envelope, ensure_ascii=False, separators=(",", ":"))
|
||
return FakeRunner(subprocess.CompletedProcess(["sandbox-exec"], returncode, stdout, stderr))
|
||
|
||
|
||
class ClaudeRuntimeTest(unittest.TestCase):
|
||
"""验证统一 runtime 的成功条件与失败优先级。"""
|
||
|
||
def test_profile_accepts_exact_model_id_with_capability_suffix(self):
|
||
"""真实 modelUsage 的方括号能力后缀必须能构造冻结 profile。"""
|
||
|
||
profile = execution_profile(resolved_model_id=CAPABILITY_MODEL_ID)
|
||
|
||
self.assertEqual(profile.resolved_model_id, CAPABILITY_MODEL_ID)
|
||
|
||
def test_model_id_capability_suffix_is_narrow_and_terminal(self):
|
||
"""能力后缀只能是末尾单个非空安全字符集合。"""
|
||
|
||
invalid_model_ids = (
|
||
"claude-opus-4-8[]",
|
||
"claude-opus-4-8[[1m]]",
|
||
"claude-opus-4-8[1m[2m]",
|
||
"claude-opus-4-8[1m/2m]",
|
||
"claude-opus-4-8[1m]suffix",
|
||
"claude-opus-4-8[1m][2m]",
|
||
)
|
||
for model_id in invalid_model_ids:
|
||
with self.subTest(model_id=model_id), self.assertRaises(ValueError):
|
||
execution_profile(resolved_model_id=model_id)
|
||
|
||
def test_model_id_length_includes_capability_suffix(self):
|
||
"""方括号后缀计入模型 ID 的 6 到 128 字符总长度。"""
|
||
|
||
max_length_model_id = "a" * 125 + "[x]"
|
||
over_max_length_model_id = "a" * 126 + "[x]"
|
||
|
||
profile = execution_profile(resolved_model_id=max_length_model_id)
|
||
self.assertEqual(len(profile.resolved_model_id), 128)
|
||
|
||
with self.assertRaises(ValueError):
|
||
execution_profile(resolved_model_id=over_max_length_model_id)
|
||
|
||
def test_capability_suffix_does_not_bypass_alias_rejection(self):
|
||
"""方括号后缀不能让 alias 与 resolved model ID 相等的 profile 通过。"""
|
||
|
||
with self.assertRaises(ValueError):
|
||
execution_profile(
|
||
model_alias=CAPABILITY_MODEL_ID,
|
||
resolved_model_id=CAPABILITY_MODEL_ID,
|
||
)
|
||
|
||
def test_success_uses_structured_output_and_emits_complete_receipt(self):
|
||
"""成功必须返回 structured_output,并形成不含正文的完整回执。"""
|
||
|
||
profile = execution_profile()
|
||
runner = runner_for(success_envelope())
|
||
result = run_claude(
|
||
profile,
|
||
{"request": "secret-body"},
|
||
runner=runner,
|
||
binding_verifier=lambda _profile: None,
|
||
source_environment={"LANG": "zh_CN.UTF-8", "DATABASE_URL": "secret"},
|
||
)
|
||
|
||
self.assertEqual(result.structured_output, {"value": "ok"})
|
||
receipt = result.receipt.as_dict()
|
||
self.assertEqual(receipt["actualModelId"], FULL_MODEL_ID)
|
||
self.assertTrue(receipt["modelMatch"])
|
||
self.assertEqual(receipt["totalCostUsd"], "0.120000")
|
||
self.assertEqual(receipt["structuredOutputSha256"], sha256_json({"value": "ok"}))
|
||
self.assertNotIn("secret-body", canonical_json(receipt))
|
||
|
||
command, kwargs = runner.calls[0]
|
||
self.assertEqual(command[0], "/usr/bin/sandbox-exec")
|
||
self.assertEqual(command[-1], SYSTEM_PROMPT)
|
||
self.assertEqual(kwargs["input"], '{"request":"secret-body"}')
|
||
self.assertEqual(kwargs["env"].get("LANG"), "zh_CN.UTF-8")
|
||
self.assertNotIn("DATABASE_URL", kwargs["env"])
|
||
self.assertNotIn("PATH", kwargs["env"])
|
||
self.assertEqual(pathlib.Path(str(kwargs["cwd"])), pathlib.Path(kwargs["env"]["HOME"]))
|
||
self.assertFalse(pathlib.Path(str(kwargs["cwd"])).exists())
|
||
|
||
def test_503_success_subtype_is_still_api_error(self):
|
||
"""subtype=success 不能覆盖 is_error、503、空模型 usage 或非零退出。"""
|
||
|
||
envelope = success_envelope()
|
||
envelope.update(
|
||
{
|
||
"is_error": True,
|
||
"terminal_reason": "api_error",
|
||
"api_error_status": 503,
|
||
"total_cost_usd": "0",
|
||
"usage": {"input_tokens": 0, "output_tokens": 0},
|
||
"modelUsage": {},
|
||
}
|
||
)
|
||
with self.assertRaises(ClaudeRuntimeError) as raised:
|
||
run_claude(
|
||
execution_profile(),
|
||
{"request": "x"},
|
||
runner=runner_for(envelope, returncode=1, stderr="gateway secret"),
|
||
binding_verifier=lambda _profile: None,
|
||
)
|
||
|
||
self.assertEqual(raised.exception.primary_code, "WRITER_API_ERROR")
|
||
self.assertIn("WRITER_NONZERO_EXIT", raised.exception.causes)
|
||
self.assertNotIn("gateway secret", str(raised.exception))
|
||
self.assertNotIn("gateway secret", canonical_json(raised.exception.details))
|
||
|
||
def test_result_never_falls_back_when_structured_output_is_missing(self):
|
||
"""即使 result 恰好是合法 JSON,也不得把它当业务对象。"""
|
||
|
||
envelope = success_envelope()
|
||
del envelope["structured_output"]
|
||
envelope["result"] = '{"value":"forbidden-fallback"}'
|
||
|
||
with self.assertRaises(ClaudeRuntimeError) as raised:
|
||
run_claude(
|
||
execution_profile(),
|
||
{"request": "x"},
|
||
runner=runner_for(envelope),
|
||
binding_verifier=lambda _profile: None,
|
||
)
|
||
|
||
self.assertEqual(raised.exception.primary_code, "WRITER_SCHEMA_INVALID")
|
||
|
||
def test_model_mismatch_and_usage_or_cost_gaps_fail_closed(self):
|
||
"""实际模型、usage 和成本任一无法核账都不能成功。"""
|
||
|
||
mismatch = success_envelope()
|
||
mismatch["modelUsage"] = {
|
||
"claude-sonnet-4-20250514": {"inputTokens": 10, "outputTokens": 2, "costUSD": "0.120000"}
|
||
}
|
||
missing_usage = success_envelope()
|
||
del missing_usage["usage"]
|
||
missing_cost = success_envelope()
|
||
del missing_cost["total_cost_usd"]
|
||
cases = (
|
||
(mismatch, "WRITER_MODEL_MISMATCH"),
|
||
(missing_usage, "WRITER_RECEIPT_INVALID"),
|
||
(missing_cost, "WRITER_RECEIPT_INVALID"),
|
||
)
|
||
for envelope, expected in cases:
|
||
with self.subTest(expected=expected), self.assertRaises(ClaudeRuntimeError) as raised:
|
||
run_claude(
|
||
execution_profile(),
|
||
{"request": "x"},
|
||
runner=runner_for(envelope),
|
||
binding_verifier=lambda _profile: None,
|
||
)
|
||
self.assertEqual(raised.exception.primary_code, expected)
|
||
|
||
def test_timeout_has_highest_priority_and_never_exposes_stderr(self):
|
||
"""硬 deadline 的稳定码优先于所有无法取得的 envelope 字段。"""
|
||
|
||
runner = FakeRunner(
|
||
subprocess.TimeoutExpired(["sandbox-exec"], 1, output="raw response", stderr="secret stderr")
|
||
)
|
||
with self.assertRaises(ClaudeRuntimeError) as raised:
|
||
run_claude(
|
||
execution_profile(),
|
||
{"request": "x"},
|
||
runner=runner,
|
||
binding_verifier=lambda _profile: None,
|
||
)
|
||
|
||
self.assertEqual(raised.exception.primary_code, "WRITER_TIMEOUT")
|
||
self.assertIsNotNone(raised.exception.receipt)
|
||
self.assertIsNone(raised.exception.receipt.exit_code)
|
||
self.assertNotIn("secret stderr", str(raised.exception))
|
||
self.assertNotIn("raw response", canonical_json(raised.exception.details))
|
||
|
||
def test_sandbox_command_contains_all_frozen_flags(self):
|
||
"""sandbox 命令必须固定完整模型、预算、schema、无工具和无持久化参数。"""
|
||
|
||
profile = execution_profile()
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
command = build_sandbox_command(profile, pathlib.Path(directory))
|
||
|
||
required_pairs = {
|
||
"--model": FULL_MODEL_ID,
|
||
"--effort": "high",
|
||
"--max-budget-usd": "1.500000",
|
||
"--output-format": "json",
|
||
"--json-schema": canonical_json(OUTPUT_SCHEMA),
|
||
"--tools": "",
|
||
"--mcp-config": '{"mcpServers":{}}',
|
||
"--system-prompt": SYSTEM_PROMPT,
|
||
}
|
||
self.assertEqual(command[0:2], ["/usr/bin/sandbox-exec", "-p"])
|
||
self.assertIn("(deny default)", command[2])
|
||
self.assertIn("network-outbound", command[2])
|
||
self.assertIn("file-write*", command[2])
|
||
self.assertIn("--print", command)
|
||
self.assertIn("--bare", command)
|
||
self.assertIn("--no-session-persistence", command)
|
||
self.assertIn("--disable-slash-commands", command)
|
||
self.assertIn("--strict-mcp-config", command)
|
||
for flag, expected in required_pairs.items():
|
||
self.assertEqual(command[command.index(flag) + 1], expected)
|
||
|
||
def test_profile_verification_rejects_binary_or_version_drift(self):
|
||
"""绝对路径内容或 CLI 版本变化时必须在模型调用前失败。"""
|
||
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
executable = pathlib.Path(directory) / "claude"
|
||
executable.write_bytes(b"frozen-binary")
|
||
executable.chmod(0o700)
|
||
profile = execution_profile(str(executable))
|
||
|
||
verify_execution_profile(
|
||
profile,
|
||
version_runner=lambda *_args, **_kwargs: subprocess.CompletedProcess(
|
||
[str(executable), "--version"], 0, "2.1.211 (Claude Code)\n", ""
|
||
),
|
||
)
|
||
executable.write_bytes(b"drifted-binary")
|
||
with self.assertRaises(ClaudeRuntimeError) as raised:
|
||
verify_execution_profile(profile)
|
||
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|