zizi c9f69d9d6d 治理: Skill 测试治理第一阶段——harness 控制平面 + 实现测试迁出运行时目录
范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):

1. 新增 harness/ 控制平面
   - skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
   - run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
     空跑与 skip-only 失败关闭、AST 测试形状门
   - manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
   - manifests/test-inventory.json:81 个测试资产登记
   - specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责

2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
   - 71 个测试文件迁移并修复项目根与临时目录运行导入
   - 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
   - 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
   - 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变

3. 运行时文档清理
   - 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
     只保留运行时合同;业务运行合同、额度、授权与离线模式均保留

4. SoT 同步
   - AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
   - 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
   - humanization 覆盖矩阵:活动测试路径同步迁移

验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。

已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
2026-08-19 01:50:20 +08:00

795 lines
32 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""统一 Claude CLI 运行时的冻结配置、联合回执和隔离测试。"""
from __future__ import annotations
import copy
from dataclasses import replace
import hashlib
import json
import os
import pathlib
import signal
import subprocess
import sys
import tempfile
import unittest
from unittest import mock
from decimal import Decimal
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "execute-claude-task" / "scripts"
sys.path.insert(0, str(SCRIPT_DIR))
from claude_runtime import ( # noqa: E402
ClaudeRuntimeError,
ExecutionProfile,
_run_default_subprocess,
_minimal_environment,
_validate_usage,
build_sandbox_command,
canonical_json,
contains_path_traversal,
run_claude,
sha256_json,
sha256_text,
verify_execution_profile,
)
FULL_MODEL_ID = "claude-opus-4-1-20250805"
class DefaultSubprocessCleanupTest(unittest.TestCase):
"""真实子进程边界必须在父进程异常时回收整个模型进程组。"""
def test_parent_interruption_kills_and_waits_for_process_group(self):
class ParentInterrupted(BaseException):
pass
process = mock.Mock()
process.pid = 4321
process.communicate.side_effect = ParentInterrupted("测试父进程中断")
with (
mock.patch("claude_runtime.subprocess.Popen", return_value=process),
mock.patch("claude_runtime.os.killpg") as killpg,
self.assertRaises(ParentInterrupted),
):
_run_default_subprocess(
["/frozen/claude"],
input="{}",
text=True,
capture_output=True,
timeout=1200,
check=False,
cwd="/private/tmp/runtime-fixture",
env={},
start_new_session=True,
)
killpg.assert_called_once_with(4321, signal.SIGKILL)
process.wait.assert_called_once_with(timeout=5)
CAPABILITY_MODEL_ID = "claude-opus-4-8[1m]"
OUTPUT_SCHEMA = {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"additionalProperties": False,
"required": ["value"],
"properties": {"value": {"type": "string", "minLength": 1}},
}
SYSTEM_PROMPT = "只返回符合给定 JSON Schema 的测试对象。"
def execution_profile(
executable: str = "/usr/bin/true",
*,
model_alias: str = "opus",
resolved_model_id: str = FULL_MODEL_ID,
) -> ExecutionProfile:
"""构造不依赖真实模型、但字段完整的冻结测试 profile。"""
executable_hash = hashlib.sha256(pathlib.Path(executable).read_bytes()).hexdigest()
return ExecutionProfile(
profile_version="claude-offline-v1",
adapter_role="writer",
claude_executable_path=executable,
claude_executable_sha256=executable_hash,
claude_cli_version="2.1.211",
model_alias=model_alias,
resolved_model_id=resolved_model_id,
effort="high",
max_budget_usd_per_call=Decimal("1.500000"),
timeout_seconds=30,
max_context_chars=10000,
json_schema_id="test-output-v1",
json_schema=OUTPUT_SCHEMA,
json_schema_sha256=sha256_json(OUTPUT_SCHEMA),
system_prompt_id="test-system-v1",
system_prompt=SYSTEM_PROMPT,
system_prompt_sha256=sha256_text(SYSTEM_PROMPT),
normal_terminal_reasons=("success",),
)
def success_envelope() -> dict[str, object]:
"""构造成本、usage、模型身份都能互相核对的成功 envelope。"""
return {
"type": "result",
"subtype": "success",
"is_error": False,
"terminal_reason": "success",
"stop_reason": "end_turn",
"api_error_status": None,
"total_cost_usd": "0.120000",
"usage": {"input_tokens": 10, "output_tokens": 2},
"modelUsage": {
FULL_MODEL_ID: {
"inputTokens": 10,
"outputTokens": 2,
"costUSD": "0.120000",
}
},
"structured_output": {"value": "ok"},
"result": "此字段不是业务输出",
}
def realistic_usage() -> dict[str, object]:
"""构造包含 Claude metadata、已知计数映射和迭代数组的真实 usage 形状。"""
return {
"input_tokens": 10,
"output_tokens": 2,
"cache_creation_input_tokens": 3,
"cache_read_input_tokens": 4,
"server_tool_use": {"web_search_requests": 1, "web_fetch_requests": 0},
"cache_creation": {"ephemeral_5m_input_tokens": 5, "ephemeral_1h_input_tokens": 6},
"service_tier": "standard",
"speed": "standard",
"inference_geo": "",
"iterations": [
{"type": "tool_use", "duration_ms": 12},
{"type": "metadata", "values": [True, None, "safe"]},
],
"future_metadata": {"trace_id": "safe", "enabled": True},
}
class FakeRunner:
"""记录 subprocess 参数并返回预设结果,测试绝不调用真实模型。"""
def __init__(self, result: subprocess.CompletedProcess[str] | BaseException):
self.result = result
self.calls: list[tuple[list[str], dict[str, object]]] = []
def __call__(self, command: list[str], **kwargs: object) -> subprocess.CompletedProcess[str]:
"""保存调用现场,并按测试场景返回或抛出。"""
self.calls.append((command, kwargs))
if isinstance(self.result, BaseException):
raise self.result
return self.result
def runner_for(envelope: dict[str, object], *, returncode: int = 0, stderr: str = "") -> FakeRunner:
"""把 envelope 包装成 subprocess 完成结果。"""
stdout = json.dumps(envelope, ensure_ascii=False, separators=(",", ":"))
return FakeRunner(subprocess.CompletedProcess(["sandbox-exec"], returncode, stdout, stderr))
class ClaudeRuntimeTest(unittest.TestCase):
"""验证统一 runtime 的成功条件与失败优先级。"""
def test_profile_accepts_exact_model_id_with_capability_suffix(self):
"""真实 modelUsage 的方括号能力后缀必须能构造冻结 profile。"""
profile = execution_profile(resolved_model_id=CAPABILITY_MODEL_ID)
self.assertEqual(profile.resolved_model_id, CAPABILITY_MODEL_ID)
def test_model_id_capability_suffix_is_narrow_and_terminal(self):
"""能力后缀只能是末尾单个非空安全字符集合。"""
invalid_model_ids = (
"claude-opus-4-8[]",
"claude-opus-4-8[[1m]]",
"claude-opus-4-8[1m[2m]",
"claude-opus-4-8[1m/2m]",
"claude-opus-4-8[1m]suffix",
"claude-opus-4-8[1m][2m]",
)
for model_id in invalid_model_ids:
with self.subTest(model_id=model_id), self.assertRaises(ValueError):
execution_profile(resolved_model_id=model_id)
def test_model_id_length_includes_capability_suffix(self):
"""方括号后缀计入模型 ID 的 6 到 128 字符总长度。"""
max_length_model_id = "a" * 125 + "[x]"
over_max_length_model_id = "a" * 126 + "[x]"
profile = execution_profile(resolved_model_id=max_length_model_id)
self.assertEqual(len(profile.resolved_model_id), 128)
with self.assertRaises(ValueError):
execution_profile(resolved_model_id=over_max_length_model_id)
def test_capability_suffix_does_not_bypass_alias_rejection(self):
"""方括号后缀不能让 alias 与 resolved model ID 相等的 profile 通过。"""
with self.assertRaises(ValueError):
execution_profile(
model_alias=CAPABILITY_MODEL_ID,
resolved_model_id=CAPABILITY_MODEL_ID,
)
def test_success_uses_structured_output_and_emits_complete_receipt(self):
"""成功必须返回 structured_output,并形成不含正文的完整回执。"""
profile = execution_profile()
runner = runner_for(success_envelope())
auth_token = "auth-token-must-not-leak"
base_url = "https://gateway.example.invalid/anthropic/v1"
result = run_claude(
profile,
{"request": "secret-body"},
runner=runner,
binding_verifier=lambda _profile: None,
source_environment={
"LANG": "zh_CN.UTF-8",
"ANTHROPIC_AUTH_TOKEN": auth_token,
"ANTHROPIC_BASE_URL": base_url,
"CLAUDE_CODE_TMPDIR": "/tmp/attacker-claude-temp",
"CLAUDE_CONFIG_DIR": "/Users/attacker-claude-config",
"DATABASE_URL": "secret",
"UNRELATED_SECRET": "must-be-stripped",
},
)
self.assertEqual(result.structured_output, {"value": "ok"})
receipt = result.receipt.as_dict()
self.assertEqual(receipt["actualModelId"], FULL_MODEL_ID)
self.assertTrue(receipt["modelMatch"])
self.assertEqual(receipt["totalCostUsd"], "0.120000")
self.assertEqual(receipt["structuredOutputSha256"], sha256_json({"value": "ok"}))
self.assertNotIn("secret-body", canonical_json(receipt))
self.assertNotIn(auth_token, canonical_json(receipt))
self.assertNotIn(base_url, canonical_json(receipt))
command, kwargs = runner.calls[0]
self.assertEqual(command[0], "/usr/bin/sandbox-exec")
self.assertEqual(command[-1], SYSTEM_PROMPT)
self.assertEqual(kwargs["input"], '{"request":"secret-body"}')
self.assertEqual(kwargs["env"].get("LANG"), "zh_CN.UTF-8")
self.assertEqual(kwargs["env"].get("ANTHROPIC_AUTH_TOKEN"), auth_token)
self.assertEqual(kwargs["env"].get("ANTHROPIC_BASE_URL"), base_url)
self.assertNotIn("DATABASE_URL", kwargs["env"])
self.assertNotIn("UNRELATED_SECRET", kwargs["env"])
self.assertNotIn("PATH", kwargs["env"])
self.assertNotIn(auth_token, canonical_json(command))
self.assertNotIn(base_url, canonical_json(command))
self.assertEqual(pathlib.Path(str(kwargs["cwd"])), pathlib.Path(kwargs["env"]["HOME"]))
self.assertEqual(kwargs["env"]["TMPDIR"], kwargs["env"]["HOME"])
self.assertEqual(kwargs["env"]["CLAUDE_CODE_TMPDIR"], kwargs["env"]["HOME"])
self.assertEqual(kwargs["env"]["CLAUDE_CONFIG_DIR"], kwargs["env"]["HOME"])
self.assertFalse(pathlib.Path(str(kwargs["cwd"])).exists())
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
real_auth_environment = _minimal_environment(
{"ANTHROPIC_AUTH_TOKEN": auth_token, "ANTHROPIC_BASE_URL": base_url},
pathlib.Path(directory),
require_authentication=True,
)
self.assertEqual(real_auth_environment["ANTHROPIC_AUTH_TOKEN"], auth_token)
self.assertEqual(real_auth_environment["ANTHROPIC_BASE_URL"], base_url)
def test_structured_output_decimal_numbers_are_json_native_and_hash_bound(self):
"""精确解析出的半分小数必须先归一,再交给适配器、持久化并计算回执哈希。"""
numeric_schema = {
"type": "object",
"additionalProperties": False,
"required": ["score", "nested"],
"properties": {
"score": {"type": "number"},
"nested": {
"type": "array",
"items": {"type": "number"},
},
},
}
profile = replace(
execution_profile(),
json_schema_id="test-numeric-output-v1",
json_schema=numeric_schema,
json_schema_sha256=sha256_json(numeric_schema),
)
envelope = success_envelope()
envelope["structured_output"] = {
"score": 7.5,
"nested": [2.25, 3],
}
result = run_claude(
profile,
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(result.structured_output, {"score": 7.5, "nested": [2.25, 3]})
self.assertIsInstance(result.structured_output["score"], float)
self.assertIsInstance(result.structured_output["nested"][1], int)
json.dumps(result.structured_output, ensure_ascii=False)
self.assertEqual(
result.receipt.structured_output_sha256,
sha256_json(result.structured_output),
)
def test_auth_token_requires_a_safe_base_url(self):
"""AUTH_TOKEN 必须绑定无凭据、无查询和无片段的 HTTP(S) 网关地址。"""
auth_token = "auth-token-never-echoed"
invalid_base_urls = (
None,
"",
" ",
"ftp://gateway.example.invalid/v1",
"https:///v1",
"https://user:password@gateway.example.invalid/v1",
"https://gateway.example.invalid/v1?token=leak",
"https://gateway.example.invalid/v1#fragment",
)
for base_url in invalid_base_urls:
source_environment = {"ANTHROPIC_AUTH_TOKEN": auth_token}
if base_url is not None:
source_environment["ANTHROPIC_BASE_URL"] = base_url
with self.subTest(base_url=base_url), self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
binding_verifier=lambda _profile: None,
source_environment=source_environment,
)
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
self.assertNotIn(auth_token, str(raised.exception))
self.assertNotIn(auth_token, canonical_json(raised.exception.details))
if base_url:
self.assertNotIn(base_url, str(raised.exception))
self.assertNotIn(base_url, canonical_json(raised.exception.details))
def test_realistic_usage_metadata_is_accepted_without_changing_cost_authority(self):
"""真实 usage metadata 可通过,但模型成本仍只由 modelUsage 对账。"""
envelope = success_envelope()
envelope["usage"] = realistic_usage()
result = run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(result.structured_output, {"value": "ok"})
self.assertEqual(result.receipt.as_dict()["totalCostUsd"], "0.120000")
self.assertEqual(result.receipt.as_dict()["usage"], realistic_usage())
def test_model_usage_info_fields_are_forward_compatible(self):
"""CLI 2.1.231 起 modelUsage 附带 canonicalModel/provider 信息字段:形状合法即放行。"""
envelope = success_envelope()
envelope["modelUsage"] = {
FULL_MODEL_ID: {
"inputTokens": 10,
"outputTokens": 2,
"costUSD": "0.120000",
"canonicalModel": "claude-opus-4-8",
"provider": "firstParty",
}
}
result = run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(result.structured_output, {"value": "ok"})
self.assertEqual(result.receipt.as_dict()["totalCostUsd"], "0.120000")
def test_model_usage_info_fields_with_bad_shape_fail_closed(self):
"""信息字段形状非法(数字/空串)仍然失败关闭;计数字段不放宽。"""
for label, canonical, provider in (
("canonicalModel 非字符串", 42, "firstParty"),
("provider 空串", "claude-opus-4-8", ""),
):
envelope = success_envelope()
envelope["modelUsage"] = {
FULL_MODEL_ID: {
"inputTokens": 10,
"outputTokens": 2,
"costUSD": "0.120000",
"canonicalModel": canonical,
"provider": provider,
}
}
with self.subTest(label=label), self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
def test_usage_rejects_bad_counts_metadata_and_non_json_iterations(self):
"""usage 的必需计数、已知映射和 metadata 破坏时必须失败关闭。"""
cases: list[tuple[str, dict[str, object]]] = []
missing_required = realistic_usage()
del missing_required["input_tokens"]
cases.append(("missing input_tokens", missing_required))
negative_token = realistic_usage()
negative_token["output_tokens"] = -1
cases.append(("negative output_tokens", negative_token))
negative_server_count = realistic_usage()
negative_server_count["server_tool_use"] = {"web_search_requests": -1}
cases.append(("negative server_tool_use count", negative_server_count))
negative_cache_count = realistic_usage()
negative_cache_count["cache_creation"] = {"ephemeral_5m_input_tokens": -1}
cases.append(("negative cache_creation count", negative_cache_count))
invalid_service_tier = realistic_usage()
invalid_service_tier["service_tier"] = 1
cases.append(("non-string service_tier", invalid_service_tier))
invalid_inference_geo = realistic_usage()
invalid_inference_geo["inference_geo"] = None
cases.append(("non-string inference_geo", invalid_inference_geo))
invalid_unknown_metadata = realistic_usage()
invalid_unknown_metadata["future_metadata"] = {"ratio": float("inf")}
cases.append(("infinite unknown metadata", invalid_unknown_metadata))
invalid_iteration_number = realistic_usage()
invalid_iteration_number["iterations"] = [{"duration_ms": float("nan")}]
cases.append(("non-finite iteration metadata", invalid_iteration_number))
for label, usage in cases:
envelope = success_envelope()
envelope["usage"] = usage
with self.subTest(label=label), self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
invalid_iteration_json = copy.deepcopy(realistic_usage())
invalid_iteration_json["iterations"] = [object()]
with self.assertRaises(ValueError):
_validate_usage(invalid_iteration_json)
def test_realistic_usage_cannot_bypass_model_cost_reconciliation(self):
"""丰富 usage metadata 不能掩盖 modelUsage 与 total cost 对账失败。"""
cases = []
mismatched_model = success_envelope()
mismatched_model["usage"] = realistic_usage()
mismatched_model["modelUsage"] = {
"claude-sonnet-4-20250514": {
"inputTokens": 10,
"outputTokens": 2,
"costUSD": "0.120000",
}
}
cases.append(mismatched_model)
mismatched_total = success_envelope()
mismatched_total["usage"] = realistic_usage()
mismatched_total["total_cost_usd"] = "0.130000"
cases.append(mismatched_total)
for envelope in cases:
with self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertIn(
raised.exception.primary_code,
{"WRITER_MODEL_MISMATCH", "WRITER_RECEIPT_INVALID"},
)
def test_api_key_and_oauth_remain_supported_and_exclusive(self):
"""API key/OAuth 仍可进入最小环境,多个认证字段仍然失败关闭。"""
for auth_field in ("ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"):
runner = runner_for(success_envelope())
run_claude(
execution_profile(),
{"request": "x"},
runner=runner,
binding_verifier=lambda _profile: None,
source_environment={auth_field: "legacy-auth-secret"},
)
self.assertEqual(runner.calls[0][1]["env"].get(auth_field), "legacy-auth-secret")
with self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
binding_verifier=lambda _profile: None,
source_environment={
"ANTHROPIC_API_KEY": "api-secret",
"CLAUDE_CODE_OAUTH_TOKEN": "oauth-secret",
},
)
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
self.assertNotIn("api-secret", str(raised.exception))
self.assertNotIn("oauth-secret", str(raised.exception))
def test_real_runner_requires_an_allowlisted_authentication_field(self):
"""真实 subprocess 没有 API key、OAuth 或 AUTH_TOKEN 时必须阻断。"""
with self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
binding_verifier=lambda _profile: None,
source_environment={"ANTHROPIC_BASE_URL": "https://gateway.example.invalid/v1"},
)
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
self.assertEqual(str(raised.exception), "真实调用缺少经授权的 Claude 认证")
def test_minimal_environment_rejects_invalid_base_url_even_without_auth_token(self):
"""任何被转发的 BASE_URL 都必须是安全的 HTTP(S) 地址。"""
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
with self.assertRaises(ValueError):
_minimal_environment(
{"ANTHROPIC_BASE_URL": "https://user:password@gateway.example.invalid/v1"},
pathlib.Path(directory),
require_authentication=False,
)
def test_claude_temp_and_config_dirs_are_forced_inside_isolation(self):
"""Claude 专用目录变量不能继承外部路径,必须落在同一隔离目录。"""
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
isolation = pathlib.Path(directory)
environment = _minimal_environment(
{
"CLAUDE_CODE_TMPDIR": "/tmp/escape",
"CLAUDE_CONFIG_DIR": "/Users/escape",
},
isolation,
require_authentication=False,
)
self.assertEqual(environment["HOME"], str(isolation))
self.assertEqual(environment["TMPDIR"], str(isolation))
self.assertEqual(environment["CLAUDE_CODE_TMPDIR"], str(isolation))
self.assertEqual(environment["CLAUDE_CONFIG_DIR"], str(isolation))
def test_503_success_subtype_is_still_api_error(self):
"""subtype=success 不能覆盖 is_error、503、空模型 usage 或非零退出。"""
auth_token = "api-error-token-must-not-leak"
base_url = "https://gateway.example.invalid/anthropic/v1"
envelope = success_envelope()
envelope.update(
{
"is_error": True,
"terminal_reason": "api_error",
"api_error_status": 503,
"total_cost_usd": "0",
"usage": {"input_tokens": 0, "output_tokens": 0},
"modelUsage": {},
}
)
runner = runner_for(
envelope,
returncode=1,
stderr=f"gateway={base_url} token={auth_token}",
)
with self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner,
binding_verifier=lambda _profile: None,
source_environment={
"ANTHROPIC_AUTH_TOKEN": auth_token,
"ANTHROPIC_BASE_URL": base_url,
},
)
self.assertEqual(raised.exception.primary_code, "WRITER_API_ERROR")
self.assertIn("WRITER_NONZERO_EXIT", raised.exception.causes)
self.assertNotIn("gateway secret", str(raised.exception))
self.assertNotIn("gateway secret", canonical_json(raised.exception.details))
self.assertNotIn(auth_token, canonical_json(raised.exception.details))
self.assertNotIn(base_url, canonical_json(raised.exception.details))
self.assertNotIn(auth_token, canonical_json(raised.exception.receipt.as_dict()))
self.assertNotIn(base_url, canonical_json(raised.exception.receipt.as_dict()))
self.assertNotIn(auth_token, canonical_json(runner.calls[0][0]))
self.assertNotIn(base_url, canonical_json(runner.calls[0][0]))
def test_result_never_falls_back_when_structured_output_is_missing(self):
"""即使 result 恰好是合法 JSON,也不得把它当业务对象。"""
envelope = success_envelope()
del envelope["structured_output"]
envelope["result"] = '{"value":"forbidden-fallback"}'
with self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.primary_code, "WRITER_SCHEMA_INVALID")
def test_model_mismatch_and_usage_or_cost_gaps_fail_closed(self):
"""实际模型、usage 和成本任一无法核账都不能成功。"""
mismatch = success_envelope()
mismatch["modelUsage"] = {
"claude-sonnet-4-20250514": {"inputTokens": 10, "outputTokens": 2, "costUSD": "0.120000"}
}
missing_usage = success_envelope()
del missing_usage["usage"]
missing_cost = success_envelope()
del missing_cost["total_cost_usd"]
cases = (
(mismatch, "WRITER_MODEL_MISMATCH"),
(missing_usage, "WRITER_RECEIPT_INVALID"),
(missing_cost, "WRITER_RECEIPT_INVALID"),
)
for envelope, expected in cases:
with self.subTest(expected=expected), self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner_for(envelope),
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.primary_code, expected)
def test_timeout_has_highest_priority_and_never_exposes_stderr(self):
"""硬 deadline 的稳定码优先于所有无法取得的 envelope 字段。"""
runner = FakeRunner(
subprocess.TimeoutExpired(["sandbox-exec"], 1, output="raw response", stderr="secret stderr")
)
with self.assertRaises(ClaudeRuntimeError) as raised:
run_claude(
execution_profile(),
{"request": "x"},
runner=runner,
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.primary_code, "WRITER_TIMEOUT")
self.assertIsNotNone(raised.exception.receipt)
self.assertIsNone(raised.exception.receipt.exit_code)
self.assertNotIn("secret stderr", str(raised.exception))
self.assertNotIn("raw response", canonical_json(raised.exception.details))
def test_sandbox_command_contains_all_frozen_flags(self):
"""sandbox 命令必须固定完整模型、预算、schema、无工具和无持久化参数。"""
profile = execution_profile()
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
isolation = pathlib.Path(directory)
command = build_sandbox_command(profile, isolation)
required_pairs = {
"--model": FULL_MODEL_ID,
"--effort": "high",
"--max-budget-usd": "1.500000",
"--output-format": "json",
"--json-schema": canonical_json(
{key: value for key, value in OUTPUT_SCHEMA.items() if key != "$schema"}
),
"--tools": "",
"--mcp-config": '{"mcpServers":{}}',
"--system-prompt": SYSTEM_PROMPT,
}
self.assertEqual(command[0:2], ["/usr/bin/sandbox-exec", "-p"])
self.assertIn("(deny default)", command[2])
self.assertIn("network-outbound", command[2])
self.assertIn("file-write*", command[2])
self.assertIn(f'(allow file-write* (subpath "{isolation}"))', command[2])
self.assertIn(
f'(deny file-write* (require-not (subpath "{isolation}")))', command[2]
)
self.assertIn("--print", command)
self.assertIn("--bare", command)
self.assertIn("--no-session-persistence", command)
self.assertIn("--disable-slash-commands", command)
self.assertIn("--strict-mcp-config", command)
self.assertIn("$schema", profile.json_schema)
self.assertEqual(profile.json_schema_sha256, sha256_json(profile.json_schema))
cli_schema = json.loads(command[command.index("--json-schema") + 1])
self.assertNotIn("$schema", cli_schema)
for flag, expected in required_pairs.items():
self.assertEqual(command[command.index(flag) + 1], expected)
def test_profile_verification_rejects_binary_or_version_drift(self):
"""绝对路径内容或 CLI 版本变化时必须在模型调用前失败。"""
with tempfile.TemporaryDirectory() as directory:
executable = pathlib.Path(directory) / "claude"
executable.write_bytes(b"frozen-binary")
executable.chmod(0o700)
profile = execution_profile(str(executable))
verify_execution_profile(
profile,
version_runner=lambda *_args, **_kwargs: subprocess.CompletedProcess(
[str(executable), "--version"], 0, "2.1.211 (Claude Code)\n", ""
),
)
executable.write_bytes(b"drifted-binary")
with self.assertRaises(ClaudeRuntimeError) as raised:
verify_execution_profile(profile)
self.assertEqual(raised.exception.primary_code, "WRITER_RECEIPT_INVALID")
class PathTraversalDetectionTest(unittest.TestCase):
"""精确路径穿越判定:只认真实路径分量,不伤省略号等合法自由文本。"""
def test_ellipsis_and_normal_prose_are_not_traversal(self):
# WHY: `".." in value` 子串匹配会把省略号 `...`(含子串 `..`)误判为路径穿越,
# 冤杀含省略号的正文/理由/引文;精确路径分量判定必须放行这些合法文本。
safe_values = [
"...",
"林澈说:等等...",
"他说:这一切......都不简单",
"正常的正文,没有穿越。",
"file..name", # 文件名内部连续点,不是路径分量
"..hidden", # 以 .. 开头的普通隐藏名,后面紧跟非分隔符
"a/..b", # ..b 是普通目录名,不是父目录分量
]
for value in safe_values:
with self.subTest(value=value):
self.assertFalse(contains_path_traversal(value))
def test_real_path_traversal_is_detected(self):
traversal_values = [
"a/../b", # 中段父目录穿越
"../x", # 起始父目录穿越
"a/..", # 结尾父目录穿越
"..", # 裸父目录
"a\\..\\b", # 反斜杠分隔的父目录穿越
"..\\x", # 反斜杠起始父目录穿越
"dir/../../etc/passwd", # 多级穿越
]
for value in traversal_values:
with self.subTest(value=value):
self.assertTrue(contains_path_traversal(value))
def test_non_string_input_fails_closed(self):
with self.assertRaises(TypeError):
contains_path_traversal(None) # type: ignore[arg-type]
if __name__ == "__main__":
unittest.main()