diff --git a/.claude/skills/replay-eval/scripts/claude_runtime.py b/.claude/skills/replay-eval/scripts/claude_runtime.py index bd4e6ae..ba5daaf 100644 --- a/.claude/skills/replay-eval/scripts/claude_runtime.py +++ b/.claude/skills/replay-eval/scripts/claude_runtime.py @@ -45,7 +45,9 @@ ENVIRONMENT_ALLOWLIST = frozenset( ) AUTHENTICATION_FIELDS = frozenset({"ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"}) HASH_PATTERN = re.compile(r"^(?:sha256:)?[0-9a-f]{64}$") -MODEL_ID_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{5,127}$") +MODEL_ID_PATTERN = re.compile( + r"^(?=.{6,128}$)[A-Za-z0-9][A-Za-z0-9._:-]{5,127}(?:\[[A-Za-z0-9._:-]+\])?$" +) MONEY_QUANTUM = Decimal("0.000001") diff --git a/.claude/skills/replay-eval/scripts/test_claude_runtime.py b/.claude/skills/replay-eval/scripts/test_claude_runtime.py index f81d47e..384cc28 100644 --- a/.claude/skills/replay-eval/scripts/test_claude_runtime.py +++ b/.claude/skills/replay-eval/scripts/test_claude_runtime.py @@ -29,6 +29,7 @@ from claude_runtime import ( # noqa: E402 FULL_MODEL_ID = "claude-opus-4-1-20250805" +CAPABILITY_MODEL_ID = "claude-opus-4-8[1m]" OUTPUT_SCHEMA = { "type": "object", "additionalProperties": False, @@ -38,7 +39,12 @@ OUTPUT_SCHEMA = { SYSTEM_PROMPT = "只返回符合给定 JSON Schema 的测试对象。" -def execution_profile(executable: str = "/usr/bin/true") -> ExecutionProfile: +def execution_profile( + executable: str = "/usr/bin/true", + *, + model_alias: str = "opus", + resolved_model_id: str = FULL_MODEL_ID, +) -> ExecutionProfile: """构造不依赖真实模型、但字段完整的冻结测试 profile。""" executable_hash = hashlib.sha256(pathlib.Path(executable).read_bytes()).hexdigest() @@ -48,8 +54,8 @@ def execution_profile(executable: str = "/usr/bin/true") -> ExecutionProfile: claude_executable_path=executable, claude_executable_sha256=executable_hash, claude_cli_version="2.1.211", - model_alias="opus", - resolved_model_id=FULL_MODEL_ID, + model_alias=model_alias, + resolved_model_id=resolved_model_id, effort="high", max_budget_usd_per_call=Decimal("1.500000"), timeout_seconds=30, @@ -114,6 +120,49 @@ def runner_for(envelope: dict[str, object], *, returncode: int = 0, stderr: str class ClaudeRuntimeTest(unittest.TestCase): """验证统一 runtime 的成功条件与失败优先级。""" + def test_profile_accepts_exact_model_id_with_capability_suffix(self): + """真实 modelUsage 的方括号能力后缀必须能构造冻结 profile。""" + + profile = execution_profile(resolved_model_id=CAPABILITY_MODEL_ID) + + self.assertEqual(profile.resolved_model_id, CAPABILITY_MODEL_ID) + + def test_model_id_capability_suffix_is_narrow_and_terminal(self): + """能力后缀只能是末尾单个非空安全字符集合。""" + + invalid_model_ids = ( + "claude-opus-4-8[]", + "claude-opus-4-8[[1m]]", + "claude-opus-4-8[1m[2m]", + "claude-opus-4-8[1m/2m]", + "claude-opus-4-8[1m]suffix", + "claude-opus-4-8[1m][2m]", + ) + for model_id in invalid_model_ids: + with self.subTest(model_id=model_id), self.assertRaises(ValueError): + execution_profile(resolved_model_id=model_id) + + def test_model_id_length_includes_capability_suffix(self): + """方括号后缀计入模型 ID 的 6 到 128 字符总长度。""" + + max_length_model_id = "a" * 125 + "[x]" + over_max_length_model_id = "a" * 126 + "[x]" + + profile = execution_profile(resolved_model_id=max_length_model_id) + self.assertEqual(len(profile.resolved_model_id), 128) + + with self.assertRaises(ValueError): + execution_profile(resolved_model_id=over_max_length_model_id) + + def test_capability_suffix_does_not_bypass_alias_rejection(self): + """方括号后缀不能让 alias 与 resolved model ID 相等的 profile 通过。""" + + with self.assertRaises(ValueError): + execution_profile( + model_alias=CAPABILITY_MODEL_ID, + resolved_model_id=CAPABILITY_MODEL_ID, + ) + def test_success_uses_structured_output_and_emits_complete_receipt(self): """成功必须返回 structured_output,并形成不含正文的完整回执。"""