2434 lines
106 KiB
Python
2434 lines
106 KiB
Python
#!/usr/bin/env python3
|
||
"""正文 A/B/C 冻结回放编排的无网络、无真实模型测试。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import hashlib
|
||
import inspect
|
||
import json
|
||
import pathlib
|
||
import subprocess
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
from unittest import mock
|
||
from dataclasses import replace
|
||
from decimal import Decimal
|
||
from datetime import datetime, timedelta, timezone
|
||
|
||
SCRIPT_DIR = pathlib.Path(__file__).resolve().parent
|
||
SKILLS_DIR = SCRIPT_DIR.parents[1]
|
||
READ_CONTEXT_DIR = SKILLS_DIR / "read-context" / "scripts"
|
||
RUNTIME_DIR = SKILLS_DIR / "runtime" / "scripts"
|
||
QUALITY_GATE_DIR = SKILLS_DIR / "quality-gate" / "scripts"
|
||
for import_path in (SCRIPT_DIR, READ_CONTEXT_DIR, RUNTIME_DIR, QUALITY_GATE_DIR):
|
||
sys.path.insert(0, str(import_path))
|
||
|
||
import run_writer_replay as replay_module # noqa: E402
|
||
from run_writer_replay import ( # noqa: E402
|
||
BudgetLedgerError,
|
||
WriterReplayError,
|
||
WriterReplayProductionAdapters,
|
||
WriterReplayTestAdapters,
|
||
run_writer_replay,
|
||
)
|
||
from claude_runtime import ExecutionProfile, sha256_json, sha256_text # noqa: E402
|
||
from file_cas import CasConflictError, FileCasStore # noqa: E402
|
||
from gate_input_builder import GateInputBuildError, GateInputBuilder, canonical_sha256 # noqa: E402
|
||
from raw_vault import RawVaultError, RawVaultManager # noqa: E402
|
||
from run_writer_blind_judge import BLIND_JUDGE_REPORT_JSON_SCHEMA # noqa: E402
|
||
from run_writer_semantic_detector import SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA # noqa: E402
|
||
from run_writer import build_writer_execution_profile # noqa: E402
|
||
from writer_contract import calculate_target_chars, han_count, validate_writer_context # noqa: E402
|
||
from writer_rubric import DIMENSIONS, RUBRIC_PROFILE, adjudicate_structured_reviews # noqa: E402
|
||
|
||
|
||
AUTHORIZATION = {
|
||
"sourceStatus": "authorized",
|
||
"copyrightStatus": "research_only",
|
||
"sourceHash": "sha256:" + "a" * 64,
|
||
"sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64,
|
||
"allowedPurpose": ["offline_evaluation"],
|
||
"forbiddenPurpose": ["external_distribution", "model_training", "production_generation"],
|
||
"authorizationSnapshot": {
|
||
"id": "auth-work-8",
|
||
"version": "auth-work-8-v1",
|
||
"immutable": True,
|
||
"sourceHash": "sha256:" + "a" * 64,
|
||
"sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64,
|
||
"sourceStatus": "authorized",
|
||
"copyrightStatus": "research_only",
|
||
"authorizationBasis": "sanitized_contract_fixture",
|
||
"allowedPurpose": ["offline_evaluation"],
|
||
"forbiddenPurpose": ["external_distribution", "model_training", "production_generation"],
|
||
"checkedAt": "2026-07-20T00:00:00Z",
|
||
"revalidationAt": "2026-08-19T00:00:00Z",
|
||
},
|
||
}
|
||
|
||
|
||
def _prose(chapter: int, block_id: int, text: str) -> dict[str, object]:
|
||
"""构造只用于合同证明的脱敏合成原文片段。"""
|
||
|
||
return {
|
||
"chapter": chapter,
|
||
"sourceRef": {
|
||
"sourceId": f"fixture:chapter:{chapter}:block:{block_id}",
|
||
"sourceVersion": f"sanitized-fixture-{chapter}-v1",
|
||
"chapter": chapter,
|
||
"blockId": block_id,
|
||
"startCodePoint": 0,
|
||
"endCodePoint": len(text),
|
||
},
|
||
"text": text,
|
||
}
|
||
|
||
|
||
def _writer_context_input() -> dict[str, object]:
|
||
"""构造可让 A/B/C 都通过 WriterContext v1 的脱敏夹具。"""
|
||
|
||
recent = [
|
||
_prose(chapter, 1000 + chapter, f"脱敏合同夹具第{chapter}章,人物保持冻结状态。")
|
||
for chapter in range(485, 489)
|
||
]
|
||
supplemental = _prose(120, 1120, "脱敏补充片段,旧徽章曾经出现。")
|
||
return {
|
||
"contentMode": "sanitized_contract_fixture",
|
||
"sourceVersion": AUTHORIZATION["sourceVersion"],
|
||
"authorizationSnapshot": {
|
||
"snapshotId": "auth-work-8",
|
||
"allowedPurpose": "offline_evaluation",
|
||
"verifiedAt": "2026-07-20T00:00:00Z",
|
||
},
|
||
"sourceStatus": "authorized",
|
||
"cardIndexVersion": "cards-frozen-488-v1",
|
||
"proseIndexVersion": "prose-frozen-488-v1",
|
||
"retrievalResult": {
|
||
"cards": [
|
||
{
|
||
"name": "林澈",
|
||
"type": "character",
|
||
"sourceRefs": [copy.deepcopy(supplemental["sourceRef"])],
|
||
}
|
||
],
|
||
"factEvidence": [],
|
||
"proseEvidence": [supplemental],
|
||
"indexHints": [
|
||
{
|
||
"cardId": "card-1",
|
||
"name": "林澈",
|
||
"type": "character",
|
||
"content": "林澈在冻结点仍位于圣蒂曼",
|
||
"sourceId": "fixture:card:1",
|
||
"sourceVersion": "cards-frozen-488-v1",
|
||
"asOf": 488,
|
||
}
|
||
],
|
||
"manifest": {"omittedSources": []},
|
||
},
|
||
"fineOutline": {
|
||
"sourceRef": {
|
||
"sourceId": "fixture:fine-outline:489",
|
||
"sourceVersion": "fine-outline-fixture-v1",
|
||
"chapter": 489,
|
||
},
|
||
"hardConstraints": ["必须完成围攻突围"],
|
||
"adjustableBeats": [],
|
||
"declaredNewFacts": [],
|
||
"entities": [{"id": "character:林澈", "type": "character", "name": "林澈"}],
|
||
"relations": [],
|
||
"items": [{"id": "item:旧徽章", "type": "item", "name": "旧徽章"}],
|
||
"locations": [{"id": "location:圣蒂曼", "type": "location", "name": "圣蒂曼"}],
|
||
"powerSystems": [],
|
||
},
|
||
"narrativeState": {
|
||
"time": "围攻当日",
|
||
"location": "圣蒂曼",
|
||
"characterPositions": {"林澈": "城内"},
|
||
"immediateSituation": "围攻持续",
|
||
},
|
||
"recentChapters": recent,
|
||
"outputContract": {
|
||
"targetChars": 4000,
|
||
"minChars": 2800,
|
||
"maxChars": 5200,
|
||
"frontmatterRequired": False,
|
||
},
|
||
"tokenBudget": {"maxContextChars": 50000},
|
||
"generatedAt": "2026-07-20T00:00:00Z",
|
||
"requirements": {
|
||
"requiredEvents": [
|
||
{"requirementId": "event-1", "anchors": ["完成围攻突围"]}
|
||
],
|
||
"requiredCharacters": ["林澈"],
|
||
"foreshadowingActions": [
|
||
{"requirementId": "foreshadow-1", "anchors": ["旧徽章"]}
|
||
],
|
||
"chapterEndHook": {
|
||
"requirementId": "hook-1",
|
||
"anchors": ["城门忽然打开"],
|
||
"maxDistanceFromEnd": 20,
|
||
},
|
||
"detectedNewSettings": [],
|
||
"frozenConflicts": [],
|
||
},
|
||
}
|
||
|
||
|
||
def config() -> dict[str, object]:
|
||
"""构造不含原文全文的单样本预注册配置。"""
|
||
|
||
common = {
|
||
"workId": 8,
|
||
"asOfChapter": 488,
|
||
"targetChapter": 489,
|
||
"outlineSource": "outline:481-488",
|
||
"fineOutlineSource": "scaffold:489",
|
||
"targetChars": 4000,
|
||
"modelVersion": "opus",
|
||
"sampling": {"temperature": 0, "topP": 1, "seed": 489},
|
||
"detectorProfile": "writer-detector-report-v1",
|
||
}
|
||
value = {
|
||
"profile": "writer_replay",
|
||
"evaluationSetVersion": "writer-gate-a-test-v1",
|
||
"strategyVersion": "writer-abc-v1",
|
||
"referenceWork": {
|
||
"id": 8,
|
||
"title": "深空之影",
|
||
"version": AUTHORIZATION["sourceVersion"],
|
||
},
|
||
"authorization": copy.deepcopy(AUTHORIZATION),
|
||
"commonControls": common,
|
||
"samples": [
|
||
{
|
||
"sampleId": "deep-space-489",
|
||
"scenario": "combat",
|
||
"asOfChapter": 488,
|
||
"targetChapter": 489,
|
||
"snapshotVersion": "writer-deep-space-489-v1",
|
||
"frozenRecentHanCounts": [4600, 4600, 4800, 4800],
|
||
"targetLengthBasis": {
|
||
"algorithm": "calculate_target_chars",
|
||
"sourceChapters": [485, 486, 487, 488],
|
||
"hardEventCount": 1,
|
||
"foreshadowingActionCount": 0,
|
||
"requiredSceneCount": 0,
|
||
"minChars": 2000,
|
||
"maxChars": 10000,
|
||
"usesTargetChapterLength": False,
|
||
},
|
||
"targetChars": 4000,
|
||
"expectedLength": {
|
||
"targetChars": 4000,
|
||
"minChars": 2800,
|
||
"maxChars": 5200,
|
||
},
|
||
"newCharacterRatio": 0.0,
|
||
"newCharacterRatioStatus": "resolved",
|
||
"newCharacterBasis": {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": 488,
|
||
"requiredCharacters": ["林澈"],
|
||
"knownBeforeAsOf": ["林澈"],
|
||
"absentBeforeAsOf": [],
|
||
"genericRoles": [],
|
||
},
|
||
"sources": [
|
||
{
|
||
"sourceId": "fixture:chapter:485-488",
|
||
"sourceVersion": AUTHORIZATION["sourceVersion"],
|
||
"chapterRange": "485-488",
|
||
},
|
||
{
|
||
"sourceId": "fixture:card-index:488",
|
||
"sourceVersion": "cards-frozen-488-v1",
|
||
"chapterRange": "1-488",
|
||
},
|
||
],
|
||
"snapshotData": {
|
||
"chapters": [
|
||
{
|
||
"chapter": 488,
|
||
"sourceId": "fixture:chapter:488",
|
||
"contentSha256": "sha256:" + "1" * 64,
|
||
}
|
||
],
|
||
"cards": [],
|
||
},
|
||
"writerContextInput": _writer_context_input(),
|
||
"leakageAudit": {
|
||
"method": "target-fact-hash-and-chapter-bound-audit",
|
||
"targetFacts": {
|
||
"targetChapter": 489,
|
||
"forbiddenFacts": [
|
||
{
|
||
"id": "target-489-1",
|
||
"firstChapter": 489,
|
||
"text": "目标章专属秘密",
|
||
}
|
||
],
|
||
},
|
||
},
|
||
}
|
||
],
|
||
}
|
||
value["executionAuthorization"] = _test_execution_authorization()
|
||
return value
|
||
|
||
|
||
def _candidate_body(marker: str) -> str:
|
||
"""构造满足动态篇幅和全部机械锚点的三臂候选。"""
|
||
|
||
prefix = f"林澈{marker}完成围攻突围,又把旧徽章压回掌心。"
|
||
hook = "城门忽然打开"
|
||
filler_count = 4000 - han_count(prefix) - han_count(hook)
|
||
return prefix + "文" * filler_count + hook
|
||
|
||
|
||
def _writer_output(
|
||
context: dict[str, object], *, call_index: int
|
||
) -> dict[str, object]:
|
||
"""从 subprocess stdin 创作输入构造最小 WriterDraft v2。"""
|
||
|
||
# A/C 投影不得泄露臂策略;测试按预注册调用顺序区分候选,不向输入补旧字段。
|
||
marker = ("甲", "乙", "丙")[call_index % 3]
|
||
body = _candidate_body(marker)
|
||
return {"candidateBody": body}
|
||
|
||
|
||
class FakeSubprocessRunner:
|
||
"""只替换 subprocess runner,候选仍由 run_writer() 解析和校验。"""
|
||
|
||
def __init__(self, *, illegal_extra_field: bool = False, include_legacy_result: bool = True):
|
||
self.calls: list[tuple[list[str], dict[str, object]]] = []
|
||
self.contexts: list[dict[str, object]] = []
|
||
self.illegal_extra_field = illegal_extra_field
|
||
self.include_legacy_result = include_legacy_result
|
||
|
||
def __call__(self, command: list[str], **kwargs: object) -> subprocess.CompletedProcess[str]:
|
||
context = json.loads(str(kwargs["input"]))
|
||
self.assert_safe_projection(context)
|
||
self.calls.append((command, kwargs))
|
||
self.contexts.append(context)
|
||
output = _writer_output(context, call_index=len(self.contexts) - 1)
|
||
if self.illegal_extra_field and len(self.contexts) == 2:
|
||
output["candidateSha256"] = "sha256:" + "0" * 64
|
||
envelope = {
|
||
"type": "result",
|
||
"subtype": "success",
|
||
"is_error": False,
|
||
# CLI 2.1.211 真实 envelope:terminal_reason 是 "completed",subtype 才是 "success"
|
||
"terminal_reason": "completed",
|
||
"stop_reason": "end_turn",
|
||
"api_error_status": None,
|
||
"total_cost_usd": "0.010000",
|
||
"usage": {"input_tokens": 10, "output_tokens": 10},
|
||
"modelUsage": {
|
||
"claude-opus-4-1-20250805": {
|
||
"inputTokens": 10,
|
||
"outputTokens": 10,
|
||
"costUSD": "0.010000",
|
||
}
|
||
},
|
||
"structured_output": output,
|
||
}
|
||
if self.include_legacy_result:
|
||
# 旧边界保留诱饵字段,证明 runtime 不会回退读取 result。
|
||
envelope["result"] = "禁止作为业务 fallback"
|
||
stdout = json.dumps(envelope, ensure_ascii=False)
|
||
return subprocess.CompletedProcess(command, 0, stdout=stdout, stderr="")
|
||
|
||
@staticmethod
|
||
def assert_safe_projection(context: dict[str, object]) -> None:
|
||
"""证明模型只收到 WriterCreativeInput v2;篇幅修订调用可额外带一个 lengthRevision 块。"""
|
||
|
||
expected = {
|
||
"fineOutline",
|
||
"narrativeState",
|
||
"factConstraints",
|
||
"proseExcerpts",
|
||
"patternReferences",
|
||
"lengthContract",
|
||
"styleConstraints",
|
||
}
|
||
keys = set(context)
|
||
if keys != expected and keys != expected | {"lengthRevision"}:
|
||
raise AssertionError(f"writer 创作投影字段非法: {sorted(context)}")
|
||
if "lengthRevision" in context:
|
||
# lengthRevision 只允许携带模型自己的上一版正文、篇幅数字与固定修订指令,
|
||
# 不得混入臂策略、raw 路径等投影外字段。
|
||
allowed_revision = {
|
||
"previousDraft",
|
||
"actualHanChars",
|
||
"minChars",
|
||
"maxChars",
|
||
"targetChars",
|
||
"instruction",
|
||
}
|
||
revision = context["lengthRevision"]
|
||
if not isinstance(revision, dict) or set(revision) != allowed_revision:
|
||
raise AssertionError(
|
||
f"lengthRevision 块字段非法: {sorted(revision) if isinstance(revision, dict) else revision}"
|
||
)
|
||
|
||
|
||
class FakeSemanticDetector:
|
||
"""记录机械门输出,并返回与候选绑定的语义通过报告。"""
|
||
|
||
def __init__(self) -> None:
|
||
self.calls: list[tuple[str, bool]] = []
|
||
|
||
def __call__(self, context, candidate, mechanical_report):
|
||
strategy = str(context.get("evidenceStrategy") or "neutral_prose_projection")
|
||
self.calls.append((strategy, mechanical_report["passed"]))
|
||
return {
|
||
"status": "passed",
|
||
"candidateVersion": candidate["candidateVersion"],
|
||
"candidateSha256": candidate["candidateSha256"],
|
||
"blockingFailures": [],
|
||
"suggestions": [],
|
||
}
|
||
|
||
|
||
def judge_report(
|
||
reviewer_id: str,
|
||
sample_id: str,
|
||
blind_id: str,
|
||
order: list[str],
|
||
score: float = 8.0,
|
||
) -> dict[str, object]:
|
||
"""构造测试盲评报告。"""
|
||
|
||
return {
|
||
"profile": RUBRIC_PROFILE,
|
||
"reviewerId": reviewer_id,
|
||
"sampleId": sample_id,
|
||
"blindCandidateId": blind_id,
|
||
"candidateOrder": order,
|
||
"scores": {
|
||
dimension: {
|
||
"score": score,
|
||
"evidence": [
|
||
{
|
||
"sourceType": "judge_inference",
|
||
"sourceRef": f"candidate:{blind_id}",
|
||
"excerpt": f"证据-{dimension}",
|
||
}
|
||
],
|
||
}
|
||
for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
|
||
|
||
class FakeJudge:
|
||
"""模拟双评差异和必要第三评,不读取网络或真实模型。"""
|
||
|
||
def __init__(
|
||
self,
|
||
*,
|
||
unstable: bool = False,
|
||
invalid_third: bool = False,
|
||
irreducibly_unstable: bool = False,
|
||
):
|
||
self.calls: list[tuple[str, str]] = []
|
||
self.unstable = unstable
|
||
self.invalid_third = invalid_third
|
||
self.irreducibly_unstable = irreducibly_unstable
|
||
|
||
def __call__(self, blind_input, reviewer_id):
|
||
blind_id = blind_input["blindCandidateId"]
|
||
candidate_order = blind_input["candidateOrder"]
|
||
self.calls.append((reviewer_id, blind_id))
|
||
if self.invalid_third and reviewer_id == "judge-3":
|
||
return {"status": "malformed"}
|
||
score = 8.0
|
||
if self.unstable and blind_id == "blind-1" and reviewer_id == "judge-2":
|
||
score = 7.0
|
||
if self.unstable and blind_id == "blind-1" and reviewer_id == "judge-3":
|
||
score = 7.5
|
||
if self.irreducibly_unstable and blind_id == "blind-1":
|
||
score = {"judge-1": 8.0, "judge-2": 6.0, "judge-3": 7.0}[reviewer_id]
|
||
return judge_report(
|
||
reviewer_id,
|
||
blind_input["sample"]["sampleId"],
|
||
blind_id,
|
||
candidate_order,
|
||
score,
|
||
)
|
||
|
||
|
||
class MaliciousJudge(FakeJudge):
|
||
"""主动扫描全部可见输入,证明 judge 无法发现真实臂或原始目录。"""
|
||
|
||
def __init__(self) -> None:
|
||
super().__init__()
|
||
self.visible_inputs: list[dict[str, object]] = []
|
||
|
||
def __call__(self, blind_input, reviewer_id):
|
||
rendered = json.dumps(blind_input, ensure_ascii=False, sort_keys=True)
|
||
forbidden = (
|
||
"candidate-A",
|
||
"candidate-B",
|
||
"candidate-C",
|
||
"pipeline-A",
|
||
"pipeline-B",
|
||
"pipeline-C",
|
||
"evidenceStrategy",
|
||
"writerContextInput",
|
||
"indexHints",
|
||
"retrievalManifest",
|
||
"evidenceCoverage",
|
||
"raw_dir",
|
||
"rawDir",
|
||
"_contexts",
|
||
)
|
||
for marker in forbidden:
|
||
if marker in rendered:
|
||
raise AssertionError(f"judge 可见输入泄露: {marker}")
|
||
self.visible_inputs.append(copy.deepcopy(blind_input))
|
||
return super().__call__(blind_input, reviewer_id)
|
||
|
||
|
||
def _test_adapters(
|
||
*,
|
||
unstable: bool = False,
|
||
illegal_extra_field: bool = False,
|
||
invalid_third: bool = False,
|
||
irreducibly_unstable: bool = False,
|
||
) -> tuple[WriterReplayTestAdapters, FakeSubprocessRunner, FakeSemanticDetector, FakeJudge]:
|
||
"""集中构造测试注入对象,避免它们被误认为生产 runtime。"""
|
||
|
||
runner = FakeSubprocessRunner(illegal_extra_field=illegal_extra_field)
|
||
detector = FakeSemanticDetector()
|
||
judge = FakeJudge(
|
||
unstable=unstable,
|
||
invalid_third=invalid_third,
|
||
irreducibly_unstable=irreducibly_unstable,
|
||
)
|
||
profile = _frozen_test_profile()
|
||
return (
|
||
WriterReplayTestAdapters(runner, profile, detector, judge),
|
||
runner,
|
||
detector,
|
||
judge,
|
||
)
|
||
|
||
|
||
def _frozen_test_profile():
|
||
"""构造不访问网络且字段完整的冻结 writer 测试 profile。"""
|
||
|
||
executable = pathlib.Path("/usr/bin/true")
|
||
return build_writer_execution_profile(
|
||
claude_executable_path=str(executable),
|
||
claude_executable_sha256=hashlib.sha256(executable.read_bytes()).hexdigest(),
|
||
claude_cli_version="2.1.211",
|
||
resolved_model_id="claude-opus-4-1-20250805",
|
||
effort="high",
|
||
max_budget_usd_per_call=Decimal("1.000000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=200000,
|
||
system_prompt="只返回严格 WriterDraft v2。",
|
||
)
|
||
|
||
|
||
def _production_writer_profile():
|
||
"""使用可真实探测版本的本机 Python 模拟生产可执行文件绑定。"""
|
||
|
||
executable = pathlib.Path(sys.executable).resolve()
|
||
return build_writer_execution_profile(
|
||
claude_executable_path=str(executable),
|
||
claude_executable_sha256=hashlib.sha256(executable.read_bytes()).hexdigest(),
|
||
claude_cli_version=f"Python {sys.version_info.major}.{sys.version_info.minor}",
|
||
resolved_model_id="claude-opus-4-1-20250805",
|
||
effort="high",
|
||
max_budget_usd_per_call=Decimal("1.000000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=200000,
|
||
system_prompt="只返回严格 WriterDraft v2。",
|
||
)
|
||
|
||
|
||
def _test_execution_authorization() -> dict[str, object]:
|
||
"""给测试适配器显式审批 probe、预算、raw 和 writer profile,禁止参数隐式放行。"""
|
||
|
||
writer = _frozen_test_profile()
|
||
probe = {
|
||
"status": "successful",
|
||
"resolvedModelId": writer.resolved_model_id,
|
||
"structuredOutputProbeSha256": "sha256:" + "6" * 64,
|
||
}
|
||
budget = {
|
||
"status": "approved",
|
||
"authorizationId": "budget-test-adapter-1",
|
||
"approvedBy": "user",
|
||
"totalBudgetUsd": "9.000000",
|
||
"plannedCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
"maxCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
}
|
||
raw = {
|
||
"status": "approved",
|
||
"authorizationId": "raw-test-adapter-1",
|
||
"approvedBy": "user",
|
||
"retainUntil": (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(),
|
||
}
|
||
for item in (probe, budget, raw):
|
||
item["receiptSha256"] = canonical_sha256(item)
|
||
return {
|
||
"runtimeProbe": probe,
|
||
"budget": budget,
|
||
"rawRetention": raw,
|
||
"profileSha256": {"writer": writer.execution_profile_sha256},
|
||
}
|
||
|
||
|
||
def _role_profile(role: str, schema: dict[str, object]) -> ExecutionProfile:
|
||
"""构造语义或盲评使用的完整冻结测试 profile。"""
|
||
|
||
executable = pathlib.Path(sys.executable).resolve()
|
||
prompt = f"只返回严格 {role} structured output。"
|
||
return ExecutionProfile(
|
||
profile_version="writer-eval-profile-v1",
|
||
adapter_role=role,
|
||
claude_executable_path=str(executable),
|
||
claude_executable_sha256=hashlib.sha256(executable.read_bytes()).hexdigest(),
|
||
claude_cli_version=f"Python {sys.version_info.major}.{sys.version_info.minor}",
|
||
model_alias="opus",
|
||
resolved_model_id="claude-opus-4-1-20250805",
|
||
effort="high",
|
||
max_budget_usd_per_call=Decimal("1.000000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=200000,
|
||
json_schema_id=f"{role}-schema-v1",
|
||
json_schema=schema,
|
||
json_schema_sha256=sha256_json(schema),
|
||
system_prompt_id=f"{role}-prompt-v1",
|
||
system_prompt=prompt,
|
||
system_prompt_sha256=sha256_text(prompt),
|
||
normal_terminal_reasons=("success",),
|
||
)
|
||
|
||
|
||
def _fake_model_receipt(
|
||
role: str,
|
||
invocation: int,
|
||
model_input: dict[str, object],
|
||
structured_output: dict[str, object],
|
||
) -> dict[str, object]:
|
||
"""构造字段完整、模型匹配且输入/输出哈希均冻结的 fake ExecutionReceipt。"""
|
||
|
||
model_id = "claude-opus-4-1-20250805"
|
||
return {
|
||
"adapterRole": role,
|
||
"invocationId": f"{role}-{invocation}",
|
||
"executionProfileSha256": canonical_sha256({"role": role, "profile": 1}),
|
||
"requestedModelId": model_id,
|
||
"actualModelId": model_id,
|
||
"modelMatch": True,
|
||
"effort": "high",
|
||
"maxBudgetUsdPerCall": "1.000000",
|
||
"totalCostUsd": "0.010000",
|
||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||
"modelUsage": {model_id: {"costUSD": "0.010000"}},
|
||
"stopReason": "end_turn",
|
||
"terminalReason": "success",
|
||
"isError": False,
|
||
"apiErrorStatus": None,
|
||
"exitCode": 0,
|
||
"durationMs": 1,
|
||
"inputSha256": canonical_sha256(model_input),
|
||
"structuredOutputSha256": canonical_sha256(structured_output),
|
||
"jsonSchemaSha256": canonical_sha256({"role": role, "schema": 1}),
|
||
}
|
||
|
||
|
||
class ProductionSemanticRunner:
|
||
"""动态回放 semantic v3 的最小模型草稿。"""
|
||
|
||
def __init__(self, *, fail_on_call: int | None = None) -> None:
|
||
self.fail_on_call = fail_on_call
|
||
self.calls: list[dict[str, object]] = []
|
||
self.receipts: list[dict[str, object]] = []
|
||
self.structured_outputs: list[dict[str, object]] = []
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
"""生成合法报告;指定调用可返回真实 failed 语义终态。"""
|
||
|
||
self.calls.append(copy.deepcopy(dict(model_input)))
|
||
failed = self.fail_on_call == len(self.calls)
|
||
body = model_input["candidateBody"]
|
||
constraints = model_input["hardConstraints"]
|
||
quote = body[:3]
|
||
evidence_id = constraints[0]["constraintId"] if constraints else "candidate-body"
|
||
draft = {
|
||
"schemaVersion": "semantic-detection-draft-v3",
|
||
"claims": [],
|
||
"findings": (
|
||
[
|
||
{
|
||
"findingId": "finding-high-1",
|
||
"severity": "high",
|
||
"category": "hard_constraint",
|
||
"candidateQuote": quote,
|
||
"evidenceIds": [evidence_id],
|
||
"message": "硬约束未满足",
|
||
}
|
||
]
|
||
if failed
|
||
else []
|
||
),
|
||
"assertionVerdicts": [
|
||
{
|
||
"assertionId": item["assertionId"],
|
||
"verdict": "pass",
|
||
"candidateQuote": quote,
|
||
"evidenceIds": [item["evidenceId"]],
|
||
}
|
||
for item in model_input["factEvidence"]
|
||
],
|
||
"hardConstraintVerdicts": [
|
||
{
|
||
"constraintId": item["constraintId"],
|
||
"verdict": "fail" if failed else "pass",
|
||
"candidateQuote": quote,
|
||
"evidenceIds": [item["constraintId"]],
|
||
}
|
||
for item in constraints
|
||
],
|
||
"newSettingCandidates": [],
|
||
"evidenceGaps": [],
|
||
}
|
||
receipt = _fake_model_receipt(
|
||
adapter_role,
|
||
len(self.calls),
|
||
dict(model_input),
|
||
draft,
|
||
)
|
||
self.receipts.append(receipt)
|
||
self.structured_outputs.append(copy.deepcopy(draft))
|
||
receipt_hash = sha256_json(receipt)
|
||
self.assertEqual(output_schema, SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
return {"structuredOutput": draft, "modelReceiptSha256": receipt_hash}
|
||
|
||
@staticmethod
|
||
def assertEqual(left, right):
|
||
"""在 fake 内保持断言失败信息直接。"""
|
||
|
||
if left != right:
|
||
raise AssertionError("semantic adapter 未传正式 v3 model schema")
|
||
|
||
|
||
class ProductionJudgeRunner:
|
||
"""动态回放 blind judge v3 模型草稿,并可制造第三评不稳定。"""
|
||
|
||
def __init__(self, *, unstable: bool = False) -> None:
|
||
self.unstable = unstable
|
||
self.calls: list[dict[str, object]] = []
|
||
self.receipts: list[dict[str, object]] = []
|
||
self.structured_outputs: list[dict[str, object]] = []
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
"""按 reviewer 次序生成严格绑定的五维评分与 verdict。"""
|
||
|
||
self.calls.append(copy.deepcopy(dict(model_input)))
|
||
call_index = len(self.calls)
|
||
score = 8.0
|
||
if self.unstable:
|
||
score = {1: 8.0, 2: 6.0, 3: 7.0}.get(call_index, 7.0)
|
||
assertion_ids = [
|
||
item["assertionId"]
|
||
for field in ("historicalAssertions", "targetAssertions")
|
||
for item in model_input["oracleTruthPack"][field]
|
||
]
|
||
constraint_ids = [
|
||
item["constraintId"] for item in model_input["fineOutline"]["hardConstraints"]
|
||
]
|
||
candidate_ids = [item["blindCandidateId"] for item in model_input["candidates"]]
|
||
draft = {
|
||
"schemaVersion": "blind-judge-draft-v3",
|
||
"candidateScores": [
|
||
{
|
||
"blindCandidateId": candidate["blindCandidateId"],
|
||
"scores": {
|
||
dimension: {
|
||
"score": score,
|
||
"reason": f"{dimension} 维度满足要求",
|
||
"candidateQuote": candidate["candidateBody"][:3],
|
||
"evidenceRefs": [{"sourceType": "candidate", "sourceId": candidate["blindCandidateId"]}],
|
||
}
|
||
for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
for candidate in model_input["candidates"]
|
||
],
|
||
"dimensionPreferences": [
|
||
{
|
||
"dimension": dimension,
|
||
"orderedCandidateIds": list(candidate_ids),
|
||
"reason": f"按 {dimension} 维度比较",
|
||
}
|
||
for dimension in DIMENSIONS
|
||
],
|
||
"oracleAssertionVerdicts": [
|
||
{
|
||
"assertionId": assertion_id,
|
||
"blindCandidateId": candidate["blindCandidateId"],
|
||
"verdict": "pass",
|
||
"reason": "候选与 oracle 断言一致",
|
||
"candidateQuote": candidate["candidateBody"][:3],
|
||
"evidenceRefs": [{"sourceType": "oracle_assertion", "sourceId": assertion_id}],
|
||
}
|
||
for candidate in model_input["candidates"]
|
||
for assertion_id in assertion_ids
|
||
],
|
||
"hardConstraintVerdicts": [
|
||
{
|
||
"constraintId": constraint_id,
|
||
"blindCandidateId": candidate["blindCandidateId"],
|
||
"verdict": "pass",
|
||
"reason": "候选满足细纲硬约束",
|
||
"candidateQuote": candidate["candidateBody"][:3],
|
||
"evidenceRefs": [{"sourceType": "fine_outline", "sourceId": constraint_id}],
|
||
}
|
||
for candidate in model_input["candidates"]
|
||
for constraint_id in constraint_ids
|
||
],
|
||
}
|
||
receipt = _fake_model_receipt(
|
||
adapter_role,
|
||
call_index,
|
||
dict(model_input),
|
||
draft,
|
||
)
|
||
self.receipts.append(receipt)
|
||
self.structured_outputs.append(copy.deepcopy(draft))
|
||
receipt_hash = sha256_json(receipt)
|
||
if output_schema != BLIND_JUDGE_REPORT_JSON_SCHEMA:
|
||
raise AssertionError("judge adapter 未传正式 v3 model schema")
|
||
return {"structuredOutput": draft, "modelReceiptSha256": receipt_hash}
|
||
|
||
|
||
def _production_config(*, budget_approved: bool = True) -> dict[str, object]:
|
||
"""构造 probe、预算、raw 和三 profile 哈希均可复核的执行配置。"""
|
||
|
||
value = config()
|
||
context_input = value["samples"][0]["writerContextInput"]
|
||
context_input["contentMode"] = "canonical_frozen_prose"
|
||
prose_a = {**_prose(120, 2120, "甲侧旧徽章发出微光。"), "retrievalArm": "A"}
|
||
prose_c = {**_prose(121, 2121, "丙侧旧徽章传来回响。"), "retrievalArm": "C"}
|
||
context_input["retrievalResult"]["proseEvidence"] = [prose_a, prose_c]
|
||
value["samples"][0]["proseCharBudget"] = len(prose_a["text"])
|
||
value["commonControls"]["adapterVersion"] = "writer-runtime-v1"
|
||
writer = _production_writer_profile()
|
||
semantic = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
judge = _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA)
|
||
value["commonControls"]["maxContextChars"] = writer.max_context_chars
|
||
probe = {
|
||
"status": "successful",
|
||
"claudeExecutablePath": writer.claude_executable_path,
|
||
"claudeExecutableSha256": writer.claude_executable_sha256,
|
||
"claudeCliVersion": writer.claude_cli_version,
|
||
"modelAlias": writer.model_alias,
|
||
"resolvedModelId": writer.resolved_model_id,
|
||
# 探针合同绑定:身份哈希绑定当前 writer 合同,证明它是按新 schema/prompt 重跑的;
|
||
# structuredOutputSha256 是合规结构化输出内容哈希,证明探针确实跑出过合规输出。
|
||
"executionProfileSha256": writer.execution_profile_sha256,
|
||
"structuredOutputSha256": "sha256:" + "6" * 64,
|
||
}
|
||
budget = {
|
||
"status": "approved" if budget_approved else "pending",
|
||
"authorizationId": "budget-test-1",
|
||
"approvedBy": "user",
|
||
"totalBudgetUsd": "9.000000",
|
||
"plannedCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
"maxCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
}
|
||
raw = {
|
||
"status": "approved",
|
||
"authorizationId": "raw-test-1",
|
||
"approvedBy": "user",
|
||
"retainUntil": (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(),
|
||
}
|
||
for item in (probe, budget, raw):
|
||
item["receiptSha256"] = canonical_sha256(item)
|
||
value["executionAuthorization"] = {
|
||
"runtimeProbe": probe,
|
||
"budget": budget,
|
||
"rawRetention": raw,
|
||
"profileSha256": {
|
||
"writer": writer.execution_profile_sha256,
|
||
"semantic_detector": semantic.execution_profile_sha256,
|
||
"blind_judge": judge.execution_profile_sha256,
|
||
},
|
||
}
|
||
return value
|
||
|
||
|
||
def _production_config_with_writer_budget(
|
||
*, planned: int, max_calls: int, total_usd: str
|
||
) -> dict[str, object]:
|
||
"""在 _production_config 基础上单独抬高 writer 预算,供篇幅修订环的多调用测试使用。
|
||
|
||
修订环会让每臂的 writer 调用数超过基础的 1 次,因此需要更大的 plannedCalls/maxCalls;
|
||
totalBudgetUsd 必须覆盖 caps*planned 的最坏预留,否则预算门会失败关闭。改动 budget
|
||
字段后必须按授权门规则重签 receiptSha256。
|
||
"""
|
||
|
||
value = _production_config()
|
||
budget = value["executionAuthorization"]["budget"]
|
||
budget["plannedCalls"]["writer"] = planned
|
||
budget["maxCalls"]["writer"] = max_calls
|
||
budget["totalBudgetUsd"] = total_usd
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: val for key, val in budget.items() if key != "receiptSha256"}
|
||
)
|
||
return value
|
||
|
||
|
||
def _oracle_pack() -> dict[str, object]:
|
||
"""构造由所有 reviewer 共用、但 detector 不可见的最小 oracle。"""
|
||
|
||
return {
|
||
"schemaVersion": "oracle-truth-pack-v1",
|
||
"evaluationSetVersion": "writer-gate-a-test-v1",
|
||
"sampleId": "deep-space-489",
|
||
"workId": 8,
|
||
"asOf": 488,
|
||
"sourceSnapshotSha256": "sha256:" + "7" * 64,
|
||
"authorizationSnapshotId": "auth-work-8",
|
||
"historicalAssertions": [
|
||
{
|
||
"assertionId": "assertion-history-1",
|
||
"text": "林澈仍在圣蒂曼",
|
||
"sourceVersion": "canonical-v488",
|
||
"chapterStart": 488,
|
||
"chapterEnd": 488,
|
||
"contentSha256": "sha256:" + "8" * 64,
|
||
}
|
||
],
|
||
"targetAssertions": [],
|
||
}
|
||
|
||
|
||
def _production_adapters(
|
||
*,
|
||
semantic_fail_on: int | None = None,
|
||
judge_unstable: bool = False,
|
||
vault_factory=RawVaultManager,
|
||
cas_factory=FileCasStore,
|
||
builder_factory=GateInputBuilder,
|
||
):
|
||
"""集中构造 production fake,所有模型结果都从 structuredOutput 返回。"""
|
||
|
||
writer_runner = FakeSubprocessRunner(include_legacy_result=False)
|
||
semantic_runner = ProductionSemanticRunner(fail_on_call=semantic_fail_on)
|
||
judge_runner = ProductionJudgeRunner(unstable=judge_unstable)
|
||
writer_profile = _production_writer_profile()
|
||
semantic_profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
judge_profile = _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA)
|
||
adapters = WriterReplayProductionAdapters(
|
||
writer_runner=writer_runner,
|
||
writer_profile=writer_profile,
|
||
semantic_model_runner=semantic_runner,
|
||
semantic_profile=semantic_profile,
|
||
judge_model_runner=judge_runner,
|
||
judge_profile=judge_profile,
|
||
oracle_truth_packs={"deep-space-489": _oracle_pack()},
|
||
vault_manager_factory=vault_factory,
|
||
cas_store_factory=cas_factory,
|
||
gate_input_builder_factory=builder_factory,
|
||
)
|
||
return adapters, writer_runner, semantic_runner, judge_runner
|
||
|
||
|
||
class WriterReplayDryRunTest(unittest.TestCase):
|
||
"""验证 dry-run 的合同、冻结、安全和控制变量。"""
|
||
|
||
@staticmethod
|
||
def _ledger(*, planned: int = 1, total: str = "3.000000"):
|
||
"""构造只用于账本边界测试的三角色 profile。"""
|
||
|
||
profiles = {
|
||
"writer": _role_profile("writer", replay_module.WRITER_OUTPUT_JSON_SCHEMA),
|
||
"semantic_detector": _role_profile(
|
||
"semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA
|
||
),
|
||
"blind_judge": _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA),
|
||
}
|
||
return replay_module._ExecutionBudgetLedger(
|
||
total_budget=Decimal(total),
|
||
planned_calls={role: planned for role in replay_module.BUDGET_ROLES},
|
||
max_calls={role: planned for role in replay_module.BUDGET_ROLES},
|
||
profiles=profiles,
|
||
)
|
||
|
||
def test_budget_ledger_settles_failed_call_receipt_before_reraising(self):
|
||
"""模型失败但带可信回执时,实际成本仍必须进入账本。"""
|
||
|
||
ledger = self._ledger()
|
||
receipt = _fake_model_receipt(
|
||
"semantic_detector", 1, {"input": "x"}, {"output": "y"}
|
||
)
|
||
|
||
class FailedDelegate:
|
||
receipts = []
|
||
structured_outputs = []
|
||
|
||
def run(self, **_kwargs):
|
||
error = RuntimeError("模型失败")
|
||
error.receipt = copy.deepcopy(receipt)
|
||
raise error
|
||
|
||
runner = replay_module._BudgetedModelRunner(
|
||
FailedDelegate(), ledger, "semantic_detector"
|
||
)
|
||
with self.assertRaisesRegex(RuntimeError, "模型失败"):
|
||
runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
snapshot = ledger.snapshot()
|
||
self.assertEqual(snapshot["usedCalls"]["semantic_detector"], 1)
|
||
self.assertEqual(snapshot["totalActualCostUsd"], "0.010000")
|
||
self.assertFalse(snapshot["costUnknown"])
|
||
|
||
def test_budget_ledger_without_failure_receipt_is_cost_unknown_and_stops_followups(self):
|
||
"""已发起调用但无可信成本时不能按零美元继续后续角色。"""
|
||
|
||
ledger = self._ledger()
|
||
|
||
class FailedDelegate:
|
||
receipts = []
|
||
structured_outputs = []
|
||
|
||
def run(self, **_kwargs):
|
||
raise RuntimeError("无回执")
|
||
|
||
runner = replay_module._BudgetedModelRunner(
|
||
FailedDelegate(), ledger, "semantic_detector"
|
||
)
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
snapshot = ledger.snapshot()
|
||
self.assertTrue(snapshot["costUnknown"])
|
||
self.assertEqual(snapshot["totalActualCostUsd"], "0.000000")
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
ledger.begin("writer")
|
||
|
||
def test_budget_ledger_rejects_duplicate_or_out_of_order_settlement(self):
|
||
"""一个调用只能由自己的 token 结算一次,不能重复计费。"""
|
||
|
||
ledger = self._ledger()
|
||
token = ledger.begin("writer")
|
||
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
|
||
ledger.complete("writer", token, receipt)
|
||
with self.assertRaisesRegex(BudgetLedgerError, "token 错序或重复"):
|
||
ledger.complete("writer", token, receipt)
|
||
self.assertEqual(ledger.snapshot()["totalActualCostUsd"], "0.010000")
|
||
|
||
def test_budget_ledger_rejects_missing_cost_and_over_cap(self):
|
||
"""缺成本和超过单次 cap 都必须 fail closed,不能静默按零计费。"""
|
||
|
||
missing = self._ledger()
|
||
token = missing.begin("writer")
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
missing.complete(
|
||
"writer",
|
||
token,
|
||
{"adapterRole": "writer", "invocationId": "writer-missing", "totalCostUsd": None},
|
||
)
|
||
self.assertTrue(missing.snapshot()["costUnknown"])
|
||
|
||
over_cap = self._ledger()
|
||
token = over_cap.begin("writer")
|
||
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
|
||
receipt["totalCostUsd"] = "2.000000"
|
||
with self.assertRaisesRegex(BudgetLedgerError, "超过单次 cap"):
|
||
over_cap.complete("writer", token, receipt)
|
||
self.assertEqual(over_cap.snapshot()["failureReason"], "EXECUTION_COST_OVER_CAP")
|
||
|
||
def test_budget_plan_requires_closed_planned_calls_and_reserves_only_375(self):
|
||
"""Gate A 启动按 writer 45(15 基础+30 修订)与两角色各 15 乘 $5 cap 预留 $375,maxCalls 不是预算计划。"""
|
||
|
||
required = {role: 15 for role in replay_module.BUDGET_ROLES}
|
||
caps = {role: Decimal("5.000000") for role in replay_module.BUDGET_ROLES}
|
||
valid = {
|
||
"status": "approved",
|
||
"totalBudgetUsd": "450.000000",
|
||
"plannedCalls": {"writer": 45, "semantic_detector": 15, "blind_judge": 15},
|
||
"maxCalls": {role: 150 for role in replay_module.BUDGET_ROLES},
|
||
}
|
||
planned, maximum, total, error = replay_module._validate_budget_plan(
|
||
valid, required_calls=required, caps=caps
|
||
)
|
||
self.assertIsNone(error)
|
||
self.assertEqual(planned, valid["plannedCalls"])
|
||
self.assertEqual(maximum, valid["maxCalls"])
|
||
self.assertEqual(total, Decimal("450.000000"))
|
||
self.assertEqual(
|
||
sum(caps[role] * planned[role] for role in replay_module.BUDGET_ROLES),
|
||
Decimal("375.000000"),
|
||
)
|
||
|
||
cases = (
|
||
("missing", lambda budget: budget.pop("plannedCalls"), "BUDGET_PLANNED_CALLS_REQUIRED"),
|
||
("invalid-shape", lambda budget: budget["plannedCalls"].update({"extra": 1}), "BUDGET_PLANNED_CALLS_INVALID"),
|
||
("invalid-type", lambda budget: budget["plannedCalls"].update({"writer": "15"}), "BUDGET_PLANNED_CALLS_INSUFFICIENT"),
|
||
("insufficient", lambda budget: budget["plannedCalls"].update({"writer": 14}), "BUDGET_PLANNED_CALLS_INSUFFICIENT"),
|
||
("planned-over-max", lambda budget: budget["plannedCalls"].update({"writer": 151}), "BUDGET_MAX_CALLS_INSUFFICIENT"),
|
||
("insufficient-total", lambda budget: budget.update({"totalBudgetUsd": "374.999999"}), "BUDGET_AUTHORIZATION_INVALID"),
|
||
)
|
||
for name, mutate, expected in cases:
|
||
with self.subTest(name=name):
|
||
tampered = copy.deepcopy(valid)
|
||
mutate(tampered)
|
||
_planned, _maximum, _total, error = replay_module._validate_budget_plan(
|
||
tampered, required_calls=required, caps=caps
|
||
)
|
||
self.assertEqual(error, expected)
|
||
|
||
def test_length_bounds_relaxed_to_30_percent(self):
|
||
"""五个预注册目标的篇幅边界都按正负 30% 计算并受 2000-10000 限幅。"""
|
||
|
||
expected = {
|
||
7500: (5250, 9750),
|
||
7600: (5320, 9880),
|
||
6700: (4690, 8710),
|
||
2000: (2000, 2600),
|
||
6100: (4270, 7930),
|
||
}
|
||
for target, bounds in expected.items():
|
||
with self.subTest(target=target):
|
||
self.assertEqual(replay_module._length_bounds(target), bounds)
|
||
|
||
def test_base_config_budget_self_hash_and_length_bounds_consistent(self):
|
||
"""base 配置 budget 自哈希重签有效,五样本 expectedLength/输出合同等于 _length_bounds(target)。"""
|
||
|
||
path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
gate_config = json.loads(path.read_text(encoding="utf-8"))
|
||
budget = gate_config["executionAuthorization"]["budget"]
|
||
# 与 _validate_authorization_records 同款复核:去掉 receiptSha256 后整段重算自哈希。
|
||
self.assertEqual(
|
||
budget["receiptSha256"],
|
||
canonical_sha256({key: value for key, value in budget.items() if key != "receiptSha256"}),
|
||
)
|
||
self.assertEqual(budget["plannedCalls"]["writer"], 45)
|
||
self.assertEqual(budget["totalBudgetUsd"], "450.000000")
|
||
for sample in gate_config["samples"]:
|
||
target = sample["targetChars"]
|
||
expected_min, expected_max = replay_module._length_bounds(target)
|
||
with self.subTest(sample=sample["sampleId"]):
|
||
self.assertEqual(
|
||
sample["expectedLength"],
|
||
{"targetChars": target, "minChars": expected_min, "maxChars": expected_max},
|
||
)
|
||
contract = sample["writerContextInput"]["outputContract"]
|
||
self.assertEqual(contract["targetChars"], target)
|
||
self.assertEqual(contract["minChars"], expected_min)
|
||
self.assertEqual(contract["maxChars"], expected_max)
|
||
|
||
def test_length_overflow_records_actual_han_chars_in_sample_result(self):
|
||
"""生产链正文越界时失败样本必须记下实际汉字数与合同边界(全是数字,不含正文)。"""
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters()
|
||
short_body = "短" * 2000 # 2000 汉字 < 合同下限 2800,必然越界
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
with mock.patch.object(
|
||
sys.modules[__name__],
|
||
"_writer_output",
|
||
lambda context, call_index: {"candidateBody": short_body},
|
||
):
|
||
result = run_writer_replay(
|
||
_production_config(),
|
||
run_id="length-violation",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
production_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
sample_result = result["samples"][0]
|
||
self.assertIn("lengthViolation", sample_result)
|
||
self.assertEqual(
|
||
sample_result["lengthViolation"],
|
||
{"actualHanChars": 2000, "minChars": 2800, "maxChars": 5200, "targetChars": 4000},
|
||
)
|
||
|
||
def test_recording_runner_keeps_adapter_fields_out_of_model_draft(self):
|
||
"""统一 runtime 只能旁路返回回执 hash,不能污染 v3 模型草稿。"""
|
||
|
||
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
draft = {
|
||
"schemaVersion": "semantic-detection-draft-v3",
|
||
"claims": [],
|
||
"findings": [],
|
||
"assertionVerdicts": [],
|
||
"hardConstraintVerdicts": [],
|
||
"newSettingCandidates": [],
|
||
"evidenceGaps": [],
|
||
}
|
||
receipt = _fake_model_receipt("semantic_detector", 1, {"candidateBody": "甲"}, draft)
|
||
|
||
class Receipt:
|
||
def as_dict(self):
|
||
return copy.deepcopy(receipt)
|
||
|
||
class Invocation:
|
||
structured_output = copy.deepcopy(draft)
|
||
receipt = Receipt()
|
||
|
||
original = replay_module.run_claude
|
||
replay_module.run_claude = lambda _profile, _input: Invocation()
|
||
try:
|
||
runner = replay_module._RecordingRuntimeModelRunner(profile)
|
||
result = runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={"candidateBody": "甲"},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
finally:
|
||
replay_module.run_claude = original
|
||
|
||
self.assertEqual(result["structuredOutput"], draft)
|
||
self.assertNotIn("modelReceiptSha256", result["structuredOutput"])
|
||
self.assertNotIn("reportSha256", result["structuredOutput"])
|
||
self.assertEqual(result["modelReceiptSha256"], sha256_json(receipt))
|
||
|
||
def test_three_arms_validate_full_context_and_only_change_evidence_strategy(self):
|
||
result = run_writer_replay(config(), run_id="dry-1")
|
||
|
||
self.assertEqual(result["status"], "ready")
|
||
sample = result["samples"][0]
|
||
arms = sample["arms"]
|
||
self.assertEqual(set(arms), {"A", "B", "C"})
|
||
self.assertEqual(arms["A"]["evidenceStrategy"], "historical_prose_only")
|
||
self.assertEqual(arms["B"]["evidenceStrategy"], "card_index_only")
|
||
self.assertEqual(arms["C"]["evidenceStrategy"], "card_index_plus_prose")
|
||
self.assertEqual(len({arm["commonControlsSha256"] for arm in arms.values()}), 1)
|
||
self.assertTrue(all(not arm["acceptanceEligible"] for arm in arms.values()))
|
||
self.assertEqual(arms["A"]["contextSummary"]["recentBaselineChapters"], [485, 486, 487, 488])
|
||
self.assertEqual(arms["B"]["contextSummary"]["proseEvidenceCount"], 0)
|
||
self.assertEqual(arms["B"]["contextSummary"]["indexHintCount"], 1)
|
||
self.assertEqual(arms["C"]["contextSummary"]["recentBaselineChapters"], [485, 486, 487, 488])
|
||
self.assertEqual(arms["C"]["contextSummary"]["indexHintCount"], 1)
|
||
self.assertTrue(
|
||
all(
|
||
arm["contextSummary"]["contentMode"] == "sanitized_contract_fixture"
|
||
for arm in arms.values()
|
||
)
|
||
)
|
||
rendered = json.dumps(result, ensure_ascii=False)
|
||
self.assertNotIn("脱敏合同夹具", rendered)
|
||
self.assertNotIn("林澈在冻结点仍位于圣蒂曼", rendered)
|
||
|
||
def test_future_index_hint_invalidates_whole_sample(self):
|
||
leaked = config()
|
||
leaked["samples"][0]["writerContextInput"]["retrievalResult"]["indexHints"][0]["asOf"] = 489
|
||
|
||
result = run_writer_replay(leaked, run_id="dry-leak")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_leakage")
|
||
|
||
def test_invalid_b_arm_contract_cannot_be_reported_ready(self):
|
||
invalid = config()
|
||
invalid["samples"][0]["writerContextInput"]["retrievalResult"]["indexHints"] = []
|
||
|
||
result = run_writer_replay(invalid, run_id="dry-invalid-context")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_writer_context")
|
||
|
||
def test_target_length_is_mechanically_recomputed_and_target_chapter_length_is_forbidden(self):
|
||
invalid = config()
|
||
invalid["samples"][0]["targetChars"] = 4100
|
||
|
||
with self.assertRaisesRegex(WriterReplayError, "机械复算"):
|
||
run_writer_replay(invalid, run_id="dry-length-tampered")
|
||
|
||
leaked = config()
|
||
leaked["samples"][0]["targetLengthBasis"]["usesTargetChapterLength"] = True
|
||
with self.assertRaisesRegex(WriterReplayError, "禁止读取目标章"):
|
||
run_writer_replay(leaked, run_id="dry-length-leaked")
|
||
|
||
def test_unresolved_generic_role_must_remain_null_instead_of_defaulting_to_zero(self):
|
||
unresolved = config()
|
||
sample = unresolved["samples"][0]
|
||
sample["writerContextInput"]["requirements"]["requiredCharacters"] = ["内应"]
|
||
sample["newCharacterRatio"] = None
|
||
sample["newCharacterRatioStatus"] = "unresolved_generic_role"
|
||
sample["newCharacterBasis"] = {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": 488,
|
||
"requiredCharacters": ["内应"],
|
||
"knownBeforeAsOf": [],
|
||
"absentBeforeAsOf": [],
|
||
"genericRoles": ["内应"],
|
||
}
|
||
self.assertTrue(run_writer_replay(unresolved, run_id="dry-generic-role")["ok"])
|
||
|
||
unresolved["samples"][0]["newCharacterRatio"] = 0
|
||
with self.assertRaisesRegex(WriterReplayError, "必须为 null"):
|
||
run_writer_replay(unresolved, run_id="dry-generic-role-zero")
|
||
|
||
def test_character_declared_absent_before_freeze_cannot_have_card_index_hint(self):
|
||
inconsistent = config()
|
||
sample = inconsistent["samples"][0]
|
||
sample["newCharacterRatio"] = 1.0
|
||
sample["newCharacterBasis"]["knownBeforeAsOf"] = []
|
||
sample["newCharacterBasis"]["absentBeforeAsOf"] = ["林澈"]
|
||
|
||
with self.assertRaisesRegex(WriterReplayError, "不得同时出现在卡索引"):
|
||
run_writer_replay(inconsistent, run_id="dry-absent-card-conflict")
|
||
|
||
def test_gate_a_preregistration_mechanically_recomputes_all_lengths_and_ratios(self):
|
||
gate_config_path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
gate_config = json.loads(gate_config_path.read_text(encoding="utf-8"))
|
||
expected_targets = {489: 7500, 321: 7600, 544: 6700, 199: 2000, 523: 6100}
|
||
expected_ratios = {489: 1.0, 321: 0.0, 544: None, 199: 0.0, 523: 0.0}
|
||
self.assertEqual(
|
||
gate_config["commonControls"]["selectorVersion"],
|
||
"writer-gate-a-deep-space-card-selectors-v4",
|
||
)
|
||
self.assertEqual(
|
||
gate_config["commonControls"]["selectorSha256"],
|
||
"sha256:99105930d7e01d32fa9faa1ceaa30663674dd8e15b84dd8ab7f5b18cd4e9d0ef",
|
||
)
|
||
self.assertNotIn("inputProvenance", gate_config["commonControls"])
|
||
self.assertEqual(gate_config["commonControls"]["maxContextChars"], 140000)
|
||
selector_path = (
|
||
SCRIPT_DIR.parent
|
||
/ "configs"
|
||
/ "writer-gate-a-deep-space-card-selectors-v1.json"
|
||
)
|
||
selector_config = json.loads(selector_path.read_text(encoding="utf-8"))
|
||
selector_by_sample = {
|
||
sample["sampleId"]: [
|
||
(card["type"], card["name"]) for card in sample["cards"]
|
||
]
|
||
for sample in selector_config["samples"]
|
||
}
|
||
sample_by_id = {sample["sampleId"]: sample for sample in gate_config["samples"]}
|
||
for sample_id in selector_by_sample:
|
||
hints = sample_by_id[sample_id]["writerContextInput"]["retrievalResult"][
|
||
"indexHints"
|
||
]
|
||
self.assertEqual(
|
||
[(hint["type"], hint["name"]) for hint in hints],
|
||
selector_by_sample[sample_id],
|
||
)
|
||
expected_card_prose_chapters = {
|
||
"deep-space-489-battle": 484,
|
||
"deep-space-321-character-dialogue": 312,
|
||
"deep-space-544-turning-point": 467,
|
||
}
|
||
for sample_id, expected_chapter in expected_card_prose_chapters.items():
|
||
prose = sample_by_id[sample_id]["writerContextInput"]["retrievalResult"][
|
||
"proseEvidence"
|
||
]
|
||
card_prose = [item for item in prose if item["retrievalArm"] == "C"]
|
||
self.assertEqual(len(card_prose), 1)
|
||
self.assertEqual(card_prose[0]["chapter"], expected_chapter)
|
||
self.assertTrue(
|
||
card_prose[0]["sourceRef"]["sourceId"].endswith(
|
||
f":{expected_chapter}"
|
||
)
|
||
)
|
||
|
||
for sample in gate_config["samples"]:
|
||
basis = sample["targetLengthBasis"]
|
||
recalculated = calculate_target_chars(
|
||
recent_chapter_han_counts=sample["frozenRecentHanCounts"],
|
||
hard_event_count=basis["hardEventCount"],
|
||
foreshadowing_action_count=basis["foreshadowingActionCount"],
|
||
required_scene_count=basis["requiredSceneCount"],
|
||
min_chars=basis["minChars"],
|
||
max_chars=basis["maxChars"],
|
||
)
|
||
target_chapter = sample["targetChapter"]
|
||
self.assertEqual(recalculated, expected_targets[target_chapter])
|
||
self.assertEqual(sample["targetChars"], recalculated)
|
||
self.assertEqual(sample["newCharacterRatio"], expected_ratios[target_chapter])
|
||
self.assertEqual(
|
||
sample["writerContextInput"]["tokenBudget"]["maxContextChars"],
|
||
gate_config["commonControls"]["maxContextChars"],
|
||
)
|
||
if target_chapter == 544:
|
||
self.assertEqual(sample["newCharacterRatioStatus"], "unresolved_generic_role")
|
||
else:
|
||
self.assertEqual(sample["newCharacterRatioStatus"], "resolved")
|
||
|
||
result = run_writer_replay(gate_config, run_id="gate-a-preregistered-dry-run")
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "ready")
|
||
self.assertEqual(len(result["samples"]), 5)
|
||
self.assertNotIn("脱敏合成历史片段", json.dumps(result, ensure_ascii=False))
|
||
for sample in result["samples"]:
|
||
receipt = sample["writerContextDiffReceipt"]
|
||
self.assertEqual(receipt["proseCharBudget"], 2000)
|
||
self.assertEqual(receipt["proseCharCount"], {"A": 2000, "C": 2000})
|
||
self.assertTrue(receipt["allowedDifferencePaths"])
|
||
self.assertNotEqual(receipt["contextSha256"]["A"], receipt["contextSha256"]["C"])
|
||
self.assertTrue(
|
||
all(
|
||
path.startswith(("$.factConstraints", "$.proseExcerpts"))
|
||
for path in receipt["allowedDifferencePaths"]
|
||
)
|
||
)
|
||
self.assertEqual(gate_config["commonControls"]["modelVersion"], "claude-opus-4-8[1m]")
|
||
self.assertIn("writer-runtime-v1", gate_config["commonControls"]["adapterVersion"])
|
||
self.assertIn("2.1.211", gate_config["commonControls"]["adapterVersion"])
|
||
executable_sha256 = gate_config["executionAuthorization"]["runtimeProbe"][
|
||
"claudeExecutableSha256"
|
||
]
|
||
self.assertEqual(len(executable_sha256), 64)
|
||
self.assertEqual(
|
||
executable_sha256,
|
||
"5a728a76198b6eca7f3c7cdbff43bab44b77b48c2108f7a3107d889773382629",
|
||
)
|
||
self.assertIn(executable_sha256, gate_config["commonControls"]["adapterVersion"])
|
||
self.assertEqual(
|
||
gate_config["executionAuthorization"]["budget"]["maxCalls"],
|
||
{"writer": 150, "semantic_detector": 150, "blind_judge": 150},
|
||
)
|
||
self.assertEqual(gate_config["executionAuthorization"]["budget"]["status"], "approved")
|
||
self.assertEqual(
|
||
gate_config["executionAuthorization"]["budget"]["totalBudgetUsd"],
|
||
"450.000000",
|
||
)
|
||
profile_cap = Decimal("5.000000")
|
||
max_calls = gate_config["executionAuthorization"]["budget"]["maxCalls"]
|
||
worst_case_budget = profile_cap * sum(max_calls.values())
|
||
self.assertEqual(worst_case_budget, Decimal("2250.000000"))
|
||
self.assertGreater(
|
||
worst_case_budget,
|
||
Decimal(gate_config["executionAuthorization"]["budget"]["totalBudgetUsd"]),
|
||
)
|
||
self.assertEqual(gate_config["executionAuthorization"]["rawRetention"]["status"], "approved")
|
||
self.assertIn("总预算 450 美元", gate_config["executionAuthorization"]["budget"]["reason"])
|
||
self.assertIn("maxCalls 为各 150 次安全上限", gate_config["executionAuthorization"]["budget"]["reason"])
|
||
|
||
def test_gate_a_formal_zero_budget_or_zero_ac_difference_fails_closed(self):
|
||
"""正式 Gate A 必须真实改变 writer 创作输入,不能只改变隐藏索引。"""
|
||
|
||
path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
zero_budget = json.loads(path.read_text(encoding="utf-8"))
|
||
zero_budget["samples"][0]["proseCharBudget"] = 0
|
||
result = run_writer_replay(zero_budget, run_id="gate-a-zero-budget")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_writer_context")
|
||
self.assertIn("proseCharBudget", result["samples"][0]["errors"][0])
|
||
|
||
zero_difference = json.loads(path.read_text(encoding="utf-8"))
|
||
evidence = zero_difference["samples"][0]["writerContextInput"]["retrievalResult"][
|
||
"proseEvidence"
|
||
]
|
||
evidence[1] = {**copy.deepcopy(evidence[0]), "retrievalArm": "C"}
|
||
result = run_writer_replay(zero_difference, run_id="gate-a-zero-difference")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_writer_context")
|
||
self.assertIn("A/C", result["samples"][0]["errors"][0])
|
||
|
||
def test_gate_a_formal_profiles_reconstruct_schema_prompt_and_runtime_bindings(self):
|
||
"""三角色 profile 必须能由 config 重建并绑定同一成功 probe。"""
|
||
|
||
gate_config_path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
gate_config = json.loads(gate_config_path.read_text(encoding="utf-8"))
|
||
profiles = gate_config["executionProfiles"]
|
||
self.assertEqual(set(profiles), {"writer", "semantic_detector", "blind_judge"})
|
||
expected_schema_ids = {
|
||
"writer": "writer-draft-v2",
|
||
"semantic_detector": "semantic-detection-draft-v3",
|
||
"blind_judge": "blind-judge-draft-v3",
|
||
}
|
||
expected_schemas = {
|
||
"writer": replay_module.WRITER_OUTPUT_JSON_SCHEMA,
|
||
"semantic_detector": SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
"blind_judge": BLIND_JUDGE_REPORT_JSON_SCHEMA,
|
||
}
|
||
probe = gate_config["executionAuthorization"]["runtimeProbe"]
|
||
bindings = {
|
||
field: {profile[field] for profile in profiles.values()}
|
||
for field in (
|
||
"claudeExecutablePath",
|
||
"claudeExecutableSha256",
|
||
"claudeCliVersion",
|
||
"modelAlias",
|
||
"resolvedModelId",
|
||
)
|
||
}
|
||
self.assertTrue(all(len(values) == 1 for values in bindings.values()))
|
||
self.assertEqual(bindings["claudeExecutablePath"].pop(), probe["claudeExecutablePath"])
|
||
self.assertEqual(bindings["claudeExecutableSha256"].pop(), probe["claudeExecutableSha256"])
|
||
self.assertEqual(bindings["claudeCliVersion"].pop(), probe["claudeCliVersion"])
|
||
self.assertEqual(bindings["modelAlias"].pop(), probe["modelAlias"])
|
||
self.assertEqual(bindings["resolvedModelId"].pop(), probe["resolvedModelId"])
|
||
|
||
profile_hashes = gate_config["executionAuthorization"]["profileSha256"]
|
||
for role, raw_profile in profiles.items():
|
||
profile = replay_module._profile_from_mapping(raw_profile, role=role)
|
||
self.assertEqual(raw_profile["maxBudgetUsdPerCall"], "5.000000")
|
||
self.assertEqual(profile.max_budget_usd_per_call, Decimal("5.000000"))
|
||
self.assertEqual(raw_profile["timeoutSeconds"], 1200)
|
||
self.assertEqual(profile.timeout_seconds, 1200.0)
|
||
self.assertEqual(profile.json_schema_id, expected_schema_ids[role])
|
||
self.assertEqual(profile.json_schema, expected_schemas[role])
|
||
self.assertEqual(profile.json_schema_sha256, sha256_json(profile.json_schema))
|
||
self.assertEqual(profile.system_prompt_sha256, sha256_text(profile.system_prompt))
|
||
self.assertEqual(profile.normal_terminal_reasons, ("completed",))
|
||
self.assertEqual(profile.execution_profile_sha256, profile_hashes[role])
|
||
self.assertEqual(profile.max_context_chars, gate_config["commonControls"]["maxContextChars"])
|
||
self.assertEqual(
|
||
len({raw_profile["systemPromptSha256"] for raw_profile in profiles.values()}),
|
||
3,
|
||
)
|
||
self.assertEqual(len({raw_profile["profileVersion"] for raw_profile in profiles.values()}), 3)
|
||
|
||
for control_name in ("budget", "rawRetention"):
|
||
control = gate_config["executionAuthorization"][control_name]
|
||
self.assertEqual(
|
||
control["receiptSha256"],
|
||
canonical_sha256(
|
||
{key: value for key, value in control.items() if key != "receiptSha256"}
|
||
),
|
||
)
|
||
|
||
def test_execute_without_formal_profiles_fails_closed_before_runner(self):
|
||
"""缺少三角色 formal profiles 时不能靠 profile hash 或测试参数放行。"""
|
||
|
||
gate_config_path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
tampered = json.loads(gate_config_path.read_text(encoding="utf-8"))
|
||
tampered.pop("executionProfiles")
|
||
tampered["oracleTruthPacks"] = {}
|
||
budget = tampered["executionAuthorization"]["budget"]
|
||
budget["status"] = "approved"
|
||
budget["totalBudgetUsd"] = "1.000000"
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
tampered,
|
||
run_id="missing-formal-profiles",
|
||
output_dir=output,
|
||
execute=True,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_profile")
|
||
self.assertEqual(result["errors"], ["EXECUTE_PROFILE_INVALID"])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_max_calls_are_sufficient_but_missing_total_budget_still_blocks(self):
|
||
"""150 次/角色满足机械调用量,但没有可靠美元总额仍必须阻断。"""
|
||
|
||
gate_config_path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
tampered = json.loads(gate_config_path.read_text(encoding="utf-8"))
|
||
tampered["oracleTruthPacks"] = {}
|
||
budget = tampered["executionAuthorization"]["budget"]
|
||
self.assertEqual(budget["status"], "approved")
|
||
self.assertEqual(budget["totalBudgetUsd"], "450.000000")
|
||
budget["status"] = "approved"
|
||
budget.pop("totalBudgetUsd", None)
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
tampered,
|
||
run_id="missing-total-budget",
|
||
output_dir=output,
|
||
execute=True,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_budget_authorization")
|
||
self.assertEqual(result["errors"], ["BUDGET_AUTHORIZATION_INVALID"])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_gate_a_execute_blocks_pending_budget_before_vault_or_runner(self):
|
||
"""正式 Gate A 已完成 runtime probe,但预算未批准时必须在副作用前阻断。"""
|
||
|
||
gate_config_path = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
gate_config = json.loads(gate_config_path.read_text(encoding="utf-8"))
|
||
budget = gate_config["executionAuthorization"]["budget"]
|
||
budget["status"] = "pending"
|
||
budget.pop("totalBudgetUsd", None)
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
gate_config,
|
||
run_id="gate-a-pending-budget",
|
||
output_dir=output,
|
||
execute=True,
|
||
production_adapters=adapters,
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_budget_authorization")
|
||
self.assertEqual(result["errors"], ["BUDGET_AUTHORIZATION_REQUIRED"])
|
||
self.assertEqual(vault_calls, [])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
|
||
class WriterReplayExecuteBoundaryTest(unittest.TestCase):
|
||
"""验证真实执行失败关闭和测试注入仍经过正式管线。"""
|
||
|
||
def test_execute_requires_private_tmp_output(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
with self.assertRaisesRegex(WriterReplayError, "/private/tmp"):
|
||
run_writer_replay(
|
||
config(),
|
||
run_id="real-bad-path",
|
||
output_dir=pathlib.Path(directory),
|
||
execute=True,
|
||
)
|
||
|
||
def test_execute_without_production_adapters_fails_closed(self):
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
pending = config()
|
||
pending["commonControls"]["modelVersion"] = "pending_probe"
|
||
pending["commonControls"]["adapterVersion"] = "pending_probe"
|
||
result = run_writer_replay(
|
||
pending,
|
||
run_id="real-not-wired",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_runtime_probe")
|
||
self.assertEqual(result["errors"], ["RUNTIME_PROBE_REQUIRED"])
|
||
|
||
def test_invalid_third_judge_report_fails_top_level(self):
|
||
adapters, _runner, _detector, _judge = _test_adapters(
|
||
unstable=True, invalid_third=True
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
result = run_writer_replay(
|
||
config(),
|
||
run_id="judge-invalid",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_invalid")
|
||
|
||
def test_third_judge_still_unstable_fails_top_level(self):
|
||
adapters, _runner, _detector, _judge = _test_adapters(
|
||
irreducibly_unstable=True
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
result = run_writer_replay(
|
||
config(),
|
||
run_id="judge-unstable",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_unstable")
|
||
|
||
def test_context_authorization_cannot_swap_snapshot_or_source_before_runner(self):
|
||
"""上下文版本、快照、用途、时间和状态都必须绑定顶层授权。"""
|
||
|
||
mutations = {
|
||
"source_version": lambda value: value["samples"][0]["writerContextInput"].update(
|
||
{"sourceVersion": "swapped-source-v2"}
|
||
),
|
||
"snapshot_id": lambda value: value["samples"][0]["writerContextInput"][
|
||
"authorizationSnapshot"
|
||
].update({"snapshotId": "auth-swapped"}),
|
||
"purpose": lambda value: value["samples"][0]["writerContextInput"][
|
||
"authorizationSnapshot"
|
||
].update({"allowedPurpose": "diagnostic"}),
|
||
"verified_at": lambda value: value["samples"][0]["writerContextInput"][
|
||
"authorizationSnapshot"
|
||
].update({"verifiedAt": "2026-07-21T00:00:00Z"}),
|
||
"source_status": lambda value: value["samples"][0]["writerContextInput"].update(
|
||
{"sourceStatus": "revoked"}
|
||
),
|
||
}
|
||
for name, mutate in mutations.items():
|
||
invalid = config()
|
||
mutate(invalid)
|
||
adapters, runner, _detector, _judge = _test_adapters()
|
||
with self.subTest(name=name), tempfile.TemporaryDirectory(
|
||
dir="/private/tmp"
|
||
) as directory:
|
||
result = run_writer_replay(
|
||
invalid,
|
||
run_id=f"auth-binding-{name}",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_authorization")
|
||
self.assertEqual(runner.calls, [])
|
||
|
||
def test_test_injection_runs_writer_pipeline_mechanical_and_semantic_detector(self):
|
||
adapters, runner, detector, judge = _test_adapters(unstable=True)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
config(),
|
||
run_id="test-pipeline",
|
||
output_dir=output_dir,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse((output_dir / "raw").exists())
|
||
lease_files = list(
|
||
(output_dir / "journal" / "raw-vault" / "leases").glob("*.json")
|
||
)
|
||
self.assertEqual(len(lease_files), 1)
|
||
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
|
||
public_manifest = (output_dir / "manifest.json").read_text(encoding="utf-8")
|
||
self.assertNotIn("candidateBody", public_manifest)
|
||
self.assertNotIn("脱敏合同夹具", public_manifest)
|
||
|
||
self.assertEqual(result["status"], "completed_test_pipeline")
|
||
self.assertTrue(all("evidenceStrategy" not in context for context in runner.contexts))
|
||
self.assertTrue(all("--model" in command for command, _kwargs in runner.calls))
|
||
self.assertEqual(detector.calls, [
|
||
("historical_prose_only", True),
|
||
("card_index_only", True),
|
||
("card_index_plus_prose", True),
|
||
])
|
||
self.assertEqual(len([call for call in judge.calls if call[0] == "judge-3"]), 1)
|
||
candidates = result["samples"][0]["candidates"]
|
||
self.assertTrue(all(not candidate["acceptanceEligible"] for candidate in candidates.values()))
|
||
|
||
def test_writer_model_cannot_forge_adapter_owned_hash(self):
|
||
adapters, _runner, detector, _judge = _test_adapters(illegal_extra_field=True)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
with self.assertRaises(WriterReplayError):
|
||
run_writer_replay(
|
||
config(),
|
||
run_id="test-writer-owned-hash",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
# B 臂模型越权输出 adapter 字段后,不能进入 semantic detector。
|
||
self.assertEqual(detector.calls, [("historical_prose_only", True)])
|
||
|
||
def test_malicious_judge_only_sees_blind_candidates_without_raw_paths_or_arm_contexts(self):
|
||
runner = FakeSubprocessRunner()
|
||
detector = FakeSemanticDetector()
|
||
judge = MaliciousJudge()
|
||
adapters = WriterReplayTestAdapters(runner, _frozen_test_profile(), detector, judge)
|
||
evaluation_config = config()
|
||
fine_outline = evaluation_config["samples"][0]["writerContextInput"]["fineOutline"]
|
||
# 恶意配置模拟把目标原文、索引、文件路径和真实臂塞进细纲对象;这些字段写手看不到,评委也不能看到。
|
||
fine_outline.update(
|
||
{
|
||
"targetOriginal": "目标章原文泄漏哨兵",
|
||
"indexHints": [{"content": "被测卡索引泄漏哨兵"}],
|
||
"path": "/private/tmp/candidate-A.json",
|
||
"arm": "A",
|
||
}
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
evaluation_config,
|
||
run_id="blind-isolation",
|
||
output_dir=output_dir,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
|
||
self.assertTrue(result["ok"])
|
||
self.assertTrue(judge.visible_inputs)
|
||
expected_writer_fine_outline = {
|
||
"hardConstraints": ["必须完成围攻突围"],
|
||
"adjustableBeats": [],
|
||
"declaredNewFacts": [],
|
||
}
|
||
self.assertTrue(
|
||
all(context["fineOutline"] == expected_writer_fine_outline for context in runner.contexts),
|
||
"三臂写手必须只看到严格细纲合同字段",
|
||
)
|
||
expected_judge_fine_outline = {
|
||
"sourceRef": {
|
||
"sourceId": "fixture:fine-outline:489",
|
||
"sourceVersion": "fine-outline-fixture-v1",
|
||
"chapter": 489,
|
||
},
|
||
**expected_writer_fine_outline,
|
||
}
|
||
shared_references = []
|
||
for visible in judge.visible_inputs:
|
||
self.assertEqual(
|
||
set(visible),
|
||
{
|
||
"schemaVersion",
|
||
"sample",
|
||
"blindCandidateId",
|
||
"candidateOrder",
|
||
"sharedEvaluationReference",
|
||
"candidates",
|
||
},
|
||
)
|
||
self.assertEqual(set(visible["candidateOrder"]), {"blind-1", "blind-2", "blind-3"})
|
||
shared = visible["sharedEvaluationReference"]
|
||
shared_references.append(copy.deepcopy(shared))
|
||
self.assertEqual(shared["fineOutline"], expected_judge_fine_outline)
|
||
self.assertNotIn("entities", shared["fineOutline"])
|
||
self.assertEqual(shared["requirements"]["requiredCharacters"], ["林澈"])
|
||
self.assertEqual(
|
||
shared["requirements"]["chapterEndHook"]["anchors"],
|
||
["城门忽然打开"],
|
||
)
|
||
self.assertEqual(
|
||
[item["chapter"] for item in shared["historicalProseBaseline"]],
|
||
[485, 486, 487, 488],
|
||
)
|
||
self.assertTrue(
|
||
all(
|
||
set(candidate)
|
||
== {"blindCandidateId", "candidateSha256", "candidateBody"}
|
||
and candidate["blindCandidateId"].startswith("blind-")
|
||
for candidate in visible["candidates"]
|
||
)
|
||
)
|
||
self.assertTrue(
|
||
all(reference == shared_references[0] for reference in shared_references[1:]),
|
||
"评委顺序变化不得改变共同评测参考",
|
||
)
|
||
|
||
|
||
class WriterReplayProductionIntegrationTest(unittest.TestCase):
|
||
"""用生产 fake 串通 Vault、CAS、v2 adapter 与 GateInputBuilder。"""
|
||
|
||
def _run(self, adapters, *, evaluation_config=None, run_id="production-fake"):
|
||
"""在安全临时目录执行一次生产链,并返回结果与输出目录内容。"""
|
||
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
result = run_writer_replay(
|
||
evaluation_config or _production_config(),
|
||
run_id=run_id,
|
||
output_dir=output,
|
||
execute=True,
|
||
production_adapters=adapters,
|
||
)
|
||
return result, output
|
||
|
||
def test_production_execute_rejects_sanitized_fixture_before_vault_or_runner(self):
|
||
"""真实 execute 只接受 loader 产出的 canonical_frozen_prose。"""
|
||
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
sanitized = _production_config()
|
||
sanitized["samples"][0]["writerContextInput"]["contentMode"] = (
|
||
"sanitized_contract_fixture"
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters, evaluation_config=sanitized, run_id="production-sanitized"
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_content_mode")
|
||
self.assertEqual(result["errors"], ["EXECUTE_CONTENT_MODE_INVALID"])
|
||
self.assertEqual(vault_calls, [])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_production_fake_happy_path_runs_v2_adapters_receipts_vault_cas_and_builder(self):
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
|
||
result, output = self._run(adapters)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual(len(writer.calls), 3)
|
||
self.assertEqual(len(semantic.calls), 3)
|
||
self.assertEqual(len(judge.calls), 2)
|
||
self.assertEqual(
|
||
result["budgetLedger"]["usedCalls"],
|
||
{"writer": 3, "semantic_detector": 3, "blind_judge": 2},
|
||
)
|
||
self.assertEqual(
|
||
result["budgetLedger"]["remainingPlannedCalls"],
|
||
{"writer": 0, "semantic_detector": 0, "blind_judge": 1},
|
||
)
|
||
self.assertEqual(result["budgetLedger"]["totalActualCostUsd"], "0.080000")
|
||
self.assertFalse(result["budgetLedger"]["costUnknown"])
|
||
self.assertTrue(all(receipt["adapterRole"] == "semantic_detector" for receipt in semantic.receipts))
|
||
self.assertTrue(all(receipt["adapterRole"] == "blind_judge" for receipt in judge.receipts))
|
||
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(gate_input["schemaVersion"], "writer-gate-input-v2")
|
||
self.assertEqual(gate_input["gateInputSha256"], result["gateInputSha256"])
|
||
self.assertTrue(
|
||
all(not sample["systemFailure"] for sample in gate_input["samples"]),
|
||
gate_input["samples"],
|
||
)
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
self.assertEqual(
|
||
{
|
||
role
|
||
for arm in execution_receipts["deep-space-489"].values()
|
||
for role in arm
|
||
},
|
||
{"writer", "semantic_detector", "blind_judge"},
|
||
)
|
||
self.assertTrue(
|
||
all(
|
||
wrapper["executionReceipts"]
|
||
for arm in execution_receipts["deep-space-489"].values()
|
||
for wrapper in arm.values()
|
||
)
|
||
)
|
||
manifest = (output / "manifest.json").read_text(encoding="utf-8")
|
||
self.assertNotIn("candidateBody", manifest)
|
||
self.assertNotIn("/private/tmp", manifest)
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
lease = json.loads(lease_files[0].read_text(encoding="utf-8"))
|
||
self.assertEqual(lease["status"], "closed")
|
||
cas_state = json.loads(
|
||
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text()
|
||
)
|
||
self.assertEqual(cas_state["state"], "COMPLETED")
|
||
self.assertEqual(cas_state["cleanupState"], "closed")
|
||
|
||
def test_length_revision_loop_recovers_out_of_range_draft(self):
|
||
"""首版越界时退回写手修订一遍即达标:样本成功,修订块带固定指令与上一版正文。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
config = _production_config_with_writer_budget(
|
||
planned=9, max_calls=9, total_usd="18.000000"
|
||
)
|
||
short_body = "短" * 2000 # 2000 汉字 < 合同下限 2800,首版必然越界
|
||
|
||
def revise_once(context, *, call_index):
|
||
# 偶数下标是每臂首版(越界),奇数下标是修订版(达标)。
|
||
if call_index % 2 == 0:
|
||
return {"candidateBody": short_body}
|
||
marker = ("甲", "乙", "丙")[(call_index // 2) % 3]
|
||
return {"candidateBody": _candidate_body(marker)}
|
||
|
||
with mock.patch.object(sys.modules[__name__], "_writer_output", revise_once):
|
||
result, output = self._run(
|
||
adapters, evaluation_config=config, run_id="length-revision-ok"
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
# 每臂 1 次首版 + 1 次修订 = 2 次 writer 调用,三臂共 6 次。
|
||
self.assertEqual(len(writer.calls), 6)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 6)
|
||
self.assertNotIn("lengthViolation", result["samples"][0])
|
||
# 每臂 writer 回执同时绑定首版与修订两条。
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
for arm in ("A", "B", "C"):
|
||
wrapper = execution_receipts["deep-space-489"][arm]["writer"]
|
||
self.assertEqual(len(wrapper["executionReceipts"]), 2)
|
||
# 首版调用不带 lengthRevision,修订调用带;下标 1/3/5 是三臂的修订输入。
|
||
for index in (0, 2, 4):
|
||
self.assertNotIn("lengthRevision", writer.contexts[index])
|
||
revision_inputs = [writer.contexts[index] for index in (1, 3, 5)]
|
||
for context in revision_inputs:
|
||
revision = context["lengthRevision"]
|
||
# 修订指令是三臂共用的固定常量(arm-invariant),不含具体字数。
|
||
self.assertEqual(revision["instruction"], replay_module.LENGTH_REVISION_INSTRUCTION)
|
||
# 修订块携带上一版正文与实际字数/区间,作为数据而非拼进指令文本。
|
||
self.assertEqual(revision["previousDraft"], short_body)
|
||
self.assertEqual(revision["actualHanChars"], 2000)
|
||
self.assertEqual(revision["minChars"], 2800)
|
||
self.assertEqual(revision["maxChars"], 5200)
|
||
self.assertEqual(revision["targetChars"], 4000)
|
||
self.assertEqual(
|
||
{context["lengthRevision"]["instruction"] for context in revision_inputs},
|
||
{replay_module.LENGTH_REVISION_INSTRUCTION},
|
||
)
|
||
|
||
def test_length_revision_exhausted_still_fails_with_actual_han_chars(self):
|
||
"""修订两遍仍越界才失败:记下实际字数,writer 调用数 = 1 首版 + 2 修订 = 3。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
short_body = "短" * 2000
|
||
with mock.patch.object(
|
||
sys.modules[__name__],
|
||
"_writer_output",
|
||
lambda context, *, call_index: {"candidateBody": short_body},
|
||
):
|
||
result, _output = self._run(adapters, run_id="length-revision-exhausted")
|
||
|
||
self.assertFalse(result["ok"])
|
||
sample = result["samples"][0]
|
||
self.assertIn("候选正文汉字数超出动态篇幅区间", sample["errors"])
|
||
self.assertIn("lengthViolation", sample)
|
||
self.assertEqual(sample["lengthViolation"]["actualHanChars"], 2000)
|
||
# 首臂耗尽 1+2=3 次 writer 调用后即终局门失败,后续臂不再运行。
|
||
self.assertEqual(len(writer.calls), 3)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 3)
|
||
|
||
def test_length_in_range_skips_revision_loop(self):
|
||
"""首版即在区间内:0 修订,每臂仅 1 次 writer 调用,输入不带 lengthRevision。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
result, output = self._run(adapters, run_id="length-no-revision")
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(len(writer.calls), 3)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 3)
|
||
for context in writer.contexts:
|
||
self.assertNotIn("lengthRevision", context)
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
for arm in ("A", "B", "C"):
|
||
wrapper = execution_receipts["deep-space-489"][arm]["writer"]
|
||
self.assertEqual(len(wrapper["executionReceipts"]), 1)
|
||
|
||
def test_bound_judge_report_binds_raw_outputs_to_receipt_hashes(self):
|
||
"""生产编排必须把 reviewer 原始输出随报告进 builder 且与回执哈希一致(正向对照)。"""
|
||
|
||
captured: dict[str, object] = {}
|
||
|
||
class CapturingBuilder:
|
||
"""记录一次生产 build 的全部来源,供篡改复现复用,本身仍走真 builder。"""
|
||
|
||
def build(self, **kwargs):
|
||
captured.update(kwargs)
|
||
return GateInputBuilder().build(**kwargs)
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
builder_factory=CapturingBuilder
|
||
)
|
||
result, _output = self._run(adapters)
|
||
self.assertTrue(result["ok"], result)
|
||
|
||
judge_report = captured["judge_reports"]["deep-space-489"]
|
||
reviewer_outputs = judge_report["reviewerStructuredOutputs"]
|
||
true_receipts = captured["execution_receipts"]["deep-space-489"]["A"][
|
||
"blind_judge"
|
||
]["executionReceipts"]
|
||
# 原始输出与 reviewerReports、回执按 reviewer 顺序一一对应,且哈希回真实回执。
|
||
self.assertEqual(len(reviewer_outputs), len(judge_report["reviewerReports"]))
|
||
self.assertEqual(len(reviewer_outputs), len(true_receipts))
|
||
for draft, receipt in zip(reviewer_outputs, true_receipts, strict=True):
|
||
self.assertEqual(
|
||
canonical_sha256(draft), receipt["structuredOutputSha256"]
|
||
)
|
||
|
||
def test_consistent_report_tampering_with_resigned_hashes_is_rejected(self):
|
||
"""一致篡改报告评分 + 重签 reportSha256 + 重推导 panel 仍被原始输出绑定拦下。
|
||
|
||
攻击场景:攻击者把每个 reviewer 报告的评分统一改成伪造高分,并把报告内携带的
|
||
reviewerStructuredOutputs 同步改成与伪造评分一致的伪造原始输出(使伪造自洽),
|
||
重签内层/外层 reportSha256,按同一 adjudicate_structured_reviews 重推导 panel。
|
||
唯一改不动的是执行回执里生产时固化的 structuredOutputSha256——它仍指向真实原始
|
||
输出,于是新的只读交叉校验 canonical(伪造输出) != 回执哈希 失败关闭。
|
||
"""
|
||
|
||
captured: dict[str, object] = {}
|
||
|
||
class CapturingBuilder:
|
||
"""记录一次生产 build 的全部来源,供篡改复现复用,本身仍走真 builder。"""
|
||
|
||
def build(self, **kwargs):
|
||
captured.update(kwargs)
|
||
return GateInputBuilder().build(**kwargs)
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
builder_factory=CapturingBuilder
|
||
)
|
||
result, _output = self._run(adapters)
|
||
self.assertTrue(result["ok"], result)
|
||
|
||
# 正向对照:未篡改的合法报告(原始输出与报告一致)必须通过 builder,无系统失败且评分入输入。
|
||
gate_input = GateInputBuilder().build(**copy.deepcopy(captured))
|
||
self.assertEqual(gate_input["schemaVersion"], "writer-gate-input-v2")
|
||
legit_sample = next(
|
||
item for item in gate_input["samples"] if item["sampleId"] == "deep-space-489"
|
||
)
|
||
self.assertFalse(legit_sample["systemFailure"])
|
||
self.assertTrue(legit_sample["schemaValid"])
|
||
self.assertIn("scores", legit_sample)
|
||
|
||
true_receipts = captured["execution_receipts"]["deep-space-489"]["A"][
|
||
"blind_judge"
|
||
]["executionReceipts"]
|
||
judge_report = copy.deepcopy(captured["judge_reports"]["deep-space-489"])
|
||
forged_score = 9.5
|
||
for index, reviewer_report in enumerate(judge_report["reviewerReports"]):
|
||
forged_output = copy.deepcopy(judge_report["reviewerStructuredOutputs"][index])
|
||
# 报告评分与伪造原始输出同步改成一致的伪造高分,使报告内部自洽。
|
||
for report_row, draft_row in zip(
|
||
reviewer_report["candidateScores"],
|
||
forged_output["candidateScores"],
|
||
strict=True,
|
||
):
|
||
for dimension in DIMENSIONS:
|
||
report_row["scores"][dimension]["score"] = forged_score
|
||
draft_row["scores"][dimension]["score"] = forged_score
|
||
judge_report["reviewerStructuredOutputs"][index] = forged_output
|
||
# 伪造输出与真实回执固化的原始输出哈希必然不同——这是攻击者改不动的锚点。
|
||
self.assertNotEqual(
|
||
canonical_sha256(forged_output),
|
||
true_receipts[index]["structuredOutputSha256"],
|
||
)
|
||
# modelReceiptSha256 保留(仍绑定未改动的真实回执),只重签内层 reportSha256。
|
||
reviewer_report["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in reviewer_report.items() if key != "reportSha256"}
|
||
)
|
||
# 按篡改后的 reviewer 报告重推导 panel,使 recomputed_panel == report 这一旧校验仍能通过。
|
||
panel = adjudicate_structured_reviews(
|
||
judge_report["reviewerReports"][0],
|
||
judge_report["reviewerReports"][1],
|
||
judge_report["reviewerReports"][2]
|
||
if len(judge_report["reviewerReports"]) == 3
|
||
else None,
|
||
)
|
||
self.assertIn(panel["status"], {"stable_report", "adjudicated_report"})
|
||
judge_report.update(panel)
|
||
judge_report["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in judge_report.items() if key != "reportSha256"}
|
||
)
|
||
tampered_judge_reports = {
|
||
**captured["judge_reports"],
|
||
"deep-space-489": judge_report,
|
||
}
|
||
# 伪造评分自洽、重签重推导都做了,仍被「原始输出 ↔ 回执哈希」交叉校验失败关闭:
|
||
# 该样本被标 systemFailure、schemaValid=False,伪造评分不进 Gate 输入的 scores。
|
||
tampered_input = GateInputBuilder().build(
|
||
**{**copy.deepcopy(captured), "judge_reports": tampered_judge_reports}
|
||
)
|
||
tampered_sample = next(
|
||
item
|
||
for item in tampered_input["samples"]
|
||
if item["sampleId"] == "deep-space-489"
|
||
)
|
||
self.assertTrue(tampered_sample["systemFailure"])
|
||
self.assertFalse(tampered_sample["schemaValid"])
|
||
self.assertNotIn("scores", tampered_sample)
|
||
self.assertTrue(
|
||
any(
|
||
"reviewerStructuredOutputs" in reason
|
||
and "未绑定模型原始 structured output" in reason
|
||
for reason in tampered_sample["systemFailureReasons"]
|
||
),
|
||
tampered_sample["systemFailureReasons"],
|
||
)
|
||
|
||
# 报告缺失 reviewerStructuredOutputs 字段时同样失败关闭:不放行无绑定的报告。
|
||
stripped = copy.deepcopy(captured["judge_reports"]["deep-space-489"])
|
||
del stripped["reviewerStructuredOutputs"]
|
||
stripped["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in stripped.items() if key != "reportSha256"}
|
||
)
|
||
stripped_input = GateInputBuilder().build(
|
||
**{
|
||
**copy.deepcopy(captured),
|
||
"judge_reports": {
|
||
**captured["judge_reports"],
|
||
"deep-space-489": stripped,
|
||
},
|
||
}
|
||
)
|
||
stripped_sample = next(
|
||
item
|
||
for item in stripped_input["samples"]
|
||
if item["sampleId"] == "deep-space-489"
|
||
)
|
||
self.assertTrue(stripped_sample["systemFailure"])
|
||
self.assertNotIn("scores", stripped_sample)
|
||
self.assertTrue(
|
||
any(
|
||
"reviewerStructuredOutputs" in reason
|
||
for reason in stripped_sample["systemFailureReasons"]
|
||
),
|
||
stripped_sample["systemFailureReasons"],
|
||
)
|
||
|
||
def test_semantic_failure_stops_sample_before_judge(self):
|
||
adapters, writer, semantic, judge = _production_adapters(semantic_fail_on=2)
|
||
|
||
result, _output = self._run(adapters, run_id="semantic-failure")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_semantic_detector")
|
||
self.assertEqual(len(writer.calls), 2)
|
||
self.assertEqual(len(semantic.calls), 2)
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_third_judge_still_unstable_fails_run(self):
|
||
adapters, _writer, _semantic, judge = _production_adapters(judge_unstable=True)
|
||
|
||
result, _output = self._run(adapters, run_id="judge-third-unstable")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_unstable")
|
||
self.assertEqual(len(judge.calls), 3)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["blind_judge"], 3)
|
||
self.assertEqual(result["budgetLedger"]["remainingPlannedCalls"]["blind_judge"], 0)
|
||
self.assertEqual(result["budgetLedger"]["totalActualCostUsd"], "0.090000")
|
||
|
||
def test_vault_cleanup_failure_overrides_success(self):
|
||
class CleanupFailureManager(RawVaultManager):
|
||
"""先实际清理 raw,再模拟清理回执失败,避免测试残留敏感目录。"""
|
||
|
||
def cleanup(self, lease):
|
||
super().cleanup(lease)
|
||
raise RawVaultError("RAW_CLEANUP_FAILED", "测试清理失败")
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
vault_factory=CleanupFailureManager
|
||
)
|
||
|
||
result, _output = self._run(adapters, run_id="cleanup-failure")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_raw_cleanup")
|
||
|
||
def test_cas_conflict_fails_sample_and_run(self):
|
||
class ConflictingCas:
|
||
"""初始化使用真实 CAS,第一次推进模拟迟到 revision 冲突。"""
|
||
|
||
def __init__(self, root):
|
||
self.delegate = FileCasStore(root)
|
||
|
||
def initialize(self, **kwargs):
|
||
return self.delegate.initialize(**kwargs)
|
||
|
||
def transition(self, **_kwargs):
|
||
raise CasConflictError("测试旧 revision")
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters(cas_factory=ConflictingCas)
|
||
|
||
result, _output = self._run(adapters, run_id="cas-conflict")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_cas_conflict")
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_builder_binding_failure_fails_after_all_sample_terminals(self):
|
||
class FailingBuilder:
|
||
"""模拟来源绑定无法闭合,证明不能手工回退聚合。"""
|
||
|
||
def build(self, **_kwargs):
|
||
raise GateInputBuildError("测试 builder 绑定失败")
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(builder_factory=FailingBuilder)
|
||
|
||
result, output = self._run(adapters, run_id="builder-failure")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_gate_input_builder")
|
||
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
|
||
def test_pending_probe_and_budget_block_before_vault_or_runner(self):
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
pending = _production_config()
|
||
pending["commonControls"]["modelVersion"] = "pending_probe"
|
||
pending["commonControls"]["adapterVersion"] = "pending_probe"
|
||
result, _output = self._run(adapters, evaluation_config=pending, run_id="pending-probe")
|
||
self.assertEqual(result["status"], "blocked_runtime_probe")
|
||
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=_production_config(budget_approved=False),
|
||
run_id="pending-budget",
|
||
)
|
||
self.assertEqual(result["status"], "blocked_budget_authorization")
|
||
self.assertEqual(vault_calls, [])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_public_test_adapter_cannot_bypass_execute_authorization(self):
|
||
"""仅传 test_adapters 不能跳过 probe、预算和 raw 审批。"""
|
||
|
||
adapters, runner, detector, judge = _test_adapters()
|
||
unauthorized = config()
|
||
unauthorized.pop("executionAuthorization", None)
|
||
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
result = run_writer_replay(
|
||
unauthorized,
|
||
run_id="test-adapter-no-authorization",
|
||
output_dir=output,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_authorization")
|
||
self.assertEqual(runner.calls, [])
|
||
self.assertEqual(detector.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "raw").exists())
|
||
|
||
def test_authorized_test_adapter_never_persists_plaintext_raw_directory(self):
|
||
"""fake 执行也必须使用受控 raw lease,并在返回前完成清理。"""
|
||
|
||
adapters, _runner, _detector, _judge = _test_adapters()
|
||
authorized = config()
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
|
||
result = run_writer_replay(
|
||
authorized,
|
||
run_id="test-adapter-vault",
|
||
output_dir=output,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertFalse((output / "raw").exists())
|
||
self.assertNotIn("candidateBody", (output / "manifest.json").read_text(encoding="utf-8"))
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
|
||
|
||
def test_production_profile_hash_mismatch_cannot_use_noop_verifier(self):
|
||
"""生产 A/C 与 B 都必须在 runner 前执行真实 CLI/hash/version 校验。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
invalid_profile = replace(
|
||
adapters.writer_profile,
|
||
claude_executable_sha256="0" * 64,
|
||
)
|
||
adapters = replace(adapters, writer_profile=invalid_profile)
|
||
evaluation_config = _production_config()
|
||
evaluation_config["executionAuthorization"]["profileSha256"]["writer"] = (
|
||
invalid_profile.execution_profile_sha256
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=evaluation_config,
|
||
run_id="invalid-cli-binding",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertNotIn("binding_verifier", inspect.signature(run_writer_replay).parameters)
|
||
|
||
def test_runtime_probe_self_hash_tamper_fails_closed_before_runner(self):
|
||
"""runtime probe receipt 自哈希被篡改时,不能进入任何模型 runner。"""
|
||
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
tampered = _production_config()
|
||
tampered["executionAuthorization"]["runtimeProbe"]["receiptSha256"] = (
|
||
"sha256:" + "0" * 64
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=tampered,
|
||
run_id="tampered-runtime-probe",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_authorization")
|
||
self.assertEqual(result["errors"], ["EXECUTE_AUTHORIZATION_HASH_INVALID"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_execute_blocks_stale_probe_bound_to_old_contract(self):
|
||
"""陈旧运行探针(身份哈希对不上新 writer 合同)必须被机械阻断在 runner 之前。
|
||
|
||
对照设计:与下方 test_execute_passes_probe_bound_to_current_contract 共用同一
|
||
_production_config,只把探针的身份哈希/输出哈希换成旧合同的值。旧探针只证明
|
||
「旧合同能产出结构化输出」,证明不了换过 schema/prompt 的新合同也可以,故必须挡下。
|
||
"""
|
||
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
stale = _production_config()
|
||
probe = stale["executionAuthorization"]["runtimeProbe"]
|
||
# 模拟探针仍是旧合同实跑的:身份哈希与结构化输出哈希都来自旧合同。
|
||
probe["executionProfileSha256"] = "sha256:" + "9" * 64
|
||
probe["structuredOutputSha256"] = "sha256:" + "4" * 64
|
||
# 重算探针自哈希,确保它卡在「合同绑定」这一关,而不是更早的自哈希这一关。
|
||
probe["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in probe.items() if key != "receiptSha256"}
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters, evaluation_config=stale, run_id="stale-probe-old-contract"
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_probe_contract")
|
||
self.assertEqual(result["errors"], ["EXECUTE_PROBE_CONTRACT_MISMATCH"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_execute_passes_probe_bound_to_current_contract(self):
|
||
"""按新合同重跑的探针(身份哈希==新 writer profile)必须放行 execute 门。
|
||
|
||
与上方陈旧探针对照:同一 _production_config,探针身份哈希绑定当前 writer 合同,
|
||
execute 门放行并真正进入模型 runner 跑完全链。
|
||
"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
bound = _production_config()
|
||
probe = bound["executionAuthorization"]["runtimeProbe"]
|
||
# 显式确认探针身份哈希已绑定当前 writer 合同(新 schema/prompt 的身份)。
|
||
self.assertEqual(
|
||
probe["executionProfileSha256"],
|
||
adapters.writer_profile.execution_profile_sha256,
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters, evaluation_config=bound, run_id="probe-current-contract"
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual(len(writer.calls), 3)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|