3867 lines
166 KiB
Python
3867 lines
166 KiB
Python
#!/usr/bin/env python3
|
||
"""正文 A/B/C 冻结回放编排的无网络、无真实模型测试。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import hashlib
|
||
import inspect
|
||
import json
|
||
import os
|
||
import pathlib
|
||
import signal
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
from dataclasses import replace
|
||
from datetime import UTC, datetime, timedelta
|
||
from decimal import Decimal
|
||
from unittest import mock
|
||
|
||
TEST_DIR = pathlib.Path(__file__).resolve().parent
|
||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||
SKILLS_DIR = PROJECT_ROOT / ".agent" / "skills"
|
||
SCRIPT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "replay" / "回放评估正文质量" / "scripts"
|
||
READ_CONTEXT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "context" / "skills" / "准备任务上下文" / "scripts"
|
||
EVIDENCE_DIR = PROJECT_ROOT / "muse" / "authority" / "evidence" / "skills" / "记录运行证据" / "scripts"
|
||
QUALITY_GATE_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "judge" / "评估内容质量" / "scripts"
|
||
GATE_ADJUDICATION_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "mechanical" / "判定质量是否合格" / "scripts"
|
||
for import_path in (
|
||
SCRIPT_DIR,
|
||
READ_CONTEXT_DIR,
|
||
EVIDENCE_DIR,
|
||
QUALITY_GATE_DIR,
|
||
GATE_ADJUDICATION_DIR,
|
||
):
|
||
if str(import_path) not in sys.path:
|
||
sys.path.insert(0, str(import_path))
|
||
|
||
import run_writer_replay as replay_module # noqa: E402
|
||
import run_writer_replay.execute as execute_module # noqa: E402
|
||
from file_cas import CasConflictError, FileCasStore # noqa: E402
|
||
from gate_input_builder import GateInputBuilder, GateInputBuildError, canonical_sha256 # noqa: E402
|
||
from muse_role import ( # noqa: E402
|
||
FIXED_OPUS_MODEL_ID,
|
||
FIXED_OPUS_POLICY_ALIAS,
|
||
FIXED_OPUS_POLICY_VERSION,
|
||
MODEL_POLICY_VERSION,
|
||
RUNTIME_ADAPTER,
|
||
RUNTIME_ADAPTER_VERSION,
|
||
RoleExecutionProfile,
|
||
RoleRuntimeError,
|
||
build_dispatch_system_prompt,
|
||
sha256_json,
|
||
sha256_text,
|
||
)
|
||
from raw_vault import RawVaultError, RawVaultManager # noqa: E402
|
||
from run_writer import build_writer_execution_profile # noqa: E402
|
||
from run_writer_blind_judge import BLIND_JUDGE_REPORT_JSON_SCHEMA # noqa: E402
|
||
from run_writer_replay import ( # noqa: E402
|
||
BudgetLedgerError,
|
||
WriterReplayError,
|
||
WriterReplayProductionAdapters,
|
||
WriterReplayTestAdapters,
|
||
run_writer_replay,
|
||
)
|
||
from run_writer_replay.blind import ( # noqa: E402
|
||
_SOURCE_REF_ALLOWED,
|
||
_clean_source_ref,
|
||
_project_oracle_pack,
|
||
_semantic_input_v3,
|
||
)
|
||
from run_writer_semantic_detector import ( # noqa: E402
|
||
SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
is_semantic_schema_specialization,
|
||
)
|
||
from writer_contract import calculate_target_chars, han_count # noqa: E402
|
||
from writer_eval_preregister import build_balanced_preregistration # noqa: E402
|
||
from writer_gate import decide_gate # noqa: E402
|
||
from writer_rubric import ( # noqa: E402
|
||
COMMON_DIMENSIONS,
|
||
DIMENSIONS,
|
||
RUBRIC_POLICY_VERSION,
|
||
RUBRIC_PROFILE,
|
||
SCENARIO_DIMENSION,
|
||
adjudicate_structured_reviews,
|
||
)
|
||
|
||
|
||
def _utc_stamp(delta: timedelta) -> str:
|
||
"""相对当前时刻生成 UTC 时间戳。"""
|
||
|
||
return (
|
||
(datetime.now(UTC) + delta)
|
||
.replace(microsecond=0)
|
||
.isoformat()
|
||
.replace("+00:00", "Z")
|
||
)
|
||
|
||
|
||
# 授权快照的 checkedAt 必须早于现在、revalidationAt 必须晚于现在,否则 check_snapshot
|
||
# 会 fail-closed 成 blocked_authorization。写死日期会让这份脱敏夹具到期后集体变红,
|
||
# 因此改成相对当前时钟派生。
|
||
AUTHORIZATION_CHECKED_AT = _utc_stamp(timedelta(days=-30))
|
||
AUTHORIZATION_REVALIDATION_AT = _utc_stamp(timedelta(days=30))
|
||
|
||
AUTHORIZATION = {
|
||
"sourceStatus": "authorized",
|
||
"copyrightStatus": "research_only",
|
||
"sourceHash": "sha256:" + "a" * 64,
|
||
"sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64,
|
||
"allowedPurpose": ["offline_evaluation"],
|
||
"forbiddenPurpose": ["external_distribution", "model_training", "production_generation"],
|
||
"authorizationSnapshot": {
|
||
"id": "auth-work-8",
|
||
"version": "auth-work-8-v1",
|
||
"immutable": True,
|
||
"sourceHash": "sha256:" + "a" * 64,
|
||
"sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64,
|
||
"sourceStatus": "authorized",
|
||
"copyrightStatus": "research_only",
|
||
"authorizationBasis": "sanitized_contract_fixture",
|
||
"allowedPurpose": ["offline_evaluation"],
|
||
"forbiddenPurpose": ["external_distribution", "model_training", "production_generation"],
|
||
"checkedAt": AUTHORIZATION_CHECKED_AT,
|
||
"revalidationAt": AUTHORIZATION_REVALIDATION_AT,
|
||
},
|
||
}
|
||
|
||
|
||
GATE_A_CONFIG_PATH = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1-cn-skills.json"
|
||
|
||
|
||
def _load_gate_a_config() -> dict[str, object]:
|
||
"""读取 Gate A 配置并把授权重验窗口对齐到当前时钟。
|
||
|
||
磁盘上的窗口是一次真实登记的结果,会随时间自然到期;结构性合同测试不应该被
|
||
这个运维事实带红,所以在内存副本里刷新 checkedAt/verifiedAt/revalidationAt。
|
||
"""
|
||
|
||
gate_config = json.loads(GATE_A_CONFIG_PATH.read_text(encoding="utf-8"))
|
||
snapshot = gate_config["authorization"]["authorizationSnapshot"]
|
||
snapshot["checkedAt"] = AUTHORIZATION_CHECKED_AT
|
||
snapshot["revalidationAt"] = AUTHORIZATION_REVALIDATION_AT
|
||
for sample in gate_config["samples"]:
|
||
context_snapshot = sample["writerContextInput"].get("authorizationSnapshot")
|
||
if isinstance(context_snapshot, dict):
|
||
context_snapshot["verifiedAt"] = AUTHORIZATION_CHECKED_AT
|
||
return gate_config
|
||
|
||
|
||
def _authorize_gate_config_for_test(gate_config: dict[str, object]) -> dict[str, object]:
|
||
"""为门禁顺序测试注入 provider-neutral 合成探针,不冒充生产能力证明。"""
|
||
|
||
writer = replay_module.profile_from_mapping(
|
||
gate_config["executionProfiles"]["writer"],
|
||
role="writer",
|
||
)
|
||
probe = {
|
||
"schemaVersion": "runtime-probe-v2-test-fixture",
|
||
"status": "successful",
|
||
"checkedAt": _utc_stamp(timedelta(minutes=-1)),
|
||
"runtimeAdapter": RUNTIME_ADAPTER,
|
||
"runtimeAdapterVersion": RUNTIME_ADAPTER_VERSION,
|
||
"modelPolicyVersion": MODEL_POLICY_VERSION,
|
||
"role": "writer",
|
||
"profileVersion": writer.profile_version,
|
||
"modelAlias": writer.model_alias,
|
||
"executionProfileSha256": writer.execution_profile_sha256,
|
||
"jsonSchemaId": writer.json_schema_id,
|
||
"jsonSchemaSha256": writer.json_schema_sha256,
|
||
"systemPromptId": writer.system_prompt_id,
|
||
"systemPromptSha256": writer.system_prompt_sha256,
|
||
"inputSha256": "sha256:" + "5" * 64,
|
||
"requestedModelId": writer.model_alias,
|
||
"actualModelId": FIXED_OPUS_MODEL_ID,
|
||
"modelMatch": True,
|
||
"executionReceiptSha256": "sha256:" + "7" * 64,
|
||
"structuredOutputSha256": "sha256:" + "6" * 64,
|
||
"terminalReason": "completed",
|
||
"totalCostUsd": "0.010000",
|
||
}
|
||
probe["receiptSha256"] = canonical_sha256(probe)
|
||
gate_config["executionAuthorization"]["runtimeProbe"] = probe
|
||
return gate_config
|
||
|
||
|
||
def _prose(chapter: int, block_id: int, text: str) -> dict[str, object]:
|
||
"""构造只用于合同证明的脱敏合成原文片段。"""
|
||
|
||
return {
|
||
"chapter": chapter,
|
||
"sourceRef": {
|
||
"sourceId": f"fixture:chapter:{chapter}:block:{block_id}",
|
||
"sourceVersion": f"sanitized-fixture-{chapter}-v1",
|
||
"chapter": chapter,
|
||
"blockId": block_id,
|
||
"startCodePoint": 0,
|
||
"endCodePoint": len(text),
|
||
},
|
||
"text": text,
|
||
}
|
||
|
||
|
||
def _writer_context_input() -> dict[str, object]:
|
||
"""构造可让 A/B/C 都通过 WriterContext v1 的脱敏夹具。"""
|
||
|
||
recent = [
|
||
_prose(chapter, 1000 + chapter, f"脱敏合同夹具第{chapter}章,人物保持冻结状态。")
|
||
for chapter in range(485, 489)
|
||
]
|
||
supplemental = _prose(120, 1120, "脱敏补充片段,旧徽章曾经出现。")
|
||
return {
|
||
"contentMode": "sanitized_contract_fixture",
|
||
"sourceVersion": AUTHORIZATION["sourceVersion"],
|
||
"authorizationSnapshot": {
|
||
"snapshotId": "auth-work-8",
|
||
"allowedPurpose": "offline_evaluation",
|
||
"verifiedAt": AUTHORIZATION_CHECKED_AT,
|
||
},
|
||
"sourceStatus": "authorized",
|
||
"cardIndexVersion": "cards-frozen-488-v1",
|
||
"proseIndexVersion": "prose-frozen-488-v1",
|
||
"retrievalResult": {
|
||
"cards": [
|
||
{
|
||
"name": "林澈",
|
||
"type": "character",
|
||
"sourceRefs": [copy.deepcopy(supplemental["sourceRef"])],
|
||
}
|
||
],
|
||
"factEvidence": [],
|
||
"proseEvidence": [supplemental],
|
||
"indexHints": [
|
||
{
|
||
"cardId": "card-1",
|
||
"name": "林澈",
|
||
"type": "character",
|
||
"content": "林澈在冻结点仍位于圣蒂曼",
|
||
"sourceId": "fixture:card:1",
|
||
"sourceVersion": "cards-frozen-488-v1",
|
||
"asOf": 488,
|
||
}
|
||
],
|
||
"manifest": {"omittedSources": []},
|
||
},
|
||
"fineOutline": {
|
||
"sourceRef": {
|
||
"sourceId": "fixture:fine-outline:489",
|
||
"sourceVersion": "fine-outline-fixture-v1",
|
||
"chapter": 489,
|
||
},
|
||
"hardConstraints": ["必须完成围攻突围"],
|
||
"adjustableBeats": [],
|
||
"declaredNewFacts": [],
|
||
"entities": [{"id": "character:林澈", "type": "character", "name": "林澈"}],
|
||
"relations": [],
|
||
"items": [{"id": "item:旧徽章", "type": "item", "name": "旧徽章"}],
|
||
"locations": [{"id": "location:圣蒂曼", "type": "location", "name": "圣蒂曼"}],
|
||
"powerSystems": [],
|
||
},
|
||
"narrativeState": {
|
||
"time": "围攻当日",
|
||
"location": "圣蒂曼",
|
||
"characterPositions": {"林澈": "城内"},
|
||
"immediateSituation": "围攻持续",
|
||
},
|
||
"recentChapters": recent,
|
||
"outputContract": {
|
||
"targetChars": 4000,
|
||
"minChars": 2800,
|
||
"maxChars": 5200,
|
||
"frontmatterRequired": False,
|
||
},
|
||
"tokenBudget": {"maxContextChars": 50000},
|
||
"generatedAt": AUTHORIZATION_CHECKED_AT,
|
||
"requirements": {
|
||
"requiredEvents": [
|
||
{"requirementId": "event-1", "anchors": ["完成围攻突围"]}
|
||
],
|
||
"requiredCharacters": ["林澈"],
|
||
"foreshadowingActions": [
|
||
{"requirementId": "foreshadow-1", "anchors": ["旧徽章"]}
|
||
],
|
||
"chapterEndHook": {
|
||
"requirementId": "hook-1",
|
||
"anchors": ["城门忽然打开"],
|
||
"maxDistanceFromEnd": 20,
|
||
},
|
||
"detectedNewSettings": [],
|
||
"frozenConflicts": [],
|
||
},
|
||
}
|
||
|
||
|
||
def config() -> dict[str, object]:
|
||
"""构造不含原文全文的单样本预注册配置。"""
|
||
|
||
common = {
|
||
"workId": 8,
|
||
"asOfChapter": 488,
|
||
"targetChapter": 489,
|
||
"outlineSource": "outline:481-488",
|
||
"fineOutlineSource": "scaffold:489",
|
||
"targetChars": 4000,
|
||
"modelVersion": MODEL_POLICY_VERSION,
|
||
"sampling": {"temperature": 0, "topP": 1, "seed": 489},
|
||
"detectorProfile": "writer-detector-report-v1",
|
||
}
|
||
value = {
|
||
"profile": "writer_replay",
|
||
"evaluationSetVersion": "writer-gate-a-test-v1",
|
||
"strategyVersion": "writer-abc-v1",
|
||
"referenceWork": {
|
||
"id": 8,
|
||
"title": "深空之影",
|
||
"version": AUTHORIZATION["sourceVersion"],
|
||
},
|
||
"authorization": copy.deepcopy(AUTHORIZATION),
|
||
"commonControls": common,
|
||
"samples": [
|
||
{
|
||
"sampleId": "deep-space-489",
|
||
"scenario": "battle",
|
||
"asOfChapter": 488,
|
||
"targetChapter": 489,
|
||
"snapshotVersion": "writer-deep-space-489-v1",
|
||
"frozenRecentHanCounts": [4600, 4600, 4800, 4800],
|
||
"targetLengthBasis": {
|
||
"algorithm": "calculate_target_chars",
|
||
"sourceChapters": [485, 486, 487, 488],
|
||
"hardEventCount": 1,
|
||
"foreshadowingActionCount": 0,
|
||
"requiredSceneCount": 0,
|
||
"minChars": 2000,
|
||
"maxChars": 10000,
|
||
"usesTargetChapterLength": False,
|
||
},
|
||
"targetChars": 4000,
|
||
"expectedLength": {
|
||
"targetChars": 4000,
|
||
"minChars": 2800,
|
||
"maxChars": 5200,
|
||
},
|
||
"newCharacterRatio": 0.0,
|
||
"newCharacterRatioStatus": "resolved",
|
||
"newCharacterBasis": {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": 488,
|
||
"requiredCharacters": ["林澈"],
|
||
"knownBeforeAsOf": ["林澈"],
|
||
"absentBeforeAsOf": [],
|
||
"genericRoles": [],
|
||
},
|
||
"sources": [
|
||
{
|
||
"sourceId": "fixture:chapter:485-488",
|
||
"sourceVersion": AUTHORIZATION["sourceVersion"],
|
||
"chapterRange": "485-488",
|
||
},
|
||
{
|
||
"sourceId": "fixture:card-index:488",
|
||
"sourceVersion": "cards-frozen-488-v1",
|
||
"chapterRange": "1-488",
|
||
},
|
||
],
|
||
"snapshotData": {
|
||
"chapters": [
|
||
{
|
||
"chapter": 488,
|
||
"sourceId": "fixture:chapter:488",
|
||
"contentSha256": "sha256:" + "1" * 64,
|
||
}
|
||
],
|
||
"cards": [],
|
||
},
|
||
"writerContextInput": _writer_context_input(),
|
||
"leakageAudit": {
|
||
"method": "target-fact-hash-and-chapter-bound-audit",
|
||
"targetFacts": {
|
||
"targetChapter": 489,
|
||
"forbiddenFacts": [
|
||
{
|
||
"id": "target-489-1",
|
||
"firstChapter": 489,
|
||
"text": "目标章专属秘密",
|
||
}
|
||
],
|
||
},
|
||
},
|
||
}
|
||
],
|
||
}
|
||
value["executionAuthorization"] = _test_execution_authorization()
|
||
return value
|
||
|
||
|
||
def _candidate_body(marker: str) -> str:
|
||
"""构造满足动态篇幅和全部机械锚点的三臂候选。"""
|
||
|
||
prefix = f"林澈{marker}完成围攻突围,又把旧徽章压回掌心。"
|
||
hook = "城门忽然打开"
|
||
filler_count = 4000 - han_count(prefix) - han_count(hook)
|
||
return prefix + "文" * filler_count + hook
|
||
|
||
|
||
def _writer_output(
|
||
context: dict[str, object], *, call_index: int
|
||
) -> dict[str, object]:
|
||
"""从治理调用的冻结 JSON 输入构造最小 WriterDraft v2。"""
|
||
|
||
# A/C 投影不得泄露臂策略;测试按预注册调用顺序区分候选,不向输入补旧字段。
|
||
marker = ("甲", "乙", "丙")[call_index % 3]
|
||
body = _candidate_body(marker)
|
||
return {"candidateBody": body}
|
||
|
||
|
||
class FakeGovernedChat:
|
||
"""只替换治理模型调用,候选仍由 run_writer() 解析和校验。"""
|
||
|
||
def __init__(self, *, illegal_extra_field: bool = False):
|
||
self.calls: list[tuple[str, dict[str, object]]] = []
|
||
self.contexts: list[dict[str, object]] = []
|
||
self.illegal_extra_field = illegal_extra_field
|
||
|
||
def __call__(self, prompt: str, **kwargs: object):
|
||
context = json.loads(prompt)
|
||
self.assert_safe_projection(context)
|
||
self.calls.append((prompt, kwargs))
|
||
self.contexts.append(context)
|
||
output = _writer_output(context, call_index=len(self.contexts) - 1)
|
||
if self.illegal_extra_field and len(self.contexts) == 2:
|
||
output["candidateSha256"] = "sha256:" + "0" * 64
|
||
return (
|
||
json.dumps(output, ensure_ascii=False),
|
||
{"input_tokens": 10, "output_tokens": 10},
|
||
FIXED_OPUS_MODEL_ID,
|
||
)
|
||
|
||
@staticmethod
|
||
def assert_safe_projection(context: dict[str, object]) -> None:
|
||
"""证明模型只收到 WriterCreativeInput v2;篇幅修订调用可额外带一个 lengthRevision 块。"""
|
||
|
||
expected = {
|
||
"fineOutline",
|
||
"narrativeState",
|
||
"factConstraints",
|
||
"proseExcerpts",
|
||
"patternReferences",
|
||
"lengthContract",
|
||
"styleConstraints",
|
||
}
|
||
keys = set(context)
|
||
if keys != expected and keys != expected | {"lengthRevision"}:
|
||
raise AssertionError(f"writer 创作投影字段非法: {sorted(context)}")
|
||
if "lengthRevision" in context:
|
||
# lengthRevision 只允许携带模型自己的上一版正文、篇幅数字与固定修订指令,
|
||
# 不得混入臂策略、raw 路径等投影外字段。
|
||
allowed_revision = {
|
||
"previousDraft",
|
||
"actualHanChars",
|
||
"minChars",
|
||
"maxChars",
|
||
"targetChars",
|
||
"countingRule",
|
||
"revisionDirection",
|
||
"targetDeltaHanChars",
|
||
"instruction",
|
||
}
|
||
revision = context["lengthRevision"]
|
||
if not isinstance(revision, dict) or set(revision) != allowed_revision:
|
||
raise AssertionError(
|
||
f"lengthRevision 块字段非法: {sorted(revision) if isinstance(revision, dict) else revision}"
|
||
)
|
||
if revision["countingRule"] != "han_chars_only":
|
||
raise AssertionError("lengthRevision 计数口径非法")
|
||
if revision["revisionDirection"] not in {"expand", "trim"}:
|
||
raise AssertionError("lengthRevision 修订方向非法")
|
||
numeric_fields = {
|
||
"actualHanChars",
|
||
"minChars",
|
||
"maxChars",
|
||
"targetChars",
|
||
"targetDeltaHanChars",
|
||
}
|
||
if any(
|
||
not isinstance(revision[field], int) or isinstance(revision[field], bool)
|
||
for field in numeric_fields
|
||
):
|
||
raise AssertionError("lengthRevision 数字字段非法")
|
||
expected_delta = abs(revision["targetChars"] - revision["actualHanChars"])
|
||
if revision["targetDeltaHanChars"] != expected_delta:
|
||
raise AssertionError("lengthRevision 距目标差值未按当前正文重算")
|
||
expected_direction = (
|
||
"expand"
|
||
if revision["actualHanChars"] < revision["targetChars"]
|
||
else "trim"
|
||
)
|
||
if revision["revisionDirection"] != expected_direction:
|
||
raise AssertionError("lengthRevision 方向未按当前正文重算")
|
||
|
||
|
||
class InterruptingChat(FakeGovernedChat):
|
||
"""模拟外部终止落在模型调用中间。"""
|
||
|
||
def __init__(self, *, interrupt_on_call: int) -> None:
|
||
super().__init__()
|
||
self.interrupt_on_call = interrupt_on_call
|
||
|
||
def __call__(self, prompt: str, **kwargs: object):
|
||
if len(self.calls) + 1 == self.interrupt_on_call:
|
||
signal.raise_signal(signal.SIGTERM)
|
||
return super().__call__(prompt, **kwargs)
|
||
|
||
|
||
class FakeSemanticDetector:
|
||
"""记录机械门输出,并返回与候选绑定的语义通过报告。"""
|
||
|
||
def __init__(self) -> None:
|
||
self.calls: list[tuple[str, bool]] = []
|
||
|
||
def __call__(self, context, candidate, mechanical_report):
|
||
strategy = str(context.get("evidenceStrategy") or "neutral_prose_projection")
|
||
self.calls.append((strategy, mechanical_report["passed"]))
|
||
return {
|
||
"status": "passed",
|
||
"candidateVersion": candidate["candidateVersion"],
|
||
"candidateSha256": candidate["candidateSha256"],
|
||
"blockingFailures": [],
|
||
"suggestions": [],
|
||
}
|
||
|
||
|
||
def judge_report(
|
||
reviewer_id: str,
|
||
sample_id: str,
|
||
blind_id: str,
|
||
order: list[str],
|
||
score: float = 8.0,
|
||
) -> dict[str, object]:
|
||
"""构造测试盲评报告。"""
|
||
|
||
return {
|
||
"profile": RUBRIC_PROFILE,
|
||
"reviewerId": reviewer_id,
|
||
"sampleId": sample_id,
|
||
"blindCandidateId": blind_id,
|
||
"candidateOrder": order,
|
||
"scores": {
|
||
dimension: {
|
||
"score": score,
|
||
"evidence": [
|
||
{
|
||
"sourceType": "judge_inference",
|
||
"sourceRef": f"candidate:{blind_id}",
|
||
"excerpt": f"证据-{dimension}",
|
||
}
|
||
],
|
||
}
|
||
for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
|
||
|
||
class FakeJudge:
|
||
"""模拟双评差异和必要第三评,不读取网络或真实模型。"""
|
||
|
||
def __init__(
|
||
self,
|
||
*,
|
||
unstable: bool = False,
|
||
invalid_third: bool = False,
|
||
irreducibly_unstable: bool = False,
|
||
):
|
||
self.calls: list[tuple[str, str]] = []
|
||
self.unstable = unstable
|
||
self.invalid_third = invalid_third
|
||
self.irreducibly_unstable = irreducibly_unstable
|
||
|
||
def __call__(self, blind_input, reviewer_id):
|
||
blind_id = blind_input["blindCandidateId"]
|
||
candidate_order = blind_input["candidateOrder"]
|
||
self.calls.append((reviewer_id, blind_id))
|
||
if self.invalid_third and reviewer_id == "judge-3":
|
||
return {"status": "malformed"}
|
||
score = 8.0
|
||
if self.unstable and blind_id == "blind-1" and reviewer_id == "judge-2":
|
||
score = 7.0
|
||
if self.unstable and blind_id == "blind-1" and reviewer_id == "judge-3":
|
||
score = 7.5
|
||
if self.irreducibly_unstable and blind_id == "blind-1":
|
||
score = {"judge-1": 8.0, "judge-2": 6.0, "judge-3": 7.0}[reviewer_id]
|
||
return judge_report(
|
||
reviewer_id,
|
||
blind_input["sample"]["sampleId"],
|
||
blind_id,
|
||
candidate_order,
|
||
score,
|
||
)
|
||
|
||
|
||
class MaliciousJudge(FakeJudge):
|
||
"""主动扫描全部可见输入,证明 judge 无法发现真实臂或原始目录。"""
|
||
|
||
def __init__(self) -> None:
|
||
super().__init__()
|
||
self.visible_inputs: list[dict[str, object]] = []
|
||
|
||
def __call__(self, blind_input, reviewer_id):
|
||
rendered = json.dumps(blind_input, ensure_ascii=False, sort_keys=True)
|
||
forbidden = (
|
||
"candidate-A",
|
||
"candidate-B",
|
||
"candidate-C",
|
||
"pipeline-A",
|
||
"pipeline-B",
|
||
"pipeline-C",
|
||
"evidenceStrategy",
|
||
"writerContextInput",
|
||
"indexHints",
|
||
"retrievalManifest",
|
||
"evidenceCoverage",
|
||
"raw_dir",
|
||
"rawDir",
|
||
"_contexts",
|
||
)
|
||
for marker in forbidden:
|
||
if marker in rendered:
|
||
raise AssertionError(f"judge 可见输入泄露: {marker}")
|
||
self.visible_inputs.append(copy.deepcopy(blind_input))
|
||
return super().__call__(blind_input, reviewer_id)
|
||
|
||
|
||
def _test_adapters(
|
||
*,
|
||
unstable: bool = False,
|
||
illegal_extra_field: bool = False,
|
||
invalid_third: bool = False,
|
||
irreducibly_unstable: bool = False,
|
||
) -> tuple[WriterReplayTestAdapters, FakeGovernedChat, FakeSemanticDetector, FakeJudge]:
|
||
"""集中构造测试注入对象,避免它们被误认为生产 runtime。"""
|
||
|
||
runner = FakeGovernedChat(illegal_extra_field=illegal_extra_field)
|
||
detector = FakeSemanticDetector()
|
||
judge = FakeJudge(
|
||
unstable=unstable,
|
||
invalid_third=invalid_third,
|
||
irreducibly_unstable=irreducibly_unstable,
|
||
)
|
||
profile = _frozen_test_profile()
|
||
return (
|
||
WriterReplayTestAdapters(runner, profile, detector, judge),
|
||
runner,
|
||
detector,
|
||
judge,
|
||
)
|
||
|
||
|
||
def _frozen_test_profile():
|
||
"""构造不访问网络且字段完整的冻结 writer 测试 profile。"""
|
||
|
||
return build_writer_execution_profile(
|
||
max_budget_usd_per_call=Decimal("1.000000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=200000,
|
||
system_prompt="只返回严格 WriterDraft v2。",
|
||
)
|
||
|
||
|
||
def _production_writer_profile():
|
||
"""构造生产编排测试使用的 provider-neutral writer profile。"""
|
||
|
||
return build_writer_execution_profile(
|
||
max_budget_usd_per_call=Decimal("1.000000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=200000,
|
||
system_prompt="只返回严格 WriterDraft v2。",
|
||
)
|
||
|
||
|
||
def _test_execution_authorization() -> dict[str, object]:
|
||
"""给测试适配器显式审批 probe、预算、raw 和 writer profile,禁止参数隐式放行。"""
|
||
|
||
writer = _frozen_test_profile()
|
||
probe = {
|
||
"status": "successful",
|
||
"runtimeAdapter": RUNTIME_ADAPTER,
|
||
"runtimeAdapterVersion": RUNTIME_ADAPTER_VERSION,
|
||
"modelPolicyVersion": MODEL_POLICY_VERSION,
|
||
"modelAlias": writer.model_alias,
|
||
"executionProfileSha256": writer.execution_profile_sha256,
|
||
"structuredOutputSha256": "sha256:" + "6" * 64,
|
||
}
|
||
budget = {
|
||
"status": "approved",
|
||
"authorizationId": "budget-test-adapter-1",
|
||
"approvedBy": "user",
|
||
"totalBudgetUsd": "9.000000",
|
||
"plannedCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
"maxCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
}
|
||
raw = {
|
||
"status": "approved",
|
||
"authorizationId": "raw-test-adapter-1",
|
||
"approvedBy": "user",
|
||
"retainUntil": (datetime.now(UTC) + timedelta(hours=1)).isoformat(),
|
||
}
|
||
for item in (probe, budget, raw):
|
||
item["receiptSha256"] = canonical_sha256(item)
|
||
return {
|
||
"runtimeProbe": probe,
|
||
"budget": budget,
|
||
"rawRetention": raw,
|
||
"profileSha256": {"writer": writer.execution_profile_sha256},
|
||
}
|
||
|
||
|
||
def _role_profile(role: str, schema: dict[str, object]) -> RoleExecutionProfile:
|
||
"""构造语义或盲评使用的完整冻结测试 profile。"""
|
||
|
||
prompt = f"只返回严格 {role} structured output。"
|
||
return RoleExecutionProfile(
|
||
profile_version="writer-eval-profile-v2",
|
||
adapter_role=role,
|
||
model_alias=FIXED_OPUS_POLICY_ALIAS,
|
||
model_policy_version=FIXED_OPUS_POLICY_VERSION,
|
||
resolved_model_id=FIXED_OPUS_MODEL_ID,
|
||
max_budget_usd_per_call=Decimal("1.000000"),
|
||
timeout_seconds=30,
|
||
max_context_chars=200000,
|
||
json_schema_id=f"{role}-schema-v1",
|
||
json_schema=schema,
|
||
json_schema_sha256=sha256_json(schema),
|
||
system_prompt_id=f"{role}-prompt-v1",
|
||
system_prompt=prompt,
|
||
system_prompt_sha256=sha256_text(prompt),
|
||
)
|
||
|
||
|
||
def _expected_total_cost(*, fixed_role_calls: int, writer_calls: int = 3) -> str:
|
||
"""按当前治理费率计算测试总成本,避免把 provider 费率硬编码进编排断言。"""
|
||
|
||
writer_cost = Decimal(
|
||
str(
|
||
replay_module.muse_llm.cost_fixed_opus(
|
||
{"input_tokens": 10, "output_tokens": 10},
|
||
)
|
||
)
|
||
)
|
||
total = Decimal("0.010000") * fixed_role_calls + writer_cost * writer_calls
|
||
return format(total.quantize(Decimal("0.000001")), "f")
|
||
|
||
|
||
def _fake_model_receipt(
|
||
role: str,
|
||
invocation: int,
|
||
model_input: dict[str, object],
|
||
structured_output: dict[str, object],
|
||
) -> dict[str, object]:
|
||
"""构造字段完整、模型匹配且输入/输出哈希均冻结的 fake ExecutionReceipt。"""
|
||
|
||
model_id = FIXED_OPUS_MODEL_ID
|
||
return {
|
||
"adapterRole": role,
|
||
"invocationId": f"{role}-{invocation}",
|
||
"executionProfileSha256": canonical_sha256({"role": role, "profile": 1}),
|
||
"requestedModelId": FIXED_OPUS_POLICY_ALIAS,
|
||
"actualModelId": model_id,
|
||
"modelMatch": True,
|
||
"effort": "governed",
|
||
"maxBudgetUsdPerCall": "1.000000",
|
||
"totalCostUsd": "0.010000",
|
||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||
"modelUsage": {model_id: {"costUSD": "0.010000"}},
|
||
"stopReason": "stop",
|
||
"terminalReason": "completed",
|
||
"isError": False,
|
||
"apiErrorStatus": None,
|
||
"exitCode": None,
|
||
"durationMs": 1,
|
||
"inputSha256": canonical_sha256(model_input),
|
||
"structuredOutputSha256": canonical_sha256(structured_output),
|
||
"jsonSchemaSha256": canonical_sha256({"role": role, "schema": 1}),
|
||
}
|
||
|
||
|
||
class ProductionSemanticRunner:
|
||
"""动态回放 semantic v3 的最小模型草稿。"""
|
||
|
||
def __init__(self, *, fail_on_call: int | None = None) -> None:
|
||
self.fail_on_call = fail_on_call
|
||
self.calls: list[dict[str, object]] = []
|
||
self.receipts: list[dict[str, object]] = []
|
||
self.structured_outputs: list[dict[str, object]] = []
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
"""生成合法报告;指定调用可返回真实 failed 语义终态。"""
|
||
|
||
self.calls.append(copy.deepcopy(dict(model_input)))
|
||
failed = self.fail_on_call == len(self.calls)
|
||
body = model_input["candidateBody"]
|
||
constraints = model_input["hardConstraints"]
|
||
quote = body[:3]
|
||
evidence_id = constraints[0]["constraintId"] if constraints else "candidate-body"
|
||
draft = {
|
||
"schemaVersion": "semantic-detection-draft-v3",
|
||
"claims": [],
|
||
"findings": (
|
||
[
|
||
{
|
||
"findingId": "finding-high-1",
|
||
"severity": "high",
|
||
"category": "hard_constraint",
|
||
"candidateQuote": quote,
|
||
"evidenceIds": [evidence_id],
|
||
"message": "硬约束未满足",
|
||
}
|
||
]
|
||
if failed
|
||
else []
|
||
),
|
||
"assertionVerdicts": [
|
||
{
|
||
"assertionId": item["assertionId"],
|
||
"verdict": "pass",
|
||
"candidateQuote": quote,
|
||
"evidenceIds": [item["evidenceId"]],
|
||
}
|
||
for item in model_input["factEvidence"]
|
||
],
|
||
"hardConstraintVerdicts": [
|
||
{
|
||
"constraintId": item["constraintId"],
|
||
"verdict": "fail" if failed else "pass",
|
||
"candidateQuote": quote,
|
||
"evidenceIds": [item["constraintId"]],
|
||
}
|
||
for item in constraints
|
||
],
|
||
"newSettingCandidates": [],
|
||
"evidenceGaps": [],
|
||
}
|
||
receipt = _fake_model_receipt(
|
||
adapter_role,
|
||
len(self.calls),
|
||
dict(model_input),
|
||
draft,
|
||
)
|
||
self.receipts.append(receipt)
|
||
self.structured_outputs.append(copy.deepcopy(draft))
|
||
receipt_hash = sha256_json(receipt)
|
||
if not is_semantic_schema_specialization(
|
||
SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
output_schema,
|
||
):
|
||
raise AssertionError("semantic adapter 未传正式 v3 闭集特化 schema")
|
||
return {"structuredOutput": draft, "modelReceiptSha256": receipt_hash}
|
||
|
||
|
||
class NeedsEvidenceSemanticRunner(ProductionSemanticRunner):
|
||
"""在指定调用生成合同合法但需要补证的 detector 草稿。"""
|
||
|
||
def __init__(self, *, needs_evidence_on_call: int | None = None) -> None:
|
||
super().__init__()
|
||
self.needs_evidence_on_call = needs_evidence_on_call
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
result = super().run(
|
||
adapter_role=adapter_role,
|
||
model_input=model_input,
|
||
output_schema=output_schema,
|
||
)
|
||
if (
|
||
self.needs_evidence_on_call is not None
|
||
and len(self.calls) != self.needs_evidence_on_call
|
||
):
|
||
return result
|
||
draft = result["structuredOutput"]
|
||
draft["evidenceGaps"] = [
|
||
{
|
||
"gapId": "gap-1",
|
||
"query": "补查候选新事实",
|
||
"reason": "冻结证据不足",
|
||
"priority": "high",
|
||
"candidateQuote": model_input["candidateBody"][:3],
|
||
}
|
||
]
|
||
self.structured_outputs[-1] = copy.deepcopy(draft)
|
||
self.receipts[-1]["structuredOutputSha256"] = canonical_sha256(draft)
|
||
result["modelReceiptSha256"] = sha256_json(self.receipts[-1])
|
||
return result
|
||
|
||
|
||
class InvalidQuoteSemanticRunner(ProductionSemanticRunner):
|
||
"""每轮都返回候选中不存在的引文,稳定耗尽 detector 纠错。"""
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
result = super().run(
|
||
adapter_role=adapter_role,
|
||
model_input=model_input,
|
||
output_schema=output_schema,
|
||
)
|
||
draft = result["structuredOutput"]
|
||
for field in ("assertionVerdicts", "hardConstraintVerdicts"):
|
||
for item in draft[field]:
|
||
item["candidateQuote"] = "候选中不存在的引文"
|
||
self.structured_outputs[-1] = copy.deepcopy(draft)
|
||
self.receipts[-1]["structuredOutputSha256"] = canonical_sha256(draft)
|
||
result["modelReceiptSha256"] = sha256_json(self.receipts[-1])
|
||
return result
|
||
|
||
|
||
class CorrectingSemanticRunner(ProductionSemanticRunner):
|
||
"""首轮引文非法、带 correction 的第二轮恢复合法。"""
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
result = super().run(
|
||
adapter_role=adapter_role,
|
||
model_input=model_input,
|
||
output_schema=output_schema,
|
||
)
|
||
if len(self.calls) != 1:
|
||
return result
|
||
draft = result["structuredOutput"]
|
||
for field in ("assertionVerdicts", "hardConstraintVerdicts"):
|
||
for item in draft[field]:
|
||
item["candidateQuote"] = "候选中不存在的首轮引文"
|
||
self.structured_outputs[-1] = copy.deepcopy(draft)
|
||
self.receipts[-1]["structuredOutputSha256"] = canonical_sha256(draft)
|
||
result["modelReceiptSha256"] = sha256_json(self.receipts[-1])
|
||
return result
|
||
|
||
|
||
class ProductionJudgeRunner:
|
||
"""动态回放 blind judge v3 模型草稿,并可制造第三评不稳定。"""
|
||
|
||
def __init__(
|
||
self,
|
||
*,
|
||
unstable: bool = False,
|
||
invalid_first_quote: bool = False,
|
||
api_error_first: bool = False,
|
||
) -> None:
|
||
self.unstable = unstable
|
||
self.invalid_first_quote = invalid_first_quote
|
||
self.api_error_first = api_error_first
|
||
self.calls: list[dict[str, object]] = []
|
||
self.receipts: list[dict[str, object]] = []
|
||
self.structured_outputs: list[dict[str, object]] = []
|
||
|
||
def run(self, *, adapter_role, model_input, output_schema):
|
||
"""按 reviewer 次序生成严格绑定的五维评分与 verdict。"""
|
||
|
||
self.calls.append(copy.deepcopy(dict(model_input)))
|
||
call_index = len(self.calls)
|
||
score = 8.0
|
||
if self.unstable:
|
||
score = {1: 8.0, 2: 6.0, 3: 7.0}.get(call_index, 7.0)
|
||
assertion_ids = [
|
||
item["assertionId"]
|
||
for field in ("historicalAssertions", "targetAssertions")
|
||
for item in model_input["oracleTruthPack"][field]
|
||
]
|
||
constraint_ids = [
|
||
item["constraintId"] for item in model_input["fineOutline"]["hardConstraints"]
|
||
]
|
||
candidate_ids = [item["blindCandidateId"] for item in model_input["candidates"]]
|
||
draft = {
|
||
"schemaVersion": "blind-judge-draft-v3",
|
||
"candidateScores": [
|
||
{
|
||
"blindCandidateId": candidate["blindCandidateId"],
|
||
"scores": {
|
||
dimension: {
|
||
"score": score,
|
||
"reason": f"{dimension} 维度满足要求",
|
||
"candidateQuote": candidate["candidateBody"][:3],
|
||
"evidenceRefs": [{"sourceType": "candidate", "sourceId": candidate["blindCandidateId"]}],
|
||
}
|
||
for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
for candidate in model_input["candidates"]
|
||
],
|
||
"dimensionPreferences": [
|
||
{
|
||
"dimension": dimension,
|
||
"orderedCandidateIds": list(candidate_ids),
|
||
"reason": f"按 {dimension} 维度比较",
|
||
}
|
||
for dimension in DIMENSIONS
|
||
],
|
||
"oracleAssertionVerdicts": [
|
||
{
|
||
"assertionId": assertion_id,
|
||
"blindCandidateId": candidate["blindCandidateId"],
|
||
"verdict": "pass",
|
||
"reason": "候选与 oracle 断言一致",
|
||
"candidateQuote": candidate["candidateBody"][:3],
|
||
"evidenceRefs": [{"sourceType": "oracle_assertion", "sourceId": assertion_id}],
|
||
}
|
||
for candidate in model_input["candidates"]
|
||
for assertion_id in assertion_ids
|
||
],
|
||
"hardConstraintVerdicts": [
|
||
{
|
||
"constraintId": constraint_id,
|
||
"blindCandidateId": candidate["blindCandidateId"],
|
||
"verdict": "pass",
|
||
"reason": "候选满足细纲硬约束",
|
||
"candidateQuote": candidate["candidateBody"][:3],
|
||
"evidenceRefs": [{"sourceType": "fine_outline", "sourceId": constraint_id}],
|
||
}
|
||
for candidate in model_input["candidates"]
|
||
for constraint_id in constraint_ids
|
||
],
|
||
}
|
||
if self.invalid_first_quote and call_index == 1:
|
||
draft["candidateScores"][0]["scores"][DIMENSIONS[0]][
|
||
"candidateQuote"
|
||
] = "候选正文中不存在的首轮引文"
|
||
receipt = _fake_model_receipt(
|
||
adapter_role,
|
||
call_index,
|
||
dict(model_input),
|
||
draft,
|
||
)
|
||
if self.api_error_first and call_index == 1:
|
||
receipt.update(
|
||
{
|
||
"actualModelId": None,
|
||
"modelMatch": False,
|
||
"totalCostUsd": "0.000000",
|
||
"terminalReason": "api_error",
|
||
"isError": True,
|
||
"apiErrorStatus": None,
|
||
"exitCode": 1,
|
||
"structuredOutputSha256": None,
|
||
}
|
||
)
|
||
self.receipts.append(receipt)
|
||
raise RoleRuntimeError(
|
||
"BLIND_JUDGE_API_ERROR",
|
||
"测试瞬时 API 错误",
|
||
receipt=receipt,
|
||
)
|
||
self.receipts.append(receipt)
|
||
self.structured_outputs.append(copy.deepcopy(draft))
|
||
receipt_hash = sha256_json(receipt)
|
||
if output_schema != BLIND_JUDGE_REPORT_JSON_SCHEMA:
|
||
raise AssertionError("judge adapter 未传正式 v3 model schema")
|
||
return {"structuredOutput": draft, "modelReceiptSha256": receipt_hash}
|
||
|
||
|
||
def _production_config(*, budget_approved: bool = True) -> dict[str, object]:
|
||
"""构造 probe、预算、raw 和三 profile 哈希均可复核的执行配置。"""
|
||
|
||
value = config()
|
||
context_input = value["samples"][0]["writerContextInput"]
|
||
context_input["contentMode"] = "canonical_frozen_prose"
|
||
prose_a = {**_prose(120, 2120, "甲侧旧徽章发出微光。"), "retrievalArm": "A"}
|
||
prose_c = {**_prose(121, 2121, "丙侧旧徽章传来回响。"), "retrievalArm": "C"}
|
||
context_input["retrievalResult"]["proseEvidence"] = [prose_a, prose_c]
|
||
value["samples"][0]["proseCharBudget"] = len(prose_a["text"])
|
||
value["commonControls"]["modelVersion"] = MODEL_POLICY_VERSION
|
||
value["commonControls"]["adapterVersion"] = RUNTIME_ADAPTER_VERSION
|
||
writer = _production_writer_profile()
|
||
semantic = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
judge = _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA)
|
||
value["commonControls"]["maxContextChars"] = writer.max_context_chars
|
||
probe = {
|
||
"status": "successful",
|
||
"runtimeAdapter": RUNTIME_ADAPTER,
|
||
"runtimeAdapterVersion": RUNTIME_ADAPTER_VERSION,
|
||
"modelPolicyVersion": MODEL_POLICY_VERSION,
|
||
"modelAlias": writer.model_alias,
|
||
# 探针合同绑定:身份哈希绑定当前 writer 合同,证明它是按新 schema/prompt 重跑的;
|
||
# structuredOutputSha256 是合规结构化输出内容哈希,证明探针确实跑出过合规输出。
|
||
"executionProfileSha256": writer.execution_profile_sha256,
|
||
"structuredOutputSha256": "sha256:" + "6" * 64,
|
||
}
|
||
budget = {
|
||
"status": "approved" if budget_approved else "pending",
|
||
"authorizationId": "budget-test-1",
|
||
"approvedBy": "user",
|
||
"totalBudgetUsd": "9.000000",
|
||
"plannedCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
"maxCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
|
||
}
|
||
raw = {
|
||
"status": "approved",
|
||
"authorizationId": "raw-test-1",
|
||
"approvedBy": "user",
|
||
"retainUntil": (datetime.now(UTC) + timedelta(hours=1)).isoformat(),
|
||
}
|
||
for item in (probe, budget, raw):
|
||
item["receiptSha256"] = canonical_sha256(item)
|
||
value["executionAuthorization"] = {
|
||
"runtimeProbe": probe,
|
||
"budget": budget,
|
||
"rawRetention": raw,
|
||
"profileSha256": {
|
||
"writer": writer.execution_profile_sha256,
|
||
"semantic_detector": semantic.execution_profile_sha256,
|
||
"blind_judge": judge.execution_profile_sha256,
|
||
},
|
||
}
|
||
return value
|
||
|
||
|
||
def _production_config_with_writer_budget(
|
||
*, planned: int, max_calls: int, total_usd: str
|
||
) -> dict[str, object]:
|
||
"""在 _production_config 基础上单独抬高 writer 预算,供篇幅修订环的多调用测试使用。
|
||
|
||
修订环会让每臂的 writer 调用数超过基础的 1 次,因此需要更大的 plannedCalls/maxCalls;
|
||
totalBudgetUsd 必须覆盖 caps*planned 的最坏预留,否则预算门会失败关闭。改动 budget
|
||
字段后必须按授权门规则重签 receiptSha256。
|
||
"""
|
||
|
||
value = _production_config()
|
||
budget = value["executionAuthorization"]["budget"]
|
||
budget["plannedCalls"]["writer"] = planned
|
||
budget["maxCalls"]["writer"] = max_calls
|
||
budget["totalBudgetUsd"] = total_usd
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: val for key, val in budget.items() if key != "receiptSha256"}
|
||
)
|
||
return value
|
||
|
||
|
||
def _semantic_call_index_for_arm(evaluation_config: dict[str, object], arm: str) -> int:
|
||
"""返回单样本预注册顺序中指定臂对应的 detector 调用序号。"""
|
||
|
||
preregistration = build_balanced_preregistration(
|
||
evaluation_set_version=str(evaluation_config["evaluationSetVersion"]),
|
||
sample_ids=[str(evaluation_config["samples"][0]["sampleId"])],
|
||
)
|
||
order = preregistration["armOrderTable"][0]["armOrder"]
|
||
return order.index(arm) + 1
|
||
|
||
|
||
def _five_scenario_gate_input(gate_input: dict[str, object]) -> dict[str, object]:
|
||
"""把单样本生产链结果扩成五场景,只用于机械验证 Gate A 的 C 臂裁决。"""
|
||
|
||
scenarios = (
|
||
"battle",
|
||
"character_dialogue",
|
||
"turning_point",
|
||
"information_reveal",
|
||
"returning_character",
|
||
)
|
||
expanded = copy.deepcopy(gate_input)
|
||
template = expanded["samples"][0]
|
||
expanded["samples"] = []
|
||
for index, scenario in enumerate(scenarios, start=1):
|
||
sample = copy.deepcopy(template)
|
||
sample["sampleId"] = f"gate-a-c-failed-{index}"
|
||
sample["scenario"] = scenario
|
||
expanded["samples"].append(sample)
|
||
expanded.pop("builderReceiptSha256", None)
|
||
expanded.pop("gateInputSha256", None)
|
||
expanded["builderReceiptSha256"] = canonical_sha256(expanded)
|
||
expanded["gateInputSha256"] = canonical_sha256(expanded)
|
||
return expanded
|
||
|
||
|
||
def _oracle_pack() -> dict[str, object]:
|
||
"""构造正式 loader 形状的 oracle;执行时再投影成 reviewer 最小合同。"""
|
||
|
||
target_statement = "目标章不得提前泄露终局真相"
|
||
historical = []
|
||
for chapter in range(485, 489):
|
||
statement = f"林澈在第{chapter}章仍位于圣蒂曼"
|
||
historical.append(
|
||
{
|
||
"assertionId": f"assertion-history-{chapter}",
|
||
"assertionType": "canonical_history",
|
||
"statement": statement,
|
||
"sourceVersion": f"canonical-v{chapter}",
|
||
"chapterBoundary": {"minChapter": chapter, "maxChapter": chapter},
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
|
||
}
|
||
)
|
||
payload = {
|
||
"schemaVersion": "oracle-truth-pack-v1",
|
||
"evaluationSetVersion": "writer-gate-a-test-v1",
|
||
"sampleId": "deep-space-489",
|
||
"workId": 8,
|
||
"asOf": 488,
|
||
"sourceSnapshotSha256": "sha256:" + "7" * 64,
|
||
"authorizationSnapshotId": "auth-work-8",
|
||
"authorization": {
|
||
"allowedPurpose": "offline_evaluation",
|
||
"sourceStatus": "authorized",
|
||
"sourceVersion": "canonical-v488",
|
||
"revalidationAt": "2026-07-25T00:00:00+00:00",
|
||
"evaluatorOnly": True,
|
||
"snapshotId": "auth-work-8",
|
||
},
|
||
"historicalAssertions": historical,
|
||
"targetAssertions": [
|
||
{
|
||
"assertionId": "assertion-target-1",
|
||
"assertionType": "target_reference_scaffold",
|
||
"statement": target_statement,
|
||
"sourceVersion": "canonical-v488",
|
||
"chapterBoundary": {"minChapter": 489, "maxChapter": 489},
|
||
"contentSha256": "sha256:" + hashlib.sha256(target_statement.encode("utf-8")).hexdigest(),
|
||
}
|
||
],
|
||
}
|
||
return {**payload, "packSha256": canonical_sha256(payload)}
|
||
|
||
|
||
def _production_adapters(
|
||
*,
|
||
semantic_fail_on: int | None = None,
|
||
judge_unstable: bool = False,
|
||
judge_invalid_first_quote: bool = False,
|
||
judge_api_error_first: bool = False,
|
||
vault_factory=RawVaultManager,
|
||
cas_factory=FileCasStore,
|
||
builder_factory=GateInputBuilder,
|
||
):
|
||
"""集中构造 production fake,所有模型结果都从 structuredOutput 返回。"""
|
||
|
||
writer_runner = FakeGovernedChat()
|
||
semantic_runner = ProductionSemanticRunner(fail_on_call=semantic_fail_on)
|
||
judge_runner = ProductionJudgeRunner(
|
||
unstable=judge_unstable,
|
||
invalid_first_quote=judge_invalid_first_quote,
|
||
api_error_first=judge_api_error_first,
|
||
)
|
||
writer_profile = _production_writer_profile()
|
||
semantic_profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
judge_profile = _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA)
|
||
adapters = WriterReplayProductionAdapters(
|
||
writer_runner=writer_runner,
|
||
writer_profile=writer_profile,
|
||
semantic_model_runner=semantic_runner,
|
||
semantic_profile=semantic_profile,
|
||
judge_model_runner=judge_runner,
|
||
judge_profile=judge_profile,
|
||
oracle_truth_packs={"deep-space-489": _oracle_pack()},
|
||
vault_manager_factory=vault_factory,
|
||
cas_store_factory=cas_factory,
|
||
gate_input_builder_factory=builder_factory,
|
||
)
|
||
return adapters, writer_runner, semantic_runner, judge_runner
|
||
|
||
|
||
class WriterReplayDryRunTest(unittest.TestCase):
|
||
"""验证 dry-run 的合同、冻结、安全和控制变量。"""
|
||
|
||
def test_pattern_references_are_included_in_leakage_audit(self):
|
||
"""范式卡若含目标章禁用事实,dry-run 必须在装配各臂前失败关闭。"""
|
||
|
||
evaluation_config = config()
|
||
evaluation_config["samples"][0]["writerContextInput"]["patternReferences"] = [
|
||
{
|
||
"sourceId": "fixture:public-pattern:1",
|
||
"sourceVersion": "public-pattern-v1",
|
||
"name": "转折范式",
|
||
"summary": "目标章专属秘密",
|
||
"writingPoints": {"转折": "先压后扬"},
|
||
}
|
||
]
|
||
|
||
result = run_writer_replay(evaluation_config, run_id="pattern-leakage-dry-run")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_leakage")
|
||
self.assertEqual(result["samples"][0]["status"], "invalid_snapshot")
|
||
findings = result["samples"][0]["leakageAudit"]["findings"]
|
||
self.assertTrue(
|
||
any("patternReferences" in finding["path"] for finding in findings),
|
||
findings,
|
||
)
|
||
|
||
@staticmethod
|
||
def _ledger(
|
||
*,
|
||
planned: int = 1,
|
||
total: str = "3.000000",
|
||
pre_call_guard=None,
|
||
):
|
||
"""构造只用于账本边界测试的三角色 profile。"""
|
||
|
||
profiles = {
|
||
"writer": _role_profile("writer", replay_module.WRITER_OUTPUT_JSON_SCHEMA),
|
||
"semantic_detector": _role_profile(
|
||
"semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA
|
||
),
|
||
"blind_judge": _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA),
|
||
}
|
||
return replay_module._ExecutionBudgetLedger(
|
||
total_budget=Decimal(total),
|
||
planned_calls={role: planned for role in replay_module.BUDGET_ROLES},
|
||
max_calls={role: planned for role in replay_module.BUDGET_ROLES},
|
||
profiles=profiles,
|
||
pre_call_guard=pre_call_guard,
|
||
)
|
||
|
||
def test_raw_guard_runs_before_budget_slot_is_consumed(self):
|
||
"""下一调用租期不足时,不得消耗计划槽位或制造在途成本。"""
|
||
|
||
checked_roles = []
|
||
|
||
def reject(role):
|
||
checked_roles.append(role)
|
||
raise RawVaultError(
|
||
"RAW_LEASE_INSUFFICIENT_RETENTION",
|
||
"测试下一调用租期不足",
|
||
)
|
||
|
||
ledger = self._ledger(pre_call_guard=reject)
|
||
|
||
with self.assertRaisesRegex(RawVaultError, "测试下一调用租期不足"):
|
||
ledger.begin("writer")
|
||
|
||
self.assertEqual(checked_roles, ["writer"])
|
||
self.assertEqual(
|
||
ledger.snapshot()["usedCalls"],
|
||
{"writer": 0, "semantic_detector": 0, "blind_judge": 0},
|
||
)
|
||
self.assertEqual(ledger.snapshot()["inFlightRoles"], [])
|
||
self.assertFalse(ledger.snapshot()["costUnknown"])
|
||
|
||
def test_raw_call_window_uses_one_role_timeout_plus_cleanup_margin(self):
|
||
"""固定时钟证明单次 30 秒 timeout 加 60 秒清理余量的精确边界。"""
|
||
|
||
now = datetime(2026, 7, 27, 0, 0, tzinfo=UTC)
|
||
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
|
||
replay_module._assert_raw_call_window(
|
||
(now + timedelta(seconds=90)).isoformat(),
|
||
profile,
|
||
now=now,
|
||
)
|
||
with self.assertRaises(RawVaultError) as raised:
|
||
replay_module._assert_raw_call_window(
|
||
(now + timedelta(seconds=89)).isoformat(),
|
||
profile,
|
||
now=now,
|
||
)
|
||
self.assertEqual(raised.exception.code, "RAW_LEASE_INSUFFICIENT_RETENTION")
|
||
|
||
def test_budget_ledger_settles_failed_call_receipt_before_reraising(self):
|
||
"""模型失败但带可信回执时,实际成本仍必须进入账本。"""
|
||
|
||
ledger = self._ledger()
|
||
receipt = _fake_model_receipt(
|
||
"semantic_detector", 1, {"input": "x"}, {"output": "y"}
|
||
)
|
||
|
||
class FailedDelegate:
|
||
receipts = []
|
||
structured_outputs = []
|
||
|
||
def run(self, **_kwargs):
|
||
error = RuntimeError("模型失败")
|
||
error.receipt = copy.deepcopy(receipt)
|
||
raise error
|
||
|
||
runner = replay_module._BudgetedModelRunner(
|
||
FailedDelegate(), ledger, "semantic_detector"
|
||
)
|
||
with self.assertRaisesRegex(RuntimeError, "模型失败"):
|
||
runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
snapshot = ledger.snapshot()
|
||
self.assertEqual(snapshot["usedCalls"]["semantic_detector"], 1)
|
||
self.assertEqual(snapshot["totalActualCostUsd"], "0.010000")
|
||
self.assertFalse(snapshot["costUnknown"])
|
||
|
||
def test_budget_ledger_without_failure_receipt_is_cost_unknown_and_stops_followups(self):
|
||
"""已发起调用但无可信成本时不能按零美元继续后续角色。"""
|
||
|
||
ledger = self._ledger()
|
||
|
||
class FailedDelegate:
|
||
receipts = []
|
||
structured_outputs = []
|
||
|
||
def run(self, **_kwargs):
|
||
raise RuntimeError("无回执")
|
||
|
||
runner = replay_module._BudgetedModelRunner(
|
||
FailedDelegate(), ledger, "semantic_detector"
|
||
)
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
snapshot = ledger.snapshot()
|
||
self.assertTrue(snapshot["costUnknown"])
|
||
self.assertEqual(snapshot["totalActualCostUsd"], "0.000000")
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
ledger.begin("writer")
|
||
|
||
def test_budget_ledger_rejects_duplicate_or_out_of_order_settlement(self):
|
||
"""一个调用只能由自己的 token 结算一次,不能重复计费。"""
|
||
|
||
ledger = self._ledger()
|
||
token = ledger.begin("writer")
|
||
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
|
||
ledger.complete("writer", token, receipt)
|
||
with self.assertRaisesRegex(BudgetLedgerError, "token 错序或重复"):
|
||
ledger.complete("writer", token, receipt)
|
||
self.assertEqual(ledger.snapshot()["totalActualCostUsd"], "0.010000")
|
||
|
||
def test_budget_settlement_keeps_in_flight_until_cost_is_recorded(self):
|
||
"""结算中断时在途标记必须仍在,供信号处理转成成本未知。"""
|
||
|
||
class InterruptingSet(set):
|
||
def add(self, _value):
|
||
raise replay_module.ReplayInterrupted("测试结算窗口")
|
||
|
||
ledger = self._ledger()
|
||
token = ledger.begin("writer")
|
||
ledger._settled_invocations = InterruptingSet()
|
||
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
|
||
|
||
with self.assertRaises(replay_module.ReplayInterrupted):
|
||
ledger.complete("writer", token, receipt)
|
||
|
||
self.assertEqual(ledger.snapshot()["inFlightRoles"], ["writer"])
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
ledger.mark_cost_unknown("writer", token)
|
||
self.assertTrue(ledger.snapshot()["costUnknown"])
|
||
|
||
def test_budget_ledger_rejects_missing_cost_and_over_cap(self):
|
||
"""缺成本和超过单次 cap 都必须 fail closed,不能静默按零计费。"""
|
||
|
||
missing = self._ledger()
|
||
token = missing.begin("writer")
|
||
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
|
||
missing.complete(
|
||
"writer",
|
||
token,
|
||
{"adapterRole": "writer", "invocationId": "writer-missing", "totalCostUsd": None},
|
||
)
|
||
self.assertTrue(missing.snapshot()["costUnknown"])
|
||
|
||
over_cap = self._ledger()
|
||
token = over_cap.begin("writer")
|
||
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
|
||
receipt["totalCostUsd"] = "2.000000"
|
||
with self.assertRaisesRegex(BudgetLedgerError, "超过单次 cap"):
|
||
over_cap.complete("writer", token, receipt)
|
||
self.assertEqual(over_cap.snapshot()["failureReason"], "EXECUTION_COST_OVER_CAP")
|
||
|
||
def test_budget_plan_requires_closed_planned_calls_and_reserves_570(self):
|
||
"""45/24/45 计划乘 $5 预留 $570;maxCalls 仍只是安全容量上限。"""
|
||
|
||
required = {role: 15 for role in replay_module.BUDGET_ROLES}
|
||
caps = {role: Decimal("5.000000") for role in replay_module.BUDGET_ROLES}
|
||
valid = {
|
||
"status": "approved",
|
||
"totalBudgetUsd": "2250.000000",
|
||
"plannedCalls": {"writer": 60, "semantic_detector": 24, "blind_judge": 45},
|
||
"maxCalls": {role: 150 for role in replay_module.BUDGET_ROLES},
|
||
}
|
||
planned, maximum, total, error = replay_module._validate_budget_plan(
|
||
valid, required_calls=required, caps=caps
|
||
)
|
||
self.assertIsNone(error)
|
||
self.assertEqual(planned, valid["plannedCalls"])
|
||
self.assertEqual(maximum, valid["maxCalls"])
|
||
self.assertEqual(total, Decimal("2250.000000"))
|
||
self.assertEqual(
|
||
sum(caps[role] * planned[role] for role in replay_module.BUDGET_ROLES),
|
||
Decimal("645.000000"),
|
||
)
|
||
|
||
cases = (
|
||
("missing", lambda budget: budget.pop("plannedCalls"), "BUDGET_PLANNED_CALLS_REQUIRED"),
|
||
("invalid-shape", lambda budget: budget["plannedCalls"].update({"extra": 1}), "BUDGET_PLANNED_CALLS_INVALID"),
|
||
("invalid-type", lambda budget: budget["plannedCalls"].update({"writer": "15"}), "BUDGET_PLANNED_CALLS_INSUFFICIENT"),
|
||
("insufficient", lambda budget: budget["plannedCalls"].update({"writer": 14}), "BUDGET_PLANNED_CALLS_INSUFFICIENT"),
|
||
("planned-over-max", lambda budget: budget["plannedCalls"].update({"writer": 151}), "BUDGET_MAX_CALLS_INSUFFICIENT"),
|
||
("insufficient-total", lambda budget: budget.update({"totalBudgetUsd": "644.999999"}), "BUDGET_AUTHORIZATION_INVALID"),
|
||
)
|
||
for name, mutate, expected in cases:
|
||
with self.subTest(name=name):
|
||
tampered = copy.deepcopy(valid)
|
||
mutate(tampered)
|
||
_planned, _maximum, _total, error = replay_module._validate_budget_plan(
|
||
tampered, required_calls=required, caps=caps
|
||
)
|
||
self.assertEqual(error, expected)
|
||
|
||
def test_length_bounds_relaxed_to_30_percent(self):
|
||
"""五个预注册目标的篇幅边界都按正负 30% 计算并受 2000-10000 限幅。"""
|
||
|
||
expected = {
|
||
7500: (5250, 9750),
|
||
7600: (5320, 9880),
|
||
6700: (4690, 8710),
|
||
2000: (2000, 2600),
|
||
6100: (4270, 7930),
|
||
}
|
||
for target, bounds in expected.items():
|
||
with self.subTest(target=target):
|
||
self.assertEqual(replay_module._length_bounds(target), bounds)
|
||
|
||
def test_base_config_budget_self_hash_and_length_bounds_consistent(self):
|
||
"""base 配置 budget 自哈希重签有效,五样本 expectedLength/输出合同等于 _length_bounds(target)。"""
|
||
|
||
gate_config = _load_gate_a_config()
|
||
budget = gate_config["executionAuthorization"]["budget"]
|
||
# 与 _validate_authorization_records 同款复核:去掉 receiptSha256 后整段重算自哈希。
|
||
self.assertEqual(
|
||
budget["receiptSha256"],
|
||
canonical_sha256({key: value for key, value in budget.items() if key != "receiptSha256"}),
|
||
)
|
||
self.assertEqual(budget["plannedCalls"]["writer"], 60)
|
||
self.assertEqual(budget["totalBudgetUsd"], "2250.000000")
|
||
for sample in gate_config["samples"]:
|
||
target = sample["targetChars"]
|
||
expected_min, expected_max = replay_module._length_bounds(target)
|
||
with self.subTest(sample=sample["sampleId"]):
|
||
self.assertEqual(
|
||
sample["expectedLength"],
|
||
{"targetChars": target, "minChars": expected_min, "maxChars": expected_max},
|
||
)
|
||
contract = sample["writerContextInput"]["outputContract"]
|
||
self.assertEqual(contract["targetChars"], target)
|
||
self.assertEqual(contract["minChars"], expected_min)
|
||
self.assertEqual(contract["maxChars"], expected_max)
|
||
|
||
def test_length_overflow_records_actual_han_chars_in_sample_result(self):
|
||
"""生产链正文越界时失败样本必须记下实际汉字数与合同边界(全是数字,不含正文)。"""
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters()
|
||
short_body = "短" * 2000 # 2000 汉字 < 合同下限 2800,必然越界
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
with mock.patch.object(
|
||
sys.modules[__name__],
|
||
"_writer_output",
|
||
lambda context, call_index: {"candidateBody": short_body},
|
||
):
|
||
result = run_writer_replay(
|
||
_production_config_with_writer_budget(
|
||
planned=4,
|
||
max_calls=4,
|
||
total_usd="12.000000",
|
||
),
|
||
run_id="length-violation",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
production_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
sample_result = result["samples"][0]
|
||
self.assertIn("lengthViolation", sample_result)
|
||
self.assertEqual(
|
||
sample_result["lengthViolation"],
|
||
{"actualHanChars": 2000, "minChars": 2800, "maxChars": 5200, "targetChars": 4000},
|
||
)
|
||
|
||
def test_recording_runner_keeps_adapter_fields_out_of_model_draft(self):
|
||
"""统一 runtime 只能旁路返回回执 hash,不能污染 v3 模型草稿。"""
|
||
|
||
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
draft = {
|
||
"schemaVersion": "semantic-detection-draft-v3",
|
||
"claims": [],
|
||
"findings": [],
|
||
"assertionVerdicts": [],
|
||
"hardConstraintVerdicts": [],
|
||
"newSettingCandidates": [],
|
||
"evidenceGaps": [],
|
||
}
|
||
receipt = _fake_model_receipt("semantic_detector", 1, {"candidateBody": "甲"}, draft)
|
||
|
||
class Receipt:
|
||
def as_dict(self):
|
||
return copy.deepcopy(receipt)
|
||
|
||
class Invocation:
|
||
structured_output = copy.deepcopy(draft)
|
||
receipt = Receipt()
|
||
|
||
original = replay_module.run_role
|
||
replay_module.run_role = lambda _profile, _input: Invocation()
|
||
try:
|
||
runner = replay_module._RecordingRuntimeModelRunner(profile)
|
||
result = runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={"candidateBody": "甲"},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
finally:
|
||
replay_module.run_role = original
|
||
|
||
self.assertEqual(result["structuredOutput"], draft)
|
||
self.assertNotIn("modelReceiptSha256", result["structuredOutput"])
|
||
self.assertNotIn("reportSha256", result["structuredOutput"])
|
||
self.assertEqual(result["modelReceiptSha256"], sha256_json(receipt))
|
||
|
||
def test_recording_runner_preserves_failure_receipt_before_reraising(self):
|
||
"""失败回执必须留在 runner,外层才能保留真实错误码并结算可信费用。"""
|
||
|
||
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
|
||
receipt = _fake_model_receipt(
|
||
"semantic_detector", 1, {"candidateBody": "甲"}, {"error": "api"}
|
||
)
|
||
|
||
class Receipt:
|
||
def as_dict(self):
|
||
return copy.deepcopy(receipt)
|
||
|
||
original = replay_module.run_role
|
||
|
||
def fail(_profile, _input):
|
||
raise RoleRuntimeError(
|
||
"SEMANTIC_DETECTOR_API_ERROR",
|
||
"模型调用失败",
|
||
receipt=Receipt(),
|
||
)
|
||
|
||
replay_module.run_role = fail
|
||
try:
|
||
runner = replay_module._RecordingRuntimeModelRunner(profile)
|
||
with self.assertRaises(RoleRuntimeError) as caught:
|
||
runner.run(
|
||
adapter_role="semantic_detector",
|
||
model_input={"candidateBody": "甲"},
|
||
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
)
|
||
finally:
|
||
replay_module.run_role = original
|
||
|
||
self.assertEqual(caught.exception.primary_code, "SEMANTIC_DETECTOR_API_ERROR")
|
||
self.assertEqual(runner.receipts, [receipt])
|
||
self.assertEqual(runner.structured_outputs, [])
|
||
|
||
def test_three_arms_validate_full_context_and_only_change_evidence_strategy(self):
|
||
result = run_writer_replay(config(), run_id="dry-1")
|
||
|
||
self.assertEqual(result["status"], "ready")
|
||
sample = result["samples"][0]
|
||
arms = sample["arms"]
|
||
self.assertEqual(set(arms), {"A", "B", "C"})
|
||
self.assertEqual(arms["A"]["evidenceStrategy"], "historical_prose_only")
|
||
self.assertEqual(arms["B"]["evidenceStrategy"], "card_index_only")
|
||
self.assertEqual(arms["C"]["evidenceStrategy"], "card_index_plus_prose")
|
||
self.assertEqual(len({arm["commonControlsSha256"] for arm in arms.values()}), 1)
|
||
self.assertTrue(all(not arm["acceptanceEligible"] for arm in arms.values()))
|
||
self.assertEqual(arms["A"]["contextSummary"]["recentBaselineChapters"], [485, 486, 487, 488])
|
||
self.assertEqual(arms["B"]["contextSummary"]["proseEvidenceCount"], 0)
|
||
self.assertEqual(arms["B"]["contextSummary"]["indexHintCount"], 1)
|
||
self.assertEqual(arms["C"]["contextSummary"]["recentBaselineChapters"], [485, 486, 487, 488])
|
||
self.assertEqual(arms["C"]["contextSummary"]["indexHintCount"], 1)
|
||
self.assertTrue(
|
||
all(
|
||
arm["contextSummary"]["contentMode"] == "sanitized_contract_fixture"
|
||
for arm in arms.values()
|
||
)
|
||
)
|
||
rendered = json.dumps(result, ensure_ascii=False)
|
||
self.assertNotIn("脱敏合同夹具", rendered)
|
||
self.assertNotIn("林澈在冻结点仍位于圣蒂曼", rendered)
|
||
|
||
def test_future_index_hint_invalidates_whole_sample(self):
|
||
leaked = config()
|
||
leaked["samples"][0]["writerContextInput"]["retrievalResult"]["indexHints"][0]["asOf"] = 489
|
||
|
||
result = run_writer_replay(leaked, run_id="dry-leak")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_leakage")
|
||
|
||
def test_invalid_b_arm_contract_cannot_be_reported_ready(self):
|
||
invalid = config()
|
||
invalid["samples"][0]["writerContextInput"]["retrievalResult"]["indexHints"] = []
|
||
|
||
result = run_writer_replay(invalid, run_id="dry-invalid-context")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_writer_context")
|
||
|
||
def test_target_length_is_mechanically_recomputed_and_target_chapter_length_is_forbidden(self):
|
||
invalid = config()
|
||
invalid["samples"][0]["targetChars"] = 4100
|
||
|
||
with self.assertRaisesRegex(WriterReplayError, "机械复算"):
|
||
run_writer_replay(invalid, run_id="dry-length-tampered")
|
||
|
||
leaked = config()
|
||
leaked["samples"][0]["targetLengthBasis"]["usesTargetChapterLength"] = True
|
||
with self.assertRaisesRegex(WriterReplayError, "禁止读取目标章"):
|
||
run_writer_replay(leaked, run_id="dry-length-leaked")
|
||
|
||
def test_unresolved_generic_role_must_remain_null_instead_of_defaulting_to_zero(self):
|
||
unresolved = config()
|
||
sample = unresolved["samples"][0]
|
||
sample["writerContextInput"]["requirements"]["requiredCharacters"] = ["内应"]
|
||
sample["newCharacterRatio"] = None
|
||
sample["newCharacterRatioStatus"] = "unresolved_generic_role"
|
||
sample["newCharacterBasis"] = {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": 488,
|
||
"requiredCharacters": ["内应"],
|
||
"knownBeforeAsOf": [],
|
||
"absentBeforeAsOf": [],
|
||
"genericRoles": ["内应"],
|
||
}
|
||
self.assertTrue(run_writer_replay(unresolved, run_id="dry-generic-role")["ok"])
|
||
|
||
unresolved["samples"][0]["newCharacterRatio"] = 0
|
||
with self.assertRaisesRegex(WriterReplayError, "必须为 null"):
|
||
run_writer_replay(unresolved, run_id="dry-generic-role-zero")
|
||
|
||
def test_character_declared_absent_before_freeze_cannot_have_card_index_hint(self):
|
||
inconsistent = config()
|
||
sample = inconsistent["samples"][0]
|
||
sample["newCharacterRatio"] = 1.0
|
||
sample["newCharacterBasis"]["knownBeforeAsOf"] = []
|
||
sample["newCharacterBasis"]["absentBeforeAsOf"] = ["林澈"]
|
||
|
||
with self.assertRaisesRegex(WriterReplayError, "不得同时出现在卡索引"):
|
||
run_writer_replay(inconsistent, run_id="dry-absent-card-conflict")
|
||
|
||
def test_gate_a_preregistration_mechanically_recomputes_all_lengths_and_ratios(self):
|
||
gate_config = _load_gate_a_config()
|
||
expected_targets = {489: 7500, 321: 7600, 544: 6700, 199: 2000, 523: 6100}
|
||
expected_ratios = {489: 1.0, 321: 0.0, 544: None, 199: 0.0, 523: 0.0}
|
||
self.assertEqual(
|
||
gate_config["commonControls"]["selectorVersion"],
|
||
"writer-gate-a-deep-space-card-selectors-v4",
|
||
)
|
||
self.assertEqual(
|
||
gate_config["commonControls"]["selectorSha256"],
|
||
"sha256:99105930d7e01d32fa9faa1ceaa30663674dd8e15b84dd8ab7f5b18cd4e9d0ef",
|
||
)
|
||
self.assertNotIn("inputProvenance", gate_config["commonControls"])
|
||
self.assertEqual(gate_config["commonControls"]["maxContextChars"], 140000)
|
||
selector_path = (
|
||
SCRIPT_DIR.parent
|
||
/ "configs"
|
||
/ "writer-gate-a-deep-space-card-selectors-v1.json"
|
||
)
|
||
selector_config = json.loads(selector_path.read_text(encoding="utf-8"))
|
||
selector_by_sample = {
|
||
sample["sampleId"]: [
|
||
(card["type"], card["name"]) for card in sample["cards"]
|
||
]
|
||
for sample in selector_config["samples"]
|
||
}
|
||
sample_by_id = {sample["sampleId"]: sample for sample in gate_config["samples"]}
|
||
for sample_id in selector_by_sample:
|
||
hints = sample_by_id[sample_id]["writerContextInput"]["retrievalResult"][
|
||
"indexHints"
|
||
]
|
||
self.assertEqual(
|
||
[(hint["type"], hint["name"]) for hint in hints],
|
||
selector_by_sample[sample_id],
|
||
)
|
||
expected_card_prose_chapters = {
|
||
"deep-space-489-battle": 484,
|
||
"deep-space-321-character-dialogue": 312,
|
||
"deep-space-544-turning-point": 467,
|
||
}
|
||
for sample_id, expected_chapter in expected_card_prose_chapters.items():
|
||
prose = sample_by_id[sample_id]["writerContextInput"]["retrievalResult"][
|
||
"proseEvidence"
|
||
]
|
||
card_prose = [item for item in prose if item["retrievalArm"] == "C"]
|
||
self.assertEqual(len(card_prose), 1)
|
||
self.assertEqual(card_prose[0]["chapter"], expected_chapter)
|
||
self.assertTrue(
|
||
card_prose[0]["sourceRef"]["sourceId"].endswith(
|
||
f":{expected_chapter}"
|
||
)
|
||
)
|
||
|
||
for sample in gate_config["samples"]:
|
||
basis = sample["targetLengthBasis"]
|
||
recalculated = calculate_target_chars(
|
||
recent_chapter_han_counts=sample["frozenRecentHanCounts"],
|
||
hard_event_count=basis["hardEventCount"],
|
||
foreshadowing_action_count=basis["foreshadowingActionCount"],
|
||
required_scene_count=basis["requiredSceneCount"],
|
||
min_chars=basis["minChars"],
|
||
max_chars=basis["maxChars"],
|
||
)
|
||
target_chapter = sample["targetChapter"]
|
||
self.assertEqual(recalculated, expected_targets[target_chapter])
|
||
self.assertEqual(sample["targetChars"], recalculated)
|
||
self.assertEqual(sample["newCharacterRatio"], expected_ratios[target_chapter])
|
||
self.assertEqual(
|
||
sample["writerContextInput"]["tokenBudget"]["maxContextChars"],
|
||
gate_config["commonControls"]["maxContextChars"],
|
||
)
|
||
if target_chapter == 544:
|
||
self.assertEqual(sample["newCharacterRatioStatus"], "unresolved_generic_role")
|
||
else:
|
||
self.assertEqual(sample["newCharacterRatioStatus"], "resolved")
|
||
|
||
result = run_writer_replay(gate_config, run_id="gate-a-preregistered-dry-run")
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "ready")
|
||
self.assertEqual(len(result["samples"]), 5)
|
||
self.assertNotIn("脱敏合成历史片段", json.dumps(result, ensure_ascii=False))
|
||
for sample in result["samples"]:
|
||
receipt = sample["writerContextDiffReceipt"]
|
||
self.assertEqual(receipt["proseCharBudget"], 2000)
|
||
self.assertEqual(receipt["proseCharCount"], {"A": 2000, "C": 2000})
|
||
self.assertTrue(receipt["allowedDifferencePaths"])
|
||
self.assertNotEqual(receipt["contextSha256"]["A"], receipt["contextSha256"]["C"])
|
||
self.assertTrue(
|
||
all(
|
||
path.startswith(("$.factConstraints", "$.proseExcerpts"))
|
||
for path in receipt["allowedDifferencePaths"]
|
||
)
|
||
)
|
||
self.assertEqual(gate_config["commonControls"]["modelVersion"], MODEL_POLICY_VERSION)
|
||
self.assertEqual(gate_config["commonControls"]["adapterVersion"], RUNTIME_ADAPTER_VERSION)
|
||
probe = gate_config["executionAuthorization"]["runtimeProbe"]
|
||
self.assertEqual(probe["schemaVersion"], "runtime-probe-v2")
|
||
self.assertEqual(probe["status"], "dry_run")
|
||
self.assertEqual(probe["runtimeAdapter"], RUNTIME_ADAPTER)
|
||
self.assertEqual(probe["runtimeAdapterVersion"], RUNTIME_ADAPTER_VERSION)
|
||
self.assertEqual(probe["modelPolicyVersion"], MODEL_POLICY_VERSION)
|
||
self.assertNotIn("reason", probe)
|
||
for field in ("inputSha256", "executionReceiptSha256", "structuredOutputSha256"):
|
||
self.assertTrue(probe[field].startswith("sha256:"))
|
||
self.assertEqual(
|
||
gate_config["executionAuthorization"]["budget"]["maxCalls"],
|
||
{"writer": 150, "semantic_detector": 150, "blind_judge": 150},
|
||
)
|
||
self.assertEqual(gate_config["executionAuthorization"]["budget"]["status"], "approved")
|
||
self.assertEqual(
|
||
gate_config["executionAuthorization"]["budget"]["totalBudgetUsd"],
|
||
"2250.000000",
|
||
)
|
||
profile_cap = Decimal("5.000000")
|
||
self.assertEqual(
|
||
gate_config["executionAuthorization"]["budget"]["plannedCalls"],
|
||
{"writer": 60, "semantic_detector": 24, "blind_judge": 45},
|
||
)
|
||
self.assertEqual(
|
||
profile_cap * sum(
|
||
gate_config["executionAuthorization"]["budget"]["plannedCalls"].values()
|
||
),
|
||
Decimal("645.000000"),
|
||
)
|
||
max_calls = gate_config["executionAuthorization"]["budget"]["maxCalls"]
|
||
worst_case_budget = profile_cap * sum(max_calls.values())
|
||
self.assertEqual(worst_case_budget, Decimal("2250.000000"))
|
||
self.assertEqual(
|
||
worst_case_budget,
|
||
Decimal(gate_config["executionAuthorization"]["budget"]["totalBudgetUsd"]),
|
||
)
|
||
self.assertEqual(gate_config["executionAuthorization"]["rawRetention"]["status"], "approved")
|
||
self.assertIn("2250 美元覆盖三角色各 150 次", gate_config["executionAuthorization"]["budget"]["reason"])
|
||
self.assertIn("安全上限", gate_config["executionAuthorization"]["budget"]["reason"])
|
||
|
||
def test_gate_a_formal_zero_budget_or_zero_ac_difference_fails_closed(self):
|
||
"""正式 Gate A 必须真实改变 writer 创作输入,不能只改变隐藏索引。"""
|
||
|
||
zero_budget = _load_gate_a_config()
|
||
zero_budget["samples"][0]["proseCharBudget"] = 0
|
||
result = run_writer_replay(zero_budget, run_id="gate-a-zero-budget")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_writer_context")
|
||
self.assertIn("proseCharBudget", result["samples"][0]["errors"][0])
|
||
|
||
zero_difference = _load_gate_a_config()
|
||
evidence = zero_difference["samples"][0]["writerContextInput"]["retrievalResult"][
|
||
"proseEvidence"
|
||
]
|
||
evidence[1] = {**copy.deepcopy(evidence[0]), "retrievalArm": "C"}
|
||
result = run_writer_replay(zero_difference, run_id="gate-a-zero-difference")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_writer_context")
|
||
self.assertIn("A/C", result["samples"][0]["errors"][0])
|
||
|
||
def test_gate_a_formal_profiles_reconstruct_schema_prompt_and_runtime_bindings(self):
|
||
"""三角色 profile 必须能由 config 重建并绑定当前固定 Opus dry-run 探针。"""
|
||
|
||
gate_config = _load_gate_a_config()
|
||
profiles = gate_config["executionProfiles"]
|
||
self.assertEqual(set(profiles), {"writer", "semantic_detector", "blind_judge"})
|
||
expected_schema_ids = {
|
||
"writer": "writer-draft-v2",
|
||
"semantic_detector": "semantic-detection-draft-v3",
|
||
"blind_judge": "blind-judge-draft-v3",
|
||
}
|
||
expected_schemas = {
|
||
"writer": replay_module.WRITER_OUTPUT_JSON_SCHEMA,
|
||
"semantic_detector": SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
"blind_judge": BLIND_JUDGE_REPORT_JSON_SCHEMA,
|
||
}
|
||
probe = gate_config["executionAuthorization"]["runtimeProbe"]
|
||
profile_hashes = gate_config["executionAuthorization"]["profileSha256"]
|
||
model_aliases = {profile["modelAlias"] for profile in profiles.values()}
|
||
self.assertEqual(model_aliases, {probe["modelAlias"]})
|
||
self.assertEqual(probe["runtimeAdapter"], RUNTIME_ADAPTER)
|
||
self.assertEqual(probe["runtimeAdapterVersion"], RUNTIME_ADAPTER_VERSION)
|
||
self.assertEqual(probe["modelPolicyVersion"], MODEL_POLICY_VERSION)
|
||
self.assertEqual(probe["status"], "dry_run")
|
||
self.assertEqual(probe["executionProfileSha256"], profile_hashes["writer"])
|
||
self.assertEqual(gate_config["commonControls"]["adapterVersion"], RUNTIME_ADAPTER_VERSION)
|
||
cli_only_fields = {
|
||
"claudeExecutablePath",
|
||
"claudeExecutableSha256",
|
||
"claudeCliVersion",
|
||
"normalTerminalReasons",
|
||
}
|
||
self.assertTrue(all(cli_only_fields.isdisjoint(profile) for profile in profiles.values()))
|
||
self.assertTrue(cli_only_fields.isdisjoint(probe))
|
||
|
||
for role, raw_profile in profiles.items():
|
||
profile = replay_module.profile_from_mapping(raw_profile, role=role)
|
||
self.assertEqual(raw_profile["maxBudgetUsdPerCall"], "5.000000")
|
||
self.assertEqual(profile.max_budget_usd_per_call, Decimal("5.000000"))
|
||
self.assertEqual(raw_profile["timeoutSeconds"], 1200)
|
||
self.assertEqual(profile.timeout_seconds, 1200.0)
|
||
self.assertEqual(profile.json_schema_id, expected_schema_ids[role])
|
||
self.assertEqual(profile.json_schema, expected_schemas[role])
|
||
self.assertEqual(profile.json_schema_sha256, sha256_json(profile.json_schema))
|
||
self.assertEqual(profile.system_prompt_sha256, sha256_text(profile.system_prompt))
|
||
self.assertEqual(profile.execution_profile_sha256, profile_hashes[role])
|
||
self.assertEqual(profile.max_context_chars, gate_config["commonControls"]["maxContextChars"])
|
||
self.assertEqual(
|
||
len({raw_profile["systemPromptSha256"] for raw_profile in profiles.values()}),
|
||
3,
|
||
)
|
||
self.assertEqual(len({raw_profile["profileVersion"] for raw_profile in profiles.values()}), 3)
|
||
|
||
for control_name in ("budget", "rawRetention"):
|
||
control = gate_config["executionAuthorization"][control_name]
|
||
self.assertEqual(
|
||
control["receiptSha256"],
|
||
canonical_sha256(
|
||
{key: value for key, value in control.items() if key != "receiptSha256"}
|
||
),
|
||
)
|
||
|
||
def test_execute_without_formal_profiles_fails_closed_before_runner(self):
|
||
"""缺少三角色 formal profiles 时不能靠 profile hash 或测试参数放行。"""
|
||
|
||
tampered = _authorize_gate_config_for_test(_load_gate_a_config())
|
||
tampered.pop("executionProfiles")
|
||
tampered["oracleTruthPacks"] = {}
|
||
budget = tampered["executionAuthorization"]["budget"]
|
||
budget["status"] = "approved"
|
||
budget["totalBudgetUsd"] = "1.000000"
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
tampered,
|
||
run_id="missing-formal-profiles",
|
||
output_dir=output,
|
||
execute=True,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_profile")
|
||
self.assertEqual(result["errors"], ["EXECUTE_PROFILE_INVALID"])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_max_calls_are_sufficient_but_missing_total_budget_still_blocks(self):
|
||
"""150 次/角色满足机械调用量,但没有可靠美元总额仍必须阻断。"""
|
||
|
||
tampered = _authorize_gate_config_for_test(_load_gate_a_config())
|
||
tampered["oracleTruthPacks"] = {}
|
||
budget = tampered["executionAuthorization"]["budget"]
|
||
self.assertEqual(budget["status"], "approved")
|
||
self.assertEqual(budget["totalBudgetUsd"], "2250.000000")
|
||
budget["status"] = "approved"
|
||
budget.pop("totalBudgetUsd", None)
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
tampered,
|
||
run_id="missing-total-budget",
|
||
output_dir=output,
|
||
execute=True,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_budget_authorization")
|
||
self.assertEqual(result["errors"], ["BUDGET_AUTHORIZATION_INVALID"])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_gate_a_execute_blocks_pending_budget_before_vault_or_runner(self):
|
||
"""正式 Gate A 已完成 runtime probe,但预算未批准时必须在副作用前阻断。"""
|
||
|
||
gate_config = _authorize_gate_config_for_test(_load_gate_a_config())
|
||
budget = gate_config["executionAuthorization"]["budget"]
|
||
budget["status"] = "pending"
|
||
budget.pop("totalBudgetUsd", None)
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
gate_config,
|
||
run_id="gate-a-pending-budget",
|
||
output_dir=output,
|
||
execute=True,
|
||
production_adapters=adapters,
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_budget_authorization")
|
||
self.assertEqual(result["errors"], ["BUDGET_AUTHORIZATION_REQUIRED"])
|
||
self.assertEqual(vault_calls, [])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
|
||
class WriterReplayExecuteBoundaryTest(unittest.TestCase):
|
||
"""验证真实执行失败关闭和测试注入仍经过正式管线。"""
|
||
|
||
def test_execute_requires_private_tmp_output(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
with self.assertRaisesRegex(WriterReplayError, "/private/tmp"):
|
||
run_writer_replay(
|
||
config(),
|
||
run_id="real-bad-path",
|
||
output_dir=pathlib.Path(directory),
|
||
execute=True,
|
||
)
|
||
|
||
def test_execute_without_production_adapters_fails_closed(self):
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
pending = config()
|
||
pending["commonControls"]["modelVersion"] = "pending_probe"
|
||
pending["commonControls"]["adapterVersion"] = "pending_probe"
|
||
result = run_writer_replay(
|
||
pending,
|
||
run_id="real-not-wired",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_runtime_probe")
|
||
self.assertEqual(result["errors"], ["RUNTIME_PROBE_REQUIRED"])
|
||
|
||
def test_invalid_third_judge_report_fails_top_level(self):
|
||
adapters, _runner, _detector, _judge = _test_adapters(
|
||
unstable=True, invalid_third=True
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
result = run_writer_replay(
|
||
config(),
|
||
run_id="judge-invalid",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_invalid")
|
||
|
||
def test_third_judge_still_unstable_fails_top_level(self):
|
||
adapters, _runner, _detector, _judge = _test_adapters(
|
||
irreducibly_unstable=True
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
result = run_writer_replay(
|
||
config(),
|
||
run_id="judge-unstable",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_unstable")
|
||
|
||
def test_context_authorization_cannot_swap_snapshot_or_source_before_runner(self):
|
||
"""上下文版本、快照、用途、时间和状态都必须绑定顶层授权。"""
|
||
|
||
mutations = {
|
||
"source_version": lambda value: value["samples"][0]["writerContextInput"].update(
|
||
{"sourceVersion": "swapped-source-v2"}
|
||
),
|
||
"snapshot_id": lambda value: value["samples"][0]["writerContextInput"][
|
||
"authorizationSnapshot"
|
||
].update({"snapshotId": "auth-swapped"}),
|
||
"purpose": lambda value: value["samples"][0]["writerContextInput"][
|
||
"authorizationSnapshot"
|
||
].update({"allowedPurpose": "diagnostic"}),
|
||
"verified_at": lambda value: value["samples"][0]["writerContextInput"][
|
||
"authorizationSnapshot"
|
||
].update({"verifiedAt": "2026-07-21T00:00:00Z"}),
|
||
"source_status": lambda value: value["samples"][0]["writerContextInput"].update(
|
||
{"sourceStatus": "revoked"}
|
||
),
|
||
}
|
||
for name, mutate in mutations.items():
|
||
invalid = config()
|
||
mutate(invalid)
|
||
adapters, runner, _detector, _judge = _test_adapters()
|
||
with self.subTest(name=name), tempfile.TemporaryDirectory(
|
||
dir="/private/tmp"
|
||
) as directory:
|
||
result = run_writer_replay(
|
||
invalid,
|
||
run_id=f"auth-binding-{name}",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_authorization")
|
||
self.assertEqual(runner.calls, [])
|
||
|
||
def test_test_injection_runs_writer_pipeline_mechanical_and_semantic_detector(self):
|
||
adapters, runner, detector, judge = _test_adapters(unstable=True)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
config(),
|
||
run_id="test-pipeline",
|
||
output_dir=output_dir,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
self.assertFalse((output_dir / "raw").exists())
|
||
lease_files = list(
|
||
(output_dir / "journal" / "raw-vault" / "leases").glob("*.json")
|
||
)
|
||
self.assertEqual(len(lease_files), 1)
|
||
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
|
||
public_manifest = (output_dir / "manifest.json").read_text(encoding="utf-8")
|
||
self.assertNotIn("candidateBody", public_manifest)
|
||
self.assertNotIn("脱敏合同夹具", public_manifest)
|
||
|
||
self.assertEqual(result["status"], "completed_test_pipeline")
|
||
self.assertTrue(all("evidenceStrategy" not in context for context in runner.contexts))
|
||
self.assertTrue(
|
||
all(
|
||
kwargs.get("system") == build_dispatch_system_prompt(adapters.writer_profile)
|
||
for _prompt, kwargs in runner.calls
|
||
)
|
||
)
|
||
self.assertEqual(detector.calls, [
|
||
("historical_prose_only", True),
|
||
("card_index_only", True),
|
||
("card_index_plus_prose", True),
|
||
])
|
||
self.assertEqual(len([call for call in judge.calls if call[0] == "judge-3"]), 1)
|
||
candidates = result["samples"][0]["candidates"]
|
||
self.assertTrue(all(not candidate["acceptanceEligible"] for candidate in candidates.values()))
|
||
|
||
def test_writer_model_cannot_forge_adapter_owned_hash(self):
|
||
adapters, _runner, detector, _judge = _test_adapters(illegal_extra_field=True)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
with self.assertRaises(WriterReplayError):
|
||
run_writer_replay(
|
||
config(),
|
||
run_id="test-writer-owned-hash",
|
||
output_dir=pathlib.Path(directory) / "run",
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
# B 臂模型越权输出 adapter 字段后,不能进入 semantic detector。
|
||
self.assertEqual(detector.calls, [("historical_prose_only", True)])
|
||
|
||
def test_malicious_judge_only_sees_blind_candidates_without_raw_paths_or_arm_contexts(self):
|
||
runner = FakeGovernedChat()
|
||
detector = FakeSemanticDetector()
|
||
judge = MaliciousJudge()
|
||
adapters = WriterReplayTestAdapters(runner, _frozen_test_profile(), detector, judge)
|
||
evaluation_config = config()
|
||
fine_outline = evaluation_config["samples"][0]["writerContextInput"]["fineOutline"]
|
||
# 恶意配置模拟把目标原文、索引、文件路径和真实臂塞进细纲对象;这些字段写手看不到,评委也不能看到。
|
||
fine_outline.update(
|
||
{
|
||
"targetOriginal": "目标章原文泄漏哨兵",
|
||
"indexHints": [{"content": "被测卡索引泄漏哨兵"}],
|
||
"path": "/private/tmp/candidate-A.json",
|
||
"arm": "A",
|
||
}
|
||
)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
evaluation_config,
|
||
run_id="blind-isolation",
|
||
output_dir=output_dir,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
|
||
self.assertTrue(result["ok"])
|
||
self.assertTrue(judge.visible_inputs)
|
||
expected_writer_fine_outline = {
|
||
"hardConstraints": ["必须完成围攻突围"],
|
||
"adjustableBeats": [],
|
||
"declaredNewFacts": [],
|
||
}
|
||
self.assertTrue(
|
||
all(context["fineOutline"] == expected_writer_fine_outline for context in runner.contexts),
|
||
"三臂写手必须只看到严格细纲合同字段",
|
||
)
|
||
expected_judge_fine_outline = {
|
||
"sourceRef": {
|
||
"sourceId": "fixture:fine-outline:489",
|
||
"sourceVersion": "fine-outline-fixture-v1",
|
||
"chapter": 489,
|
||
},
|
||
**expected_writer_fine_outline,
|
||
}
|
||
shared_references = []
|
||
for visible in judge.visible_inputs:
|
||
self.assertEqual(
|
||
set(visible),
|
||
{
|
||
"schemaVersion",
|
||
"sample",
|
||
"blindCandidateId",
|
||
"candidateOrder",
|
||
"sharedEvaluationReference",
|
||
"candidates",
|
||
},
|
||
)
|
||
self.assertEqual(set(visible["candidateOrder"]), {"blind-1", "blind-2", "blind-3"})
|
||
shared = visible["sharedEvaluationReference"]
|
||
shared_references.append(copy.deepcopy(shared))
|
||
self.assertEqual(shared["fineOutline"], expected_judge_fine_outline)
|
||
self.assertNotIn("entities", shared["fineOutline"])
|
||
self.assertEqual(shared["requirements"]["requiredCharacters"], ["林澈"])
|
||
self.assertEqual(
|
||
shared["requirements"]["chapterEndHook"]["anchors"],
|
||
["城门忽然打开"],
|
||
)
|
||
self.assertEqual(
|
||
[item["chapter"] for item in shared["historicalProseBaseline"]],
|
||
[485, 486, 487, 488],
|
||
)
|
||
self.assertTrue(
|
||
all(
|
||
set(candidate)
|
||
== {"blindCandidateId", "candidateSha256", "candidateBody"}
|
||
and candidate["blindCandidateId"].startswith("blind-")
|
||
for candidate in visible["candidates"]
|
||
)
|
||
)
|
||
self.assertTrue(
|
||
all(reference == shared_references[0] for reference in shared_references[1:]),
|
||
"评委顺序变化不得改变共同评测参考",
|
||
)
|
||
|
||
|
||
class WriterReplayProductionIntegrationTest(unittest.TestCase):
|
||
"""用生产 fake 串通 Vault、CAS、v2 adapter 与 GateInputBuilder。"""
|
||
|
||
def _run(
|
||
self,
|
||
adapters,
|
||
*,
|
||
evaluation_config=None,
|
||
run_id="production-fake",
|
||
raw_archive_dir=None,
|
||
):
|
||
"""在安全临时目录执行一次生产链,并返回结果与输出目录内容。"""
|
||
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
result = run_writer_replay(
|
||
evaluation_config or _production_config(),
|
||
run_id=run_id,
|
||
output_dir=output,
|
||
raw_archive_dir=raw_archive_dir,
|
||
execute=True,
|
||
production_adapters=adapters,
|
||
)
|
||
return result, output
|
||
|
||
def test_default_runtime_requires_explicit_raw_archive_before_vault_or_model(self):
|
||
"""正式默认 runtime 未给仓外归档根时必须在建 vault 和调模型前失败关闭。"""
|
||
|
||
evaluation_config = _production_config()
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
authorized, blocked_status, blocked_code = replay_module._validate_execute_authorization(
|
||
evaluation_config, adapters
|
||
)
|
||
self.assertIsNotNone(authorized, (blocked_status, blocked_code))
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
|
||
with mock.patch.object(
|
||
replay_module,
|
||
"_validate_execute_authorization",
|
||
return_value=(authorized, None, None),
|
||
):
|
||
result = run_writer_replay(
|
||
evaluation_config,
|
||
run_id="missing-raw-archive",
|
||
output_dir=output,
|
||
execute=True,
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_raw_archive")
|
||
self.assertEqual(result["errors"], ["RAW_ARCHIVE_REQUIRED"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_relative_raw_archive_blocks_before_vault_or_model(self):
|
||
"""显式归档根必须是绝对路径,不能退回仓库相对目录。"""
|
||
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
run_id="relative-raw-archive",
|
||
raw_archive_dir=pathlib.Path("relative/archive"),
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_raw_archive")
|
||
self.assertEqual(result["errors"], ["RAW_ARCHIVE_INVALID"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
@staticmethod
|
||
def _cas_safe_summary(output: pathlib.Path, sample_id: str) -> dict[str, object]:
|
||
cas_root = output / "journal" / "cas" / sample_id
|
||
state = json.loads((cas_root / "state.json").read_text(encoding="utf-8"))
|
||
return json.loads(
|
||
(cas_root / "artifacts" / state["safeSummaryArtifact"]).read_text(
|
||
encoding="utf-8"
|
||
)
|
||
)
|
||
|
||
def test_oracle_projection_keeps_only_common_recent_four_chapters(self):
|
||
"""完整源包可验真全历史,但 blind judge 只能收到共同的最近四章与目标断言。"""
|
||
|
||
source = _oracle_pack()
|
||
statement = "第484章的更早历史只留在受控 oracle 源包"
|
||
source["historicalAssertions"].insert(
|
||
0,
|
||
{
|
||
"assertionId": "assertion-history-484",
|
||
"assertionType": "canonical_history",
|
||
"statement": statement,
|
||
"sourceVersion": "canonical-v484",
|
||
"chapterBoundary": {"minChapter": 484, "maxChapter": 484},
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
|
||
},
|
||
)
|
||
source["packSha256"] = canonical_sha256(
|
||
{key: value for key, value in source.items() if key != "packSha256"}
|
||
)
|
||
|
||
projected = _project_oracle_pack(source)
|
||
|
||
self.assertEqual(
|
||
[item["chapterStart"] for item in projected["historicalAssertions"]],
|
||
[485, 486, 487, 488],
|
||
)
|
||
self.assertTrue(projected["targetAssertions"])
|
||
self.assertNotIn("authorization", projected)
|
||
self.assertNotIn("assertionType", json.dumps(projected, ensure_ascii=False))
|
||
self.assertNotIn("第484章", json.dumps(projected, ensure_ascii=False))
|
||
|
||
def test_production_execute_rejects_sanitized_fixture_before_vault_or_runner(self):
|
||
"""真实 execute 只接受 loader 产出的 canonical_frozen_prose。"""
|
||
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
sanitized = _production_config()
|
||
sanitized["samples"][0]["writerContextInput"]["contentMode"] = (
|
||
"sanitized_contract_fixture"
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters, evaluation_config=sanitized, run_id="production-sanitized"
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_content_mode")
|
||
self.assertEqual(result["errors"], ["EXECUTE_CONTENT_MODE_INVALID"])
|
||
self.assertEqual(vault_calls, [])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
def test_default_runtime_missing_auth_blocks_before_vault_or_budget_slot(self):
|
||
"""真实 runtime 缺显式 token 时必须在任何副作用前返回准确阻断码。"""
|
||
|
||
config = _production_config()
|
||
adapters, _writer, _semantic, _judge = _production_adapters()
|
||
with (
|
||
mock.patch.dict(
|
||
os.environ,
|
||
{"MUSE_ROLE_OPUS_BASE_URL": "", "MUSE_ROLE_OPUS_AUTH_TOKEN": ""},
|
||
clear=False,
|
||
),
|
||
mock.patch(
|
||
"run_writer_replay._production_adapters_from_config",
|
||
return_value=adapters,
|
||
),
|
||
tempfile.TemporaryDirectory(dir="/private/tmp") as directory,
|
||
):
|
||
output = pathlib.Path(directory) / "run"
|
||
result = run_writer_replay(
|
||
config,
|
||
run_id="runtime-auth-missing",
|
||
output_dir=output,
|
||
execute=True,
|
||
)
|
||
self.assertFalse((output / "journal" / "raw-vault").exists())
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_runtime_authentication")
|
||
self.assertEqual(result["errors"], ["RUNTIME_AUTHENTICATION_REQUIRED"])
|
||
self.assertEqual(result["samples"][0]["status"], "ready")
|
||
self.assertNotIn("budgetLedger", result)
|
||
|
||
def test_planned_timeout_sum_does_not_block_next_call_safe_run(self):
|
||
"""理论调用总时长超过租期时,只要每笔调用窗口充足就允许真实编排继续。"""
|
||
|
||
vault_calls = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
config = _production_config()
|
||
raw = config["executionAuthorization"]["rawRetention"]
|
||
raw["retainUntil"] = (datetime.now(UTC) + timedelta(minutes=4)).isoformat()
|
||
raw["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in raw.items() if key != "receiptSha256"}
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=config,
|
||
run_id="next-call-retention-allowed",
|
||
)
|
||
|
||
# 测试 profile 每角色 plannedCalls=3、timeout=30 秒,理论总和加清理为 330 秒,
|
||
# 高于 4 分钟租期;单次调用只需 30+60=90 秒,所以不应被总和误阻断。
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual(len(vault_calls), 1)
|
||
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
|
||
self.assertTrue((output / "gate-input.json").exists())
|
||
|
||
def test_b_needs_evidence_keeps_candidate_and_reaches_judge_and_gate(self):
|
||
"""B 臂合法补证信号属于对照质量,不得误报为执行系统失败。"""
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters()
|
||
config = _production_config()
|
||
semantic = NeedsEvidenceSemanticRunner(
|
||
needs_evidence_on_call=_semantic_call_index_for_arm(config, "B")
|
||
)
|
||
adapters = replace(adapters, semantic_model_runner=semantic)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=config,
|
||
run_id="semantic-needs-evidence-b",
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
sample = result["samples"][0]
|
||
self.assertEqual(set(sample["candidates"]), {"A", "B", "C"})
|
||
self.assertNotIn("detector", sample)
|
||
diagnostic = sample["semanticDiagnostics"]["B"]
|
||
self.assertEqual(diagnostic["outcome"], "needs_evidence")
|
||
self.assertEqual(diagnostic["blockingCounts"]["evidenceGaps"], 1)
|
||
self.assertEqual(diagnostic["blockingCounts"]["unknownAssertions"], 0)
|
||
self.assertEqual(len(semantic.calls), 3)
|
||
self.assertEqual(len(judge.calls), 2)
|
||
rubric = judge.calls[0]["rubric"]
|
||
self.assertEqual(rubric["policyVersion"], RUBRIC_POLICY_VERSION)
|
||
self.assertEqual(rubric["scenarioType"], "battle")
|
||
self.assertEqual(
|
||
[item["dimensionId"] for item in rubric["commonDimensions"]],
|
||
list(COMMON_DIMENSIONS),
|
||
)
|
||
self.assertEqual(
|
||
rubric["scenarioDimension"]["dimensionId"], SCENARIO_DIMENSION
|
||
)
|
||
self.assertEqual(rubric["scenarioDimension"]["displayName"], "战斗执行")
|
||
self.assertTrue((output / "gate-input.json").exists())
|
||
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(manifest["samples"][0]["semanticDiagnostics"], {"B": diagnostic})
|
||
self.assertEqual(
|
||
self._cas_safe_summary(output, "deep-space-489")["semanticDiagnostics"],
|
||
{"B": diagnostic},
|
||
)
|
||
serialized = json.dumps(manifest, ensure_ascii=False)
|
||
for forbidden in (
|
||
"candidateQuote",
|
||
"补查候选新事实",
|
||
"冻结证据不足",
|
||
"gap-1",
|
||
'"query"',
|
||
'"message"',
|
||
"/private/tmp",
|
||
):
|
||
self.assertNotIn(forbidden, serialized)
|
||
|
||
def test_pre_call_judge_failure_preserves_primary_code_without_receipt(self):
|
||
"""评委输入在首调前失败时,不能用“缺回执”覆盖真正错误码。"""
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters()
|
||
invalid = {
|
||
"ok": False,
|
||
"status": "failed_judge_invalid",
|
||
"primaryCode": "BLIND_INPUT_INVALID",
|
||
"reviewCount": 0,
|
||
"attemptCount": 0,
|
||
"safeDiagnostic": {
|
||
"schemaVersion": "blind-judge-safe-diagnostic-v1",
|
||
"errorCode": "BLIND_INPUT_INVALID",
|
||
"errorMessage": "输入绑定非法",
|
||
"draftAvailable": False,
|
||
},
|
||
}
|
||
|
||
with mock.patch.object(
|
||
execute_module,
|
||
"run_writer_blind_judge_panel",
|
||
return_value=invalid,
|
||
):
|
||
result, output = self._run(adapters, run_id="judge-pre-call-invalid")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_invalid")
|
||
self.assertEqual(
|
||
result["samples"][0]["errors"],
|
||
["failed_judge_invalid:BLIND_INPUT_INVALID"],
|
||
)
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
diagnostic = result["samples"][0]["judgeDiagnostic"]
|
||
self.assertEqual(diagnostic["primaryCode"], "BLIND_INPUT_INVALID")
|
||
self.assertEqual(diagnostic["modelCallCount"], 0)
|
||
self.assertEqual(
|
||
diagnostic["modelOutput"]["errorCode"], "BLIND_INPUT_INVALID"
|
||
)
|
||
self.assertNotIn("candidateBody", json.dumps(diagnostic, ensure_ascii=False))
|
||
|
||
def test_detector_correction_exhaustion_keeps_failed_sample(self):
|
||
"""detector 不合约必须保留首样本并停止,不能继续启动第二样本。"""
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters()
|
||
semantic = InvalidQuoteSemanticRunner()
|
||
evaluation_config = _production_config()
|
||
second_sample = copy.deepcopy(evaluation_config["samples"][0])
|
||
second_sample["sampleId"] = "deep-space-490"
|
||
evaluation_config["samples"].append(second_sample)
|
||
budget = evaluation_config["executionAuthorization"]["budget"]
|
||
for role in replay_module.BUDGET_ROLES:
|
||
budget["plannedCalls"][role] = 6
|
||
budget["maxCalls"][role] = 6
|
||
budget["totalBudgetUsd"] = "18.000000"
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
second_oracle = copy.deepcopy(_oracle_pack())
|
||
second_oracle["sampleId"] = "deep-space-490"
|
||
second_oracle["packSha256"] = canonical_sha256(
|
||
{key: value for key, value in second_oracle.items() if key != "packSha256"}
|
||
)
|
||
adapters = replace(
|
||
adapters,
|
||
semantic_model_runner=semantic,
|
||
oracle_truth_packs={
|
||
**adapters.oracle_truth_packs,
|
||
"deep-space-490": second_oracle,
|
||
},
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=evaluation_config,
|
||
run_id="semantic-invalid-exhausted",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_semantic_detector")
|
||
self.assertEqual(len(result["samples"]), 1)
|
||
self.assertEqual(result["samples"][0]["status"], "failed_semantic_detector")
|
||
self.assertEqual(
|
||
result["samples"][0]["errors"],
|
||
["SEMANTIC_DETECTOR_QUOTE_NOT_FOUND"],
|
||
)
|
||
diagnostic = result["samples"][0]["semanticDiagnostic"]
|
||
self.assertEqual(diagnostic["reasonCode"], "QUOTE_NOT_FOUND")
|
||
self.assertEqual(diagnostic["attemptCount"], 3)
|
||
self.assertEqual(diagnostic["correctionCount"], 2)
|
||
self.assertEqual(len(semantic.calls), 3)
|
||
self.assertEqual(judge.calls, [])
|
||
cas_dirs = sorted(
|
||
path.name
|
||
for path in (output / "journal" / "cas").iterdir()
|
||
if path.is_dir()
|
||
)
|
||
self.assertEqual(cas_dirs, [result["samples"][0]["sampleId"]])
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
manifest_text = (output / "manifest.json").read_text(encoding="utf-8")
|
||
for forbidden in ("candidateQuote", "候选中不存在的引文", "previousDraft"):
|
||
self.assertNotIn(forbidden, manifest_text)
|
||
self.assertEqual(
|
||
self._cas_safe_summary(output, result["samples"][0]["sampleId"])[
|
||
"semanticDiagnostic"
|
||
],
|
||
diagnostic,
|
||
)
|
||
|
||
def test_detector_correction_receipts_bind_final_success_in_gate(self):
|
||
"""首轮非法、第二轮纠正成功时,Gate 绑定最后回执且整轮可完成。"""
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters()
|
||
semantic = CorrectingSemanticRunner()
|
||
adapters = replace(adapters, semantic_model_runner=semantic)
|
||
config = _production_config()
|
||
budget = config["executionAuthorization"]["budget"]
|
||
budget["plannedCalls"]["semantic_detector"] = 4
|
||
budget["maxCalls"]["semantic_detector"] = 4
|
||
budget["totalBudgetUsd"] = "10.000000"
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=config,
|
||
run_id="semantic-correction-success",
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(len(semantic.calls), 4)
|
||
receipts = json.loads((output / "execution-receipts.json").read_text())
|
||
self.assertEqual(
|
||
len(receipts["deep-space-489"]["A"]["semantic_detector"]["executionReceipts"]),
|
||
2,
|
||
)
|
||
gate_input = json.loads((output / "gate-input.json").read_text())
|
||
self.assertFalse(gate_input["samples"][0]["systemFailure"])
|
||
self.assertTrue(all("semanticDiagnostic" not in sample for sample in result["samples"]))
|
||
self.assertNotIn(
|
||
"semanticDiagnostic",
|
||
self._cas_safe_summary(output, "deep-space-489"),
|
||
)
|
||
|
||
def test_raw_expiry_inside_sample_is_recorded_and_cas_closed(self):
|
||
"""样本执行中的 raw 异常必须保留样本安全终态并收口 CAS。"""
|
||
|
||
class ExpiringManager(RawVaultManager):
|
||
def __init__(self, root):
|
||
super().__init__(root)
|
||
self.write_count = 0
|
||
|
||
def write_bytes(self, lease, relative_path, content):
|
||
self.write_count += 1
|
||
if self.write_count == 2:
|
||
raise RawVaultError("RAW_LEASE_EXPIRED", "测试租约到期")
|
||
return super().write_bytes(lease, relative_path, content)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=ExpiringManager)
|
||
|
||
result, output = self._run(adapters, run_id="raw-expired-in-sample")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_raw_vault")
|
||
self.assertEqual(len(result["samples"]), 1)
|
||
self.assertEqual(result["samples"][0]["errors"], ["RAW_LEASE_EXPIRED"])
|
||
self.assertEqual((writer.calls, semantic.calls, judge.calls), ([], [], []))
|
||
cas = json.loads(
|
||
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text()
|
||
)
|
||
self.assertEqual(cas["state"], "FAILED")
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
|
||
def test_gate_publish_failure_removes_partially_published_input(self):
|
||
"""Gate 文件原子写后若后续发布动作失败,失败 manifest 前必须撤销文件。"""
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters()
|
||
original_write = replay_module.atomic_write_json
|
||
|
||
def fail_after_gate_write(path, value):
|
||
original_write(path, value)
|
||
if path.name == "gate-input.json":
|
||
raise OSError("测试 Gate 发布失败")
|
||
|
||
with mock.patch("run_writer_replay.execute.atomic_write_json", side_effect=fail_after_gate_write):
|
||
result, output = self._run(adapters, run_id="gate-publish-failure")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_system")
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
|
||
def test_judge_quote_correction_keeps_full_receipts_and_binds_final_reports(self):
|
||
"""评委纠错必须保留全量调用,并用最终索引通过 Gate builder 反篡改复核。"""
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters(
|
||
judge_invalid_first_quote=True
|
||
)
|
||
|
||
result, output = self._run(adapters, run_id="judge-quote-correction")
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(len(judge.calls), 3)
|
||
self.assertNotIn("correction", judge.calls[0])
|
||
self.assertIn("correction", judge.calls[1])
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
panel_receipts = execution_receipts["deep-space-489"]["A"][
|
||
"blind_judge"
|
||
]["executionReceipts"]
|
||
self.assertEqual(len(panel_receipts), 3)
|
||
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
|
||
self.assertTrue(
|
||
all(not sample["systemFailure"] for sample in gate_input["samples"]),
|
||
gate_input["samples"],
|
||
)
|
||
|
||
def test_production_fake_happy_path_runs_v2_adapters_receipts_vault_cas_and_builder(self):
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
|
||
result, output = self._run(adapters)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual(len(writer.calls), 3)
|
||
self.assertEqual(len(semantic.calls), 3)
|
||
self.assertEqual(len(judge.calls), 2)
|
||
self.assertEqual(
|
||
result["budgetLedger"]["usedCalls"],
|
||
{"writer": 3, "semantic_detector": 3, "blind_judge": 2},
|
||
)
|
||
self.assertEqual(
|
||
result["budgetLedger"]["remainingPlannedCalls"],
|
||
{"writer": 0, "semantic_detector": 0, "blind_judge": 1},
|
||
)
|
||
self.assertEqual(
|
||
result["budgetLedger"]["totalActualCostUsd"],
|
||
_expected_total_cost(fixed_role_calls=5),
|
||
)
|
||
self.assertFalse(result["budgetLedger"]["costUnknown"])
|
||
self.assertTrue(all(receipt["adapterRole"] == "semantic_detector" for receipt in semantic.receipts))
|
||
self.assertTrue(all(receipt["adapterRole"] == "blind_judge" for receipt in judge.receipts))
|
||
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(gate_input["schemaVersion"], "writer-gate-input-v3")
|
||
self.assertEqual(gate_input["rubricPolicyVersion"], RUBRIC_POLICY_VERSION)
|
||
self.assertEqual(gate_input["gateInputSha256"], result["gateInputSha256"])
|
||
self.assertTrue(
|
||
all(not sample["systemFailure"] for sample in gate_input["samples"]),
|
||
gate_input["samples"],
|
||
)
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
self.assertEqual(
|
||
{
|
||
role
|
||
for arm in execution_receipts["deep-space-489"].values()
|
||
for role in arm
|
||
},
|
||
{"writer", "semantic_detector", "blind_judge"},
|
||
)
|
||
self.assertTrue(
|
||
all(
|
||
wrapper["executionReceipts"]
|
||
for arm in execution_receipts["deep-space-489"].values()
|
||
for wrapper in arm.values()
|
||
)
|
||
)
|
||
manifest = (output / "manifest.json").read_text(encoding="utf-8")
|
||
self.assertNotIn("candidateBody", manifest)
|
||
self.assertNotIn("/private/tmp", manifest)
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
lease = json.loads(lease_files[0].read_text(encoding="utf-8"))
|
||
self.assertEqual(lease["status"], "migrated")
|
||
self.assertEqual(result["rawDisposition"]["status"], "migrated")
|
||
self.assertEqual(result["rawDisposition"]["archiveId"], lease["archiveId"])
|
||
archive_root = output.parent / "production-fake-raw-archive"
|
||
archived = archive_root / f"muse-raw-archive-{lease['archiveId']}"
|
||
self.assertTrue((archived / "run" / "config.json").is_file())
|
||
self.assertTrue((archived / ".migration-receipt.json").is_file())
|
||
cas_state = json.loads(
|
||
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text()
|
||
)
|
||
self.assertEqual(cas_state["state"], "COMPLETED")
|
||
self.assertEqual(cas_state["cleanupState"], "migrated")
|
||
|
||
def test_length_revision_loop_recovers_out_of_range_draft(self):
|
||
"""首版越界时退回写手修订一遍即达标:样本成功,修订块带固定指令与上一版正文。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
config = _production_config_with_writer_budget(
|
||
planned=9, max_calls=9, total_usd="18.000000"
|
||
)
|
||
short_body = "短" * 2000 # 2000 汉字 < 合同下限 2800,首版必然越界
|
||
|
||
def revise_once(context, *, call_index):
|
||
# 偶数下标是每臂首版(越界),奇数下标是修订版(达标)。
|
||
if call_index % 2 == 0:
|
||
return {"candidateBody": short_body}
|
||
marker = ("甲", "乙", "丙")[(call_index // 2) % 3]
|
||
return {"candidateBody": _candidate_body(marker)}
|
||
|
||
with mock.patch.object(sys.modules[__name__], "_writer_output", revise_once):
|
||
result, output = self._run(
|
||
adapters, evaluation_config=config, run_id="length-revision-ok"
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
# 每臂 1 次首版 + 1 次修订 = 2 次 writer 调用,三臂共 6 次。
|
||
self.assertEqual(len(writer.calls), 6)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 6)
|
||
self.assertNotIn("lengthViolation", result["samples"][0])
|
||
# 每臂 writer 回执同时绑定首版与修订两条。
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
for arm in ("A", "B", "C"):
|
||
wrapper = execution_receipts["deep-space-489"][arm]["writer"]
|
||
self.assertEqual(len(wrapper["executionReceipts"]), 2)
|
||
# 首版调用不带 lengthRevision,修订调用带;下标 1/3/5 是三臂的修订输入。
|
||
for index in (0, 2, 4):
|
||
self.assertNotIn("lengthRevision", writer.contexts[index])
|
||
revision_inputs = [writer.contexts[index] for index in (1, 3, 5)]
|
||
for context in revision_inputs:
|
||
revision = context["lengthRevision"]
|
||
# 修订指令是三臂共用的固定常量(arm-invariant),不含具体字数。
|
||
self.assertEqual(revision["instruction"], replay_module.LENGTH_REVISION_INSTRUCTION)
|
||
# 修订块携带上一版正文与实际字数/区间,作为数据而非拼进指令文本。
|
||
self.assertEqual(revision["previousDraft"], short_body)
|
||
self.assertEqual(revision["actualHanChars"], 2000)
|
||
self.assertEqual(revision["minChars"], 2800)
|
||
self.assertEqual(revision["maxChars"], 5200)
|
||
self.assertEqual(revision["targetChars"], 4000)
|
||
self.assertEqual(revision["countingRule"], "han_chars_only")
|
||
self.assertEqual(revision["revisionDirection"], "expand")
|
||
self.assertEqual(revision["targetDeltaHanChars"], 2000)
|
||
self.assertEqual(
|
||
{context["lengthRevision"]["instruction"] for context in revision_inputs},
|
||
{replay_module.LENGTH_REVISION_INSTRUCTION},
|
||
)
|
||
instruction = replay_module.LENGTH_REVISION_INSTRUCTION
|
||
self.assertIn("只统计 candidateBody 中的汉字", instruction)
|
||
self.assertIn("不统计标点、空格、数字或拉丁字母", instruction)
|
||
self.assertIn("以 targetChars 为修订目标", instruction)
|
||
self.assertIn("不要只擦到", instruction)
|
||
self.assertIn("必须完整保留上一版已有内容", instruction)
|
||
self.assertIn("不得压缩、删除或合并已有内容", instruction)
|
||
self.assertIn("revisionDirection=trim", instruction)
|
||
self.assertIn("删减约 targetDeltaHanChars 个汉字", instruction)
|
||
self.assertIn("不得继续扩写", instruction)
|
||
|
||
def test_length_revision_recomputes_direction_and_target_delta_each_round(self):
|
||
"""同一臂每轮都按当前正文重算方向与距目标差值,不能复用首版数字。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
config = _production_config_with_writer_budget(
|
||
planned=9, max_calls=9, total_usd="18.000000"
|
||
)
|
||
short_body = "短" * 2000
|
||
long_body = "长" * 6000
|
||
|
||
original_writer_output = _writer_output
|
||
|
||
def output_without_recursion(context, *, call_index):
|
||
if call_index == 0:
|
||
return {"candidateBody": short_body}
|
||
if call_index == 1:
|
||
return {"candidateBody": long_body}
|
||
return original_writer_output(context, call_index=call_index)
|
||
|
||
with mock.patch.object(
|
||
sys.modules[__name__], "_writer_output", output_without_recursion
|
||
):
|
||
result, _output = self._run(
|
||
adapters, evaluation_config=config, run_id="length-revision-recomputed"
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
first_revision = writer.contexts[1]["lengthRevision"]
|
||
second_revision = writer.contexts[2]["lengthRevision"]
|
||
self.assertEqual(first_revision["previousDraft"], short_body)
|
||
self.assertEqual(first_revision["actualHanChars"], 2000)
|
||
self.assertEqual(first_revision["revisionDirection"], "expand")
|
||
self.assertEqual(first_revision["targetDeltaHanChars"], 2000)
|
||
self.assertEqual(second_revision["previousDraft"], long_body)
|
||
self.assertEqual(second_revision["actualHanChars"], 6000)
|
||
self.assertEqual(second_revision["revisionDirection"], "trim")
|
||
self.assertEqual(second_revision["targetDeltaHanChars"], 2000)
|
||
self.assertEqual(
|
||
{first_revision["countingRule"], second_revision["countingRule"]},
|
||
{"han_chars_only"},
|
||
)
|
||
|
||
def test_length_revision_exhausted_still_fails_with_actual_han_chars(self):
|
||
"""修订三遍仍越界才失败:记下实际字数,writer 调用数 = 1 首版 + 3 修订 = 4。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
config = _production_config_with_writer_budget(
|
||
planned=4,
|
||
max_calls=4,
|
||
total_usd="12.000000",
|
||
)
|
||
short_body = "短" * 2000
|
||
with mock.patch.object(
|
||
sys.modules[__name__],
|
||
"_writer_output",
|
||
lambda context, *, call_index: {"candidateBody": short_body},
|
||
):
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=config,
|
||
run_id="length-revision-exhausted",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
sample = result["samples"][0]
|
||
self.assertIn("候选正文汉字数超出动态篇幅区间", sample["errors"])
|
||
self.assertIn("lengthViolation", sample)
|
||
self.assertEqual(sample["lengthViolation"]["actualHanChars"], 2000)
|
||
# 首臂耗尽 1+3=4 次 writer 调用后即终局门失败,后续臂不再运行。
|
||
self.assertEqual(len(writer.calls), 4)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 4)
|
||
|
||
def test_length_in_range_skips_revision_loop(self):
|
||
"""首版即在区间内:0 修订,每臂仅 1 次 writer 调用,输入不带 lengthRevision。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
result, output = self._run(adapters, run_id="length-no-revision")
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(len(writer.calls), 3)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 3)
|
||
for context in writer.contexts:
|
||
self.assertNotIn("lengthRevision", context)
|
||
execution_receipts = json.loads(
|
||
(output / "execution-receipts.json").read_text(encoding="utf-8")
|
||
)
|
||
for arm in ("A", "B", "C"):
|
||
wrapper = execution_receipts["deep-space-489"][arm]["writer"]
|
||
self.assertEqual(len(wrapper["executionReceipts"]), 1)
|
||
|
||
def test_bound_judge_report_binds_raw_outputs_to_receipt_hashes(self):
|
||
"""生产编排必须把 reviewer 原始输出随报告进 builder 且与回执哈希一致(正向对照)。"""
|
||
|
||
captured: dict[str, object] = {}
|
||
|
||
class CapturingBuilder:
|
||
"""记录一次生产 build 的全部来源,供篡改复现复用,本身仍走真 builder。"""
|
||
|
||
def build(self, **kwargs):
|
||
captured.update(kwargs)
|
||
return GateInputBuilder().build(**kwargs)
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
builder_factory=CapturingBuilder
|
||
)
|
||
result, _output = self._run(adapters)
|
||
self.assertTrue(result["ok"], result)
|
||
|
||
judge_report = captured["judge_reports"]["deep-space-489"]
|
||
reviewer_outputs = judge_report["reviewerStructuredOutputs"]
|
||
true_receipts = captured["execution_receipts"]["deep-space-489"]["A"][
|
||
"blind_judge"
|
||
]["executionReceipts"]
|
||
# 原始输出与 reviewerReports、回执按 reviewer 顺序一一对应,且哈希回真实回执。
|
||
self.assertEqual(len(reviewer_outputs), len(judge_report["reviewerReports"]))
|
||
self.assertEqual(len(reviewer_outputs), len(true_receipts))
|
||
for draft, receipt in zip(reviewer_outputs, true_receipts, strict=True):
|
||
self.assertEqual(
|
||
canonical_sha256(draft), receipt["structuredOutputSha256"]
|
||
)
|
||
|
||
def test_consistent_report_tampering_with_resigned_hashes_is_rejected(self):
|
||
"""一致篡改报告评分 + 重签 reportSha256 + 重推导 panel 仍被原始输出绑定拦下。
|
||
|
||
攻击场景:攻击者把每个 reviewer 报告的评分统一改成伪造高分,并把报告内携带的
|
||
reviewerStructuredOutputs 同步改成与伪造评分一致的伪造原始输出(使伪造自洽),
|
||
重签内层/外层 reportSha256,按同一 adjudicate_structured_reviews 重推导 panel。
|
||
唯一改不动的是执行回执里生产时固化的 structuredOutputSha256——它仍指向真实原始
|
||
输出,于是新的只读交叉校验 canonical(伪造输出) != 回执哈希 失败关闭。
|
||
"""
|
||
|
||
captured: dict[str, object] = {}
|
||
|
||
class CapturingBuilder:
|
||
"""记录一次生产 build 的全部来源,供篡改复现复用,本身仍走真 builder。"""
|
||
|
||
def build(self, **kwargs):
|
||
captured.update(kwargs)
|
||
return GateInputBuilder().build(**kwargs)
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
builder_factory=CapturingBuilder
|
||
)
|
||
result, _output = self._run(adapters)
|
||
self.assertTrue(result["ok"], result)
|
||
|
||
# 正向对照:未篡改的合法报告(原始输出与报告一致)必须通过 builder,无系统失败且评分入输入。
|
||
gate_input = GateInputBuilder().build(**copy.deepcopy(captured))
|
||
self.assertEqual(gate_input["schemaVersion"], "writer-gate-input-v3")
|
||
self.assertEqual(gate_input["rubricPolicyVersion"], RUBRIC_POLICY_VERSION)
|
||
legit_sample = next(
|
||
item for item in gate_input["samples"] if item["sampleId"] == "deep-space-489"
|
||
)
|
||
self.assertFalse(legit_sample["systemFailure"])
|
||
self.assertTrue(legit_sample["schemaValid"])
|
||
self.assertIn("scores", legit_sample)
|
||
|
||
true_receipts = captured["execution_receipts"]["deep-space-489"]["A"][
|
||
"blind_judge"
|
||
]["executionReceipts"]
|
||
judge_report = copy.deepcopy(captured["judge_reports"]["deep-space-489"])
|
||
forged_score = 9.5
|
||
for index, reviewer_report in enumerate(judge_report["reviewerReports"]):
|
||
forged_output = copy.deepcopy(judge_report["reviewerStructuredOutputs"][index])
|
||
# 报告评分与伪造原始输出同步改成一致的伪造高分,使报告内部自洽。
|
||
for report_row, draft_row in zip(
|
||
reviewer_report["candidateScores"],
|
||
forged_output["candidateScores"],
|
||
strict=True,
|
||
):
|
||
for dimension in DIMENSIONS:
|
||
report_row["scores"][dimension]["score"] = forged_score
|
||
draft_row["scores"][dimension]["score"] = forged_score
|
||
judge_report["reviewerStructuredOutputs"][index] = forged_output
|
||
# 伪造输出与真实回执固化的原始输出哈希必然不同——这是攻击者改不动的锚点。
|
||
self.assertNotEqual(
|
||
canonical_sha256(forged_output),
|
||
true_receipts[index]["structuredOutputSha256"],
|
||
)
|
||
# modelReceiptSha256 保留(仍绑定未改动的真实回执),只重签内层 reportSha256。
|
||
reviewer_report["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in reviewer_report.items() if key != "reportSha256"}
|
||
)
|
||
# 按篡改后的 reviewer 报告重推导 panel,使 recomputed_panel == report 这一旧校验仍能通过。
|
||
panel = adjudicate_structured_reviews(
|
||
judge_report["reviewerReports"][0],
|
||
judge_report["reviewerReports"][1],
|
||
judge_report["reviewerReports"][2]
|
||
if len(judge_report["reviewerReports"]) == 3
|
||
else None,
|
||
)
|
||
self.assertIn(panel["status"], {"stable_report", "adjudicated_report"})
|
||
judge_report.update(panel)
|
||
judge_report["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in judge_report.items() if key != "reportSha256"}
|
||
)
|
||
tampered_judge_reports = {
|
||
**captured["judge_reports"],
|
||
"deep-space-489": judge_report,
|
||
}
|
||
# 伪造评分自洽、重签重推导都做了,仍被「原始输出 ↔ 回执哈希」交叉校验失败关闭:
|
||
# 该样本被标 systemFailure、schemaValid=False,伪造评分不进 Gate 输入的 scores。
|
||
tampered_input = GateInputBuilder().build(
|
||
**{**copy.deepcopy(captured), "judge_reports": tampered_judge_reports}
|
||
)
|
||
tampered_sample = next(
|
||
item
|
||
for item in tampered_input["samples"]
|
||
if item["sampleId"] == "deep-space-489"
|
||
)
|
||
self.assertTrue(tampered_sample["systemFailure"])
|
||
self.assertFalse(tampered_sample["schemaValid"])
|
||
self.assertNotIn("scores", tampered_sample)
|
||
self.assertTrue(
|
||
any(
|
||
"reviewerStructuredOutputs" in reason
|
||
and "未绑定模型原始 structured output" in reason
|
||
for reason in tampered_sample["systemFailureReasons"]
|
||
),
|
||
tampered_sample["systemFailureReasons"],
|
||
)
|
||
|
||
# 报告缺失 reviewerStructuredOutputs 字段时同样失败关闭:不放行无绑定的报告。
|
||
stripped = copy.deepcopy(captured["judge_reports"]["deep-space-489"])
|
||
del stripped["reviewerStructuredOutputs"]
|
||
stripped["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in stripped.items() if key != "reportSha256"}
|
||
)
|
||
stripped_input = GateInputBuilder().build(
|
||
**{
|
||
**copy.deepcopy(captured),
|
||
"judge_reports": {
|
||
**captured["judge_reports"],
|
||
"deep-space-489": stripped,
|
||
},
|
||
}
|
||
)
|
||
stripped_sample = next(
|
||
item
|
||
for item in stripped_input["samples"]
|
||
if item["sampleId"] == "deep-space-489"
|
||
)
|
||
self.assertTrue(stripped_sample["systemFailure"])
|
||
self.assertNotIn("scores", stripped_sample)
|
||
self.assertTrue(
|
||
any(
|
||
"reviewerStructuredOutputs" in reason
|
||
for reason in stripped_sample["systemFailureReasons"]
|
||
),
|
||
stripped_sample["systemFailureReasons"],
|
||
)
|
||
|
||
def test_a_semantic_failure_keeps_candidate_and_reaches_judge_and_gate(self):
|
||
"""A 臂合法 failed 是对照质量信号,不得阻断剩余臂和 Gate builder。"""
|
||
|
||
config = _production_config()
|
||
adapters, writer, semantic, judge = _production_adapters(
|
||
semantic_fail_on=_semantic_call_index_for_arm(config, "A")
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=config,
|
||
run_id="semantic-failure-a",
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
|
||
sample = result["samples"][0]
|
||
self.assertEqual(set(sample["candidates"]), {"A", "B", "C"})
|
||
self.assertNotIn("detector", sample)
|
||
diagnostic = sample["semanticDiagnostics"]["A"]
|
||
self.assertEqual(diagnostic["outcome"], "failed")
|
||
self.assertEqual(diagnostic["blockingCounts"]["highFindings"], 1)
|
||
self.assertEqual(diagnostic["blockingCounts"]["failedHardConstraints"], 1)
|
||
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
|
||
self.assertFalse(gate_input["samples"][0]["systemFailure"])
|
||
self.assertEqual(gate_input["samples"][0]["cArm"]["hardConstraintCoverage"], 1.0)
|
||
self.assertEqual(gate_input["samples"][0]["cArm"]["highSeverityResidualCount"], 0)
|
||
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(manifest["samples"][0]["semanticDiagnostics"], {"A": diagnostic})
|
||
self.assertEqual(
|
||
self._cas_safe_summary(output, "deep-space-489")["semanticDiagnostics"],
|
||
{"A": diagnostic},
|
||
)
|
||
serialized = json.dumps(manifest, ensure_ascii=False)
|
||
for forbidden in (
|
||
"candidateQuote",
|
||
"硬约束未满足",
|
||
"finding-high-1",
|
||
'"findingId"',
|
||
'"message"',
|
||
"/private/tmp",
|
||
):
|
||
self.assertNotIn(forbidden, serialized)
|
||
|
||
def test_judge_api_retry_aligns_null_output_and_binds_final_reports(self):
|
||
"""前置 API 失败无输出;null 占位与全回执对齐后,终稿索引仍可反篡改复核。"""
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters(
|
||
judge_api_error_first=True
|
||
)
|
||
|
||
result, output = self._run(adapters, run_id="judge-api-retry")
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(len(judge.calls), 3)
|
||
receipts = json.loads((output / "execution-receipts.json").read_text())
|
||
panel_receipts = receipts["deep-space-489"]["A"]["blind_judge"][
|
||
"executionReceipts"
|
||
]
|
||
self.assertEqual(len(panel_receipts), 3)
|
||
self.assertEqual(panel_receipts[0]["terminalReason"], "api_error")
|
||
gate_input = json.loads((output / "gate-input.json").read_text())
|
||
self.assertFalse(gate_input["samples"][0]["systemFailure"])
|
||
|
||
def test_c_semantic_failure_reaches_gate_and_gate_a_fails_quality(self):
|
||
"""C 臂合法 failed 必须走完整链,并由 Gate A 的 C 臂硬门判为 failed。"""
|
||
|
||
config = _production_config()
|
||
adapters, writer, semantic, judge = _production_adapters(
|
||
semantic_fail_on=_semantic_call_index_for_arm(config, "C")
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=config,
|
||
run_id="semantic-failure-c",
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
|
||
sample = result["samples"][0]
|
||
self.assertEqual(sample["semanticDiagnostics"]["C"]["outcome"], "failed")
|
||
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
|
||
self.assertFalse(gate_input["samples"][0]["systemFailure"])
|
||
self.assertEqual(gate_input["samples"][0]["cArm"]["hardConstraintCoverage"], 0.0)
|
||
self.assertEqual(gate_input["samples"][0]["cArm"]["highSeverityResidualCount"], 1)
|
||
|
||
gate_report = decide_gate(_five_scenario_gate_input(gate_input))
|
||
|
||
self.assertEqual(gate_report["status"], "failed")
|
||
self.assertIn("c_arm_high_severity_residual", gate_report["reasons"])
|
||
self.assertIn(
|
||
"c_arm_hard_constraint_coverage_below_100_percent",
|
||
gate_report["reasons"],
|
||
)
|
||
|
||
def test_third_judge_still_unstable_fails_run(self):
|
||
adapters, _writer, _semantic, judge = _production_adapters(judge_unstable=True)
|
||
|
||
result, _output = self._run(adapters, run_id="judge-third-unstable")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_judge_unstable")
|
||
self.assertEqual(len(judge.calls), 3)
|
||
self.assertEqual(result["budgetLedger"]["usedCalls"]["blind_judge"], 3)
|
||
self.assertEqual(result["budgetLedger"]["remainingPlannedCalls"]["blind_judge"], 0)
|
||
self.assertEqual(
|
||
result["budgetLedger"]["totalActualCostUsd"],
|
||
_expected_total_cost(fixed_role_calls=6),
|
||
)
|
||
|
||
def test_vault_migration_failure_overrides_success(self):
|
||
class MigrationFailureManager(RawVaultManager):
|
||
"""先实际迁移 raw,再模拟迁移回执失败;归档内容仍必须保留。"""
|
||
|
||
def migrate(self, lease, *, archive_root):
|
||
super().migrate(lease, archive_root=archive_root)
|
||
raise RawVaultError("RAW_MIGRATION_FAILED", "测试迁移失败")
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
vault_factory=MigrationFailureManager
|
||
)
|
||
|
||
result, output = self._run(adapters, run_id="migration-failure")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_raw_migration")
|
||
cas_state = json.loads(
|
||
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text(
|
||
encoding="utf-8"
|
||
)
|
||
)
|
||
self.assertEqual(cas_state["state"], "FAILED")
|
||
self.assertEqual(cas_state["cleanupState"], "failed")
|
||
|
||
def test_cas_conflict_fails_sample_and_run(self):
|
||
class ConflictingCas:
|
||
"""初始化使用真实 CAS,第一次推进模拟迟到 revision 冲突。"""
|
||
|
||
def __init__(self, root):
|
||
self.delegate = FileCasStore(root)
|
||
|
||
def initialize(self, **kwargs):
|
||
return self.delegate.initialize(**kwargs)
|
||
|
||
def transition(self, **_kwargs):
|
||
raise CasConflictError("测试旧 revision")
|
||
|
||
adapters, _writer, _semantic, judge = _production_adapters(cas_factory=ConflictingCas)
|
||
|
||
result, _output = self._run(adapters, run_id="cas-conflict")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_cas_conflict")
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_builder_binding_failure_fails_after_all_sample_terminals(self):
|
||
class FailingBuilder:
|
||
"""模拟来源绑定无法闭合,证明不能手工回退聚合。"""
|
||
|
||
def build(self, **_kwargs):
|
||
raise GateInputBuildError("测试 builder 绑定失败")
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(builder_factory=FailingBuilder)
|
||
|
||
result, output = self._run(adapters, run_id="builder-failure")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_gate_input_builder")
|
||
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
|
||
def test_builder_failure_closes_all_multi_sample_judge_terminals_as_failed(self):
|
||
"""整轮 builder 失败时,所有已到 JUDGE_COMPLETED 的样本仍必须失败关闭。"""
|
||
|
||
class FailingBuilder:
|
||
def build(self, **_kwargs):
|
||
raise GateInputBuildError("测试多样本 builder 绑定失败")
|
||
|
||
evaluation_config = _production_config()
|
||
second_sample = copy.deepcopy(evaluation_config["samples"][0])
|
||
second_sample["sampleId"] = "deep-space-490"
|
||
evaluation_config["samples"].append(second_sample)
|
||
budget = evaluation_config["executionAuthorization"]["budget"]
|
||
for role in replay_module.BUDGET_ROLES:
|
||
budget["plannedCalls"][role] = 6
|
||
budget["maxCalls"][role] = 6
|
||
budget["totalBudgetUsd"] = "18.000000"
|
||
budget["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in budget.items() if key != "receiptSha256"}
|
||
)
|
||
second_oracle = copy.deepcopy(_oracle_pack())
|
||
second_oracle["sampleId"] = "deep-space-490"
|
||
second_oracle["packSha256"] = canonical_sha256(
|
||
{key: value for key, value in second_oracle.items() if key != "packSha256"}
|
||
)
|
||
adapters, writer, semantic, judge = _production_adapters(
|
||
builder_factory=FailingBuilder
|
||
)
|
||
adapters = replace(
|
||
adapters,
|
||
oracle_truth_packs={
|
||
**adapters.oracle_truth_packs,
|
||
"deep-space-490": second_oracle,
|
||
},
|
||
)
|
||
|
||
result, output = self._run(
|
||
adapters,
|
||
evaluation_config=evaluation_config,
|
||
run_id="builder-failure-multi-sample",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_gate_input_builder")
|
||
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (6, 6, 4))
|
||
self.assertFalse((output / "gate-input.json").exists())
|
||
for sample_id in ("deep-space-489", "deep-space-490"):
|
||
cas_state = json.loads(
|
||
(output / "journal" / "cas" / sample_id / "state.json").read_text(
|
||
encoding="utf-8"
|
||
)
|
||
)
|
||
self.assertEqual(cas_state["state"], "FAILED")
|
||
self.assertEqual(cas_state["cleanupState"], "migrated")
|
||
|
||
def test_sigterm_after_vault_creation_recovers_open_lease_and_raw_vault(self):
|
||
"""create_vault 返回前被首次 SIGTERM 打断时必须恢复并关闭 raw lease。"""
|
||
|
||
class InterruptingCreateVaultManager(RawVaultManager):
|
||
def create_vault(self, **kwargs):
|
||
super().create_vault(**kwargs)
|
||
signal.raise_signal(signal.SIGTERM)
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
vault_factory=InterruptingCreateVaultManager
|
||
)
|
||
|
||
result, output = self._run(adapters, run_id="sigterm-vault-create")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_interrupted")
|
||
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(manifest["status"], "failed_interrupted")
|
||
self.assertNotIn("candidateBody", json.dumps(manifest, ensure_ascii=False))
|
||
self.assertNotIn("/private/tmp", json.dumps(manifest, ensure_ascii=False))
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
|
||
self.assertEqual(
|
||
sorted(path.name for path in (output / "journal" / "raw-vault").iterdir()),
|
||
["leases"],
|
||
)
|
||
|
||
def test_sigterm_after_cas_initialize_closes_started_cas_as_failed(self):
|
||
"""CAS.initialize 提交 STARTED 后首次 SIGTERM 也必须失败收口。"""
|
||
|
||
class InterruptingInitializeCas:
|
||
def __init__(self, root):
|
||
self.delegate = FileCasStore(root)
|
||
|
||
def initialize(self, **kwargs):
|
||
self.delegate.initialize(**kwargs)
|
||
signal.raise_signal(signal.SIGTERM)
|
||
|
||
def latest(self):
|
||
return self.delegate.latest()
|
||
|
||
def transition(self, **kwargs):
|
||
return self.delegate.transition(**kwargs)
|
||
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
cas_factory=InterruptingInitializeCas
|
||
)
|
||
|
||
result, output = self._run(adapters, run_id="sigterm-cas-initialize")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_interrupted")
|
||
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(manifest["status"], "failed_interrupted")
|
||
self.assertNotIn("candidateBody", json.dumps(manifest, ensure_ascii=False))
|
||
self.assertNotIn("/private/tmp", json.dumps(manifest, ensure_ascii=False))
|
||
cas_state = json.loads(
|
||
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text(
|
||
encoding="utf-8"
|
||
)
|
||
)
|
||
self.assertEqual(cas_state["state"], "FAILED")
|
||
self.assertEqual(cas_state["cleanupState"], "migrated")
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "migrated")
|
||
|
||
def test_pending_probe_and_budget_block_before_vault_or_runner(self):
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
pending = _production_config()
|
||
pending["commonControls"]["modelVersion"] = "pending_probe"
|
||
pending["commonControls"]["adapterVersion"] = "pending_probe"
|
||
result, _output = self._run(adapters, evaluation_config=pending, run_id="pending-probe")
|
||
self.assertEqual(result["status"], "blocked_runtime_probe")
|
||
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=_production_config(budget_approved=False),
|
||
run_id="pending-budget",
|
||
)
|
||
self.assertEqual(result["status"], "blocked_budget_authorization")
|
||
self.assertEqual(vault_calls, [])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_public_test_adapter_cannot_bypass_execute_authorization(self):
|
||
"""仅传 test_adapters 不能跳过 probe、预算和 raw 审批。"""
|
||
|
||
adapters, runner, detector, judge = _test_adapters()
|
||
unauthorized = config()
|
||
unauthorized.pop("executionAuthorization", None)
|
||
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
result = run_writer_replay(
|
||
unauthorized,
|
||
run_id="test-adapter-no-authorization",
|
||
output_dir=output,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_authorization")
|
||
self.assertEqual(runner.calls, [])
|
||
self.assertEqual(detector.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
self.assertFalse((output / "raw").exists())
|
||
|
||
def test_authorized_test_adapter_never_persists_plaintext_raw_directory(self):
|
||
"""fake 执行也必须使用受控 raw lease,并在返回前完成清理。"""
|
||
|
||
adapters, _runner, _detector, _judge = _test_adapters()
|
||
authorized = config()
|
||
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
|
||
self.addCleanup(directory.cleanup)
|
||
output = pathlib.Path(directory.name) / "run"
|
||
|
||
result = run_writer_replay(
|
||
authorized,
|
||
run_id="test-adapter-vault",
|
||
output_dir=output,
|
||
execute=True,
|
||
test_adapters=adapters,
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertFalse((output / "raw").exists())
|
||
self.assertNotIn("candidateBody", (output / "manifest.json").read_text(encoding="utf-8"))
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
|
||
|
||
def test_production_profile_prompt_tamper_cannot_use_noop_verifier(self):
|
||
"""生产 A/C 与 B 都必须在治理调用前执行 prompt/schema 哈希校验。"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
invalid_profile = copy.deepcopy(adapters.writer_profile)
|
||
# 模拟配置构造后内存中的 prompt 被替换;身份哈希仍绑定旧 prompt 哈希,
|
||
# 只有真实 verify_role_profile 才能在模型调用前发现。
|
||
object.__setattr__(invalid_profile, "system_prompt", "被篡改的 writer prompt")
|
||
adapters = replace(adapters, writer_profile=invalid_profile)
|
||
evaluation_config = _production_config()
|
||
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=evaluation_config,
|
||
run_id="invalid-prompt-binding",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertNotIn("binding_verifier", inspect.signature(run_writer_replay).parameters)
|
||
|
||
def test_runtime_probe_self_hash_tamper_fails_closed_before_runner(self):
|
||
"""runtime probe receipt 自哈希被篡改时,不能进入任何模型 runner。"""
|
||
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
tampered = _production_config()
|
||
tampered["executionAuthorization"]["runtimeProbe"]["receiptSha256"] = (
|
||
"sha256:" + "0" * 64
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters,
|
||
evaluation_config=tampered,
|
||
run_id="tampered-runtime-probe",
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_authorization")
|
||
self.assertEqual(result["errors"], ["EXECUTE_AUTHORIZATION_HASH_INVALID"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_execute_blocks_stale_probe_bound_to_old_contract(self):
|
||
"""陈旧运行探针(身份哈希对不上新 writer 合同)必须被机械阻断在 runner 之前。
|
||
|
||
对照设计:与下方 test_execute_passes_probe_bound_to_current_contract 共用同一
|
||
_production_config,只把探针的身份哈希/输出哈希换成旧合同的值。旧探针只证明
|
||
「旧合同能产出结构化输出」,证明不了换过 schema/prompt 的新合同也可以,故必须挡下。
|
||
"""
|
||
|
||
adapters, writer, semantic, judge = _production_adapters()
|
||
stale = _production_config()
|
||
probe = stale["executionAuthorization"]["runtimeProbe"]
|
||
# 模拟探针仍是旧合同实跑的:身份哈希与结构化输出哈希都来自旧合同。
|
||
probe["executionProfileSha256"] = "sha256:" + "9" * 64
|
||
probe["structuredOutputSha256"] = "sha256:" + "4" * 64
|
||
# 重算探针自哈希,确保它卡在「合同绑定」这一关,而不是更早的自哈希这一关。
|
||
probe["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in probe.items() if key != "receiptSha256"}
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters, evaluation_config=stale, run_id="stale-probe-old-contract"
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_execute_probe_contract")
|
||
self.assertEqual(result["errors"], ["EXECUTE_PROBE_CONTRACT_MISMATCH"])
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
|
||
def test_execute_passes_probe_bound_to_current_contract(self):
|
||
"""按新合同重跑的探针(身份哈希==新 writer profile)必须放行 execute 门。
|
||
|
||
与上方陈旧探针对照:同一 _production_config,探针身份哈希绑定当前 writer 合同,
|
||
execute 门放行并真正进入模型 runner 跑完全链。
|
||
"""
|
||
|
||
adapters, writer, _semantic, _judge = _production_adapters()
|
||
bound = _production_config()
|
||
probe = bound["executionAuthorization"]["runtimeProbe"]
|
||
# 显式确认探针身份哈希已绑定当前 writer 合同(新 schema/prompt 的身份)。
|
||
self.assertEqual(
|
||
probe["executionProfileSha256"],
|
||
adapters.writer_profile.execution_profile_sha256,
|
||
)
|
||
|
||
result, _output = self._run(
|
||
adapters, evaluation_config=bound, run_id="probe-current-contract"
|
||
)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual(len(writer.calls), 3)
|
||
|
||
def test_termination_handlers_install_and_restore_without_killing_process(self):
|
||
"""handler 安装/恢复必须是确定性的:装上后信号被接管,恢复后还原调用方原 handler。"""
|
||
|
||
old_term = signal.getsignal(signal.SIGTERM)
|
||
old_int = signal.getsignal(signal.SIGINT)
|
||
previous = replay_module.install_termination_handlers()
|
||
try:
|
||
self.assertIn(signal.SIGTERM, previous)
|
||
self.assertIn(signal.SIGINT, previous)
|
||
self.assertIsNot(signal.getsignal(signal.SIGTERM), old_term)
|
||
self.assertIsNot(signal.getsignal(signal.SIGINT), old_int)
|
||
finally:
|
||
replay_module.restore_termination_handlers(previous)
|
||
self.assertIs(signal.getsignal(signal.SIGTERM), old_term)
|
||
self.assertIs(signal.getsignal(signal.SIGINT), old_int)
|
||
|
||
def test_sigterm_during_writer_call_fails_closed_with_cost_unknown(self):
|
||
"""终止落在已发起但无回执的 writer 调用中:成本未知、raw 仍迁移、manifest 安全。"""
|
||
|
||
old_term = signal.getsignal(signal.SIGTERM)
|
||
old_int = signal.getsignal(signal.SIGINT)
|
||
adapters, _writer, semantic, judge = _production_adapters()
|
||
adapters = replace(adapters, writer_runner=InterruptingChat(interrupt_on_call=1))
|
||
|
||
result, output = self._run(adapters, run_id="sigterm-writer")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_interrupted")
|
||
self.assertEqual(result["errors"], ["EXECUTION_COST_UNKNOWN"])
|
||
self.assertTrue(result["budgetLedger"]["costUnknown"])
|
||
self.assertEqual(result["budgetLedger"]["failureReason"], "EXECUTION_COST_UNKNOWN")
|
||
# 终止之后不得再发起任何 detector / judge 调用。
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
# manifest 必须是安全字段:不含正文、不含 raw 路径。
|
||
manifest_text = (output / "manifest.json").read_text(encoding="utf-8")
|
||
self.assertNotIn("candidateBody", manifest_text)
|
||
self.assertNotIn("/private/tmp", manifest_text)
|
||
# open lease 必须被收敛为 migrated,临时 vault 消失但受控归档保留。
|
||
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
|
||
self.assertEqual(len(lease_files), 1)
|
||
lease = json.loads(lease_files[0].read_text(encoding="utf-8"))
|
||
self.assertEqual(lease["status"], "migrated")
|
||
archive_root = output.parent / "sigterm-writer-raw-archive"
|
||
self.assertTrue(
|
||
(archive_root / f"muse-raw-archive-{lease['archiveId']}").is_dir()
|
||
)
|
||
# 运行 journal 只留非敏感 lease;raw 已迁往独立受控归档。
|
||
self.assertEqual(
|
||
[path.name for path in (output / "journal" / "raw-vault").iterdir()],
|
||
["leases"],
|
||
)
|
||
cas_state = json.loads(
|
||
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text(
|
||
encoding="utf-8"
|
||
)
|
||
)
|
||
self.assertEqual(cas_state["state"], "FAILED")
|
||
self.assertEqual(cas_state["cleanupState"], "migrated")
|
||
self.assertIs(signal.getsignal(signal.SIGTERM), old_term)
|
||
self.assertIs(signal.getsignal(signal.SIGINT), old_int)
|
||
|
||
def test_termination_is_deferred_during_migration_and_restored_after_manifest(self):
|
||
"""最终迁移和 manifest 写入期间必须忽略后续终止,返回后恢复原 handler。"""
|
||
|
||
handler_states: list[tuple[object, object]] = []
|
||
|
||
class HandlerObservingManager(RawVaultManager):
|
||
def migrate(self, lease, *, archive_root):
|
||
handler_states.append(
|
||
(
|
||
signal.getsignal(signal.SIGTERM),
|
||
signal.getsignal(signal.SIGINT),
|
||
)
|
||
)
|
||
return super().migrate(lease, archive_root=archive_root)
|
||
|
||
old_term = signal.getsignal(signal.SIGTERM)
|
||
old_int = signal.getsignal(signal.SIGINT)
|
||
adapters, _writer, _semantic, _judge = _production_adapters(
|
||
vault_factory=HandlerObservingManager
|
||
)
|
||
adapters = replace(adapters, writer_runner=InterruptingChat(interrupt_on_call=1))
|
||
|
||
result, _output = self._run(adapters, run_id="sigterm-migration-handler")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(handler_states, [(signal.SIG_IGN, signal.SIG_IGN)])
|
||
self.assertIs(signal.getsignal(signal.SIGTERM), old_term)
|
||
self.assertIs(signal.getsignal(signal.SIGINT), old_int)
|
||
|
||
def test_short_retention_lease_blocks_before_vault_contents_or_runner(self):
|
||
"""27 秒短租约必须在落 lease / 建 vault / 调模型前稳定失败关闭。"""
|
||
|
||
vault_calls: list[pathlib.Path] = []
|
||
|
||
def vault_factory(path):
|
||
vault_calls.append(path)
|
||
return RawVaultManager(path)
|
||
|
||
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
|
||
short = _production_config()
|
||
raw = short["executionAuthorization"]["rawRetention"]
|
||
raw["retainUntil"] = (datetime.now(UTC) + timedelta(seconds=27)).isoformat()
|
||
raw["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in raw.items() if key != "receiptSha256"}
|
||
)
|
||
|
||
result, output = self._run(adapters, evaluation_config=short, run_id="short-lease")
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "failed_raw_vault")
|
||
self.assertEqual(result["errors"], ["RAW_LEASE_INSUFFICIENT_RETENTION"])
|
||
# 失败关闭在任何模型 runner 之前。
|
||
self.assertEqual(writer.calls, [])
|
||
self.assertEqual(semantic.calls, [])
|
||
self.assertEqual(judge.calls, [])
|
||
# 不能留下 lease journal 或 raw vault 目录(manager 外壳目录可有,内容必须空)。
|
||
self.assertEqual(list((output / "journal" / "raw-vault" / "leases").glob("*.json")), [])
|
||
|
||
|
||
class SemanticInputSourceRefCleaningTest(unittest.TestCase):
|
||
"""验证 _semantic_input_v3 投影检测输入时清洗证据 sourceRef 的多余字段。
|
||
|
||
C 臂(卡索引+原文臂)的 proseEvidence sourceRef 带 sourceType(如 card_chapter_proxy),
|
||
检测输入校验把 sourceRef 当闭集会拒收多余字段。投影时必须清洗成合同形状,且 writer_context
|
||
本身不动(writer 看到的 proseEvidence 仍带 sourceType,那是 writer 侧合同)。
|
||
"""
|
||
|
||
PROSE_TEXT = "脱敏原文片段,C 臂知识卡投影。"
|
||
|
||
@classmethod
|
||
def _writer_context(cls) -> dict[str, object]:
|
||
prose_hash = "sha256:" + hashlib.sha256(cls.PROSE_TEXT.encode("utf-8")).hexdigest()
|
||
prose_source_ref = {
|
||
"sourceId": "fixture:card-chapter-proxy:1",
|
||
"sourceVersion": "cards-frozen-488-v1",
|
||
"chapter": 488,
|
||
"blockId": 1120,
|
||
"startCodePoint": 0,
|
||
"endCodePoint": len(cls.PROSE_TEXT),
|
||
"contentSha256": prose_hash,
|
||
# writer 侧合同允许、但检测闭集拒收的多余字段:
|
||
"sourceType": "card_chapter_proxy",
|
||
"cardId": "card-1",
|
||
}
|
||
fact_source_ref = {
|
||
"sourceId": "fixture:fact:1",
|
||
"sourceVersion": "canonical-v488",
|
||
"contentSha256": "sha256:" + "8" * 64,
|
||
"sourceType": "canonical_state",
|
||
}
|
||
return {
|
||
"fineOutline": {
|
||
"sourceRef": {
|
||
"sourceId": "fixture:outline:489",
|
||
"sourceVersion": "outline-frozen-489-v1",
|
||
},
|
||
"hardConstraints": ["人物保持冻结状态"],
|
||
"adjustableBeats": ["过场节奏可调"],
|
||
"declaredNewFacts": [],
|
||
},
|
||
"contextSnapshot": {"contextSha256": "sha256:" + "5" * 64},
|
||
"asOf": 488,
|
||
"authorizationSnapshot": {"snapshotId": "auth-work-8"},
|
||
"factEvidence": [
|
||
{
|
||
"evidenceId": "fact-ev-1",
|
||
"fact": "林澈仍在圣蒂曼",
|
||
"sourceType": "canonical_state",
|
||
"sourceRef": fact_source_ref,
|
||
"contentSha256": "sha256:" + "8" * 64,
|
||
"riskLevel": "low",
|
||
}
|
||
],
|
||
"proseEvidence": [
|
||
{
|
||
"evidenceId": "prose-ev-1",
|
||
"chapter": 488,
|
||
"sourceRef": prose_source_ref,
|
||
"contentSha256": prose_hash,
|
||
"purpose": "style_baseline",
|
||
"text": cls.PROSE_TEXT,
|
||
"isRecentBaseline": True,
|
||
}
|
||
],
|
||
}
|
||
|
||
@staticmethod
|
||
def _candidate() -> dict[str, object]:
|
||
body = "候选正文。"
|
||
return {
|
||
"candidateVersion": 1,
|
||
"candidateSha256": "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest(),
|
||
"candidateBody": body,
|
||
}
|
||
|
||
def _project(self, writer_context: dict[str, object], opaque_arm_id: str) -> dict[str, object]:
|
||
return _semantic_input_v3(
|
||
run_id="run-clean",
|
||
sample_id="deep-space-321",
|
||
opaque_arm_id=opaque_arm_id,
|
||
writer_context=writer_context,
|
||
candidate=self._candidate(),
|
||
)
|
||
|
||
def test_prose_evidence_source_ref_drops_extra_fields(self):
|
||
result = self._project(self._writer_context(), "blind-1")
|
||
cleaned_ref = result["proseEvidence"][0]["sourceRef"]
|
||
# 多余字段(sourceType/cardId)被清洗掉。
|
||
self.assertNotIn("sourceType", cleaned_ref)
|
||
self.assertNotIn("cardId", cleaned_ref)
|
||
# 只保留检测闭集允许的字段。
|
||
self.assertLessEqual(set(cleaned_ref), _SOURCE_REF_ALLOWED)
|
||
# 关键定位字段保留。
|
||
self.assertEqual(cleaned_ref["sourceId"], "fixture:card-chapter-proxy:1")
|
||
self.assertEqual(cleaned_ref["sourceVersion"], "cards-frozen-488-v1")
|
||
self.assertEqual(cleaned_ref["chapter"], 488)
|
||
self.assertEqual(cleaned_ref["blockId"], 1120)
|
||
# 证据其它字段原样保留(只清洗 sourceRef 子对象)。
|
||
prose_item = result["proseEvidence"][0]
|
||
self.assertEqual(prose_item["evidenceId"], "prose-ev-1")
|
||
self.assertEqual(prose_item["purpose"], "style_baseline")
|
||
self.assertIs(prose_item["isRecentBaseline"], True)
|
||
self.assertEqual(prose_item["text"], self.PROSE_TEXT)
|
||
|
||
def test_fact_evidence_source_ref_drops_extra_fields(self):
|
||
result = self._project(self._writer_context(), "blind-2")
|
||
cleaned_ref = result["factEvidence"][0]["sourceRef"]
|
||
self.assertNotIn("sourceType", cleaned_ref)
|
||
self.assertLessEqual(set(cleaned_ref), _SOURCE_REF_ALLOWED)
|
||
self.assertEqual(cleaned_ref["sourceId"], "fixture:fact:1")
|
||
self.assertEqual(cleaned_ref["sourceVersion"], "canonical-v488")
|
||
# factEvidence 条目自身的 sourceType(检测合同要求)保留,只有 sourceRef 子对象被清洗。
|
||
self.assertEqual(result["factEvidence"][0]["sourceType"], "canonical_state")
|
||
|
||
def test_writer_context_source_ref_untouched(self):
|
||
writer_context = self._writer_context()
|
||
self._project(writer_context, "blind-3")
|
||
# writer 侧合同不变:投影后 writer_context 的 sourceRef 仍带 sourceType。
|
||
self.assertEqual(
|
||
writer_context["proseEvidence"][0]["sourceRef"]["sourceType"],
|
||
"card_chapter_proxy",
|
||
)
|
||
self.assertEqual(
|
||
writer_context["factEvidence"][0]["sourceRef"]["sourceType"],
|
||
"canonical_state",
|
||
)
|
||
|
||
def test_clean_source_ref_non_mapping_passthrough(self):
|
||
# ref 不是 dict 时原样返回,交由下游检测校验按合同处理。
|
||
self.assertEqual(_clean_source_ref("fixture:plain"), "fixture:plain")
|
||
self.assertIsNone(_clean_source_ref(None))
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|