3867 lines
166 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""正文 A/B/C 冻结回放编排的无网络、无真实模型测试。"""
from __future__ import annotations
import copy
import hashlib
import inspect
import json
import os
import pathlib
import signal
import sys
import tempfile
import unittest
from dataclasses import replace
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from unittest import mock
TEST_DIR = pathlib.Path(__file__).resolve().parent
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
SKILLS_DIR = PROJECT_ROOT / ".agent" / "skills"
SCRIPT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "replay" / "回放评估正文质量" / "scripts"
READ_CONTEXT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "context" / "skills" / "准备任务上下文" / "scripts"
EVIDENCE_DIR = PROJECT_ROOT / "muse" / "authority" / "evidence" / "skills" / "记录运行证据" / "scripts"
QUALITY_GATE_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "judge" / "评估内容质量" / "scripts"
GATE_ADJUDICATION_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "mechanical" / "判定质量是否合格" / "scripts"
for import_path in (
SCRIPT_DIR,
READ_CONTEXT_DIR,
EVIDENCE_DIR,
QUALITY_GATE_DIR,
GATE_ADJUDICATION_DIR,
):
if str(import_path) not in sys.path:
sys.path.insert(0, str(import_path))
import run_writer_replay as replay_module # noqa: E402
import run_writer_replay.execute as execute_module # noqa: E402
from file_cas import CasConflictError, FileCasStore # noqa: E402
from gate_input_builder import GateInputBuilder, GateInputBuildError, canonical_sha256 # noqa: E402
from muse_role import ( # noqa: E402
FIXED_OPUS_MODEL_ID,
FIXED_OPUS_POLICY_ALIAS,
FIXED_OPUS_POLICY_VERSION,
MODEL_POLICY_VERSION,
RUNTIME_ADAPTER,
RUNTIME_ADAPTER_VERSION,
RoleExecutionProfile,
RoleRuntimeError,
build_dispatch_system_prompt,
sha256_json,
sha256_text,
)
from raw_vault import RawVaultError, RawVaultManager # noqa: E402
from run_writer import build_writer_execution_profile # noqa: E402
from run_writer_blind_judge import BLIND_JUDGE_REPORT_JSON_SCHEMA # noqa: E402
from run_writer_replay import ( # noqa: E402
BudgetLedgerError,
WriterReplayError,
WriterReplayProductionAdapters,
WriterReplayTestAdapters,
run_writer_replay,
)
from run_writer_replay.blind import ( # noqa: E402
_SOURCE_REF_ALLOWED,
_clean_source_ref,
_project_oracle_pack,
_semantic_input_v3,
)
from run_writer_semantic_detector import ( # noqa: E402
SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
is_semantic_schema_specialization,
)
from writer_contract import calculate_target_chars, han_count # noqa: E402
from writer_eval_preregister import build_balanced_preregistration # noqa: E402
from writer_gate import decide_gate # noqa: E402
from writer_rubric import ( # noqa: E402
COMMON_DIMENSIONS,
DIMENSIONS,
RUBRIC_POLICY_VERSION,
RUBRIC_PROFILE,
SCENARIO_DIMENSION,
adjudicate_structured_reviews,
)
def _utc_stamp(delta: timedelta) -> str:
"""相对当前时刻生成 UTC 时间戳。"""
return (
(datetime.now(UTC) + delta)
.replace(microsecond=0)
.isoformat()
.replace("+00:00", "Z")
)
# 授权快照的 checkedAt 必须早于现在、revalidationAt 必须晚于现在,否则 check_snapshot
# 会 fail-closed 成 blocked_authorization。写死日期会让这份脱敏夹具到期后集体变红,
# 因此改成相对当前时钟派生。
AUTHORIZATION_CHECKED_AT = _utc_stamp(timedelta(days=-30))
AUTHORIZATION_REVALIDATION_AT = _utc_stamp(timedelta(days=30))
AUTHORIZATION = {
"sourceStatus": "authorized",
"copyrightStatus": "research_only",
"sourceHash": "sha256:" + "a" * 64,
"sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64,
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution", "model_training", "production_generation"],
"authorizationSnapshot": {
"id": "auth-work-8",
"version": "auth-work-8-v1",
"immutable": True,
"sourceHash": "sha256:" + "a" * 64,
"sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64,
"sourceStatus": "authorized",
"copyrightStatus": "research_only",
"authorizationBasis": "sanitized_contract_fixture",
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution", "model_training", "production_generation"],
"checkedAt": AUTHORIZATION_CHECKED_AT,
"revalidationAt": AUTHORIZATION_REVALIDATION_AT,
},
}
GATE_A_CONFIG_PATH = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1-cn-skills.json"
def _load_gate_a_config() -> dict[str, object]:
"""读取 Gate A 配置并把授权重验窗口对齐到当前时钟。
磁盘上的窗口是一次真实登记的结果,会随时间自然到期;结构性合同测试不应该被
这个运维事实带红,所以在内存副本里刷新 checkedAt/verifiedAt/revalidationAt。
"""
gate_config = json.loads(GATE_A_CONFIG_PATH.read_text(encoding="utf-8"))
snapshot = gate_config["authorization"]["authorizationSnapshot"]
snapshot["checkedAt"] = AUTHORIZATION_CHECKED_AT
snapshot["revalidationAt"] = AUTHORIZATION_REVALIDATION_AT
for sample in gate_config["samples"]:
context_snapshot = sample["writerContextInput"].get("authorizationSnapshot")
if isinstance(context_snapshot, dict):
context_snapshot["verifiedAt"] = AUTHORIZATION_CHECKED_AT
return gate_config
def _authorize_gate_config_for_test(gate_config: dict[str, object]) -> dict[str, object]:
"""为门禁顺序测试注入 provider-neutral 合成探针,不冒充生产能力证明。"""
writer = replay_module.profile_from_mapping(
gate_config["executionProfiles"]["writer"],
role="writer",
)
probe = {
"schemaVersion": "runtime-probe-v2-test-fixture",
"status": "successful",
"checkedAt": _utc_stamp(timedelta(minutes=-1)),
"runtimeAdapter": RUNTIME_ADAPTER,
"runtimeAdapterVersion": RUNTIME_ADAPTER_VERSION,
"modelPolicyVersion": MODEL_POLICY_VERSION,
"role": "writer",
"profileVersion": writer.profile_version,
"modelAlias": writer.model_alias,
"executionProfileSha256": writer.execution_profile_sha256,
"jsonSchemaId": writer.json_schema_id,
"jsonSchemaSha256": writer.json_schema_sha256,
"systemPromptId": writer.system_prompt_id,
"systemPromptSha256": writer.system_prompt_sha256,
"inputSha256": "sha256:" + "5" * 64,
"requestedModelId": writer.model_alias,
"actualModelId": FIXED_OPUS_MODEL_ID,
"modelMatch": True,
"executionReceiptSha256": "sha256:" + "7" * 64,
"structuredOutputSha256": "sha256:" + "6" * 64,
"terminalReason": "completed",
"totalCostUsd": "0.010000",
}
probe["receiptSha256"] = canonical_sha256(probe)
gate_config["executionAuthorization"]["runtimeProbe"] = probe
return gate_config
def _prose(chapter: int, block_id: int, text: str) -> dict[str, object]:
"""构造只用于合同证明的脱敏合成原文片段。"""
return {
"chapter": chapter,
"sourceRef": {
"sourceId": f"fixture:chapter:{chapter}:block:{block_id}",
"sourceVersion": f"sanitized-fixture-{chapter}-v1",
"chapter": chapter,
"blockId": block_id,
"startCodePoint": 0,
"endCodePoint": len(text),
},
"text": text,
}
def _writer_context_input() -> dict[str, object]:
"""构造可让 A/B/C 都通过 WriterContext v1 的脱敏夹具。"""
recent = [
_prose(chapter, 1000 + chapter, f"脱敏合同夹具第{chapter}章,人物保持冻结状态。")
for chapter in range(485, 489)
]
supplemental = _prose(120, 1120, "脱敏补充片段,旧徽章曾经出现。")
return {
"contentMode": "sanitized_contract_fixture",
"sourceVersion": AUTHORIZATION["sourceVersion"],
"authorizationSnapshot": {
"snapshotId": "auth-work-8",
"allowedPurpose": "offline_evaluation",
"verifiedAt": AUTHORIZATION_CHECKED_AT,
},
"sourceStatus": "authorized",
"cardIndexVersion": "cards-frozen-488-v1",
"proseIndexVersion": "prose-frozen-488-v1",
"retrievalResult": {
"cards": [
{
"name": "林澈",
"type": "character",
"sourceRefs": [copy.deepcopy(supplemental["sourceRef"])],
}
],
"factEvidence": [],
"proseEvidence": [supplemental],
"indexHints": [
{
"cardId": "card-1",
"name": "林澈",
"type": "character",
"content": "林澈在冻结点仍位于圣蒂曼",
"sourceId": "fixture:card:1",
"sourceVersion": "cards-frozen-488-v1",
"asOf": 488,
}
],
"manifest": {"omittedSources": []},
},
"fineOutline": {
"sourceRef": {
"sourceId": "fixture:fine-outline:489",
"sourceVersion": "fine-outline-fixture-v1",
"chapter": 489,
},
"hardConstraints": ["必须完成围攻突围"],
"adjustableBeats": [],
"declaredNewFacts": [],
"entities": [{"id": "character:林澈", "type": "character", "name": "林澈"}],
"relations": [],
"items": [{"id": "item:旧徽章", "type": "item", "name": "旧徽章"}],
"locations": [{"id": "location:圣蒂曼", "type": "location", "name": "圣蒂曼"}],
"powerSystems": [],
},
"narrativeState": {
"time": "围攻当日",
"location": "圣蒂曼",
"characterPositions": {"林澈": "城内"},
"immediateSituation": "围攻持续",
},
"recentChapters": recent,
"outputContract": {
"targetChars": 4000,
"minChars": 2800,
"maxChars": 5200,
"frontmatterRequired": False,
},
"tokenBudget": {"maxContextChars": 50000},
"generatedAt": AUTHORIZATION_CHECKED_AT,
"requirements": {
"requiredEvents": [
{"requirementId": "event-1", "anchors": ["完成围攻突围"]}
],
"requiredCharacters": ["林澈"],
"foreshadowingActions": [
{"requirementId": "foreshadow-1", "anchors": ["旧徽章"]}
],
"chapterEndHook": {
"requirementId": "hook-1",
"anchors": ["城门忽然打开"],
"maxDistanceFromEnd": 20,
},
"detectedNewSettings": [],
"frozenConflicts": [],
},
}
def config() -> dict[str, object]:
"""构造不含原文全文的单样本预注册配置。"""
common = {
"workId": 8,
"asOfChapter": 488,
"targetChapter": 489,
"outlineSource": "outline:481-488",
"fineOutlineSource": "scaffold:489",
"targetChars": 4000,
"modelVersion": MODEL_POLICY_VERSION,
"sampling": {"temperature": 0, "topP": 1, "seed": 489},
"detectorProfile": "writer-detector-report-v1",
}
value = {
"profile": "writer_replay",
"evaluationSetVersion": "writer-gate-a-test-v1",
"strategyVersion": "writer-abc-v1",
"referenceWork": {
"id": 8,
"title": "深空之影",
"version": AUTHORIZATION["sourceVersion"],
},
"authorization": copy.deepcopy(AUTHORIZATION),
"commonControls": common,
"samples": [
{
"sampleId": "deep-space-489",
"scenario": "battle",
"asOfChapter": 488,
"targetChapter": 489,
"snapshotVersion": "writer-deep-space-489-v1",
"frozenRecentHanCounts": [4600, 4600, 4800, 4800],
"targetLengthBasis": {
"algorithm": "calculate_target_chars",
"sourceChapters": [485, 486, 487, 488],
"hardEventCount": 1,
"foreshadowingActionCount": 0,
"requiredSceneCount": 0,
"minChars": 2000,
"maxChars": 10000,
"usesTargetChapterLength": False,
},
"targetChars": 4000,
"expectedLength": {
"targetChars": 4000,
"minChars": 2800,
"maxChars": 5200,
},
"newCharacterRatio": 0.0,
"newCharacterRatioStatus": "resolved",
"newCharacterBasis": {
"definition": "named_required_characters_absent_before_as_of_ratio",
"asOfChapter": 488,
"requiredCharacters": ["林澈"],
"knownBeforeAsOf": ["林澈"],
"absentBeforeAsOf": [],
"genericRoles": [],
},
"sources": [
{
"sourceId": "fixture:chapter:485-488",
"sourceVersion": AUTHORIZATION["sourceVersion"],
"chapterRange": "485-488",
},
{
"sourceId": "fixture:card-index:488",
"sourceVersion": "cards-frozen-488-v1",
"chapterRange": "1-488",
},
],
"snapshotData": {
"chapters": [
{
"chapter": 488,
"sourceId": "fixture:chapter:488",
"contentSha256": "sha256:" + "1" * 64,
}
],
"cards": [],
},
"writerContextInput": _writer_context_input(),
"leakageAudit": {
"method": "target-fact-hash-and-chapter-bound-audit",
"targetFacts": {
"targetChapter": 489,
"forbiddenFacts": [
{
"id": "target-489-1",
"firstChapter": 489,
"text": "目标章专属秘密",
}
],
},
},
}
],
}
value["executionAuthorization"] = _test_execution_authorization()
return value
def _candidate_body(marker: str) -> str:
"""构造满足动态篇幅和全部机械锚点的三臂候选。"""
prefix = f"林澈{marker}完成围攻突围,又把旧徽章压回掌心。"
hook = "城门忽然打开"
filler_count = 4000 - han_count(prefix) - han_count(hook)
return prefix + "文" * filler_count + hook
def _writer_output(
context: dict[str, object], *, call_index: int
) -> dict[str, object]:
"""从治理调用的冻结 JSON 输入构造最小 WriterDraft v2。"""
# A/C 投影不得泄露臂策略;测试按预注册调用顺序区分候选,不向输入补旧字段。
marker = ("甲", "乙", "丙")[call_index % 3]
body = _candidate_body(marker)
return {"candidateBody": body}
class FakeGovernedChat:
"""只替换治理模型调用,候选仍由 run_writer() 解析和校验。"""
def __init__(self, *, illegal_extra_field: bool = False):
self.calls: list[tuple[str, dict[str, object]]] = []
self.contexts: list[dict[str, object]] = []
self.illegal_extra_field = illegal_extra_field
def __call__(self, prompt: str, **kwargs: object):
context = json.loads(prompt)
self.assert_safe_projection(context)
self.calls.append((prompt, kwargs))
self.contexts.append(context)
output = _writer_output(context, call_index=len(self.contexts) - 1)
if self.illegal_extra_field and len(self.contexts) == 2:
output["candidateSha256"] = "sha256:" + "0" * 64
return (
json.dumps(output, ensure_ascii=False),
{"input_tokens": 10, "output_tokens": 10},
FIXED_OPUS_MODEL_ID,
)
@staticmethod
def assert_safe_projection(context: dict[str, object]) -> None:
"""证明模型只收到 WriterCreativeInput v2;篇幅修订调用可额外带一个 lengthRevision 块。"""
expected = {
"fineOutline",
"narrativeState",
"factConstraints",
"proseExcerpts",
"patternReferences",
"lengthContract",
"styleConstraints",
}
keys = set(context)
if keys != expected and keys != expected | {"lengthRevision"}:
raise AssertionError(f"writer 创作投影字段非法: {sorted(context)}")
if "lengthRevision" in context:
# lengthRevision 只允许携带模型自己的上一版正文、篇幅数字与固定修订指令,
# 不得混入臂策略、raw 路径等投影外字段。
allowed_revision = {
"previousDraft",
"actualHanChars",
"minChars",
"maxChars",
"targetChars",
"countingRule",
"revisionDirection",
"targetDeltaHanChars",
"instruction",
}
revision = context["lengthRevision"]
if not isinstance(revision, dict) or set(revision) != allowed_revision:
raise AssertionError(
f"lengthRevision 块字段非法: {sorted(revision) if isinstance(revision, dict) else revision}"
)
if revision["countingRule"] != "han_chars_only":
raise AssertionError("lengthRevision 计数口径非法")
if revision["revisionDirection"] not in {"expand", "trim"}:
raise AssertionError("lengthRevision 修订方向非法")
numeric_fields = {
"actualHanChars",
"minChars",
"maxChars",
"targetChars",
"targetDeltaHanChars",
}
if any(
not isinstance(revision[field], int) or isinstance(revision[field], bool)
for field in numeric_fields
):
raise AssertionError("lengthRevision 数字字段非法")
expected_delta = abs(revision["targetChars"] - revision["actualHanChars"])
if revision["targetDeltaHanChars"] != expected_delta:
raise AssertionError("lengthRevision 距目标差值未按当前正文重算")
expected_direction = (
"expand"
if revision["actualHanChars"] < revision["targetChars"]
else "trim"
)
if revision["revisionDirection"] != expected_direction:
raise AssertionError("lengthRevision 方向未按当前正文重算")
class InterruptingChat(FakeGovernedChat):
"""模拟外部终止落在模型调用中间。"""
def __init__(self, *, interrupt_on_call: int) -> None:
super().__init__()
self.interrupt_on_call = interrupt_on_call
def __call__(self, prompt: str, **kwargs: object):
if len(self.calls) + 1 == self.interrupt_on_call:
signal.raise_signal(signal.SIGTERM)
return super().__call__(prompt, **kwargs)
class FakeSemanticDetector:
"""记录机械门输出,并返回与候选绑定的语义通过报告。"""
def __init__(self) -> None:
self.calls: list[tuple[str, bool]] = []
def __call__(self, context, candidate, mechanical_report):
strategy = str(context.get("evidenceStrategy") or "neutral_prose_projection")
self.calls.append((strategy, mechanical_report["passed"]))
return {
"status": "passed",
"candidateVersion": candidate["candidateVersion"],
"candidateSha256": candidate["candidateSha256"],
"blockingFailures": [],
"suggestions": [],
}
def judge_report(
reviewer_id: str,
sample_id: str,
blind_id: str,
order: list[str],
score: float = 8.0,
) -> dict[str, object]:
"""构造测试盲评报告。"""
return {
"profile": RUBRIC_PROFILE,
"reviewerId": reviewer_id,
"sampleId": sample_id,
"blindCandidateId": blind_id,
"candidateOrder": order,
"scores": {
dimension: {
"score": score,
"evidence": [
{
"sourceType": "judge_inference",
"sourceRef": f"candidate:{blind_id}",
"excerpt": f"证据-{dimension}",
}
],
}
for dimension in DIMENSIONS
},
}
class FakeJudge:
"""模拟双评差异和必要第三评,不读取网络或真实模型。"""
def __init__(
self,
*,
unstable: bool = False,
invalid_third: bool = False,
irreducibly_unstable: bool = False,
):
self.calls: list[tuple[str, str]] = []
self.unstable = unstable
self.invalid_third = invalid_third
self.irreducibly_unstable = irreducibly_unstable
def __call__(self, blind_input, reviewer_id):
blind_id = blind_input["blindCandidateId"]
candidate_order = blind_input["candidateOrder"]
self.calls.append((reviewer_id, blind_id))
if self.invalid_third and reviewer_id == "judge-3":
return {"status": "malformed"}
score = 8.0
if self.unstable and blind_id == "blind-1" and reviewer_id == "judge-2":
score = 7.0
if self.unstable and blind_id == "blind-1" and reviewer_id == "judge-3":
score = 7.5
if self.irreducibly_unstable and blind_id == "blind-1":
score = {"judge-1": 8.0, "judge-2": 6.0, "judge-3": 7.0}[reviewer_id]
return judge_report(
reviewer_id,
blind_input["sample"]["sampleId"],
blind_id,
candidate_order,
score,
)
class MaliciousJudge(FakeJudge):
"""主动扫描全部可见输入,证明 judge 无法发现真实臂或原始目录。"""
def __init__(self) -> None:
super().__init__()
self.visible_inputs: list[dict[str, object]] = []
def __call__(self, blind_input, reviewer_id):
rendered = json.dumps(blind_input, ensure_ascii=False, sort_keys=True)
forbidden = (
"candidate-A",
"candidate-B",
"candidate-C",
"pipeline-A",
"pipeline-B",
"pipeline-C",
"evidenceStrategy",
"writerContextInput",
"indexHints",
"retrievalManifest",
"evidenceCoverage",
"raw_dir",
"rawDir",
"_contexts",
)
for marker in forbidden:
if marker in rendered:
raise AssertionError(f"judge 可见输入泄露: {marker}")
self.visible_inputs.append(copy.deepcopy(blind_input))
return super().__call__(blind_input, reviewer_id)
def _test_adapters(
*,
unstable: bool = False,
illegal_extra_field: bool = False,
invalid_third: bool = False,
irreducibly_unstable: bool = False,
) -> tuple[WriterReplayTestAdapters, FakeGovernedChat, FakeSemanticDetector, FakeJudge]:
"""集中构造测试注入对象,避免它们被误认为生产 runtime。"""
runner = FakeGovernedChat(illegal_extra_field=illegal_extra_field)
detector = FakeSemanticDetector()
judge = FakeJudge(
unstable=unstable,
invalid_third=invalid_third,
irreducibly_unstable=irreducibly_unstable,
)
profile = _frozen_test_profile()
return (
WriterReplayTestAdapters(runner, profile, detector, judge),
runner,
detector,
judge,
)
def _frozen_test_profile():
"""构造不访问网络且字段完整的冻结 writer 测试 profile。"""
return build_writer_execution_profile(
max_budget_usd_per_call=Decimal("1.000000"),
timeout_seconds=30,
max_context_chars=200000,
system_prompt="只返回严格 WriterDraft v2。",
)
def _production_writer_profile():
"""构造生产编排测试使用的 provider-neutral writer profile。"""
return build_writer_execution_profile(
max_budget_usd_per_call=Decimal("1.000000"),
timeout_seconds=30,
max_context_chars=200000,
system_prompt="只返回严格 WriterDraft v2。",
)
def _test_execution_authorization() -> dict[str, object]:
"""给测试适配器显式审批 probe、预算、raw 和 writer profile,禁止参数隐式放行。"""
writer = _frozen_test_profile()
probe = {
"status": "successful",
"runtimeAdapter": RUNTIME_ADAPTER,
"runtimeAdapterVersion": RUNTIME_ADAPTER_VERSION,
"modelPolicyVersion": MODEL_POLICY_VERSION,
"modelAlias": writer.model_alias,
"executionProfileSha256": writer.execution_profile_sha256,
"structuredOutputSha256": "sha256:" + "6" * 64,
}
budget = {
"status": "approved",
"authorizationId": "budget-test-adapter-1",
"approvedBy": "user",
"totalBudgetUsd": "9.000000",
"plannedCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
"maxCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
}
raw = {
"status": "approved",
"authorizationId": "raw-test-adapter-1",
"approvedBy": "user",
"retainUntil": (datetime.now(UTC) + timedelta(hours=1)).isoformat(),
}
for item in (probe, budget, raw):
item["receiptSha256"] = canonical_sha256(item)
return {
"runtimeProbe": probe,
"budget": budget,
"rawRetention": raw,
"profileSha256": {"writer": writer.execution_profile_sha256},
}
def _role_profile(role: str, schema: dict[str, object]) -> RoleExecutionProfile:
"""构造语义或盲评使用的完整冻结测试 profile。"""
prompt = f"只返回严格 {role} structured output。"
return RoleExecutionProfile(
profile_version="writer-eval-profile-v2",
adapter_role=role,
model_alias=FIXED_OPUS_POLICY_ALIAS,
model_policy_version=FIXED_OPUS_POLICY_VERSION,
resolved_model_id=FIXED_OPUS_MODEL_ID,
max_budget_usd_per_call=Decimal("1.000000"),
timeout_seconds=30,
max_context_chars=200000,
json_schema_id=f"{role}-schema-v1",
json_schema=schema,
json_schema_sha256=sha256_json(schema),
system_prompt_id=f"{role}-prompt-v1",
system_prompt=prompt,
system_prompt_sha256=sha256_text(prompt),
)
def _expected_total_cost(*, fixed_role_calls: int, writer_calls: int = 3) -> str:
"""按当前治理费率计算测试总成本,避免把 provider 费率硬编码进编排断言。"""
writer_cost = Decimal(
str(
replay_module.muse_llm.cost_fixed_opus(
{"input_tokens": 10, "output_tokens": 10},
)
)
)
total = Decimal("0.010000") * fixed_role_calls + writer_cost * writer_calls
return format(total.quantize(Decimal("0.000001")), "f")
def _fake_model_receipt(
role: str,
invocation: int,
model_input: dict[str, object],
structured_output: dict[str, object],
) -> dict[str, object]:
"""构造字段完整、模型匹配且输入/输出哈希均冻结的 fake ExecutionReceipt。"""
model_id = FIXED_OPUS_MODEL_ID
return {
"adapterRole": role,
"invocationId": f"{role}-{invocation}",
"executionProfileSha256": canonical_sha256({"role": role, "profile": 1}),
"requestedModelId": FIXED_OPUS_POLICY_ALIAS,
"actualModelId": model_id,
"modelMatch": True,
"effort": "governed",
"maxBudgetUsdPerCall": "1.000000",
"totalCostUsd": "0.010000",
"usage": {"input_tokens": 1, "output_tokens": 1},
"modelUsage": {model_id: {"costUSD": "0.010000"}},
"stopReason": "stop",
"terminalReason": "completed",
"isError": False,
"apiErrorStatus": None,
"exitCode": None,
"durationMs": 1,
"inputSha256": canonical_sha256(model_input),
"structuredOutputSha256": canonical_sha256(structured_output),
"jsonSchemaSha256": canonical_sha256({"role": role, "schema": 1}),
}
class ProductionSemanticRunner:
"""动态回放 semantic v3 的最小模型草稿。"""
def __init__(self, *, fail_on_call: int | None = None) -> None:
self.fail_on_call = fail_on_call
self.calls: list[dict[str, object]] = []
self.receipts: list[dict[str, object]] = []
self.structured_outputs: list[dict[str, object]] = []
def run(self, *, adapter_role, model_input, output_schema):
"""生成合法报告;指定调用可返回真实 failed 语义终态。"""
self.calls.append(copy.deepcopy(dict(model_input)))
failed = self.fail_on_call == len(self.calls)
body = model_input["candidateBody"]
constraints = model_input["hardConstraints"]
quote = body[:3]
evidence_id = constraints[0]["constraintId"] if constraints else "candidate-body"
draft = {
"schemaVersion": "semantic-detection-draft-v3",
"claims": [],
"findings": (
[
{
"findingId": "finding-high-1",
"severity": "high",
"category": "hard_constraint",
"candidateQuote": quote,
"evidenceIds": [evidence_id],
"message": "硬约束未满足",
}
]
if failed
else []
),
"assertionVerdicts": [
{
"assertionId": item["assertionId"],
"verdict": "pass",
"candidateQuote": quote,
"evidenceIds": [item["evidenceId"]],
}
for item in model_input["factEvidence"]
],
"hardConstraintVerdicts": [
{
"constraintId": item["constraintId"],
"verdict": "fail" if failed else "pass",
"candidateQuote": quote,
"evidenceIds": [item["constraintId"]],
}
for item in constraints
],
"newSettingCandidates": [],
"evidenceGaps": [],
}
receipt = _fake_model_receipt(
adapter_role,
len(self.calls),
dict(model_input),
draft,
)
self.receipts.append(receipt)
self.structured_outputs.append(copy.deepcopy(draft))
receipt_hash = sha256_json(receipt)
if not is_semantic_schema_specialization(
SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
output_schema,
):
raise AssertionError("semantic adapter 未传正式 v3 闭集特化 schema")
return {"structuredOutput": draft, "modelReceiptSha256": receipt_hash}
class NeedsEvidenceSemanticRunner(ProductionSemanticRunner):
"""在指定调用生成合同合法但需要补证的 detector 草稿。"""
def __init__(self, *, needs_evidence_on_call: int | None = None) -> None:
super().__init__()
self.needs_evidence_on_call = needs_evidence_on_call
def run(self, *, adapter_role, model_input, output_schema):
result = super().run(
adapter_role=adapter_role,
model_input=model_input,
output_schema=output_schema,
)
if (
self.needs_evidence_on_call is not None
and len(self.calls) != self.needs_evidence_on_call
):
return result
draft = result["structuredOutput"]
draft["evidenceGaps"] = [
{
"gapId": "gap-1",
"query": "补查候选新事实",
"reason": "冻结证据不足",
"priority": "high",
"candidateQuote": model_input["candidateBody"][:3],
}
]
self.structured_outputs[-1] = copy.deepcopy(draft)
self.receipts[-1]["structuredOutputSha256"] = canonical_sha256(draft)
result["modelReceiptSha256"] = sha256_json(self.receipts[-1])
return result
class InvalidQuoteSemanticRunner(ProductionSemanticRunner):
"""每轮都返回候选中不存在的引文,稳定耗尽 detector 纠错。"""
def run(self, *, adapter_role, model_input, output_schema):
result = super().run(
adapter_role=adapter_role,
model_input=model_input,
output_schema=output_schema,
)
draft = result["structuredOutput"]
for field in ("assertionVerdicts", "hardConstraintVerdicts"):
for item in draft[field]:
item["candidateQuote"] = "候选中不存在的引文"
self.structured_outputs[-1] = copy.deepcopy(draft)
self.receipts[-1]["structuredOutputSha256"] = canonical_sha256(draft)
result["modelReceiptSha256"] = sha256_json(self.receipts[-1])
return result
class CorrectingSemanticRunner(ProductionSemanticRunner):
"""首轮引文非法、带 correction 的第二轮恢复合法。"""
def run(self, *, adapter_role, model_input, output_schema):
result = super().run(
adapter_role=adapter_role,
model_input=model_input,
output_schema=output_schema,
)
if len(self.calls) != 1:
return result
draft = result["structuredOutput"]
for field in ("assertionVerdicts", "hardConstraintVerdicts"):
for item in draft[field]:
item["candidateQuote"] = "候选中不存在的首轮引文"
self.structured_outputs[-1] = copy.deepcopy(draft)
self.receipts[-1]["structuredOutputSha256"] = canonical_sha256(draft)
result["modelReceiptSha256"] = sha256_json(self.receipts[-1])
return result
class ProductionJudgeRunner:
"""动态回放 blind judge v3 模型草稿,并可制造第三评不稳定。"""
def __init__(
self,
*,
unstable: bool = False,
invalid_first_quote: bool = False,
api_error_first: bool = False,
) -> None:
self.unstable = unstable
self.invalid_first_quote = invalid_first_quote
self.api_error_first = api_error_first
self.calls: list[dict[str, object]] = []
self.receipts: list[dict[str, object]] = []
self.structured_outputs: list[dict[str, object]] = []
def run(self, *, adapter_role, model_input, output_schema):
"""按 reviewer 次序生成严格绑定的五维评分与 verdict。"""
self.calls.append(copy.deepcopy(dict(model_input)))
call_index = len(self.calls)
score = 8.0
if self.unstable:
score = {1: 8.0, 2: 6.0, 3: 7.0}.get(call_index, 7.0)
assertion_ids = [
item["assertionId"]
for field in ("historicalAssertions", "targetAssertions")
for item in model_input["oracleTruthPack"][field]
]
constraint_ids = [
item["constraintId"] for item in model_input["fineOutline"]["hardConstraints"]
]
candidate_ids = [item["blindCandidateId"] for item in model_input["candidates"]]
draft = {
"schemaVersion": "blind-judge-draft-v3",
"candidateScores": [
{
"blindCandidateId": candidate["blindCandidateId"],
"scores": {
dimension: {
"score": score,
"reason": f"{dimension} 维度满足要求",
"candidateQuote": candidate["candidateBody"][:3],
"evidenceRefs": [{"sourceType": "candidate", "sourceId": candidate["blindCandidateId"]}],
}
for dimension in DIMENSIONS
},
}
for candidate in model_input["candidates"]
],
"dimensionPreferences": [
{
"dimension": dimension,
"orderedCandidateIds": list(candidate_ids),
"reason": f"按 {dimension} 维度比较",
}
for dimension in DIMENSIONS
],
"oracleAssertionVerdicts": [
{
"assertionId": assertion_id,
"blindCandidateId": candidate["blindCandidateId"],
"verdict": "pass",
"reason": "候选与 oracle 断言一致",
"candidateQuote": candidate["candidateBody"][:3],
"evidenceRefs": [{"sourceType": "oracle_assertion", "sourceId": assertion_id}],
}
for candidate in model_input["candidates"]
for assertion_id in assertion_ids
],
"hardConstraintVerdicts": [
{
"constraintId": constraint_id,
"blindCandidateId": candidate["blindCandidateId"],
"verdict": "pass",
"reason": "候选满足细纲硬约束",
"candidateQuote": candidate["candidateBody"][:3],
"evidenceRefs": [{"sourceType": "fine_outline", "sourceId": constraint_id}],
}
for candidate in model_input["candidates"]
for constraint_id in constraint_ids
],
}
if self.invalid_first_quote and call_index == 1:
draft["candidateScores"][0]["scores"][DIMENSIONS[0]][
"candidateQuote"
] = "候选正文中不存在的首轮引文"
receipt = _fake_model_receipt(
adapter_role,
call_index,
dict(model_input),
draft,
)
if self.api_error_first and call_index == 1:
receipt.update(
{
"actualModelId": None,
"modelMatch": False,
"totalCostUsd": "0.000000",
"terminalReason": "api_error",
"isError": True,
"apiErrorStatus": None,
"exitCode": 1,
"structuredOutputSha256": None,
}
)
self.receipts.append(receipt)
raise RoleRuntimeError(
"BLIND_JUDGE_API_ERROR",
"测试瞬时 API 错误",
receipt=receipt,
)
self.receipts.append(receipt)
self.structured_outputs.append(copy.deepcopy(draft))
receipt_hash = sha256_json(receipt)
if output_schema != BLIND_JUDGE_REPORT_JSON_SCHEMA:
raise AssertionError("judge adapter 未传正式 v3 model schema")
return {"structuredOutput": draft, "modelReceiptSha256": receipt_hash}
def _production_config(*, budget_approved: bool = True) -> dict[str, object]:
"""构造 probe、预算、raw 和三 profile 哈希均可复核的执行配置。"""
value = config()
context_input = value["samples"][0]["writerContextInput"]
context_input["contentMode"] = "canonical_frozen_prose"
prose_a = {**_prose(120, 2120, "甲侧旧徽章发出微光。"), "retrievalArm": "A"}
prose_c = {**_prose(121, 2121, "丙侧旧徽章传来回响。"), "retrievalArm": "C"}
context_input["retrievalResult"]["proseEvidence"] = [prose_a, prose_c]
value["samples"][0]["proseCharBudget"] = len(prose_a["text"])
value["commonControls"]["modelVersion"] = MODEL_POLICY_VERSION
value["commonControls"]["adapterVersion"] = RUNTIME_ADAPTER_VERSION
writer = _production_writer_profile()
semantic = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
judge = _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA)
value["commonControls"]["maxContextChars"] = writer.max_context_chars
probe = {
"status": "successful",
"runtimeAdapter": RUNTIME_ADAPTER,
"runtimeAdapterVersion": RUNTIME_ADAPTER_VERSION,
"modelPolicyVersion": MODEL_POLICY_VERSION,
"modelAlias": writer.model_alias,
# 探针合同绑定:身份哈希绑定当前 writer 合同,证明它是按新 schema/prompt 重跑的;
# structuredOutputSha256 是合规结构化输出内容哈希,证明探针确实跑出过合规输出。
"executionProfileSha256": writer.execution_profile_sha256,
"structuredOutputSha256": "sha256:" + "6" * 64,
}
budget = {
"status": "approved" if budget_approved else "pending",
"authorizationId": "budget-test-1",
"approvedBy": "user",
"totalBudgetUsd": "9.000000",
"plannedCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
"maxCalls": {"writer": 3, "semantic_detector": 3, "blind_judge": 3},
}
raw = {
"status": "approved",
"authorizationId": "raw-test-1",
"approvedBy": "user",
"retainUntil": (datetime.now(UTC) + timedelta(hours=1)).isoformat(),
}
for item in (probe, budget, raw):
item["receiptSha256"] = canonical_sha256(item)
value["executionAuthorization"] = {
"runtimeProbe": probe,
"budget": budget,
"rawRetention": raw,
"profileSha256": {
"writer": writer.execution_profile_sha256,
"semantic_detector": semantic.execution_profile_sha256,
"blind_judge": judge.execution_profile_sha256,
},
}
return value
def _production_config_with_writer_budget(
*, planned: int, max_calls: int, total_usd: str
) -> dict[str, object]:
"""在 _production_config 基础上单独抬高 writer 预算,供篇幅修订环的多调用测试使用。
修订环会让每臂的 writer 调用数超过基础的 1 次,因此需要更大的 plannedCalls/maxCalls;
totalBudgetUsd 必须覆盖 caps*planned 的最坏预留,否则预算门会失败关闭。改动 budget
字段后必须按授权门规则重签 receiptSha256。
"""
value = _production_config()
budget = value["executionAuthorization"]["budget"]
budget["plannedCalls"]["writer"] = planned
budget["maxCalls"]["writer"] = max_calls
budget["totalBudgetUsd"] = total_usd
budget["receiptSha256"] = canonical_sha256(
{key: val for key, val in budget.items() if key != "receiptSha256"}
)
return value
def _semantic_call_index_for_arm(evaluation_config: dict[str, object], arm: str) -> int:
"""返回单样本预注册顺序中指定臂对应的 detector 调用序号。"""
preregistration = build_balanced_preregistration(
evaluation_set_version=str(evaluation_config["evaluationSetVersion"]),
sample_ids=[str(evaluation_config["samples"][0]["sampleId"])],
)
order = preregistration["armOrderTable"][0]["armOrder"]
return order.index(arm) + 1
def _five_scenario_gate_input(gate_input: dict[str, object]) -> dict[str, object]:
"""把单样本生产链结果扩成五场景,只用于机械验证 Gate A 的 C 臂裁决。"""
scenarios = (
"battle",
"character_dialogue",
"turning_point",
"information_reveal",
"returning_character",
)
expanded = copy.deepcopy(gate_input)
template = expanded["samples"][0]
expanded["samples"] = []
for index, scenario in enumerate(scenarios, start=1):
sample = copy.deepcopy(template)
sample["sampleId"] = f"gate-a-c-failed-{index}"
sample["scenario"] = scenario
expanded["samples"].append(sample)
expanded.pop("builderReceiptSha256", None)
expanded.pop("gateInputSha256", None)
expanded["builderReceiptSha256"] = canonical_sha256(expanded)
expanded["gateInputSha256"] = canonical_sha256(expanded)
return expanded
def _oracle_pack() -> dict[str, object]:
"""构造正式 loader 形状的 oracle;执行时再投影成 reviewer 最小合同。"""
target_statement = "目标章不得提前泄露终局真相"
historical = []
for chapter in range(485, 489):
statement = f"林澈在第{chapter}章仍位于圣蒂曼"
historical.append(
{
"assertionId": f"assertion-history-{chapter}",
"assertionType": "canonical_history",
"statement": statement,
"sourceVersion": f"canonical-v{chapter}",
"chapterBoundary": {"minChapter": chapter, "maxChapter": chapter},
"contentSha256": "sha256:"
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
}
)
payload = {
"schemaVersion": "oracle-truth-pack-v1",
"evaluationSetVersion": "writer-gate-a-test-v1",
"sampleId": "deep-space-489",
"workId": 8,
"asOf": 488,
"sourceSnapshotSha256": "sha256:" + "7" * 64,
"authorizationSnapshotId": "auth-work-8",
"authorization": {
"allowedPurpose": "offline_evaluation",
"sourceStatus": "authorized",
"sourceVersion": "canonical-v488",
"revalidationAt": "2026-07-25T00:00:00+00:00",
"evaluatorOnly": True,
"snapshotId": "auth-work-8",
},
"historicalAssertions": historical,
"targetAssertions": [
{
"assertionId": "assertion-target-1",
"assertionType": "target_reference_scaffold",
"statement": target_statement,
"sourceVersion": "canonical-v488",
"chapterBoundary": {"minChapter": 489, "maxChapter": 489},
"contentSha256": "sha256:" + hashlib.sha256(target_statement.encode("utf-8")).hexdigest(),
}
],
}
return {**payload, "packSha256": canonical_sha256(payload)}
def _production_adapters(
*,
semantic_fail_on: int | None = None,
judge_unstable: bool = False,
judge_invalid_first_quote: bool = False,
judge_api_error_first: bool = False,
vault_factory=RawVaultManager,
cas_factory=FileCasStore,
builder_factory=GateInputBuilder,
):
"""集中构造 production fake,所有模型结果都从 structuredOutput 返回。"""
writer_runner = FakeGovernedChat()
semantic_runner = ProductionSemanticRunner(fail_on_call=semantic_fail_on)
judge_runner = ProductionJudgeRunner(
unstable=judge_unstable,
invalid_first_quote=judge_invalid_first_quote,
api_error_first=judge_api_error_first,
)
writer_profile = _production_writer_profile()
semantic_profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
judge_profile = _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA)
adapters = WriterReplayProductionAdapters(
writer_runner=writer_runner,
writer_profile=writer_profile,
semantic_model_runner=semantic_runner,
semantic_profile=semantic_profile,
judge_model_runner=judge_runner,
judge_profile=judge_profile,
oracle_truth_packs={"deep-space-489": _oracle_pack()},
vault_manager_factory=vault_factory,
cas_store_factory=cas_factory,
gate_input_builder_factory=builder_factory,
)
return adapters, writer_runner, semantic_runner, judge_runner
class WriterReplayDryRunTest(unittest.TestCase):
"""验证 dry-run 的合同、冻结、安全和控制变量。"""
def test_pattern_references_are_included_in_leakage_audit(self):
"""范式卡若含目标章禁用事实,dry-run 必须在装配各臂前失败关闭。"""
evaluation_config = config()
evaluation_config["samples"][0]["writerContextInput"]["patternReferences"] = [
{
"sourceId": "fixture:public-pattern:1",
"sourceVersion": "public-pattern-v1",
"name": "转折范式",
"summary": "目标章专属秘密",
"writingPoints": {"转折": "先压后扬"},
}
]
result = run_writer_replay(evaluation_config, run_id="pattern-leakage-dry-run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_leakage")
self.assertEqual(result["samples"][0]["status"], "invalid_snapshot")
findings = result["samples"][0]["leakageAudit"]["findings"]
self.assertTrue(
any("patternReferences" in finding["path"] for finding in findings),
findings,
)
@staticmethod
def _ledger(
*,
planned: int = 1,
total: str = "3.000000",
pre_call_guard=None,
):
"""构造只用于账本边界测试的三角色 profile。"""
profiles = {
"writer": _role_profile("writer", replay_module.WRITER_OUTPUT_JSON_SCHEMA),
"semantic_detector": _role_profile(
"semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA
),
"blind_judge": _role_profile("blind_judge", BLIND_JUDGE_REPORT_JSON_SCHEMA),
}
return replay_module._ExecutionBudgetLedger(
total_budget=Decimal(total),
planned_calls={role: planned for role in replay_module.BUDGET_ROLES},
max_calls={role: planned for role in replay_module.BUDGET_ROLES},
profiles=profiles,
pre_call_guard=pre_call_guard,
)
def test_raw_guard_runs_before_budget_slot_is_consumed(self):
"""下一调用租期不足时,不得消耗计划槽位或制造在途成本。"""
checked_roles = []
def reject(role):
checked_roles.append(role)
raise RawVaultError(
"RAW_LEASE_INSUFFICIENT_RETENTION",
"测试下一调用租期不足",
)
ledger = self._ledger(pre_call_guard=reject)
with self.assertRaisesRegex(RawVaultError, "测试下一调用租期不足"):
ledger.begin("writer")
self.assertEqual(checked_roles, ["writer"])
self.assertEqual(
ledger.snapshot()["usedCalls"],
{"writer": 0, "semantic_detector": 0, "blind_judge": 0},
)
self.assertEqual(ledger.snapshot()["inFlightRoles"], [])
self.assertFalse(ledger.snapshot()["costUnknown"])
def test_raw_call_window_uses_one_role_timeout_plus_cleanup_margin(self):
"""固定时钟证明单次 30 秒 timeout 加 60 秒清理余量的精确边界。"""
now = datetime(2026, 7, 27, 0, 0, tzinfo=UTC)
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
replay_module._assert_raw_call_window(
(now + timedelta(seconds=90)).isoformat(),
profile,
now=now,
)
with self.assertRaises(RawVaultError) as raised:
replay_module._assert_raw_call_window(
(now + timedelta(seconds=89)).isoformat(),
profile,
now=now,
)
self.assertEqual(raised.exception.code, "RAW_LEASE_INSUFFICIENT_RETENTION")
def test_budget_ledger_settles_failed_call_receipt_before_reraising(self):
"""模型失败但带可信回执时,实际成本仍必须进入账本。"""
ledger = self._ledger()
receipt = _fake_model_receipt(
"semantic_detector", 1, {"input": "x"}, {"output": "y"}
)
class FailedDelegate:
receipts = []
structured_outputs = []
def run(self, **_kwargs):
error = RuntimeError("模型失败")
error.receipt = copy.deepcopy(receipt)
raise error
runner = replay_module._BudgetedModelRunner(
FailedDelegate(), ledger, "semantic_detector"
)
with self.assertRaisesRegex(RuntimeError, "模型失败"):
runner.run(
adapter_role="semantic_detector",
model_input={},
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
)
snapshot = ledger.snapshot()
self.assertEqual(snapshot["usedCalls"]["semantic_detector"], 1)
self.assertEqual(snapshot["totalActualCostUsd"], "0.010000")
self.assertFalse(snapshot["costUnknown"])
def test_budget_ledger_without_failure_receipt_is_cost_unknown_and_stops_followups(self):
"""已发起调用但无可信成本时不能按零美元继续后续角色。"""
ledger = self._ledger()
class FailedDelegate:
receipts = []
structured_outputs = []
def run(self, **_kwargs):
raise RuntimeError("无回执")
runner = replay_module._BudgetedModelRunner(
FailedDelegate(), ledger, "semantic_detector"
)
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
runner.run(
adapter_role="semantic_detector",
model_input={},
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
)
snapshot = ledger.snapshot()
self.assertTrue(snapshot["costUnknown"])
self.assertEqual(snapshot["totalActualCostUsd"], "0.000000")
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
ledger.begin("writer")
def test_budget_ledger_rejects_duplicate_or_out_of_order_settlement(self):
"""一个调用只能由自己的 token 结算一次,不能重复计费。"""
ledger = self._ledger()
token = ledger.begin("writer")
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
ledger.complete("writer", token, receipt)
with self.assertRaisesRegex(BudgetLedgerError, "token 错序或重复"):
ledger.complete("writer", token, receipt)
self.assertEqual(ledger.snapshot()["totalActualCostUsd"], "0.010000")
def test_budget_settlement_keeps_in_flight_until_cost_is_recorded(self):
"""结算中断时在途标记必须仍在,供信号处理转成成本未知。"""
class InterruptingSet(set):
def add(self, _value):
raise replay_module.ReplayInterrupted("测试结算窗口")
ledger = self._ledger()
token = ledger.begin("writer")
ledger._settled_invocations = InterruptingSet()
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
with self.assertRaises(replay_module.ReplayInterrupted):
ledger.complete("writer", token, receipt)
self.assertEqual(ledger.snapshot()["inFlightRoles"], ["writer"])
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
ledger.mark_cost_unknown("writer", token)
self.assertTrue(ledger.snapshot()["costUnknown"])
def test_budget_ledger_rejects_missing_cost_and_over_cap(self):
"""缺成本和超过单次 cap 都必须 fail closed,不能静默按零计费。"""
missing = self._ledger()
token = missing.begin("writer")
with self.assertRaisesRegex(BudgetLedgerError, "EXECUTION_COST_UNKNOWN"):
missing.complete(
"writer",
token,
{"adapterRole": "writer", "invocationId": "writer-missing", "totalCostUsd": None},
)
self.assertTrue(missing.snapshot()["costUnknown"])
over_cap = self._ledger()
token = over_cap.begin("writer")
receipt = _fake_model_receipt("writer", 1, {"input": "x"}, {"output": "y"})
receipt["totalCostUsd"] = "2.000000"
with self.assertRaisesRegex(BudgetLedgerError, "超过单次 cap"):
over_cap.complete("writer", token, receipt)
self.assertEqual(over_cap.snapshot()["failureReason"], "EXECUTION_COST_OVER_CAP")
def test_budget_plan_requires_closed_planned_calls_and_reserves_570(self):
"""45/24/45 计划乘 $5 预留 $570;maxCalls 仍只是安全容量上限。"""
required = {role: 15 for role in replay_module.BUDGET_ROLES}
caps = {role: Decimal("5.000000") for role in replay_module.BUDGET_ROLES}
valid = {
"status": "approved",
"totalBudgetUsd": "2250.000000",
"plannedCalls": {"writer": 60, "semantic_detector": 24, "blind_judge": 45},
"maxCalls": {role: 150 for role in replay_module.BUDGET_ROLES},
}
planned, maximum, total, error = replay_module._validate_budget_plan(
valid, required_calls=required, caps=caps
)
self.assertIsNone(error)
self.assertEqual(planned, valid["plannedCalls"])
self.assertEqual(maximum, valid["maxCalls"])
self.assertEqual(total, Decimal("2250.000000"))
self.assertEqual(
sum(caps[role] * planned[role] for role in replay_module.BUDGET_ROLES),
Decimal("645.000000"),
)
cases = (
("missing", lambda budget: budget.pop("plannedCalls"), "BUDGET_PLANNED_CALLS_REQUIRED"),
("invalid-shape", lambda budget: budget["plannedCalls"].update({"extra": 1}), "BUDGET_PLANNED_CALLS_INVALID"),
("invalid-type", lambda budget: budget["plannedCalls"].update({"writer": "15"}), "BUDGET_PLANNED_CALLS_INSUFFICIENT"),
("insufficient", lambda budget: budget["plannedCalls"].update({"writer": 14}), "BUDGET_PLANNED_CALLS_INSUFFICIENT"),
("planned-over-max", lambda budget: budget["plannedCalls"].update({"writer": 151}), "BUDGET_MAX_CALLS_INSUFFICIENT"),
("insufficient-total", lambda budget: budget.update({"totalBudgetUsd": "644.999999"}), "BUDGET_AUTHORIZATION_INVALID"),
)
for name, mutate, expected in cases:
with self.subTest(name=name):
tampered = copy.deepcopy(valid)
mutate(tampered)
_planned, _maximum, _total, error = replay_module._validate_budget_plan(
tampered, required_calls=required, caps=caps
)
self.assertEqual(error, expected)
def test_length_bounds_relaxed_to_30_percent(self):
"""五个预注册目标的篇幅边界都按正负 30% 计算并受 2000-10000 限幅。"""
expected = {
7500: (5250, 9750),
7600: (5320, 9880),
6700: (4690, 8710),
2000: (2000, 2600),
6100: (4270, 7930),
}
for target, bounds in expected.items():
with self.subTest(target=target):
self.assertEqual(replay_module._length_bounds(target), bounds)
def test_base_config_budget_self_hash_and_length_bounds_consistent(self):
"""base 配置 budget 自哈希重签有效,五样本 expectedLength/输出合同等于 _length_bounds(target)。"""
gate_config = _load_gate_a_config()
budget = gate_config["executionAuthorization"]["budget"]
# 与 _validate_authorization_records 同款复核:去掉 receiptSha256 后整段重算自哈希。
self.assertEqual(
budget["receiptSha256"],
canonical_sha256({key: value for key, value in budget.items() if key != "receiptSha256"}),
)
self.assertEqual(budget["plannedCalls"]["writer"], 60)
self.assertEqual(budget["totalBudgetUsd"], "2250.000000")
for sample in gate_config["samples"]:
target = sample["targetChars"]
expected_min, expected_max = replay_module._length_bounds(target)
with self.subTest(sample=sample["sampleId"]):
self.assertEqual(
sample["expectedLength"],
{"targetChars": target, "minChars": expected_min, "maxChars": expected_max},
)
contract = sample["writerContextInput"]["outputContract"]
self.assertEqual(contract["targetChars"], target)
self.assertEqual(contract["minChars"], expected_min)
self.assertEqual(contract["maxChars"], expected_max)
def test_length_overflow_records_actual_han_chars_in_sample_result(self):
"""生产链正文越界时失败样本必须记下实际汉字数与合同边界(全是数字,不含正文)。"""
adapters, _writer, _semantic, _judge = _production_adapters()
short_body = "短" * 2000 # 2000 汉字 < 合同下限 2800,必然越界
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
with mock.patch.object(
sys.modules[__name__],
"_writer_output",
lambda context, call_index: {"candidateBody": short_body},
):
result = run_writer_replay(
_production_config_with_writer_budget(
planned=4,
max_calls=4,
total_usd="12.000000",
),
run_id="length-violation",
output_dir=pathlib.Path(directory) / "run",
execute=True,
production_adapters=adapters,
)
self.assertFalse(result["ok"])
sample_result = result["samples"][0]
self.assertIn("lengthViolation", sample_result)
self.assertEqual(
sample_result["lengthViolation"],
{"actualHanChars": 2000, "minChars": 2800, "maxChars": 5200, "targetChars": 4000},
)
def test_recording_runner_keeps_adapter_fields_out_of_model_draft(self):
"""统一 runtime 只能旁路返回回执 hash,不能污染 v3 模型草稿。"""
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
draft = {
"schemaVersion": "semantic-detection-draft-v3",
"claims": [],
"findings": [],
"assertionVerdicts": [],
"hardConstraintVerdicts": [],
"newSettingCandidates": [],
"evidenceGaps": [],
}
receipt = _fake_model_receipt("semantic_detector", 1, {"candidateBody": "甲"}, draft)
class Receipt:
def as_dict(self):
return copy.deepcopy(receipt)
class Invocation:
structured_output = copy.deepcopy(draft)
receipt = Receipt()
original = replay_module.run_role
replay_module.run_role = lambda _profile, _input: Invocation()
try:
runner = replay_module._RecordingRuntimeModelRunner(profile)
result = runner.run(
adapter_role="semantic_detector",
model_input={"candidateBody": "甲"},
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
)
finally:
replay_module.run_role = original
self.assertEqual(result["structuredOutput"], draft)
self.assertNotIn("modelReceiptSha256", result["structuredOutput"])
self.assertNotIn("reportSha256", result["structuredOutput"])
self.assertEqual(result["modelReceiptSha256"], sha256_json(receipt))
def test_recording_runner_preserves_failure_receipt_before_reraising(self):
"""失败回执必须留在 runner,外层才能保留真实错误码并结算可信费用。"""
profile = _role_profile("semantic_detector", SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA)
receipt = _fake_model_receipt(
"semantic_detector", 1, {"candidateBody": "甲"}, {"error": "api"}
)
class Receipt:
def as_dict(self):
return copy.deepcopy(receipt)
original = replay_module.run_role
def fail(_profile, _input):
raise RoleRuntimeError(
"SEMANTIC_DETECTOR_API_ERROR",
"模型调用失败",
receipt=Receipt(),
)
replay_module.run_role = fail
try:
runner = replay_module._RecordingRuntimeModelRunner(profile)
with self.assertRaises(RoleRuntimeError) as caught:
runner.run(
adapter_role="semantic_detector",
model_input={"candidateBody": "甲"},
output_schema=SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
)
finally:
replay_module.run_role = original
self.assertEqual(caught.exception.primary_code, "SEMANTIC_DETECTOR_API_ERROR")
self.assertEqual(runner.receipts, [receipt])
self.assertEqual(runner.structured_outputs, [])
def test_three_arms_validate_full_context_and_only_change_evidence_strategy(self):
result = run_writer_replay(config(), run_id="dry-1")
self.assertEqual(result["status"], "ready")
sample = result["samples"][0]
arms = sample["arms"]
self.assertEqual(set(arms), {"A", "B", "C"})
self.assertEqual(arms["A"]["evidenceStrategy"], "historical_prose_only")
self.assertEqual(arms["B"]["evidenceStrategy"], "card_index_only")
self.assertEqual(arms["C"]["evidenceStrategy"], "card_index_plus_prose")
self.assertEqual(len({arm["commonControlsSha256"] for arm in arms.values()}), 1)
self.assertTrue(all(not arm["acceptanceEligible"] for arm in arms.values()))
self.assertEqual(arms["A"]["contextSummary"]["recentBaselineChapters"], [485, 486, 487, 488])
self.assertEqual(arms["B"]["contextSummary"]["proseEvidenceCount"], 0)
self.assertEqual(arms["B"]["contextSummary"]["indexHintCount"], 1)
self.assertEqual(arms["C"]["contextSummary"]["recentBaselineChapters"], [485, 486, 487, 488])
self.assertEqual(arms["C"]["contextSummary"]["indexHintCount"], 1)
self.assertTrue(
all(
arm["contextSummary"]["contentMode"] == "sanitized_contract_fixture"
for arm in arms.values()
)
)
rendered = json.dumps(result, ensure_ascii=False)
self.assertNotIn("脱敏合同夹具", rendered)
self.assertNotIn("林澈在冻结点仍位于圣蒂曼", rendered)
def test_future_index_hint_invalidates_whole_sample(self):
leaked = config()
leaked["samples"][0]["writerContextInput"]["retrievalResult"]["indexHints"][0]["asOf"] = 489
result = run_writer_replay(leaked, run_id="dry-leak")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_leakage")
def test_invalid_b_arm_contract_cannot_be_reported_ready(self):
invalid = config()
invalid["samples"][0]["writerContextInput"]["retrievalResult"]["indexHints"] = []
result = run_writer_replay(invalid, run_id="dry-invalid-context")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_writer_context")
def test_target_length_is_mechanically_recomputed_and_target_chapter_length_is_forbidden(self):
invalid = config()
invalid["samples"][0]["targetChars"] = 4100
with self.assertRaisesRegex(WriterReplayError, "机械复算"):
run_writer_replay(invalid, run_id="dry-length-tampered")
leaked = config()
leaked["samples"][0]["targetLengthBasis"]["usesTargetChapterLength"] = True
with self.assertRaisesRegex(WriterReplayError, "禁止读取目标章"):
run_writer_replay(leaked, run_id="dry-length-leaked")
def test_unresolved_generic_role_must_remain_null_instead_of_defaulting_to_zero(self):
unresolved = config()
sample = unresolved["samples"][0]
sample["writerContextInput"]["requirements"]["requiredCharacters"] = ["内应"]
sample["newCharacterRatio"] = None
sample["newCharacterRatioStatus"] = "unresolved_generic_role"
sample["newCharacterBasis"] = {
"definition": "named_required_characters_absent_before_as_of_ratio",
"asOfChapter": 488,
"requiredCharacters": ["内应"],
"knownBeforeAsOf": [],
"absentBeforeAsOf": [],
"genericRoles": ["内应"],
}
self.assertTrue(run_writer_replay(unresolved, run_id="dry-generic-role")["ok"])
unresolved["samples"][0]["newCharacterRatio"] = 0
with self.assertRaisesRegex(WriterReplayError, "必须为 null"):
run_writer_replay(unresolved, run_id="dry-generic-role-zero")
def test_character_declared_absent_before_freeze_cannot_have_card_index_hint(self):
inconsistent = config()
sample = inconsistent["samples"][0]
sample["newCharacterRatio"] = 1.0
sample["newCharacterBasis"]["knownBeforeAsOf"] = []
sample["newCharacterBasis"]["absentBeforeAsOf"] = ["林澈"]
with self.assertRaisesRegex(WriterReplayError, "不得同时出现在卡索引"):
run_writer_replay(inconsistent, run_id="dry-absent-card-conflict")
def test_gate_a_preregistration_mechanically_recomputes_all_lengths_and_ratios(self):
gate_config = _load_gate_a_config()
expected_targets = {489: 7500, 321: 7600, 544: 6700, 199: 2000, 523: 6100}
expected_ratios = {489: 1.0, 321: 0.0, 544: None, 199: 0.0, 523: 0.0}
self.assertEqual(
gate_config["commonControls"]["selectorVersion"],
"writer-gate-a-deep-space-card-selectors-v4",
)
self.assertEqual(
gate_config["commonControls"]["selectorSha256"],
"sha256:99105930d7e01d32fa9faa1ceaa30663674dd8e15b84dd8ab7f5b18cd4e9d0ef",
)
self.assertNotIn("inputProvenance", gate_config["commonControls"])
self.assertEqual(gate_config["commonControls"]["maxContextChars"], 140000)
selector_path = (
SCRIPT_DIR.parent
/ "configs"
/ "writer-gate-a-deep-space-card-selectors-v1.json"
)
selector_config = json.loads(selector_path.read_text(encoding="utf-8"))
selector_by_sample = {
sample["sampleId"]: [
(card["type"], card["name"]) for card in sample["cards"]
]
for sample in selector_config["samples"]
}
sample_by_id = {sample["sampleId"]: sample for sample in gate_config["samples"]}
for sample_id in selector_by_sample:
hints = sample_by_id[sample_id]["writerContextInput"]["retrievalResult"][
"indexHints"
]
self.assertEqual(
[(hint["type"], hint["name"]) for hint in hints],
selector_by_sample[sample_id],
)
expected_card_prose_chapters = {
"deep-space-489-battle": 484,
"deep-space-321-character-dialogue": 312,
"deep-space-544-turning-point": 467,
}
for sample_id, expected_chapter in expected_card_prose_chapters.items():
prose = sample_by_id[sample_id]["writerContextInput"]["retrievalResult"][
"proseEvidence"
]
card_prose = [item for item in prose if item["retrievalArm"] == "C"]
self.assertEqual(len(card_prose), 1)
self.assertEqual(card_prose[0]["chapter"], expected_chapter)
self.assertTrue(
card_prose[0]["sourceRef"]["sourceId"].endswith(
f":{expected_chapter}"
)
)
for sample in gate_config["samples"]:
basis = sample["targetLengthBasis"]
recalculated = calculate_target_chars(
recent_chapter_han_counts=sample["frozenRecentHanCounts"],
hard_event_count=basis["hardEventCount"],
foreshadowing_action_count=basis["foreshadowingActionCount"],
required_scene_count=basis["requiredSceneCount"],
min_chars=basis["minChars"],
max_chars=basis["maxChars"],
)
target_chapter = sample["targetChapter"]
self.assertEqual(recalculated, expected_targets[target_chapter])
self.assertEqual(sample["targetChars"], recalculated)
self.assertEqual(sample["newCharacterRatio"], expected_ratios[target_chapter])
self.assertEqual(
sample["writerContextInput"]["tokenBudget"]["maxContextChars"],
gate_config["commonControls"]["maxContextChars"],
)
if target_chapter == 544:
self.assertEqual(sample["newCharacterRatioStatus"], "unresolved_generic_role")
else:
self.assertEqual(sample["newCharacterRatioStatus"], "resolved")
result = run_writer_replay(gate_config, run_id="gate-a-preregistered-dry-run")
self.assertTrue(result["ok"])
self.assertEqual(result["status"], "ready")
self.assertEqual(len(result["samples"]), 5)
self.assertNotIn("脱敏合成历史片段", json.dumps(result, ensure_ascii=False))
for sample in result["samples"]:
receipt = sample["writerContextDiffReceipt"]
self.assertEqual(receipt["proseCharBudget"], 2000)
self.assertEqual(receipt["proseCharCount"], {"A": 2000, "C": 2000})
self.assertTrue(receipt["allowedDifferencePaths"])
self.assertNotEqual(receipt["contextSha256"]["A"], receipt["contextSha256"]["C"])
self.assertTrue(
all(
path.startswith(("$.factConstraints", "$.proseExcerpts"))
for path in receipt["allowedDifferencePaths"]
)
)
self.assertEqual(gate_config["commonControls"]["modelVersion"], MODEL_POLICY_VERSION)
self.assertEqual(gate_config["commonControls"]["adapterVersion"], RUNTIME_ADAPTER_VERSION)
probe = gate_config["executionAuthorization"]["runtimeProbe"]
self.assertEqual(probe["schemaVersion"], "runtime-probe-v2")
self.assertEqual(probe["status"], "dry_run")
self.assertEqual(probe["runtimeAdapter"], RUNTIME_ADAPTER)
self.assertEqual(probe["runtimeAdapterVersion"], RUNTIME_ADAPTER_VERSION)
self.assertEqual(probe["modelPolicyVersion"], MODEL_POLICY_VERSION)
self.assertNotIn("reason", probe)
for field in ("inputSha256", "executionReceiptSha256", "structuredOutputSha256"):
self.assertTrue(probe[field].startswith("sha256:"))
self.assertEqual(
gate_config["executionAuthorization"]["budget"]["maxCalls"],
{"writer": 150, "semantic_detector": 150, "blind_judge": 150},
)
self.assertEqual(gate_config["executionAuthorization"]["budget"]["status"], "approved")
self.assertEqual(
gate_config["executionAuthorization"]["budget"]["totalBudgetUsd"],
"2250.000000",
)
profile_cap = Decimal("5.000000")
self.assertEqual(
gate_config["executionAuthorization"]["budget"]["plannedCalls"],
{"writer": 60, "semantic_detector": 24, "blind_judge": 45},
)
self.assertEqual(
profile_cap * sum(
gate_config["executionAuthorization"]["budget"]["plannedCalls"].values()
),
Decimal("645.000000"),
)
max_calls = gate_config["executionAuthorization"]["budget"]["maxCalls"]
worst_case_budget = profile_cap * sum(max_calls.values())
self.assertEqual(worst_case_budget, Decimal("2250.000000"))
self.assertEqual(
worst_case_budget,
Decimal(gate_config["executionAuthorization"]["budget"]["totalBudgetUsd"]),
)
self.assertEqual(gate_config["executionAuthorization"]["rawRetention"]["status"], "approved")
self.assertIn("2250 美元覆盖三角色各 150 次", gate_config["executionAuthorization"]["budget"]["reason"])
self.assertIn("安全上限", gate_config["executionAuthorization"]["budget"]["reason"])
def test_gate_a_formal_zero_budget_or_zero_ac_difference_fails_closed(self):
"""正式 Gate A 必须真实改变 writer 创作输入,不能只改变隐藏索引。"""
zero_budget = _load_gate_a_config()
zero_budget["samples"][0]["proseCharBudget"] = 0
result = run_writer_replay(zero_budget, run_id="gate-a-zero-budget")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_writer_context")
self.assertIn("proseCharBudget", result["samples"][0]["errors"][0])
zero_difference = _load_gate_a_config()
evidence = zero_difference["samples"][0]["writerContextInput"]["retrievalResult"][
"proseEvidence"
]
evidence[1] = {**copy.deepcopy(evidence[0]), "retrievalArm": "C"}
result = run_writer_replay(zero_difference, run_id="gate-a-zero-difference")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_writer_context")
self.assertIn("A/C", result["samples"][0]["errors"][0])
def test_gate_a_formal_profiles_reconstruct_schema_prompt_and_runtime_bindings(self):
"""三角色 profile 必须能由 config 重建并绑定当前固定 Opus dry-run 探针。"""
gate_config = _load_gate_a_config()
profiles = gate_config["executionProfiles"]
self.assertEqual(set(profiles), {"writer", "semantic_detector", "blind_judge"})
expected_schema_ids = {
"writer": "writer-draft-v2",
"semantic_detector": "semantic-detection-draft-v3",
"blind_judge": "blind-judge-draft-v3",
}
expected_schemas = {
"writer": replay_module.WRITER_OUTPUT_JSON_SCHEMA,
"semantic_detector": SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
"blind_judge": BLIND_JUDGE_REPORT_JSON_SCHEMA,
}
probe = gate_config["executionAuthorization"]["runtimeProbe"]
profile_hashes = gate_config["executionAuthorization"]["profileSha256"]
model_aliases = {profile["modelAlias"] for profile in profiles.values()}
self.assertEqual(model_aliases, {probe["modelAlias"]})
self.assertEqual(probe["runtimeAdapter"], RUNTIME_ADAPTER)
self.assertEqual(probe["runtimeAdapterVersion"], RUNTIME_ADAPTER_VERSION)
self.assertEqual(probe["modelPolicyVersion"], MODEL_POLICY_VERSION)
self.assertEqual(probe["status"], "dry_run")
self.assertEqual(probe["executionProfileSha256"], profile_hashes["writer"])
self.assertEqual(gate_config["commonControls"]["adapterVersion"], RUNTIME_ADAPTER_VERSION)
cli_only_fields = {
"claudeExecutablePath",
"claudeExecutableSha256",
"claudeCliVersion",
"normalTerminalReasons",
}
self.assertTrue(all(cli_only_fields.isdisjoint(profile) for profile in profiles.values()))
self.assertTrue(cli_only_fields.isdisjoint(probe))
for role, raw_profile in profiles.items():
profile = replay_module.profile_from_mapping(raw_profile, role=role)
self.assertEqual(raw_profile["maxBudgetUsdPerCall"], "5.000000")
self.assertEqual(profile.max_budget_usd_per_call, Decimal("5.000000"))
self.assertEqual(raw_profile["timeoutSeconds"], 1200)
self.assertEqual(profile.timeout_seconds, 1200.0)
self.assertEqual(profile.json_schema_id, expected_schema_ids[role])
self.assertEqual(profile.json_schema, expected_schemas[role])
self.assertEqual(profile.json_schema_sha256, sha256_json(profile.json_schema))
self.assertEqual(profile.system_prompt_sha256, sha256_text(profile.system_prompt))
self.assertEqual(profile.execution_profile_sha256, profile_hashes[role])
self.assertEqual(profile.max_context_chars, gate_config["commonControls"]["maxContextChars"])
self.assertEqual(
len({raw_profile["systemPromptSha256"] for raw_profile in profiles.values()}),
3,
)
self.assertEqual(len({raw_profile["profileVersion"] for raw_profile in profiles.values()}), 3)
for control_name in ("budget", "rawRetention"):
control = gate_config["executionAuthorization"][control_name]
self.assertEqual(
control["receiptSha256"],
canonical_sha256(
{key: value for key, value in control.items() if key != "receiptSha256"}
),
)
def test_execute_without_formal_profiles_fails_closed_before_runner(self):
"""缺少三角色 formal profiles 时不能靠 profile hash 或测试参数放行。"""
tampered = _authorize_gate_config_for_test(_load_gate_a_config())
tampered.pop("executionProfiles")
tampered["oracleTruthPacks"] = {}
budget = tampered["executionAuthorization"]["budget"]
budget["status"] = "approved"
budget["totalBudgetUsd"] = "1.000000"
budget["receiptSha256"] = canonical_sha256(
{key: value for key, value in budget.items() if key != "receiptSha256"}
)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
output = pathlib.Path(directory) / "run"
result = run_writer_replay(
tampered,
run_id="missing-formal-profiles",
output_dir=output,
execute=True,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_execute_profile")
self.assertEqual(result["errors"], ["EXECUTE_PROFILE_INVALID"])
self.assertFalse((output / "journal" / "raw-vault").exists())
def test_max_calls_are_sufficient_but_missing_total_budget_still_blocks(self):
"""150 次/角色满足机械调用量,但没有可靠美元总额仍必须阻断。"""
tampered = _authorize_gate_config_for_test(_load_gate_a_config())
tampered["oracleTruthPacks"] = {}
budget = tampered["executionAuthorization"]["budget"]
self.assertEqual(budget["status"], "approved")
self.assertEqual(budget["totalBudgetUsd"], "2250.000000")
budget["status"] = "approved"
budget.pop("totalBudgetUsd", None)
budget["receiptSha256"] = canonical_sha256(
{key: value for key, value in budget.items() if key != "receiptSha256"}
)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
output = pathlib.Path(directory) / "run"
result = run_writer_replay(
tampered,
run_id="missing-total-budget",
output_dir=output,
execute=True,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_budget_authorization")
self.assertEqual(result["errors"], ["BUDGET_AUTHORIZATION_INVALID"])
self.assertFalse((output / "journal" / "raw-vault").exists())
def test_gate_a_execute_blocks_pending_budget_before_vault_or_runner(self):
"""正式 Gate A 已完成 runtime probe,但预算未批准时必须在副作用前阻断。"""
gate_config = _authorize_gate_config_for_test(_load_gate_a_config())
budget = gate_config["executionAuthorization"]["budget"]
budget["status"] = "pending"
budget.pop("totalBudgetUsd", None)
budget["receiptSha256"] = canonical_sha256(
{key: value for key, value in budget.items() if key != "receiptSha256"}
)
vault_calls: list[pathlib.Path] = []
def vault_factory(path):
vault_calls.append(path)
return RawVaultManager(path)
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
output = pathlib.Path(directory) / "run"
result = run_writer_replay(
gate_config,
run_id="gate-a-pending-budget",
output_dir=output,
execute=True,
production_adapters=adapters,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_budget_authorization")
self.assertEqual(result["errors"], ["BUDGET_AUTHORIZATION_REQUIRED"])
self.assertEqual(vault_calls, [])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
class WriterReplayExecuteBoundaryTest(unittest.TestCase):
"""验证真实执行失败关闭和测试注入仍经过正式管线。"""
def test_execute_requires_private_tmp_output(self):
with tempfile.TemporaryDirectory() as directory:
with self.assertRaisesRegex(WriterReplayError, "/private/tmp"):
run_writer_replay(
config(),
run_id="real-bad-path",
output_dir=pathlib.Path(directory),
execute=True,
)
def test_execute_without_production_adapters_fails_closed(self):
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
pending = config()
pending["commonControls"]["modelVersion"] = "pending_probe"
pending["commonControls"]["adapterVersion"] = "pending_probe"
result = run_writer_replay(
pending,
run_id="real-not-wired",
output_dir=pathlib.Path(directory) / "run",
execute=True,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_runtime_probe")
self.assertEqual(result["errors"], ["RUNTIME_PROBE_REQUIRED"])
def test_invalid_third_judge_report_fails_top_level(self):
adapters, _runner, _detector, _judge = _test_adapters(
unstable=True, invalid_third=True
)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
result = run_writer_replay(
config(),
run_id="judge-invalid",
output_dir=pathlib.Path(directory) / "run",
execute=True,
test_adapters=adapters,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_judge_invalid")
def test_third_judge_still_unstable_fails_top_level(self):
adapters, _runner, _detector, _judge = _test_adapters(
irreducibly_unstable=True
)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
result = run_writer_replay(
config(),
run_id="judge-unstable",
output_dir=pathlib.Path(directory) / "run",
execute=True,
test_adapters=adapters,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_judge_unstable")
def test_context_authorization_cannot_swap_snapshot_or_source_before_runner(self):
"""上下文版本、快照、用途、时间和状态都必须绑定顶层授权。"""
mutations = {
"source_version": lambda value: value["samples"][0]["writerContextInput"].update(
{"sourceVersion": "swapped-source-v2"}
),
"snapshot_id": lambda value: value["samples"][0]["writerContextInput"][
"authorizationSnapshot"
].update({"snapshotId": "auth-swapped"}),
"purpose": lambda value: value["samples"][0]["writerContextInput"][
"authorizationSnapshot"
].update({"allowedPurpose": "diagnostic"}),
"verified_at": lambda value: value["samples"][0]["writerContextInput"][
"authorizationSnapshot"
].update({"verifiedAt": "2026-07-21T00:00:00Z"}),
"source_status": lambda value: value["samples"][0]["writerContextInput"].update(
{"sourceStatus": "revoked"}
),
}
for name, mutate in mutations.items():
invalid = config()
mutate(invalid)
adapters, runner, _detector, _judge = _test_adapters()
with self.subTest(name=name), tempfile.TemporaryDirectory(
dir="/private/tmp"
) as directory:
result = run_writer_replay(
invalid,
run_id=f"auth-binding-{name}",
output_dir=pathlib.Path(directory) / "run",
execute=True,
test_adapters=adapters,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_authorization")
self.assertEqual(runner.calls, [])
def test_test_injection_runs_writer_pipeline_mechanical_and_semantic_detector(self):
adapters, runner, detector, judge = _test_adapters(unstable=True)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
output_dir = pathlib.Path(directory) / "run"
result = run_writer_replay(
config(),
run_id="test-pipeline",
output_dir=output_dir,
execute=True,
test_adapters=adapters,
)
self.assertFalse((output_dir / "raw").exists())
lease_files = list(
(output_dir / "journal" / "raw-vault" / "leases").glob("*.json")
)
self.assertEqual(len(lease_files), 1)
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
public_manifest = (output_dir / "manifest.json").read_text(encoding="utf-8")
self.assertNotIn("candidateBody", public_manifest)
self.assertNotIn("脱敏合同夹具", public_manifest)
self.assertEqual(result["status"], "completed_test_pipeline")
self.assertTrue(all("evidenceStrategy" not in context for context in runner.contexts))
self.assertTrue(
all(
kwargs.get("system") == build_dispatch_system_prompt(adapters.writer_profile)
for _prompt, kwargs in runner.calls
)
)
self.assertEqual(detector.calls, [
("historical_prose_only", True),
("card_index_only", True),
("card_index_plus_prose", True),
])
self.assertEqual(len([call for call in judge.calls if call[0] == "judge-3"]), 1)
candidates = result["samples"][0]["candidates"]
self.assertTrue(all(not candidate["acceptanceEligible"] for candidate in candidates.values()))
def test_writer_model_cannot_forge_adapter_owned_hash(self):
adapters, _runner, detector, _judge = _test_adapters(illegal_extra_field=True)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
with self.assertRaises(WriterReplayError):
run_writer_replay(
config(),
run_id="test-writer-owned-hash",
output_dir=pathlib.Path(directory) / "run",
execute=True,
test_adapters=adapters,
)
# B 臂模型越权输出 adapter 字段后,不能进入 semantic detector。
self.assertEqual(detector.calls, [("historical_prose_only", True)])
def test_malicious_judge_only_sees_blind_candidates_without_raw_paths_or_arm_contexts(self):
runner = FakeGovernedChat()
detector = FakeSemanticDetector()
judge = MaliciousJudge()
adapters = WriterReplayTestAdapters(runner, _frozen_test_profile(), detector, judge)
evaluation_config = config()
fine_outline = evaluation_config["samples"][0]["writerContextInput"]["fineOutline"]
# 恶意配置模拟把目标原文、索引、文件路径和真实臂塞进细纲对象;这些字段写手看不到,评委也不能看到。
fine_outline.update(
{
"targetOriginal": "目标章原文泄漏哨兵",
"indexHints": [{"content": "被测卡索引泄漏哨兵"}],
"path": "/private/tmp/candidate-A.json",
"arm": "A",
}
)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
output_dir = pathlib.Path(directory) / "run"
result = run_writer_replay(
evaluation_config,
run_id="blind-isolation",
output_dir=output_dir,
execute=True,
test_adapters=adapters,
)
self.assertTrue(result["ok"])
self.assertTrue(judge.visible_inputs)
expected_writer_fine_outline = {
"hardConstraints": ["必须完成围攻突围"],
"adjustableBeats": [],
"declaredNewFacts": [],
}
self.assertTrue(
all(context["fineOutline"] == expected_writer_fine_outline for context in runner.contexts),
"三臂写手必须只看到严格细纲合同字段",
)
expected_judge_fine_outline = {
"sourceRef": {
"sourceId": "fixture:fine-outline:489",
"sourceVersion": "fine-outline-fixture-v1",
"chapter": 489,
},
**expected_writer_fine_outline,
}
shared_references = []
for visible in judge.visible_inputs:
self.assertEqual(
set(visible),
{
"schemaVersion",
"sample",
"blindCandidateId",
"candidateOrder",
"sharedEvaluationReference",
"candidates",
},
)
self.assertEqual(set(visible["candidateOrder"]), {"blind-1", "blind-2", "blind-3"})
shared = visible["sharedEvaluationReference"]
shared_references.append(copy.deepcopy(shared))
self.assertEqual(shared["fineOutline"], expected_judge_fine_outline)
self.assertNotIn("entities", shared["fineOutline"])
self.assertEqual(shared["requirements"]["requiredCharacters"], ["林澈"])
self.assertEqual(
shared["requirements"]["chapterEndHook"]["anchors"],
["城门忽然打开"],
)
self.assertEqual(
[item["chapter"] for item in shared["historicalProseBaseline"]],
[485, 486, 487, 488],
)
self.assertTrue(
all(
set(candidate)
== {"blindCandidateId", "candidateSha256", "candidateBody"}
and candidate["blindCandidateId"].startswith("blind-")
for candidate in visible["candidates"]
)
)
self.assertTrue(
all(reference == shared_references[0] for reference in shared_references[1:]),
"评委顺序变化不得改变共同评测参考",
)
class WriterReplayProductionIntegrationTest(unittest.TestCase):
"""用生产 fake 串通 Vault、CAS、v2 adapter 与 GateInputBuilder。"""
def _run(
self,
adapters,
*,
evaluation_config=None,
run_id="production-fake",
raw_archive_dir=None,
):
"""在安全临时目录执行一次生产链,并返回结果与输出目录内容。"""
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
self.addCleanup(directory.cleanup)
output = pathlib.Path(directory.name) / "run"
result = run_writer_replay(
evaluation_config or _production_config(),
run_id=run_id,
output_dir=output,
raw_archive_dir=raw_archive_dir,
execute=True,
production_adapters=adapters,
)
return result, output
def test_default_runtime_requires_explicit_raw_archive_before_vault_or_model(self):
"""正式默认 runtime 未给仓外归档根时必须在建 vault 和调模型前失败关闭。"""
evaluation_config = _production_config()
adapters, writer, semantic, judge = _production_adapters()
authorized, blocked_status, blocked_code = replay_module._validate_execute_authorization(
evaluation_config, adapters
)
self.assertIsNotNone(authorized, (blocked_status, blocked_code))
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
self.addCleanup(directory.cleanup)
output = pathlib.Path(directory.name) / "run"
with mock.patch.object(
replay_module,
"_validate_execute_authorization",
return_value=(authorized, None, None),
):
result = run_writer_replay(
evaluation_config,
run_id="missing-raw-archive",
output_dir=output,
execute=True,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_raw_archive")
self.assertEqual(result["errors"], ["RAW_ARCHIVE_REQUIRED"])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
self.assertFalse((output / "journal" / "raw-vault").exists())
def test_relative_raw_archive_blocks_before_vault_or_model(self):
"""显式归档根必须是绝对路径,不能退回仓库相对目录。"""
adapters, writer, semantic, judge = _production_adapters()
result, output = self._run(
adapters,
run_id="relative-raw-archive",
raw_archive_dir=pathlib.Path("relative/archive"),
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_raw_archive")
self.assertEqual(result["errors"], ["RAW_ARCHIVE_INVALID"])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
self.assertFalse((output / "journal" / "raw-vault").exists())
@staticmethod
def _cas_safe_summary(output: pathlib.Path, sample_id: str) -> dict[str, object]:
cas_root = output / "journal" / "cas" / sample_id
state = json.loads((cas_root / "state.json").read_text(encoding="utf-8"))
return json.loads(
(cas_root / "artifacts" / state["safeSummaryArtifact"]).read_text(
encoding="utf-8"
)
)
def test_oracle_projection_keeps_only_common_recent_four_chapters(self):
"""完整源包可验真全历史,但 blind judge 只能收到共同的最近四章与目标断言。"""
source = _oracle_pack()
statement = "第484章的更早历史只留在受控 oracle 源包"
source["historicalAssertions"].insert(
0,
{
"assertionId": "assertion-history-484",
"assertionType": "canonical_history",
"statement": statement,
"sourceVersion": "canonical-v484",
"chapterBoundary": {"minChapter": 484, "maxChapter": 484},
"contentSha256": "sha256:"
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
},
)
source["packSha256"] = canonical_sha256(
{key: value for key, value in source.items() if key != "packSha256"}
)
projected = _project_oracle_pack(source)
self.assertEqual(
[item["chapterStart"] for item in projected["historicalAssertions"]],
[485, 486, 487, 488],
)
self.assertTrue(projected["targetAssertions"])
self.assertNotIn("authorization", projected)
self.assertNotIn("assertionType", json.dumps(projected, ensure_ascii=False))
self.assertNotIn("第484章", json.dumps(projected, ensure_ascii=False))
def test_production_execute_rejects_sanitized_fixture_before_vault_or_runner(self):
"""真实 execute 只接受 loader 产出的 canonical_frozen_prose。"""
vault_calls: list[pathlib.Path] = []
def vault_factory(path):
vault_calls.append(path)
return RawVaultManager(path)
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
sanitized = _production_config()
sanitized["samples"][0]["writerContextInput"]["contentMode"] = (
"sanitized_contract_fixture"
)
result, output = self._run(
adapters, evaluation_config=sanitized, run_id="production-sanitized"
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_execute_content_mode")
self.assertEqual(result["errors"], ["EXECUTE_CONTENT_MODE_INVALID"])
self.assertEqual(vault_calls, [])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
self.assertFalse((output / "journal" / "raw-vault").exists())
def test_default_runtime_missing_auth_blocks_before_vault_or_budget_slot(self):
"""真实 runtime 缺显式 token 时必须在任何副作用前返回准确阻断码。"""
config = _production_config()
adapters, _writer, _semantic, _judge = _production_adapters()
with (
mock.patch.dict(
os.environ,
{"MUSE_ROLE_OPUS_BASE_URL": "", "MUSE_ROLE_OPUS_AUTH_TOKEN": ""},
clear=False,
),
mock.patch(
"run_writer_replay._production_adapters_from_config",
return_value=adapters,
),
tempfile.TemporaryDirectory(dir="/private/tmp") as directory,
):
output = pathlib.Path(directory) / "run"
result = run_writer_replay(
config,
run_id="runtime-auth-missing",
output_dir=output,
execute=True,
)
self.assertFalse((output / "journal" / "raw-vault").exists())
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_runtime_authentication")
self.assertEqual(result["errors"], ["RUNTIME_AUTHENTICATION_REQUIRED"])
self.assertEqual(result["samples"][0]["status"], "ready")
self.assertNotIn("budgetLedger", result)
def test_planned_timeout_sum_does_not_block_next_call_safe_run(self):
"""理论调用总时长超过租期时,只要每笔调用窗口充足就允许真实编排继续。"""
vault_calls = []
def vault_factory(path):
vault_calls.append(path)
return RawVaultManager(path)
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
config = _production_config()
raw = config["executionAuthorization"]["rawRetention"]
raw["retainUntil"] = (datetime.now(UTC) + timedelta(minutes=4)).isoformat()
raw["receiptSha256"] = canonical_sha256(
{key: value for key, value in raw.items() if key != "receiptSha256"}
)
result, output = self._run(
adapters,
evaluation_config=config,
run_id="next-call-retention-allowed",
)
# 测试 profile 每角色 plannedCalls=3、timeout=30 秒,理论总和加清理为 330 秒,
# 高于 4 分钟租期;单次调用只需 30+60=90 秒,所以不应被总和误阻断。
self.assertTrue(result["ok"], result)
self.assertEqual(result["status"], "completed")
self.assertEqual(len(vault_calls), 1)
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
self.assertTrue((output / "gate-input.json").exists())
def test_b_needs_evidence_keeps_candidate_and_reaches_judge_and_gate(self):
"""B 臂合法补证信号属于对照质量,不得误报为执行系统失败。"""
adapters, _writer, _semantic, judge = _production_adapters()
config = _production_config()
semantic = NeedsEvidenceSemanticRunner(
needs_evidence_on_call=_semantic_call_index_for_arm(config, "B")
)
adapters = replace(adapters, semantic_model_runner=semantic)
result, output = self._run(
adapters,
evaluation_config=config,
run_id="semantic-needs-evidence-b",
)
self.assertTrue(result["ok"], result)
self.assertEqual(result["status"], "completed")
sample = result["samples"][0]
self.assertEqual(set(sample["candidates"]), {"A", "B", "C"})
self.assertNotIn("detector", sample)
diagnostic = sample["semanticDiagnostics"]["B"]
self.assertEqual(diagnostic["outcome"], "needs_evidence")
self.assertEqual(diagnostic["blockingCounts"]["evidenceGaps"], 1)
self.assertEqual(diagnostic["blockingCounts"]["unknownAssertions"], 0)
self.assertEqual(len(semantic.calls), 3)
self.assertEqual(len(judge.calls), 2)
rubric = judge.calls[0]["rubric"]
self.assertEqual(rubric["policyVersion"], RUBRIC_POLICY_VERSION)
self.assertEqual(rubric["scenarioType"], "battle")
self.assertEqual(
[item["dimensionId"] for item in rubric["commonDimensions"]],
list(COMMON_DIMENSIONS),
)
self.assertEqual(
rubric["scenarioDimension"]["dimensionId"], SCENARIO_DIMENSION
)
self.assertEqual(rubric["scenarioDimension"]["displayName"], "战斗执行")
self.assertTrue((output / "gate-input.json").exists())
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
self.assertEqual(manifest["samples"][0]["semanticDiagnostics"], {"B": diagnostic})
self.assertEqual(
self._cas_safe_summary(output, "deep-space-489")["semanticDiagnostics"],
{"B": diagnostic},
)
serialized = json.dumps(manifest, ensure_ascii=False)
for forbidden in (
"candidateQuote",
"补查候选新事实",
"冻结证据不足",
"gap-1",
'"query"',
'"message"',
"/private/tmp",
):
self.assertNotIn(forbidden, serialized)
def test_pre_call_judge_failure_preserves_primary_code_without_receipt(self):
"""评委输入在首调前失败时,不能用“缺回执”覆盖真正错误码。"""
adapters, _writer, _semantic, judge = _production_adapters()
invalid = {
"ok": False,
"status": "failed_judge_invalid",
"primaryCode": "BLIND_INPUT_INVALID",
"reviewCount": 0,
"attemptCount": 0,
"safeDiagnostic": {
"schemaVersion": "blind-judge-safe-diagnostic-v1",
"errorCode": "BLIND_INPUT_INVALID",
"errorMessage": "输入绑定非法",
"draftAvailable": False,
},
}
with mock.patch.object(
execute_module,
"run_writer_blind_judge_panel",
return_value=invalid,
):
result, output = self._run(adapters, run_id="judge-pre-call-invalid")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_judge_invalid")
self.assertEqual(
result["samples"][0]["errors"],
["failed_judge_invalid:BLIND_INPUT_INVALID"],
)
self.assertEqual(judge.calls, [])
self.assertFalse((output / "gate-input.json").exists())
diagnostic = result["samples"][0]["judgeDiagnostic"]
self.assertEqual(diagnostic["primaryCode"], "BLIND_INPUT_INVALID")
self.assertEqual(diagnostic["modelCallCount"], 0)
self.assertEqual(
diagnostic["modelOutput"]["errorCode"], "BLIND_INPUT_INVALID"
)
self.assertNotIn("candidateBody", json.dumps(diagnostic, ensure_ascii=False))
def test_detector_correction_exhaustion_keeps_failed_sample(self):
"""detector 不合约必须保留首样本并停止,不能继续启动第二样本。"""
adapters, _writer, _semantic, judge = _production_adapters()
semantic = InvalidQuoteSemanticRunner()
evaluation_config = _production_config()
second_sample = copy.deepcopy(evaluation_config["samples"][0])
second_sample["sampleId"] = "deep-space-490"
evaluation_config["samples"].append(second_sample)
budget = evaluation_config["executionAuthorization"]["budget"]
for role in replay_module.BUDGET_ROLES:
budget["plannedCalls"][role] = 6
budget["maxCalls"][role] = 6
budget["totalBudgetUsd"] = "18.000000"
budget["receiptSha256"] = canonical_sha256(
{key: value for key, value in budget.items() if key != "receiptSha256"}
)
second_oracle = copy.deepcopy(_oracle_pack())
second_oracle["sampleId"] = "deep-space-490"
second_oracle["packSha256"] = canonical_sha256(
{key: value for key, value in second_oracle.items() if key != "packSha256"}
)
adapters = replace(
adapters,
semantic_model_runner=semantic,
oracle_truth_packs={
**adapters.oracle_truth_packs,
"deep-space-490": second_oracle,
},
)
result, output = self._run(
adapters,
evaluation_config=evaluation_config,
run_id="semantic-invalid-exhausted",
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_semantic_detector")
self.assertEqual(len(result["samples"]), 1)
self.assertEqual(result["samples"][0]["status"], "failed_semantic_detector")
self.assertEqual(
result["samples"][0]["errors"],
["SEMANTIC_DETECTOR_QUOTE_NOT_FOUND"],
)
diagnostic = result["samples"][0]["semanticDiagnostic"]
self.assertEqual(diagnostic["reasonCode"], "QUOTE_NOT_FOUND")
self.assertEqual(diagnostic["attemptCount"], 3)
self.assertEqual(diagnostic["correctionCount"], 2)
self.assertEqual(len(semantic.calls), 3)
self.assertEqual(judge.calls, [])
cas_dirs = sorted(
path.name
for path in (output / "journal" / "cas").iterdir()
if path.is_dir()
)
self.assertEqual(cas_dirs, [result["samples"][0]["sampleId"]])
self.assertFalse((output / "gate-input.json").exists())
manifest_text = (output / "manifest.json").read_text(encoding="utf-8")
for forbidden in ("candidateQuote", "候选中不存在的引文", "previousDraft"):
self.assertNotIn(forbidden, manifest_text)
self.assertEqual(
self._cas_safe_summary(output, result["samples"][0]["sampleId"])[
"semanticDiagnostic"
],
diagnostic,
)
def test_detector_correction_receipts_bind_final_success_in_gate(self):
"""首轮非法、第二轮纠正成功时,Gate 绑定最后回执且整轮可完成。"""
adapters, _writer, _semantic, _judge = _production_adapters()
semantic = CorrectingSemanticRunner()
adapters = replace(adapters, semantic_model_runner=semantic)
config = _production_config()
budget = config["executionAuthorization"]["budget"]
budget["plannedCalls"]["semantic_detector"] = 4
budget["maxCalls"]["semantic_detector"] = 4
budget["totalBudgetUsd"] = "10.000000"
budget["receiptSha256"] = canonical_sha256(
{key: value for key, value in budget.items() if key != "receiptSha256"}
)
result, output = self._run(
adapters,
evaluation_config=config,
run_id="semantic-correction-success",
)
self.assertTrue(result["ok"], result)
self.assertEqual(len(semantic.calls), 4)
receipts = json.loads((output / "execution-receipts.json").read_text())
self.assertEqual(
len(receipts["deep-space-489"]["A"]["semantic_detector"]["executionReceipts"]),
2,
)
gate_input = json.loads((output / "gate-input.json").read_text())
self.assertFalse(gate_input["samples"][0]["systemFailure"])
self.assertTrue(all("semanticDiagnostic" not in sample for sample in result["samples"]))
self.assertNotIn(
"semanticDiagnostic",
self._cas_safe_summary(output, "deep-space-489"),
)
def test_raw_expiry_inside_sample_is_recorded_and_cas_closed(self):
"""样本执行中的 raw 异常必须保留样本安全终态并收口 CAS。"""
class ExpiringManager(RawVaultManager):
def __init__(self, root):
super().__init__(root)
self.write_count = 0
def write_bytes(self, lease, relative_path, content):
self.write_count += 1
if self.write_count == 2:
raise RawVaultError("RAW_LEASE_EXPIRED", "测试租约到期")
return super().write_bytes(lease, relative_path, content)
adapters, writer, semantic, judge = _production_adapters(vault_factory=ExpiringManager)
result, output = self._run(adapters, run_id="raw-expired-in-sample")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_raw_vault")
self.assertEqual(len(result["samples"]), 1)
self.assertEqual(result["samples"][0]["errors"], ["RAW_LEASE_EXPIRED"])
self.assertEqual((writer.calls, semantic.calls, judge.calls), ([], [], []))
cas = json.loads(
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text()
)
self.assertEqual(cas["state"], "FAILED")
self.assertFalse((output / "gate-input.json").exists())
def test_gate_publish_failure_removes_partially_published_input(self):
"""Gate 文件原子写后若后续发布动作失败,失败 manifest 前必须撤销文件。"""
adapters, _writer, _semantic, _judge = _production_adapters()
original_write = replay_module.atomic_write_json
def fail_after_gate_write(path, value):
original_write(path, value)
if path.name == "gate-input.json":
raise OSError("测试 Gate 发布失败")
with mock.patch("run_writer_replay.execute.atomic_write_json", side_effect=fail_after_gate_write):
result, output = self._run(adapters, run_id="gate-publish-failure")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_system")
self.assertFalse((output / "gate-input.json").exists())
def test_judge_quote_correction_keeps_full_receipts_and_binds_final_reports(self):
"""评委纠错必须保留全量调用,并用最终索引通过 Gate builder 反篡改复核。"""
adapters, _writer, _semantic, judge = _production_adapters(
judge_invalid_first_quote=True
)
result, output = self._run(adapters, run_id="judge-quote-correction")
self.assertTrue(result["ok"], result)
self.assertEqual(len(judge.calls), 3)
self.assertNotIn("correction", judge.calls[0])
self.assertIn("correction", judge.calls[1])
execution_receipts = json.loads(
(output / "execution-receipts.json").read_text(encoding="utf-8")
)
panel_receipts = execution_receipts["deep-space-489"]["A"][
"blind_judge"
]["executionReceipts"]
self.assertEqual(len(panel_receipts), 3)
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
self.assertTrue(
all(not sample["systemFailure"] for sample in gate_input["samples"]),
gate_input["samples"],
)
def test_production_fake_happy_path_runs_v2_adapters_receipts_vault_cas_and_builder(self):
adapters, writer, semantic, judge = _production_adapters()
result, output = self._run(adapters)
self.assertTrue(result["ok"], result)
self.assertEqual(result["status"], "completed")
self.assertEqual(len(writer.calls), 3)
self.assertEqual(len(semantic.calls), 3)
self.assertEqual(len(judge.calls), 2)
self.assertEqual(
result["budgetLedger"]["usedCalls"],
{"writer": 3, "semantic_detector": 3, "blind_judge": 2},
)
self.assertEqual(
result["budgetLedger"]["remainingPlannedCalls"],
{"writer": 0, "semantic_detector": 0, "blind_judge": 1},
)
self.assertEqual(
result["budgetLedger"]["totalActualCostUsd"],
_expected_total_cost(fixed_role_calls=5),
)
self.assertFalse(result["budgetLedger"]["costUnknown"])
self.assertTrue(all(receipt["adapterRole"] == "semantic_detector" for receipt in semantic.receipts))
self.assertTrue(all(receipt["adapterRole"] == "blind_judge" for receipt in judge.receipts))
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
self.assertEqual(gate_input["schemaVersion"], "writer-gate-input-v3")
self.assertEqual(gate_input["rubricPolicyVersion"], RUBRIC_POLICY_VERSION)
self.assertEqual(gate_input["gateInputSha256"], result["gateInputSha256"])
self.assertTrue(
all(not sample["systemFailure"] for sample in gate_input["samples"]),
gate_input["samples"],
)
execution_receipts = json.loads(
(output / "execution-receipts.json").read_text(encoding="utf-8")
)
self.assertEqual(
{
role
for arm in execution_receipts["deep-space-489"].values()
for role in arm
},
{"writer", "semantic_detector", "blind_judge"},
)
self.assertTrue(
all(
wrapper["executionReceipts"]
for arm in execution_receipts["deep-space-489"].values()
for wrapper in arm.values()
)
)
manifest = (output / "manifest.json").read_text(encoding="utf-8")
self.assertNotIn("candidateBody", manifest)
self.assertNotIn("/private/tmp", manifest)
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
self.assertEqual(len(lease_files), 1)
lease = json.loads(lease_files[0].read_text(encoding="utf-8"))
self.assertEqual(lease["status"], "migrated")
self.assertEqual(result["rawDisposition"]["status"], "migrated")
self.assertEqual(result["rawDisposition"]["archiveId"], lease["archiveId"])
archive_root = output.parent / "production-fake-raw-archive"
archived = archive_root / f"muse-raw-archive-{lease['archiveId']}"
self.assertTrue((archived / "run" / "config.json").is_file())
self.assertTrue((archived / ".migration-receipt.json").is_file())
cas_state = json.loads(
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text()
)
self.assertEqual(cas_state["state"], "COMPLETED")
self.assertEqual(cas_state["cleanupState"], "migrated")
def test_length_revision_loop_recovers_out_of_range_draft(self):
"""首版越界时退回写手修订一遍即达标:样本成功,修订块带固定指令与上一版正文。"""
adapters, writer, _semantic, _judge = _production_adapters()
config = _production_config_with_writer_budget(
planned=9, max_calls=9, total_usd="18.000000"
)
short_body = "短" * 2000 # 2000 汉字 < 合同下限 2800,首版必然越界
def revise_once(context, *, call_index):
# 偶数下标是每臂首版(越界),奇数下标是修订版(达标)。
if call_index % 2 == 0:
return {"candidateBody": short_body}
marker = ("甲", "乙", "丙")[(call_index // 2) % 3]
return {"candidateBody": _candidate_body(marker)}
with mock.patch.object(sys.modules[__name__], "_writer_output", revise_once):
result, output = self._run(
adapters, evaluation_config=config, run_id="length-revision-ok"
)
self.assertTrue(result["ok"], result)
self.assertEqual(result["status"], "completed")
# 每臂 1 次首版 + 1 次修订 = 2 次 writer 调用,三臂共 6 次。
self.assertEqual(len(writer.calls), 6)
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 6)
self.assertNotIn("lengthViolation", result["samples"][0])
# 每臂 writer 回执同时绑定首版与修订两条。
execution_receipts = json.loads(
(output / "execution-receipts.json").read_text(encoding="utf-8")
)
for arm in ("A", "B", "C"):
wrapper = execution_receipts["deep-space-489"][arm]["writer"]
self.assertEqual(len(wrapper["executionReceipts"]), 2)
# 首版调用不带 lengthRevision,修订调用带;下标 1/3/5 是三臂的修订输入。
for index in (0, 2, 4):
self.assertNotIn("lengthRevision", writer.contexts[index])
revision_inputs = [writer.contexts[index] for index in (1, 3, 5)]
for context in revision_inputs:
revision = context["lengthRevision"]
# 修订指令是三臂共用的固定常量(arm-invariant),不含具体字数。
self.assertEqual(revision["instruction"], replay_module.LENGTH_REVISION_INSTRUCTION)
# 修订块携带上一版正文与实际字数/区间,作为数据而非拼进指令文本。
self.assertEqual(revision["previousDraft"], short_body)
self.assertEqual(revision["actualHanChars"], 2000)
self.assertEqual(revision["minChars"], 2800)
self.assertEqual(revision["maxChars"], 5200)
self.assertEqual(revision["targetChars"], 4000)
self.assertEqual(revision["countingRule"], "han_chars_only")
self.assertEqual(revision["revisionDirection"], "expand")
self.assertEqual(revision["targetDeltaHanChars"], 2000)
self.assertEqual(
{context["lengthRevision"]["instruction"] for context in revision_inputs},
{replay_module.LENGTH_REVISION_INSTRUCTION},
)
instruction = replay_module.LENGTH_REVISION_INSTRUCTION
self.assertIn("只统计 candidateBody 中的汉字", instruction)
self.assertIn("不统计标点、空格、数字或拉丁字母", instruction)
self.assertIn("以 targetChars 为修订目标", instruction)
self.assertIn("不要只擦到", instruction)
self.assertIn("必须完整保留上一版已有内容", instruction)
self.assertIn("不得压缩、删除或合并已有内容", instruction)
self.assertIn("revisionDirection=trim", instruction)
self.assertIn("删减约 targetDeltaHanChars 个汉字", instruction)
self.assertIn("不得继续扩写", instruction)
def test_length_revision_recomputes_direction_and_target_delta_each_round(self):
"""同一臂每轮都按当前正文重算方向与距目标差值,不能复用首版数字。"""
adapters, writer, _semantic, _judge = _production_adapters()
config = _production_config_with_writer_budget(
planned=9, max_calls=9, total_usd="18.000000"
)
short_body = "短" * 2000
long_body = "长" * 6000
original_writer_output = _writer_output
def output_without_recursion(context, *, call_index):
if call_index == 0:
return {"candidateBody": short_body}
if call_index == 1:
return {"candidateBody": long_body}
return original_writer_output(context, call_index=call_index)
with mock.patch.object(
sys.modules[__name__], "_writer_output", output_without_recursion
):
result, _output = self._run(
adapters, evaluation_config=config, run_id="length-revision-recomputed"
)
self.assertTrue(result["ok"], result)
first_revision = writer.contexts[1]["lengthRevision"]
second_revision = writer.contexts[2]["lengthRevision"]
self.assertEqual(first_revision["previousDraft"], short_body)
self.assertEqual(first_revision["actualHanChars"], 2000)
self.assertEqual(first_revision["revisionDirection"], "expand")
self.assertEqual(first_revision["targetDeltaHanChars"], 2000)
self.assertEqual(second_revision["previousDraft"], long_body)
self.assertEqual(second_revision["actualHanChars"], 6000)
self.assertEqual(second_revision["revisionDirection"], "trim")
self.assertEqual(second_revision["targetDeltaHanChars"], 2000)
self.assertEqual(
{first_revision["countingRule"], second_revision["countingRule"]},
{"han_chars_only"},
)
def test_length_revision_exhausted_still_fails_with_actual_han_chars(self):
"""修订三遍仍越界才失败:记下实际字数,writer 调用数 = 1 首版 + 3 修订 = 4。"""
adapters, writer, _semantic, _judge = _production_adapters()
config = _production_config_with_writer_budget(
planned=4,
max_calls=4,
total_usd="12.000000",
)
short_body = "短" * 2000
with mock.patch.object(
sys.modules[__name__],
"_writer_output",
lambda context, *, call_index: {"candidateBody": short_body},
):
result, _output = self._run(
adapters,
evaluation_config=config,
run_id="length-revision-exhausted",
)
self.assertFalse(result["ok"])
sample = result["samples"][0]
self.assertIn("候选正文汉字数超出动态篇幅区间", sample["errors"])
self.assertIn("lengthViolation", sample)
self.assertEqual(sample["lengthViolation"]["actualHanChars"], 2000)
# 首臂耗尽 1+3=4 次 writer 调用后即终局门失败,后续臂不再运行。
self.assertEqual(len(writer.calls), 4)
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 4)
def test_length_in_range_skips_revision_loop(self):
"""首版即在区间内:0 修订,每臂仅 1 次 writer 调用,输入不带 lengthRevision。"""
adapters, writer, _semantic, _judge = _production_adapters()
result, output = self._run(adapters, run_id="length-no-revision")
self.assertTrue(result["ok"], result)
self.assertEqual(len(writer.calls), 3)
self.assertEqual(result["budgetLedger"]["usedCalls"]["writer"], 3)
for context in writer.contexts:
self.assertNotIn("lengthRevision", context)
execution_receipts = json.loads(
(output / "execution-receipts.json").read_text(encoding="utf-8")
)
for arm in ("A", "B", "C"):
wrapper = execution_receipts["deep-space-489"][arm]["writer"]
self.assertEqual(len(wrapper["executionReceipts"]), 1)
def test_bound_judge_report_binds_raw_outputs_to_receipt_hashes(self):
"""生产编排必须把 reviewer 原始输出随报告进 builder 且与回执哈希一致(正向对照)。"""
captured: dict[str, object] = {}
class CapturingBuilder:
"""记录一次生产 build 的全部来源,供篡改复现复用,本身仍走真 builder。"""
def build(self, **kwargs):
captured.update(kwargs)
return GateInputBuilder().build(**kwargs)
adapters, _writer, _semantic, _judge = _production_adapters(
builder_factory=CapturingBuilder
)
result, _output = self._run(adapters)
self.assertTrue(result["ok"], result)
judge_report = captured["judge_reports"]["deep-space-489"]
reviewer_outputs = judge_report["reviewerStructuredOutputs"]
true_receipts = captured["execution_receipts"]["deep-space-489"]["A"][
"blind_judge"
]["executionReceipts"]
# 原始输出与 reviewerReports、回执按 reviewer 顺序一一对应,且哈希回真实回执。
self.assertEqual(len(reviewer_outputs), len(judge_report["reviewerReports"]))
self.assertEqual(len(reviewer_outputs), len(true_receipts))
for draft, receipt in zip(reviewer_outputs, true_receipts, strict=True):
self.assertEqual(
canonical_sha256(draft), receipt["structuredOutputSha256"]
)
def test_consistent_report_tampering_with_resigned_hashes_is_rejected(self):
"""一致篡改报告评分 + 重签 reportSha256 + 重推导 panel 仍被原始输出绑定拦下。
攻击场景:攻击者把每个 reviewer 报告的评分统一改成伪造高分,并把报告内携带的
reviewerStructuredOutputs 同步改成与伪造评分一致的伪造原始输出(使伪造自洽),
重签内层/外层 reportSha256,按同一 adjudicate_structured_reviews 重推导 panel。
唯一改不动的是执行回执里生产时固化的 structuredOutputSha256——它仍指向真实原始
输出,于是新的只读交叉校验 canonical(伪造输出) != 回执哈希 失败关闭。
"""
captured: dict[str, object] = {}
class CapturingBuilder:
"""记录一次生产 build 的全部来源,供篡改复现复用,本身仍走真 builder。"""
def build(self, **kwargs):
captured.update(kwargs)
return GateInputBuilder().build(**kwargs)
adapters, _writer, _semantic, _judge = _production_adapters(
builder_factory=CapturingBuilder
)
result, _output = self._run(adapters)
self.assertTrue(result["ok"], result)
# 正向对照:未篡改的合法报告(原始输出与报告一致)必须通过 builder,无系统失败且评分入输入。
gate_input = GateInputBuilder().build(**copy.deepcopy(captured))
self.assertEqual(gate_input["schemaVersion"], "writer-gate-input-v3")
self.assertEqual(gate_input["rubricPolicyVersion"], RUBRIC_POLICY_VERSION)
legit_sample = next(
item for item in gate_input["samples"] if item["sampleId"] == "deep-space-489"
)
self.assertFalse(legit_sample["systemFailure"])
self.assertTrue(legit_sample["schemaValid"])
self.assertIn("scores", legit_sample)
true_receipts = captured["execution_receipts"]["deep-space-489"]["A"][
"blind_judge"
]["executionReceipts"]
judge_report = copy.deepcopy(captured["judge_reports"]["deep-space-489"])
forged_score = 9.5
for index, reviewer_report in enumerate(judge_report["reviewerReports"]):
forged_output = copy.deepcopy(judge_report["reviewerStructuredOutputs"][index])
# 报告评分与伪造原始输出同步改成一致的伪造高分,使报告内部自洽。
for report_row, draft_row in zip(
reviewer_report["candidateScores"],
forged_output["candidateScores"],
strict=True,
):
for dimension in DIMENSIONS:
report_row["scores"][dimension]["score"] = forged_score
draft_row["scores"][dimension]["score"] = forged_score
judge_report["reviewerStructuredOutputs"][index] = forged_output
# 伪造输出与真实回执固化的原始输出哈希必然不同——这是攻击者改不动的锚点。
self.assertNotEqual(
canonical_sha256(forged_output),
true_receipts[index]["structuredOutputSha256"],
)
# modelReceiptSha256 保留(仍绑定未改动的真实回执),只重签内层 reportSha256。
reviewer_report["reportSha256"] = canonical_sha256(
{key: value for key, value in reviewer_report.items() if key != "reportSha256"}
)
# 按篡改后的 reviewer 报告重推导 panel,使 recomputed_panel == report 这一旧校验仍能通过。
panel = adjudicate_structured_reviews(
judge_report["reviewerReports"][0],
judge_report["reviewerReports"][1],
judge_report["reviewerReports"][2]
if len(judge_report["reviewerReports"]) == 3
else None,
)
self.assertIn(panel["status"], {"stable_report", "adjudicated_report"})
judge_report.update(panel)
judge_report["reportSha256"] = canonical_sha256(
{key: value for key, value in judge_report.items() if key != "reportSha256"}
)
tampered_judge_reports = {
**captured["judge_reports"],
"deep-space-489": judge_report,
}
# 伪造评分自洽、重签重推导都做了,仍被「原始输出 ↔ 回执哈希」交叉校验失败关闭:
# 该样本被标 systemFailure、schemaValid=False,伪造评分不进 Gate 输入的 scores。
tampered_input = GateInputBuilder().build(
**{**copy.deepcopy(captured), "judge_reports": tampered_judge_reports}
)
tampered_sample = next(
item
for item in tampered_input["samples"]
if item["sampleId"] == "deep-space-489"
)
self.assertTrue(tampered_sample["systemFailure"])
self.assertFalse(tampered_sample["schemaValid"])
self.assertNotIn("scores", tampered_sample)
self.assertTrue(
any(
"reviewerStructuredOutputs" in reason
and "未绑定模型原始 structured output" in reason
for reason in tampered_sample["systemFailureReasons"]
),
tampered_sample["systemFailureReasons"],
)
# 报告缺失 reviewerStructuredOutputs 字段时同样失败关闭:不放行无绑定的报告。
stripped = copy.deepcopy(captured["judge_reports"]["deep-space-489"])
del stripped["reviewerStructuredOutputs"]
stripped["reportSha256"] = canonical_sha256(
{key: value for key, value in stripped.items() if key != "reportSha256"}
)
stripped_input = GateInputBuilder().build(
**{
**copy.deepcopy(captured),
"judge_reports": {
**captured["judge_reports"],
"deep-space-489": stripped,
},
}
)
stripped_sample = next(
item
for item in stripped_input["samples"]
if item["sampleId"] == "deep-space-489"
)
self.assertTrue(stripped_sample["systemFailure"])
self.assertNotIn("scores", stripped_sample)
self.assertTrue(
any(
"reviewerStructuredOutputs" in reason
for reason in stripped_sample["systemFailureReasons"]
),
stripped_sample["systemFailureReasons"],
)
def test_a_semantic_failure_keeps_candidate_and_reaches_judge_and_gate(self):
"""A 臂合法 failed 是对照质量信号,不得阻断剩余臂和 Gate builder。"""
config = _production_config()
adapters, writer, semantic, judge = _production_adapters(
semantic_fail_on=_semantic_call_index_for_arm(config, "A")
)
result, output = self._run(
adapters,
evaluation_config=config,
run_id="semantic-failure-a",
)
self.assertTrue(result["ok"], result)
self.assertEqual(result["status"], "completed")
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
sample = result["samples"][0]
self.assertEqual(set(sample["candidates"]), {"A", "B", "C"})
self.assertNotIn("detector", sample)
diagnostic = sample["semanticDiagnostics"]["A"]
self.assertEqual(diagnostic["outcome"], "failed")
self.assertEqual(diagnostic["blockingCounts"]["highFindings"], 1)
self.assertEqual(diagnostic["blockingCounts"]["failedHardConstraints"], 1)
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
self.assertFalse(gate_input["samples"][0]["systemFailure"])
self.assertEqual(gate_input["samples"][0]["cArm"]["hardConstraintCoverage"], 1.0)
self.assertEqual(gate_input["samples"][0]["cArm"]["highSeverityResidualCount"], 0)
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
self.assertEqual(manifest["samples"][0]["semanticDiagnostics"], {"A": diagnostic})
self.assertEqual(
self._cas_safe_summary(output, "deep-space-489")["semanticDiagnostics"],
{"A": diagnostic},
)
serialized = json.dumps(manifest, ensure_ascii=False)
for forbidden in (
"candidateQuote",
"硬约束未满足",
"finding-high-1",
'"findingId"',
'"message"',
"/private/tmp",
):
self.assertNotIn(forbidden, serialized)
def test_judge_api_retry_aligns_null_output_and_binds_final_reports(self):
"""前置 API 失败无输出;null 占位与全回执对齐后,终稿索引仍可反篡改复核。"""
adapters, _writer, _semantic, judge = _production_adapters(
judge_api_error_first=True
)
result, output = self._run(adapters, run_id="judge-api-retry")
self.assertTrue(result["ok"], result)
self.assertEqual(len(judge.calls), 3)
receipts = json.loads((output / "execution-receipts.json").read_text())
panel_receipts = receipts["deep-space-489"]["A"]["blind_judge"][
"executionReceipts"
]
self.assertEqual(len(panel_receipts), 3)
self.assertEqual(panel_receipts[0]["terminalReason"], "api_error")
gate_input = json.loads((output / "gate-input.json").read_text())
self.assertFalse(gate_input["samples"][0]["systemFailure"])
def test_c_semantic_failure_reaches_gate_and_gate_a_fails_quality(self):
"""C 臂合法 failed 必须走完整链,并由 Gate A 的 C 臂硬门判为 failed。"""
config = _production_config()
adapters, writer, semantic, judge = _production_adapters(
semantic_fail_on=_semantic_call_index_for_arm(config, "C")
)
result, output = self._run(
adapters,
evaluation_config=config,
run_id="semantic-failure-c",
)
self.assertTrue(result["ok"], result)
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
sample = result["samples"][0]
self.assertEqual(sample["semanticDiagnostics"]["C"]["outcome"], "failed")
gate_input = json.loads((output / "gate-input.json").read_text(encoding="utf-8"))
self.assertFalse(gate_input["samples"][0]["systemFailure"])
self.assertEqual(gate_input["samples"][0]["cArm"]["hardConstraintCoverage"], 0.0)
self.assertEqual(gate_input["samples"][0]["cArm"]["highSeverityResidualCount"], 1)
gate_report = decide_gate(_five_scenario_gate_input(gate_input))
self.assertEqual(gate_report["status"], "failed")
self.assertIn("c_arm_high_severity_residual", gate_report["reasons"])
self.assertIn(
"c_arm_hard_constraint_coverage_below_100_percent",
gate_report["reasons"],
)
def test_third_judge_still_unstable_fails_run(self):
adapters, _writer, _semantic, judge = _production_adapters(judge_unstable=True)
result, _output = self._run(adapters, run_id="judge-third-unstable")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_judge_unstable")
self.assertEqual(len(judge.calls), 3)
self.assertEqual(result["budgetLedger"]["usedCalls"]["blind_judge"], 3)
self.assertEqual(result["budgetLedger"]["remainingPlannedCalls"]["blind_judge"], 0)
self.assertEqual(
result["budgetLedger"]["totalActualCostUsd"],
_expected_total_cost(fixed_role_calls=6),
)
def test_vault_migration_failure_overrides_success(self):
class MigrationFailureManager(RawVaultManager):
"""先实际迁移 raw,再模拟迁移回执失败;归档内容仍必须保留。"""
def migrate(self, lease, *, archive_root):
super().migrate(lease, archive_root=archive_root)
raise RawVaultError("RAW_MIGRATION_FAILED", "测试迁移失败")
adapters, _writer, _semantic, _judge = _production_adapters(
vault_factory=MigrationFailureManager
)
result, output = self._run(adapters, run_id="migration-failure")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_raw_migration")
cas_state = json.loads(
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text(
encoding="utf-8"
)
)
self.assertEqual(cas_state["state"], "FAILED")
self.assertEqual(cas_state["cleanupState"], "failed")
def test_cas_conflict_fails_sample_and_run(self):
class ConflictingCas:
"""初始化使用真实 CAS,第一次推进模拟迟到 revision 冲突。"""
def __init__(self, root):
self.delegate = FileCasStore(root)
def initialize(self, **kwargs):
return self.delegate.initialize(**kwargs)
def transition(self, **_kwargs):
raise CasConflictError("测试旧 revision")
adapters, _writer, _semantic, judge = _production_adapters(cas_factory=ConflictingCas)
result, _output = self._run(adapters, run_id="cas-conflict")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_cas_conflict")
self.assertEqual(judge.calls, [])
def test_builder_binding_failure_fails_after_all_sample_terminals(self):
class FailingBuilder:
"""模拟来源绑定无法闭合,证明不能手工回退聚合。"""
def build(self, **_kwargs):
raise GateInputBuildError("测试 builder 绑定失败")
adapters, writer, semantic, judge = _production_adapters(builder_factory=FailingBuilder)
result, output = self._run(adapters, run_id="builder-failure")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_gate_input_builder")
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (3, 3, 2))
self.assertFalse((output / "gate-input.json").exists())
def test_builder_failure_closes_all_multi_sample_judge_terminals_as_failed(self):
"""整轮 builder 失败时,所有已到 JUDGE_COMPLETED 的样本仍必须失败关闭。"""
class FailingBuilder:
def build(self, **_kwargs):
raise GateInputBuildError("测试多样本 builder 绑定失败")
evaluation_config = _production_config()
second_sample = copy.deepcopy(evaluation_config["samples"][0])
second_sample["sampleId"] = "deep-space-490"
evaluation_config["samples"].append(second_sample)
budget = evaluation_config["executionAuthorization"]["budget"]
for role in replay_module.BUDGET_ROLES:
budget["plannedCalls"][role] = 6
budget["maxCalls"][role] = 6
budget["totalBudgetUsd"] = "18.000000"
budget["receiptSha256"] = canonical_sha256(
{key: value for key, value in budget.items() if key != "receiptSha256"}
)
second_oracle = copy.deepcopy(_oracle_pack())
second_oracle["sampleId"] = "deep-space-490"
second_oracle["packSha256"] = canonical_sha256(
{key: value for key, value in second_oracle.items() if key != "packSha256"}
)
adapters, writer, semantic, judge = _production_adapters(
builder_factory=FailingBuilder
)
adapters = replace(
adapters,
oracle_truth_packs={
**adapters.oracle_truth_packs,
"deep-space-490": second_oracle,
},
)
result, output = self._run(
adapters,
evaluation_config=evaluation_config,
run_id="builder-failure-multi-sample",
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_gate_input_builder")
self.assertEqual((len(writer.calls), len(semantic.calls), len(judge.calls)), (6, 6, 4))
self.assertFalse((output / "gate-input.json").exists())
for sample_id in ("deep-space-489", "deep-space-490"):
cas_state = json.loads(
(output / "journal" / "cas" / sample_id / "state.json").read_text(
encoding="utf-8"
)
)
self.assertEqual(cas_state["state"], "FAILED")
self.assertEqual(cas_state["cleanupState"], "migrated")
def test_sigterm_after_vault_creation_recovers_open_lease_and_raw_vault(self):
"""create_vault 返回前被首次 SIGTERM 打断时必须恢复并关闭 raw lease。"""
class InterruptingCreateVaultManager(RawVaultManager):
def create_vault(self, **kwargs):
super().create_vault(**kwargs)
signal.raise_signal(signal.SIGTERM)
adapters, _writer, _semantic, _judge = _production_adapters(
vault_factory=InterruptingCreateVaultManager
)
result, output = self._run(adapters, run_id="sigterm-vault-create")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_interrupted")
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
self.assertEqual(manifest["status"], "failed_interrupted")
self.assertNotIn("candidateBody", json.dumps(manifest, ensure_ascii=False))
self.assertNotIn("/private/tmp", json.dumps(manifest, ensure_ascii=False))
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
self.assertEqual(len(lease_files), 1)
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
self.assertEqual(
sorted(path.name for path in (output / "journal" / "raw-vault").iterdir()),
["leases"],
)
def test_sigterm_after_cas_initialize_closes_started_cas_as_failed(self):
"""CAS.initialize 提交 STARTED 后首次 SIGTERM 也必须失败收口。"""
class InterruptingInitializeCas:
def __init__(self, root):
self.delegate = FileCasStore(root)
def initialize(self, **kwargs):
self.delegate.initialize(**kwargs)
signal.raise_signal(signal.SIGTERM)
def latest(self):
return self.delegate.latest()
def transition(self, **kwargs):
return self.delegate.transition(**kwargs)
adapters, _writer, _semantic, _judge = _production_adapters(
cas_factory=InterruptingInitializeCas
)
result, output = self._run(adapters, run_id="sigterm-cas-initialize")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_interrupted")
manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8"))
self.assertEqual(manifest["status"], "failed_interrupted")
self.assertNotIn("candidateBody", json.dumps(manifest, ensure_ascii=False))
self.assertNotIn("/private/tmp", json.dumps(manifest, ensure_ascii=False))
cas_state = json.loads(
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text(
encoding="utf-8"
)
)
self.assertEqual(cas_state["state"], "FAILED")
self.assertEqual(cas_state["cleanupState"], "migrated")
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
self.assertEqual(len(lease_files), 1)
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "migrated")
def test_pending_probe_and_budget_block_before_vault_or_runner(self):
vault_calls: list[pathlib.Path] = []
def vault_factory(path):
vault_calls.append(path)
return RawVaultManager(path)
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
pending = _production_config()
pending["commonControls"]["modelVersion"] = "pending_probe"
pending["commonControls"]["adapterVersion"] = "pending_probe"
result, _output = self._run(adapters, evaluation_config=pending, run_id="pending-probe")
self.assertEqual(result["status"], "blocked_runtime_probe")
result, _output = self._run(
adapters,
evaluation_config=_production_config(budget_approved=False),
run_id="pending-budget",
)
self.assertEqual(result["status"], "blocked_budget_authorization")
self.assertEqual(vault_calls, [])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
def test_public_test_adapter_cannot_bypass_execute_authorization(self):
"""仅传 test_adapters 不能跳过 probe、预算和 raw 审批。"""
adapters, runner, detector, judge = _test_adapters()
unauthorized = config()
unauthorized.pop("executionAuthorization", None)
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
self.addCleanup(directory.cleanup)
output = pathlib.Path(directory.name) / "run"
result = run_writer_replay(
unauthorized,
run_id="test-adapter-no-authorization",
output_dir=output,
execute=True,
test_adapters=adapters,
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_execute_authorization")
self.assertEqual(runner.calls, [])
self.assertEqual(detector.calls, [])
self.assertEqual(judge.calls, [])
self.assertFalse((output / "raw").exists())
def test_authorized_test_adapter_never_persists_plaintext_raw_directory(self):
"""fake 执行也必须使用受控 raw lease,并在返回前完成清理。"""
adapters, _runner, _detector, _judge = _test_adapters()
authorized = config()
directory = tempfile.TemporaryDirectory(dir="/private/tmp")
self.addCleanup(directory.cleanup)
output = pathlib.Path(directory.name) / "run"
result = run_writer_replay(
authorized,
run_id="test-adapter-vault",
output_dir=output,
execute=True,
test_adapters=adapters,
)
self.assertTrue(result["ok"], result)
self.assertFalse((output / "raw").exists())
self.assertNotIn("candidateBody", (output / "manifest.json").read_text(encoding="utf-8"))
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
self.assertEqual(len(lease_files), 1)
self.assertEqual(json.loads(lease_files[0].read_text())["status"], "closed")
def test_production_profile_prompt_tamper_cannot_use_noop_verifier(self):
"""生产 A/C 与 B 都必须在治理调用前执行 prompt/schema 哈希校验。"""
adapters, writer, _semantic, _judge = _production_adapters()
invalid_profile = copy.deepcopy(adapters.writer_profile)
# 模拟配置构造后内存中的 prompt 被替换;身份哈希仍绑定旧 prompt 哈希,
# 只有真实 verify_role_profile 才能在模型调用前发现。
object.__setattr__(invalid_profile, "system_prompt", "被篡改的 writer prompt")
adapters = replace(adapters, writer_profile=invalid_profile)
evaluation_config = _production_config()
result, _output = self._run(
adapters,
evaluation_config=evaluation_config,
run_id="invalid-prompt-binding",
)
self.assertFalse(result["ok"])
self.assertEqual(writer.calls, [])
self.assertNotIn("binding_verifier", inspect.signature(run_writer_replay).parameters)
def test_runtime_probe_self_hash_tamper_fails_closed_before_runner(self):
"""runtime probe receipt 自哈希被篡改时,不能进入任何模型 runner。"""
adapters, writer, semantic, judge = _production_adapters()
tampered = _production_config()
tampered["executionAuthorization"]["runtimeProbe"]["receiptSha256"] = (
"sha256:" + "0" * 64
)
result, _output = self._run(
adapters,
evaluation_config=tampered,
run_id="tampered-runtime-probe",
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_execute_authorization")
self.assertEqual(result["errors"], ["EXECUTE_AUTHORIZATION_HASH_INVALID"])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
def test_execute_blocks_stale_probe_bound_to_old_contract(self):
"""陈旧运行探针(身份哈希对不上新 writer 合同)必须被机械阻断在 runner 之前。
对照设计:与下方 test_execute_passes_probe_bound_to_current_contract 共用同一
_production_config,只把探针的身份哈希/输出哈希换成旧合同的值。旧探针只证明
「旧合同能产出结构化输出」,证明不了换过 schema/prompt 的新合同也可以,故必须挡下。
"""
adapters, writer, semantic, judge = _production_adapters()
stale = _production_config()
probe = stale["executionAuthorization"]["runtimeProbe"]
# 模拟探针仍是旧合同实跑的:身份哈希与结构化输出哈希都来自旧合同。
probe["executionProfileSha256"] = "sha256:" + "9" * 64
probe["structuredOutputSha256"] = "sha256:" + "4" * 64
# 重算探针自哈希,确保它卡在「合同绑定」这一关,而不是更早的自哈希这一关。
probe["receiptSha256"] = canonical_sha256(
{key: value for key, value in probe.items() if key != "receiptSha256"}
)
result, _output = self._run(
adapters, evaluation_config=stale, run_id="stale-probe-old-contract"
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_execute_probe_contract")
self.assertEqual(result["errors"], ["EXECUTE_PROBE_CONTRACT_MISMATCH"])
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
def test_execute_passes_probe_bound_to_current_contract(self):
"""按新合同重跑的探针(身份哈希==新 writer profile)必须放行 execute 门。
与上方陈旧探针对照:同一 _production_config,探针身份哈希绑定当前 writer 合同,
execute 门放行并真正进入模型 runner 跑完全链。
"""
adapters, writer, _semantic, _judge = _production_adapters()
bound = _production_config()
probe = bound["executionAuthorization"]["runtimeProbe"]
# 显式确认探针身份哈希已绑定当前 writer 合同(新 schema/prompt 的身份)。
self.assertEqual(
probe["executionProfileSha256"],
adapters.writer_profile.execution_profile_sha256,
)
result, _output = self._run(
adapters, evaluation_config=bound, run_id="probe-current-contract"
)
self.assertTrue(result["ok"], result)
self.assertEqual(result["status"], "completed")
self.assertEqual(len(writer.calls), 3)
def test_termination_handlers_install_and_restore_without_killing_process(self):
"""handler 安装/恢复必须是确定性的:装上后信号被接管,恢复后还原调用方原 handler。"""
old_term = signal.getsignal(signal.SIGTERM)
old_int = signal.getsignal(signal.SIGINT)
previous = replay_module.install_termination_handlers()
try:
self.assertIn(signal.SIGTERM, previous)
self.assertIn(signal.SIGINT, previous)
self.assertIsNot(signal.getsignal(signal.SIGTERM), old_term)
self.assertIsNot(signal.getsignal(signal.SIGINT), old_int)
finally:
replay_module.restore_termination_handlers(previous)
self.assertIs(signal.getsignal(signal.SIGTERM), old_term)
self.assertIs(signal.getsignal(signal.SIGINT), old_int)
def test_sigterm_during_writer_call_fails_closed_with_cost_unknown(self):
"""终止落在已发起但无回执的 writer 调用中:成本未知、raw 仍迁移、manifest 安全。"""
old_term = signal.getsignal(signal.SIGTERM)
old_int = signal.getsignal(signal.SIGINT)
adapters, _writer, semantic, judge = _production_adapters()
adapters = replace(adapters, writer_runner=InterruptingChat(interrupt_on_call=1))
result, output = self._run(adapters, run_id="sigterm-writer")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_interrupted")
self.assertEqual(result["errors"], ["EXECUTION_COST_UNKNOWN"])
self.assertTrue(result["budgetLedger"]["costUnknown"])
self.assertEqual(result["budgetLedger"]["failureReason"], "EXECUTION_COST_UNKNOWN")
# 终止之后不得再发起任何 detector / judge 调用。
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
# manifest 必须是安全字段:不含正文、不含 raw 路径。
manifest_text = (output / "manifest.json").read_text(encoding="utf-8")
self.assertNotIn("candidateBody", manifest_text)
self.assertNotIn("/private/tmp", manifest_text)
# open lease 必须被收敛为 migrated,临时 vault 消失但受控归档保留。
lease_files = list((output / "journal" / "raw-vault" / "leases").glob("*.json"))
self.assertEqual(len(lease_files), 1)
lease = json.loads(lease_files[0].read_text(encoding="utf-8"))
self.assertEqual(lease["status"], "migrated")
archive_root = output.parent / "sigterm-writer-raw-archive"
self.assertTrue(
(archive_root / f"muse-raw-archive-{lease['archiveId']}").is_dir()
)
# 运行 journal 只留非敏感 lease;raw 已迁往独立受控归档。
self.assertEqual(
[path.name for path in (output / "journal" / "raw-vault").iterdir()],
["leases"],
)
cas_state = json.loads(
(output / "journal" / "cas" / "deep-space-489" / "state.json").read_text(
encoding="utf-8"
)
)
self.assertEqual(cas_state["state"], "FAILED")
self.assertEqual(cas_state["cleanupState"], "migrated")
self.assertIs(signal.getsignal(signal.SIGTERM), old_term)
self.assertIs(signal.getsignal(signal.SIGINT), old_int)
def test_termination_is_deferred_during_migration_and_restored_after_manifest(self):
"""最终迁移和 manifest 写入期间必须忽略后续终止,返回后恢复原 handler。"""
handler_states: list[tuple[object, object]] = []
class HandlerObservingManager(RawVaultManager):
def migrate(self, lease, *, archive_root):
handler_states.append(
(
signal.getsignal(signal.SIGTERM),
signal.getsignal(signal.SIGINT),
)
)
return super().migrate(lease, archive_root=archive_root)
old_term = signal.getsignal(signal.SIGTERM)
old_int = signal.getsignal(signal.SIGINT)
adapters, _writer, _semantic, _judge = _production_adapters(
vault_factory=HandlerObservingManager
)
adapters = replace(adapters, writer_runner=InterruptingChat(interrupt_on_call=1))
result, _output = self._run(adapters, run_id="sigterm-migration-handler")
self.assertFalse(result["ok"])
self.assertEqual(handler_states, [(signal.SIG_IGN, signal.SIG_IGN)])
self.assertIs(signal.getsignal(signal.SIGTERM), old_term)
self.assertIs(signal.getsignal(signal.SIGINT), old_int)
def test_short_retention_lease_blocks_before_vault_contents_or_runner(self):
"""27 秒短租约必须在落 lease / 建 vault / 调模型前稳定失败关闭。"""
vault_calls: list[pathlib.Path] = []
def vault_factory(path):
vault_calls.append(path)
return RawVaultManager(path)
adapters, writer, semantic, judge = _production_adapters(vault_factory=vault_factory)
short = _production_config()
raw = short["executionAuthorization"]["rawRetention"]
raw["retainUntil"] = (datetime.now(UTC) + timedelta(seconds=27)).isoformat()
raw["receiptSha256"] = canonical_sha256(
{key: value for key, value in raw.items() if key != "receiptSha256"}
)
result, output = self._run(adapters, evaluation_config=short, run_id="short-lease")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "failed_raw_vault")
self.assertEqual(result["errors"], ["RAW_LEASE_INSUFFICIENT_RETENTION"])
# 失败关闭在任何模型 runner 之前。
self.assertEqual(writer.calls, [])
self.assertEqual(semantic.calls, [])
self.assertEqual(judge.calls, [])
# 不能留下 lease journal 或 raw vault 目录(manager 外壳目录可有,内容必须空)。
self.assertEqual(list((output / "journal" / "raw-vault" / "leases").glob("*.json")), [])
class SemanticInputSourceRefCleaningTest(unittest.TestCase):
"""验证 _semantic_input_v3 投影检测输入时清洗证据 sourceRef 的多余字段。
C 臂(卡索引+原文臂)的 proseEvidence sourceRef 带 sourceType(如 card_chapter_proxy),
检测输入校验把 sourceRef 当闭集会拒收多余字段。投影时必须清洗成合同形状,且 writer_context
本身不动(writer 看到的 proseEvidence 仍带 sourceType,那是 writer 侧合同)。
"""
PROSE_TEXT = "脱敏原文片段,C 臂知识卡投影。"
@classmethod
def _writer_context(cls) -> dict[str, object]:
prose_hash = "sha256:" + hashlib.sha256(cls.PROSE_TEXT.encode("utf-8")).hexdigest()
prose_source_ref = {
"sourceId": "fixture:card-chapter-proxy:1",
"sourceVersion": "cards-frozen-488-v1",
"chapter": 488,
"blockId": 1120,
"startCodePoint": 0,
"endCodePoint": len(cls.PROSE_TEXT),
"contentSha256": prose_hash,
# writer 侧合同允许、但检测闭集拒收的多余字段:
"sourceType": "card_chapter_proxy",
"cardId": "card-1",
}
fact_source_ref = {
"sourceId": "fixture:fact:1",
"sourceVersion": "canonical-v488",
"contentSha256": "sha256:" + "8" * 64,
"sourceType": "canonical_state",
}
return {
"fineOutline": {
"sourceRef": {
"sourceId": "fixture:outline:489",
"sourceVersion": "outline-frozen-489-v1",
},
"hardConstraints": ["人物保持冻结状态"],
"adjustableBeats": ["过场节奏可调"],
"declaredNewFacts": [],
},
"contextSnapshot": {"contextSha256": "sha256:" + "5" * 64},
"asOf": 488,
"authorizationSnapshot": {"snapshotId": "auth-work-8"},
"factEvidence": [
{
"evidenceId": "fact-ev-1",
"fact": "林澈仍在圣蒂曼",
"sourceType": "canonical_state",
"sourceRef": fact_source_ref,
"contentSha256": "sha256:" + "8" * 64,
"riskLevel": "low",
}
],
"proseEvidence": [
{
"evidenceId": "prose-ev-1",
"chapter": 488,
"sourceRef": prose_source_ref,
"contentSha256": prose_hash,
"purpose": "style_baseline",
"text": cls.PROSE_TEXT,
"isRecentBaseline": True,
}
],
}
@staticmethod
def _candidate() -> dict[str, object]:
body = "候选正文。"
return {
"candidateVersion": 1,
"candidateSha256": "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest(),
"candidateBody": body,
}
def _project(self, writer_context: dict[str, object], opaque_arm_id: str) -> dict[str, object]:
return _semantic_input_v3(
run_id="run-clean",
sample_id="deep-space-321",
opaque_arm_id=opaque_arm_id,
writer_context=writer_context,
candidate=self._candidate(),
)
def test_prose_evidence_source_ref_drops_extra_fields(self):
result = self._project(self._writer_context(), "blind-1")
cleaned_ref = result["proseEvidence"][0]["sourceRef"]
# 多余字段(sourceType/cardId)被清洗掉。
self.assertNotIn("sourceType", cleaned_ref)
self.assertNotIn("cardId", cleaned_ref)
# 只保留检测闭集允许的字段。
self.assertLessEqual(set(cleaned_ref), _SOURCE_REF_ALLOWED)
# 关键定位字段保留。
self.assertEqual(cleaned_ref["sourceId"], "fixture:card-chapter-proxy:1")
self.assertEqual(cleaned_ref["sourceVersion"], "cards-frozen-488-v1")
self.assertEqual(cleaned_ref["chapter"], 488)
self.assertEqual(cleaned_ref["blockId"], 1120)
# 证据其它字段原样保留(只清洗 sourceRef 子对象)。
prose_item = result["proseEvidence"][0]
self.assertEqual(prose_item["evidenceId"], "prose-ev-1")
self.assertEqual(prose_item["purpose"], "style_baseline")
self.assertIs(prose_item["isRecentBaseline"], True)
self.assertEqual(prose_item["text"], self.PROSE_TEXT)
def test_fact_evidence_source_ref_drops_extra_fields(self):
result = self._project(self._writer_context(), "blind-2")
cleaned_ref = result["factEvidence"][0]["sourceRef"]
self.assertNotIn("sourceType", cleaned_ref)
self.assertLessEqual(set(cleaned_ref), _SOURCE_REF_ALLOWED)
self.assertEqual(cleaned_ref["sourceId"], "fixture:fact:1")
self.assertEqual(cleaned_ref["sourceVersion"], "canonical-v488")
# factEvidence 条目自身的 sourceType(检测合同要求)保留,只有 sourceRef 子对象被清洗。
self.assertEqual(result["factEvidence"][0]["sourceType"], "canonical_state")
def test_writer_context_source_ref_untouched(self):
writer_context = self._writer_context()
self._project(writer_context, "blind-3")
# writer 侧合同不变:投影后 writer_context 的 sourceRef 仍带 sourceType。
self.assertEqual(
writer_context["proseEvidence"][0]["sourceRef"]["sourceType"],
"card_chapter_proxy",
)
self.assertEqual(
writer_context["factEvidence"][0]["sourceRef"]["sourceType"],
"canonical_state",
)
def test_clean_source_ref_non_mapping_passthrough(self):
# ref 不是 dict 时原样返回,交由下游检测校验按合同处理。
self.assertEqual(_clean_source_ref("fixture:plain"), "fixture:plain")
self.assertIsNone(_clean_source_ref(None))
if __name__ == "__main__":
unittest.main()