656 lines
32 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""回放编排器和安全摘要的无网络测试。"""
from __future__ import annotations
import json
import pathlib
import sys
import tempfile
import unittest
from unittest.mock import patch
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
SKILLS_DIR = PROJECT_ROOT / ".agent" / "skills"
SCRIPT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "replay" / "回放评估细纲质量" / "scripts"
TEST_DIR = PROJECT_ROOT / "tests" / "skills" / "回放评估细纲质量"
QUALITY_GATE_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "judge" / "评估内容质量" / "scripts"
REPLAY_GATE_TEST_DIR = PROJECT_ROOT / "tests" / "skills" / "判定质量是否合格"
for _import_dir in (SCRIPT_DIR, TEST_DIR, QUALITY_GATE_DIR, REPLAY_GATE_TEST_DIR):
if str(_import_dir) not in sys.path:
sys.path.insert(0, str(_import_dir))
from muse_role import FIXED_OPUS_MODEL_ID # noqa: E402
from run_replay import _parse_args, _planner_prompt # noqa: E402
from run_replay import run_replay as _run_replay
from test_writer_gate import build_input, source_bundle # noqa: E402
from write_report import render_report # noqa: E402
from writer_gate import issue_gate_report_and_receipt # noqa: E402
SOURCE_HASH = "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4"
SOURCE_VERSION = f"raw-file-v1:{SOURCE_HASH}"
AUTH = {
"sourceStatus": "active",
"copyrightStatus": "research_only",
"sourceHash": SOURCE_HASH,
"sourceVersion": SOURCE_VERSION,
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution"],
"authorizationSnapshot": {
"id": "auth-1",
"version": "v1",
"immutable": True,
"sourceHash": SOURCE_HASH,
"sourceVersion": SOURCE_VERSION,
"sourceStatus": "active",
"copyrightStatus": "research_only",
"authorizationBasis": "user_authorization",
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution"],
"checkedAt": "2026-07-19T00:00:00Z",
"revalidationAt": "2099-07-20T00:00:00Z",
},
}
def run_replay(config_value, output_dir, **kwargs):
"""旧执行测试统一经过真实本地 Gate 签发链,不提供生产绕过参数。"""
if kwargs.get("mode", "dry_run") != "execute":
return _run_replay(config_value, output_dir, **kwargs)
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
gate_root = pathlib.Path(directory)
gate_input = build_input(source_bundle())
gate_input_path = gate_root / "gate-input.json"
gate_input_path.write_text(json.dumps(gate_input, ensure_ascii=False), encoding="utf-8")
issue_gate_report_and_receipt(
gate_input=gate_input,
expected_input_sha256=gate_input["gateInputSha256"],
output_dir=gate_root,
cas_dir=gate_root / "cas",
)
return _run_replay(
config_value,
output_dir,
writer_gate_receipt_path=gate_root / "writer-gate-b-passed-receipt.json",
writer_gate_report_path=gate_root / "gate-report.json",
writer_gate_input_path=gate_input_path,
writer_gate_cas_dir=gate_root / "cas",
**kwargs,
)
def config():
common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}}
return {
"runId": "smoke-001",
"referenceWork": {"id": "deep-space", "version": SOURCE_VERSION},
"evaluationSetVersion": "set-v1",
"strategyVersion": "strategy-v1",
"authorization": AUTH,
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
"leakageAudit": {
"targetFacts": {
"targetChapter": 489,
"forbiddenFacts": [
{
"id": "future-fact-1",
"firstChapter": 489,
"text": "目标章未进入快照",
}
],
}
},
"targetChapter": 489,
"snapshot": {
"asOfChapter": 488,
"snapshotVersion": "next_fine_outline_replay_v0",
"data": {
"milestones": [{"chapter": 488, "fact": "安全历史"}, {"chapter": 489, "fact": "未来"}],
"cards": [{"name": "已知实体", "milestones": [{"chapter": 488, "step": "历史"}]}],
},
},
"sources": [{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
"commonContext": common,
"arms": {
"outline_only": {"cards": [], "cardSourceIds": [], "cardStrategy": "none"},
"outline_plus_cards": {"cards": [{"name": "正确卡"}], "cardSourceIds": ["card-correct-1"], "cardStrategy": "correct"},
"outline_plus_placebo_cards": {"cards": [{"name": "错配卡"}], "cardSourceIds": ["card-placebo-1"], "cardStrategy": "placebo"},
},
}
def candidate(goal="机制 smoke", *, events=True):
"""构造满足细纲闭集合同的合成候选。"""
return {
"targetChapter": 489,
"chapterGoal": goal,
"keyEvents": (
[
{
"id": "event-1",
"order": 1,
"event": "侦察敌情",
"participants": ["测试角色"],
"trigger": "收到异常信号",
"resultDirection": "确认威胁存在",
}
]
if events
else []
),
"entities": [],
"foreshadowing": [],
"stateChanges": [],
"hook": "下一步",
"unknowns": [],
"assumptions": [],
}
def write_fake_chat(directory, mode="stable"):
"""构造与 muse_llm.chat_governed 同签名的假治理调用,测试不触发真实模型。
角色识别依据派发方注入的系统提示词(planner 身份段 / detector / judge 角色文件
frontmatter name),与 run_replay 的注入合同一致。调用记录沿用 agent-calls.jsonl。
"""
directory = pathlib.Path(directory)
log_path = directory / "agent-calls.jsonl"
count_path = directory / "planner-count.txt"
run_result_path = directory / "run" / "run_result.json"
usage = {"input_tokens": 1, "output_tokens": 1}
def fake_chat(prompt, system=None, temperature=0.2, top_p=None, *,
run_id=None, caller=None, persist_call=None, **_kwargs):
system_text = system or ""
if "planner identity" in system_text:
agent = "planner"
elif "name: detector" in system_text:
agent = "detector"
elif "name: judge" in system_text:
agent = "judge"
else:
agent = "unknown"
run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None
with log_path.open("a", encoding="utf-8") as handle:
handle.write(json.dumps({"agent": agent, "prompt": prompt, "runState": run_state}, ensure_ascii=False) + "\n")
if mode == f"{agent}_timeout":
raise TimeoutError("fake timeout")
if mode == f"{agent}_exit":
raise RuntimeError("fake execution failure")
if mode == f"{agent}_invalid_json":
return "{invalid-json", usage, FIXED_OPUS_MODEL_ID
if agent == "planner":
count = int(count_path.read_text() if count_path.exists() else "0") + 1
count_path.write_text(str(count))
events = not (mode == "invalid_candidate" and count == 2)
value = candidate()
value["chapterGoal"] = f"goal-{count}"
if not events:
value["keyEvents"] = []
return json.dumps(value, ensure_ascii=False), usage, FIXED_OPUS_MODEL_ID
if agent == "detector":
request = json.loads(prompt)
findings = []
if mode == "high_detector" and request["candidate"]["chapterGoal"] == "goal-2":
findings.append({"severity": "high", "category": "entity_state", "location": "keyEvents[0]", "evidenceSummary": "冻结事实冲突"})
if mode == "invalid_detector":
findings.append({"category": "entity_state"})
return json.dumps({"protocol": "fine_outline_detector_v0", "candidateId": request["candidateId"], "findings": findings, "coverageFindings": []}, ensure_ascii=False), usage, FIXED_OPUS_MODEL_ID
if agent == "judge":
request = json.loads(prompt)
dimensions = ["structure_completeness", "direction_causality", "order_pacing", "entity_state", "foreshadowing_action", "handoff_hook"]
evaluations = []
for item in request["candidates"]:
base = {"goal-1": 3, "goal-2": 4, "goal-3": 2}[item["candidate"]["chapterGoal"]]
scores = {dimension: {"score": base, "evidence": f"{dimension}-结构化证据"} for dimension in dimensions}
if mode == "unstable" and request["judgeId"] == "judge-secondary":
scores["order_pacing"]["score"] = min(5, base + 1)
if mode == "invalid_rubric":
scores["order_pacing"]["score"] = 6
evaluations.append({"candidateId": item["candidateId"], "scores": scores, "summary": "结构化评分摘要"})
return json.dumps({"profile": "fine_outline_replay", "judgeId": request["judgeId"], "evaluations": evaluations}, ensure_ascii=False), usage, FIXED_OPUS_MODEL_ID
raise RuntimeError(f"未知角色: {agent}")
return fake_chat, log_path
def read_calls(log_path):
if not log_path.exists():
return []
return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()]
def nested_keys(value):
"""收集嵌套 JSON 的全部字段名,供盲化边界测试使用。"""
if isinstance(value, dict):
keys = set(value)
for item in value.values():
keys.update(nested_keys(item))
return keys
if isinstance(value, list):
keys = set()
for item in value:
keys.update(nested_keys(item))
return keys
return set()
class ReplayRunTest(unittest.TestCase):
def test_execute_without_signed_writer_gate_receipt_is_rejected(self):
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
result = _run_replay(
config(),
pathlib.Path(directory) / "run",
mode="execute",
)
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_writer_gate_receipt")
def test_public_planner_context_does_not_include_snapshot_cards(self):
prompt = _planner_prompt(
target=489,
as_of=488,
snapshot={"cards": [{"name": "hidden-card"}], "safe": "public"},
common_input={},
cards=[{"name": "injected-card"}],
)
self.assertNotIn("hidden-card", prompt)
self.assertIn("injected-card", prompt)
def test_dry_run_passes_and_writes_only_metadata(self):
with tempfile.TemporaryDirectory() as directory:
result = run_replay(config(), pathlib.Path(directory), mode="dry_run")
self.assertTrue(result["ok"])
self.assertEqual(result["status"], "ready")
manifest = json.loads((pathlib.Path(directory) / "snapshot_manifest.json").read_text())
self.assertNotIn("payload", json.dumps(manifest, ensure_ascii=False))
self.assertFalse(list(pathlib.Path(directory).glob("planner_*.raw.json")))
def test_bad_authorization_stops_before_model(self):
bad = config()
bad["authorization"] = {"sourceStatus": "active"}
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_authorization")
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_reused_output_dir_bad_config_overwrites_previous_completed_state(self):
with tempfile.TemporaryDirectory() as directory:
output_dir = pathlib.Path(directory) / "run"
output_dir.mkdir()
result_path = output_dir / "run_result.json"
result_path.write_text(
json.dumps({"runId": "old-run", "status": "completed", "ok": True}),
encoding="utf-8",
)
bad = config()
bad["snapshot"]["data"]["unknownSection"] = []
with self.assertRaisesRegex(ValueError, "未登记顶层分区"):
run_replay(bad, output_dir, mode="execute")
persisted = json.loads(result_path.read_text(encoding="utf-8"))
self.assertEqual(persisted["runId"], "smoke-001")
self.assertEqual(persisted["status"], "config_invalid")
self.assertFalse(persisted["ok"])
self.assertNotEqual(persisted["status"], "completed")
def test_content_leak_stops_before_manifest(self):
bad = config()
bad["leakageAudit"]["targetFacts"]["forbiddenFacts"][0]["text"] = "安全历史"
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_snapshot")
self.assertEqual(result["leakageAudit"]["findingCount"], 1)
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_missing_content_audit_stops_before_model(self):
bad = config()
del bad["leakageAudit"]
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_leakage_audit")
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_arm_card_future_record_stops_before_model(self):
bad = config()
bad["arms"]["outline_plus_cards"]["cards"] = [
{"name": "未来卡", "milestones": [{"chapter": 489, "fact": "未来"}]}
]
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_snapshot")
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_report_contains_hashes_but_not_raw_fields(self):
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
report = render_report(result)
self.assertIn("smoke-001", report)
self.assertIn("snapshotManifestSha256", report)
self.assertNotIn('"prompt":', report)
self.assertNotIn('"payload":', report)
def test_report_rejects_text_injection_through_identifier_fields(self):
base = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
injections = {
"runId": "safe-run\n完整目标细纲:第一幕到第三幕的全部事件原文",
"referenceWork": "深空之影目标章原文与完整细纲",
}
for field, injected in injections.items():
with self.subTest(field=field):
unsafe = dict(base)
unsafe[field] = injected
with self.assertRaisesRegex(ValueError, field):
render_report(unsafe)
def test_report_does_not_render_preflight_free_text(self):
bad = config()
injected = "目标章事实文本"
bad["authorization"] = {**bad["authorization"], "sourceStatus": injected}
result = run_replay(bad, pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
report = render_report(result)
self.assertNotIn(injected, report)
self.assertIn("授权前置门未通过", report)
def test_report_accepts_registered_target_source_failure_status(self):
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
result["results"] = {
"outline_only": {"status": "target_source_forbidden", "ok": False}
}
report = render_report(result)
self.assertIn("target_source_forbidden", report)
def test_execute_uses_external_planner_and_validates_each_candidate(self):
with tempfile.TemporaryDirectory() as directory:
fake, log_path = write_fake_chat(directory)
output_dir = pathlib.Path(directory) / "run"
result = run_replay(config(), output_dir, mode="execute", governed_chat=fake)
first_call = read_calls(log_path)[0]
self.assertTrue(result["ok"])
self.assertEqual(result["status"], "completed")
self.assertFalse(first_call["runState"]["ok"])
self.assertEqual(first_call["runState"]["status"], "running_planner")
self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"})
self.assertTrue(list(output_dir.glob("candidate_*.json")))
raw_receipts = list(output_dir.glob("*.raw.json"))
self.assertTrue(raw_receipts)
for path in raw_receipts:
payload = json.loads(path.read_text(encoding="utf-8"))
self.assertEqual(
set(payload),
{"rawRequestSha256", "rawResponseSha256", "receipt"},
)
self.assertNotIn("rawRequest", payload)
self.assertNotIn("rawResponse", payload)
def test_model_channel_failure_fails_closed_and_persists_result(self):
with tempfile.TemporaryDirectory() as directory:
output_dir = pathlib.Path(directory) / "run"
def broken_chat(*_args, **_kwargs):
raise OSError("模型通道不可用")
try:
result = run_replay(config(), output_dir, mode="execute", governed_chat=broken_chat)
except OSError as error:
self.fail(f"底座启动异常不得逃逸: {error}")
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "planner_failed")
self.assertEqual(persisted["status"], "planner_failed")
self.assertFalse(persisted["ok"])
self.assertNotIn(persisted["status"], {"ready", "completed"})
def test_planner_nonzero_exit_stops_after_first_call(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "planner_exit")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
self.assertEqual(result["status"], "planner_failed")
self.assertFalse(result["ok"])
self.assertEqual(len(calls), 1)
def test_planner_invalid_json_stops_after_first_call(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "planner_invalid_json")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
self.assertEqual(result["status"], "planner_invalid")
self.assertFalse(result["ok"])
self.assertEqual(len(calls), 1)
def test_planner_timeout_fails_closed_and_persists_timeout_status(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "planner_timeout")
output_dir = pathlib.Path(directory) / "run"
result = run_replay(
config(),
output_dir,
mode="execute",
governed_chat=runner,
timeout_seconds=1.0,
)
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
self.assertEqual(result["status"], "planner_timeout")
self.assertFalse(result["ok"])
self.assertEqual(persisted["status"], "planner_timeout")
self.assertEqual(len(read_calls(log_path)), 1)
def test_detector_nonzero_exit_stops_before_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "detector_exit")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_failed")
self.assertFalse(result["ok"])
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3)
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_detector_invalid_json_stops_before_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "detector_invalid_json")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_invalid")
self.assertFalse(result["ok"])
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_detector_timeout_fails_closed_before_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "detector_timeout")
result = run_replay(
config(),
pathlib.Path(directory) / "run",
mode="execute",
governed_chat=runner,
timeout_seconds=1.0,
)
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_timeout")
self.assertFalse(result["ok"])
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_primary_judge_nonzero_exit_does_not_call_secondary(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "judge_exit")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "judge_failed")
self.assertFalse(result["ok"])
self.assertEqual(len(judge_calls), 1)
def test_primary_judge_invalid_json_does_not_call_secondary(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "judge_invalid_json")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "judge_invalid")
self.assertFalse(result["ok"])
self.assertEqual(len(judge_calls), 1)
def test_primary_judge_timeout_does_not_call_secondary(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "judge_timeout")
result = run_replay(
config(),
pathlib.Path(directory) / "run",
mode="execute",
governed_chat=runner,
timeout_seconds=1.0,
)
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "judge_timeout")
self.assertFalse(result["ok"])
self.assertEqual(len(judge_calls), 1)
def test_cli_accepts_subprocess_timeout_seconds(self):
argv = [
"run_replay.py",
"--config",
"/tmp/replay-config.json",
"--output-dir",
"/tmp/replay-output",
"--timeout-seconds",
"12.5",
]
with patch.object(sys, "argv", argv):
args = _parse_args()
self.assertEqual(args.timeout_seconds, 12.5)
def test_non_finite_timeout_is_persisted_as_config_invalid(self):
with tempfile.TemporaryDirectory() as directory:
for index, timeout_seconds in enumerate((float("nan"), float("inf"))):
with self.subTest(timeout_seconds=timeout_seconds):
output_dir = pathlib.Path(directory) / f"run-{index}"
with self.assertRaisesRegex(ValueError, "timeout_seconds"):
run_replay(
config(),
output_dir,
mode="execute",
timeout_seconds=timeout_seconds,
)
persisted = json.loads(
(output_dir / "run_result.json").read_text(encoding="utf-8")
)
self.assertEqual(persisted["status"], "config_invalid")
self.assertFalse(persisted["ok"])
def test_schema_failure_stops_before_all_detectors_and_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "invalid_candidate")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
self.assertEqual(result["status"], "planner_invalid")
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 2)
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 0)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_high_detector_finding_blocks_group_and_never_calls_judge(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "high_detector")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
detector_calls = [call for call in calls if call["agent"] == "detector"]
self.assertEqual(result["status"], "detector_blocked")
self.assertEqual(len(detector_calls), 3)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls))
self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls))
def test_detector_requests_cannot_distinguish_arm_specific_card_identity(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "stable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
self.assertEqual(result["status"], "completed")
requests = [
json.loads(call["prompt"])
for call in read_calls(log_path)
if call["agent"] == "detector"
]
self.assertEqual(len(requests), 3)
for request in requests:
self.assertTrue(
nested_keys(request).isdisjoint({"arm", "cardInjection", "cardManifest"})
)
serialized = json.dumps(request, ensure_ascii=False)
self.assertNotIn("正确卡", serialized)
self.assertNotIn("错配卡", serialized)
self.assertNotIn("card-correct-1", serialized)
self.assertNotIn("card-placebo-1", serialized)
public_parts = [
{key: value for key, value in request.items() if key not in {"candidateId", "candidate"}}
for request in requests
]
self.assertEqual(public_parts[0], public_parts[1])
self.assertEqual(public_parts[1], public_parts[2])
def test_invalid_detector_report_fails_closed_before_judge(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "invalid_detector")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_invalid")
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_two_judges_are_independent_blind_reversed_and_unblinded(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_chat(directory, "stable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "completed")
self.assertEqual(len(judge_calls), 2)
first = json.loads(judge_calls[0]["prompt"])
second = json.loads(judge_calls[1]["prompt"])
self.assertNotEqual(first["judgeId"], second["judgeId"])
self.assertEqual(
[item["candidateId"] for item in second["candidates"]],
list(reversed([item["candidateId"] for item in first["candidates"]])),
)
self.assertNotIn("outline_only", judge_calls[0]["prompt"])
self.assertNotIn("outline_plus_cards", judge_calls[1]["prompt"])
self.assertTrue(result["evaluation"]["stability"]["stable"])
self.assertEqual(result["evaluation"]["deltas"]["B-A"]["order_pacing"], 1.0)
self.assertEqual(result["evaluation"]["deltas"]["C-A"]["order_pacing"], -1.0)
def test_invalid_rubric_report_does_not_complete(self):
with tempfile.TemporaryDirectory() as directory:
runner, _ = write_fake_chat(directory, "invalid_rubric")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
self.assertEqual(result["status"], "judge_invalid")
self.assertFalse(result["ok"])
def test_unstable_judges_have_explicit_non_completed_status(self):
with tempfile.TemporaryDirectory() as directory:
runner, _ = write_fake_chat(directory, "unstable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
self.assertEqual(result["status"], "judge_unstable")
self.assertFalse(result["ok"])
self.assertFalse(result["evaluation"]["stability"]["stable"])
self.assertNotIn("deltas", result["evaluation"])
def test_report_contains_only_aggregated_evaluation(self):
with tempfile.TemporaryDirectory() as directory:
runner, _ = write_fake_chat(directory, "stable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
report = render_report(result)
self.assertIn("B-A", report)
self.assertIn("C-A", report)
self.assertIn("评委稳定性", report)
self.assertNotIn("结构化证据", report)
self.assertNotIn("目标章未进入快照", report)
if __name__ == "__main__":
unittest.main()