592 lines
29 KiB
Python
592 lines
29 KiB
Python
#!/usr/bin/env python3
|
|
"""回放编排器和安全摘要的无网络测试。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import pathlib
|
|
import sys
|
|
import tempfile
|
|
import unittest
|
|
from unittest.mock import patch
|
|
|
|
sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent))
|
|
from run_replay import _parse_args, _planner_prompt, run_replay # noqa: E402
|
|
from write_report import render_report # noqa: E402
|
|
|
|
|
|
SOURCE_HASH = "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4"
|
|
SOURCE_VERSION = f"raw-file-v1:{SOURCE_HASH}"
|
|
AUTH = {
|
|
"sourceStatus": "active",
|
|
"copyrightStatus": "research_only",
|
|
"sourceHash": SOURCE_HASH,
|
|
"sourceVersion": SOURCE_VERSION,
|
|
"allowedPurpose": ["offline_evaluation"],
|
|
"forbiddenPurpose": ["external_distribution"],
|
|
"authorizationSnapshot": {
|
|
"id": "auth-1",
|
|
"version": "v1",
|
|
"immutable": True,
|
|
"sourceHash": SOURCE_HASH,
|
|
"sourceVersion": SOURCE_VERSION,
|
|
"sourceStatus": "active",
|
|
"copyrightStatus": "research_only",
|
|
"authorizationBasis": "user_authorization",
|
|
"allowedPurpose": ["offline_evaluation"],
|
|
"forbiddenPurpose": ["external_distribution"],
|
|
"checkedAt": "2026-07-19T00:00:00Z",
|
|
"revalidationAt": "2099-07-20T00:00:00Z",
|
|
},
|
|
}
|
|
|
|
|
|
def config():
|
|
common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}}
|
|
return {
|
|
"runId": "smoke-001",
|
|
"referenceWork": {"id": "deep-space", "version": SOURCE_VERSION},
|
|
"evaluationSetVersion": "set-v1",
|
|
"strategyVersion": "strategy-v1",
|
|
"authorization": AUTH,
|
|
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
|
|
"leakageAudit": {
|
|
"targetFacts": {
|
|
"targetChapter": 489,
|
|
"forbiddenFacts": [
|
|
{
|
|
"id": "future-fact-1",
|
|
"firstChapter": 489,
|
|
"text": "目标章未进入快照",
|
|
}
|
|
],
|
|
}
|
|
},
|
|
"targetChapter": 489,
|
|
"snapshot": {
|
|
"asOfChapter": 488,
|
|
"snapshotVersion": "next_fine_outline_replay_v0",
|
|
"data": {
|
|
"milestones": [{"chapter": 488, "fact": "安全历史"}, {"chapter": 489, "fact": "未来"}],
|
|
"cards": [{"name": "已知实体", "milestones": [{"chapter": 488, "step": "历史"}]}],
|
|
},
|
|
},
|
|
"sources": [{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
|
|
"commonContext": common,
|
|
"arms": {
|
|
"outline_only": {"cards": [], "cardSourceIds": [], "cardStrategy": "none"},
|
|
"outline_plus_cards": {"cards": [{"name": "正确卡"}], "cardSourceIds": ["card-correct-1"], "cardStrategy": "correct"},
|
|
"outline_plus_placebo_cards": {"cards": [{"name": "错配卡"}], "cardSourceIds": ["card-placebo-1"], "cardStrategy": "placebo"},
|
|
},
|
|
}
|
|
|
|
|
|
def candidate(goal="机制 smoke", *, events=True):
|
|
"""构造满足细纲闭集合同的合成候选。"""
|
|
|
|
return {
|
|
"targetChapter": 489,
|
|
"chapterGoal": goal,
|
|
"keyEvents": (
|
|
[
|
|
{
|
|
"id": "event-1",
|
|
"order": 1,
|
|
"event": "侦察敌情",
|
|
"participants": ["测试角色"],
|
|
"trigger": "收到异常信号",
|
|
"resultDirection": "确认威胁存在",
|
|
}
|
|
]
|
|
if events
|
|
else []
|
|
),
|
|
"entities": [],
|
|
"foreshadowing": [],
|
|
"stateChanges": [],
|
|
"hook": "下一步",
|
|
"unknowns": [],
|
|
"assumptions": [],
|
|
}
|
|
|
|
|
|
def write_fake_runner(directory, mode="stable"):
|
|
"""生成可记录调用输入的假模型二进制,测试不触发真实模型。"""
|
|
|
|
directory = pathlib.Path(directory)
|
|
runner = directory / "fake-agent.py"
|
|
log_path = directory / "agent-calls.jsonl"
|
|
count_path = directory / "planner-count.txt"
|
|
run_result_path = directory / "run" / "run_result.json"
|
|
runner.write_text(
|
|
"#!/usr/bin/env python3\n"
|
|
"import json, pathlib, sys, time\n"
|
|
f"mode = {mode!r}\n"
|
|
f"log_path = pathlib.Path({str(log_path)!r})\n"
|
|
f"count_path = pathlib.Path({str(count_path)!r})\n"
|
|
f"run_result_path = pathlib.Path({str(run_result_path)!r})\n"
|
|
"args = sys.argv[1:]\n"
|
|
"agent = args[args.index('--agent') + 1]\n"
|
|
"prompt = args[-1]\n"
|
|
"run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None\n"
|
|
"with log_path.open('a', encoding='utf-8') as handle:\n"
|
|
" handle.write(json.dumps({'agent': agent, 'prompt': prompt, 'runState': run_state}, ensure_ascii=False) + '\\n')\n"
|
|
"if mode == f'{agent}_timeout':\n"
|
|
" time.sleep(2)\n"
|
|
"if mode == f'{agent}_exit':\n"
|
|
" raise SystemExit(7)\n"
|
|
"if mode == f'{agent}_invalid_json':\n"
|
|
" print('{invalid-json')\n"
|
|
" raise SystemExit(0)\n"
|
|
"if agent == 'planner':\n"
|
|
" count = int(count_path.read_text() if count_path.exists() else '0') + 1\n"
|
|
" count_path.write_text(str(count))\n"
|
|
" events = not (mode == 'invalid_candidate' and count == 2)\n"
|
|
f" value = {candidate()!r}\n"
|
|
" value['chapterGoal'] = f'goal-{count}'\n"
|
|
" if not events:\n"
|
|
" value['keyEvents'] = []\n"
|
|
" print(json.dumps({'result': json.dumps(value, ensure_ascii=False)}, ensure_ascii=False))\n"
|
|
"elif agent == 'detector':\n"
|
|
" request = json.loads(prompt)\n"
|
|
" findings = []\n"
|
|
" if mode == 'high_detector' and request['candidate']['chapterGoal'] == 'goal-2':\n"
|
|
" findings.append({'severity': 'high', 'category': 'entity_state', 'location': 'keyEvents[0]', 'evidenceSummary': '冻结事实冲突'})\n"
|
|
" if mode == 'invalid_detector':\n"
|
|
" findings.append({'category': 'entity_state'})\n"
|
|
" print(json.dumps({'protocol': 'fine_outline_detector_v0', 'candidateId': request['candidateId'], 'findings': findings, 'coverageFindings': []}, ensure_ascii=False))\n"
|
|
"elif agent == 'judge':\n"
|
|
" request = json.loads(prompt)\n"
|
|
" dimensions = ['structure_completeness', 'direction_causality', 'order_pacing', 'entity_state', 'foreshadowing_action', 'handoff_hook']\n"
|
|
" evaluations = []\n"
|
|
" for item in request['candidates']:\n"
|
|
" base = {'goal-1': 3, 'goal-2': 4, 'goal-3': 2}[item['candidate']['chapterGoal']]\n"
|
|
" scores = {dimension: {'score': base, 'evidence': f'{dimension}-结构化证据'} for dimension in dimensions}\n"
|
|
" if mode == 'unstable' and request['judgeId'] == 'judge-secondary':\n"
|
|
" scores['order_pacing']['score'] = min(5, base + 1)\n"
|
|
" if mode == 'invalid_rubric':\n"
|
|
" scores['order_pacing']['score'] = 6\n"
|
|
" evaluations.append({'candidateId': item['candidateId'], 'scores': scores, 'summary': '结构化评分摘要'})\n"
|
|
" print(json.dumps({'profile': 'fine_outline_replay', 'judgeId': request['judgeId'], 'evaluations': evaluations}, ensure_ascii=False))\n",
|
|
encoding="utf-8",
|
|
)
|
|
os.chmod(runner, 0o755)
|
|
return runner, log_path
|
|
|
|
|
|
def read_calls(log_path):
|
|
if not log_path.exists():
|
|
return []
|
|
return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()]
|
|
|
|
|
|
def nested_keys(value):
|
|
"""收集嵌套 JSON 的全部字段名,供盲化边界测试使用。"""
|
|
|
|
if isinstance(value, dict):
|
|
keys = set(value)
|
|
for item in value.values():
|
|
keys.update(nested_keys(item))
|
|
return keys
|
|
if isinstance(value, list):
|
|
keys = set()
|
|
for item in value:
|
|
keys.update(nested_keys(item))
|
|
return keys
|
|
return set()
|
|
|
|
|
|
class ReplayRunTest(unittest.TestCase):
|
|
def test_public_planner_context_does_not_include_snapshot_cards(self):
|
|
prompt = _planner_prompt(
|
|
target=489,
|
|
as_of=488,
|
|
snapshot={"cards": [{"name": "hidden-card"}], "safe": "public"},
|
|
common_input={},
|
|
cards=[{"name": "injected-card"}],
|
|
)
|
|
self.assertNotIn("hidden-card", prompt)
|
|
self.assertIn("injected-card", prompt)
|
|
|
|
def test_dry_run_passes_and_writes_only_metadata(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
result = run_replay(config(), pathlib.Path(directory), mode="dry_run")
|
|
self.assertTrue(result["ok"])
|
|
self.assertEqual(result["status"], "ready")
|
|
manifest = json.loads((pathlib.Path(directory) / "snapshot_manifest.json").read_text())
|
|
self.assertNotIn("payload", json.dumps(manifest, ensure_ascii=False))
|
|
self.assertFalse(list(pathlib.Path(directory).glob("planner_*.raw.json")))
|
|
|
|
def test_bad_authorization_stops_before_model(self):
|
|
bad = config()
|
|
bad["authorization"] = {"sourceStatus": "active"}
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(result["status"], "blocked_authorization")
|
|
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
|
|
|
def test_reused_output_dir_bad_config_overwrites_previous_completed_state(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
output_dir = pathlib.Path(directory) / "run"
|
|
output_dir.mkdir()
|
|
result_path = output_dir / "run_result.json"
|
|
result_path.write_text(
|
|
json.dumps({"runId": "old-run", "status": "completed", "ok": True}),
|
|
encoding="utf-8",
|
|
)
|
|
bad = config()
|
|
bad["snapshot"]["data"]["unknownSection"] = []
|
|
|
|
with self.assertRaisesRegex(ValueError, "未登记顶层分区"):
|
|
run_replay(bad, output_dir, mode="execute")
|
|
|
|
persisted = json.loads(result_path.read_text(encoding="utf-8"))
|
|
self.assertEqual(persisted["runId"], "smoke-001")
|
|
self.assertEqual(persisted["status"], "config_invalid")
|
|
self.assertFalse(persisted["ok"])
|
|
self.assertNotEqual(persisted["status"], "completed")
|
|
|
|
def test_content_leak_stops_before_manifest(self):
|
|
bad = config()
|
|
bad["leakageAudit"]["targetFacts"]["forbiddenFacts"][0]["text"] = "安全历史"
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(result["status"], "invalid_snapshot")
|
|
self.assertEqual(result["leakageAudit"]["findingCount"], 1)
|
|
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
|
|
|
def test_missing_content_audit_stops_before_model(self):
|
|
bad = config()
|
|
del bad["leakageAudit"]
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(result["status"], "blocked_leakage_audit")
|
|
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
|
|
|
def test_arm_card_future_record_stops_before_model(self):
|
|
bad = config()
|
|
bad["arms"]["outline_plus_cards"]["cards"] = [
|
|
{"name": "未来卡", "milestones": [{"chapter": 489, "fact": "未来"}]}
|
|
]
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(result["status"], "invalid_snapshot")
|
|
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
|
|
|
def test_report_contains_hashes_but_not_raw_fields(self):
|
|
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
|
report = render_report(result)
|
|
self.assertIn("smoke-001", report)
|
|
self.assertIn("snapshotManifestSha256", report)
|
|
self.assertNotIn('"prompt":', report)
|
|
self.assertNotIn('"payload":', report)
|
|
|
|
def test_report_rejects_text_injection_through_identifier_fields(self):
|
|
base = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
|
injections = {
|
|
"runId": "safe-run\n完整目标细纲:第一幕到第三幕的全部事件原文",
|
|
"referenceWork": "深空之影目标章原文与完整细纲",
|
|
}
|
|
for field, injected in injections.items():
|
|
with self.subTest(field=field):
|
|
unsafe = dict(base)
|
|
unsafe[field] = injected
|
|
with self.assertRaisesRegex(ValueError, field):
|
|
render_report(unsafe)
|
|
|
|
def test_report_does_not_render_preflight_free_text(self):
|
|
bad = config()
|
|
injected = "目标章事实文本"
|
|
bad["authorization"] = {**bad["authorization"], "sourceStatus": injected}
|
|
result = run_replay(bad, pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
|
report = render_report(result)
|
|
self.assertNotIn(injected, report)
|
|
self.assertIn("授权前置门未通过", report)
|
|
|
|
def test_report_accepts_registered_target_source_failure_status(self):
|
|
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
|
result["results"] = {
|
|
"outline_only": {"status": "target_source_forbidden", "ok": False}
|
|
}
|
|
report = render_report(result)
|
|
self.assertIn("target_source_forbidden", report)
|
|
|
|
def test_execute_uses_external_planner_and_validates_each_candidate(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
fake, log_path = write_fake_runner(directory)
|
|
output_dir = pathlib.Path(directory) / "run"
|
|
result = run_replay(config(), output_dir, mode="execute", planner_bin=str(fake))
|
|
first_call = read_calls(log_path)[0]
|
|
self.assertTrue(result["ok"])
|
|
self.assertEqual(result["status"], "completed")
|
|
self.assertFalse(first_call["runState"]["ok"])
|
|
self.assertEqual(first_call["runState"]["status"], "running_planner")
|
|
self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"})
|
|
self.assertTrue(list(output_dir.glob("candidate_*.json")))
|
|
|
|
def test_missing_planner_binary_fails_closed_and_persists_result(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
output_dir = pathlib.Path(directory) / "run"
|
|
missing = pathlib.Path(directory) / "missing-planner"
|
|
try:
|
|
result = run_replay(config(), output_dir, mode="execute", planner_bin=str(missing))
|
|
except OSError as error:
|
|
self.fail(f"runner 启动异常不得逃逸: {error}")
|
|
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(result["status"], "planner_failed")
|
|
self.assertEqual(persisted["status"], "planner_failed")
|
|
self.assertFalse(persisted["ok"])
|
|
self.assertNotIn(persisted["status"], {"ready", "completed"})
|
|
|
|
def test_planner_nonzero_exit_stops_after_first_call(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "planner_exit")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "planner_failed")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len(calls), 1)
|
|
|
|
def test_planner_invalid_json_stops_after_first_call(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "planner_invalid_json")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "planner_invalid")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len(calls), 1)
|
|
|
|
def test_planner_timeout_fails_closed_and_persists_timeout_status(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "planner_timeout")
|
|
output_dir = pathlib.Path(directory) / "run"
|
|
result = run_replay(
|
|
config(),
|
|
output_dir,
|
|
mode="execute",
|
|
planner_bin=str(runner),
|
|
timeout_seconds=1.0,
|
|
)
|
|
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
|
|
self.assertEqual(result["status"], "planner_timeout")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(persisted["status"], "planner_timeout")
|
|
self.assertEqual(len(read_calls(log_path)), 1)
|
|
|
|
def test_detector_nonzero_exit_stops_before_judges(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "detector_exit")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "detector_failed")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
|
|
|
def test_detector_invalid_json_stops_before_judges(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "detector_invalid_json")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "detector_invalid")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
|
|
|
def test_detector_timeout_fails_closed_before_judges(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "detector_timeout")
|
|
result = run_replay(
|
|
config(),
|
|
pathlib.Path(directory) / "run",
|
|
mode="execute",
|
|
planner_bin=str(runner),
|
|
timeout_seconds=1.0,
|
|
)
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "detector_timeout")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
|
|
|
def test_primary_judge_nonzero_exit_does_not_call_secondary(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "judge_exit")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
|
self.assertEqual(result["status"], "judge_failed")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len(judge_calls), 1)
|
|
|
|
def test_primary_judge_invalid_json_does_not_call_secondary(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "judge_invalid_json")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
|
self.assertEqual(result["status"], "judge_invalid")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len(judge_calls), 1)
|
|
|
|
def test_primary_judge_timeout_does_not_call_secondary(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "judge_timeout")
|
|
result = run_replay(
|
|
config(),
|
|
pathlib.Path(directory) / "run",
|
|
mode="execute",
|
|
planner_bin=str(runner),
|
|
timeout_seconds=1.0,
|
|
)
|
|
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
|
self.assertEqual(result["status"], "judge_timeout")
|
|
self.assertFalse(result["ok"])
|
|
self.assertEqual(len(judge_calls), 1)
|
|
|
|
def test_cli_accepts_subprocess_timeout_seconds(self):
|
|
argv = [
|
|
"run_replay.py",
|
|
"--config",
|
|
"/tmp/replay-config.json",
|
|
"--output-dir",
|
|
"/tmp/replay-output",
|
|
"--timeout-seconds",
|
|
"12.5",
|
|
]
|
|
with patch.object(sys, "argv", argv):
|
|
args = _parse_args()
|
|
self.assertEqual(args.timeout_seconds, 12.5)
|
|
|
|
def test_non_finite_timeout_is_persisted_as_config_invalid(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
for index, timeout_seconds in enumerate((float("nan"), float("inf"))):
|
|
with self.subTest(timeout_seconds=timeout_seconds):
|
|
output_dir = pathlib.Path(directory) / f"run-{index}"
|
|
with self.assertRaisesRegex(ValueError, "timeout_seconds"):
|
|
run_replay(
|
|
config(),
|
|
output_dir,
|
|
mode="execute",
|
|
timeout_seconds=timeout_seconds,
|
|
)
|
|
persisted = json.loads(
|
|
(output_dir / "run_result.json").read_text(encoding="utf-8")
|
|
)
|
|
self.assertEqual(persisted["status"], "config_invalid")
|
|
self.assertFalse(persisted["ok"])
|
|
|
|
def test_schema_failure_stops_before_all_detectors_and_judges(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "invalid_candidate")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "candidate_blocked")
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 0)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
|
|
|
def test_high_detector_finding_blocks_group_and_never_calls_judge(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "high_detector")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
detector_calls = [call for call in calls if call["agent"] == "detector"]
|
|
self.assertEqual(result["status"], "detector_blocked")
|
|
self.assertEqual(len(detector_calls), 3)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
|
self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls))
|
|
self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls))
|
|
|
|
def test_detector_requests_cannot_distinguish_arm_specific_card_identity(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "stable")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
self.assertEqual(result["status"], "completed")
|
|
requests = [
|
|
json.loads(call["prompt"])
|
|
for call in read_calls(log_path)
|
|
if call["agent"] == "detector"
|
|
]
|
|
self.assertEqual(len(requests), 3)
|
|
for request in requests:
|
|
self.assertTrue(
|
|
nested_keys(request).isdisjoint({"arm", "cardInjection", "cardManifest"})
|
|
)
|
|
serialized = json.dumps(request, ensure_ascii=False)
|
|
self.assertNotIn("正确卡", serialized)
|
|
self.assertNotIn("错配卡", serialized)
|
|
self.assertNotIn("card-correct-1", serialized)
|
|
self.assertNotIn("card-placebo-1", serialized)
|
|
public_parts = [
|
|
{key: value for key, value in request.items() if key not in {"candidateId", "candidate"}}
|
|
for request in requests
|
|
]
|
|
self.assertEqual(public_parts[0], public_parts[1])
|
|
self.assertEqual(public_parts[1], public_parts[2])
|
|
|
|
def test_invalid_detector_report_fails_closed_before_judge(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "invalid_detector")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
calls = read_calls(log_path)
|
|
self.assertEqual(result["status"], "detector_invalid")
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
|
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
|
|
|
def test_two_judges_are_independent_blind_reversed_and_unblinded(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, log_path = write_fake_runner(directory, "stable")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
|
self.assertEqual(result["status"], "completed")
|
|
self.assertEqual(len(judge_calls), 2)
|
|
first = json.loads(judge_calls[0]["prompt"])
|
|
second = json.loads(judge_calls[1]["prompt"])
|
|
self.assertNotEqual(first["judgeId"], second["judgeId"])
|
|
self.assertEqual(
|
|
[item["candidateId"] for item in second["candidates"]],
|
|
list(reversed([item["candidateId"] for item in first["candidates"]])),
|
|
)
|
|
self.assertNotIn("outline_only", judge_calls[0]["prompt"])
|
|
self.assertNotIn("outline_plus_cards", judge_calls[1]["prompt"])
|
|
self.assertTrue(result["evaluation"]["stability"]["stable"])
|
|
self.assertEqual(result["evaluation"]["deltas"]["B-A"]["order_pacing"], 1.0)
|
|
self.assertEqual(result["evaluation"]["deltas"]["C-A"]["order_pacing"], -1.0)
|
|
|
|
def test_invalid_rubric_report_does_not_complete(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, _ = write_fake_runner(directory, "invalid_rubric")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
self.assertEqual(result["status"], "judge_invalid")
|
|
self.assertFalse(result["ok"])
|
|
|
|
def test_unstable_judges_have_explicit_non_completed_status(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, _ = write_fake_runner(directory, "unstable")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
self.assertEqual(result["status"], "judge_unstable")
|
|
self.assertFalse(result["ok"])
|
|
self.assertFalse(result["evaluation"]["stability"]["stable"])
|
|
self.assertNotIn("deltas", result["evaluation"])
|
|
|
|
def test_report_contains_only_aggregated_evaluation(self):
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
runner, _ = write_fake_runner(directory, "stable")
|
|
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
|
|
report = render_report(result)
|
|
self.assertIn("B-A", report)
|
|
self.assertIn("C-A", report)
|
|
self.assertIn("评委稳定性", report)
|
|
self.assertNotIn("结构化证据", report)
|
|
self.assertNotIn("目标章未进入快照", report)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|