656 lines
32 KiB
Python
656 lines
32 KiB
Python
#!/usr/bin/env python3
|
||
"""回放编排器和安全摘要的无网络测试。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import pathlib
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
from unittest.mock import patch
|
||
|
||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||
SKILLS_DIR = PROJECT_ROOT / ".agent" / "skills"
|
||
SCRIPT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "replay" / "回放评估细纲质量" / "scripts"
|
||
TEST_DIR = PROJECT_ROOT / "tests" / "skills" / "回放评估细纲质量"
|
||
QUALITY_GATE_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "judge" / "评估内容质量" / "scripts"
|
||
REPLAY_GATE_TEST_DIR = PROJECT_ROOT / "tests" / "skills" / "判定质量是否合格"
|
||
for _import_dir in (SCRIPT_DIR, TEST_DIR, QUALITY_GATE_DIR, REPLAY_GATE_TEST_DIR):
|
||
if str(_import_dir) not in sys.path:
|
||
sys.path.insert(0, str(_import_dir))
|
||
from muse_role import FIXED_OPUS_MODEL_ID # noqa: E402
|
||
from run_replay import _parse_args, _planner_prompt # noqa: E402
|
||
from run_replay import run_replay as _run_replay
|
||
from test_writer_gate import build_input, source_bundle # noqa: E402
|
||
from write_report import render_report # noqa: E402
|
||
from writer_gate import issue_gate_report_and_receipt # noqa: E402
|
||
|
||
SOURCE_HASH = "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4"
|
||
SOURCE_VERSION = f"raw-file-v1:{SOURCE_HASH}"
|
||
AUTH = {
|
||
"sourceStatus": "active",
|
||
"copyrightStatus": "research_only",
|
||
"sourceHash": SOURCE_HASH,
|
||
"sourceVersion": SOURCE_VERSION,
|
||
"allowedPurpose": ["offline_evaluation"],
|
||
"forbiddenPurpose": ["external_distribution"],
|
||
"authorizationSnapshot": {
|
||
"id": "auth-1",
|
||
"version": "v1",
|
||
"immutable": True,
|
||
"sourceHash": SOURCE_HASH,
|
||
"sourceVersion": SOURCE_VERSION,
|
||
"sourceStatus": "active",
|
||
"copyrightStatus": "research_only",
|
||
"authorizationBasis": "user_authorization",
|
||
"allowedPurpose": ["offline_evaluation"],
|
||
"forbiddenPurpose": ["external_distribution"],
|
||
"checkedAt": "2026-07-19T00:00:00Z",
|
||
"revalidationAt": "2099-07-20T00:00:00Z",
|
||
},
|
||
}
|
||
|
||
|
||
def run_replay(config_value, output_dir, **kwargs):
|
||
"""旧执行测试统一经过真实本地 Gate 签发链,不提供生产绕过参数。"""
|
||
|
||
if kwargs.get("mode", "dry_run") != "execute":
|
||
return _run_replay(config_value, output_dir, **kwargs)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
gate_root = pathlib.Path(directory)
|
||
gate_input = build_input(source_bundle())
|
||
gate_input_path = gate_root / "gate-input.json"
|
||
gate_input_path.write_text(json.dumps(gate_input, ensure_ascii=False), encoding="utf-8")
|
||
issue_gate_report_and_receipt(
|
||
gate_input=gate_input,
|
||
expected_input_sha256=gate_input["gateInputSha256"],
|
||
output_dir=gate_root,
|
||
cas_dir=gate_root / "cas",
|
||
)
|
||
return _run_replay(
|
||
config_value,
|
||
output_dir,
|
||
writer_gate_receipt_path=gate_root / "writer-gate-b-passed-receipt.json",
|
||
writer_gate_report_path=gate_root / "gate-report.json",
|
||
writer_gate_input_path=gate_input_path,
|
||
writer_gate_cas_dir=gate_root / "cas",
|
||
**kwargs,
|
||
)
|
||
|
||
|
||
def config():
|
||
common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}}
|
||
return {
|
||
"runId": "smoke-001",
|
||
"referenceWork": {"id": "deep-space", "version": SOURCE_VERSION},
|
||
"evaluationSetVersion": "set-v1",
|
||
"strategyVersion": "strategy-v1",
|
||
"authorization": AUTH,
|
||
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
|
||
"leakageAudit": {
|
||
"targetFacts": {
|
||
"targetChapter": 489,
|
||
"forbiddenFacts": [
|
||
{
|
||
"id": "future-fact-1",
|
||
"firstChapter": 489,
|
||
"text": "目标章未进入快照",
|
||
}
|
||
],
|
||
}
|
||
},
|
||
"targetChapter": 489,
|
||
"snapshot": {
|
||
"asOfChapter": 488,
|
||
"snapshotVersion": "next_fine_outline_replay_v0",
|
||
"data": {
|
||
"milestones": [{"chapter": 488, "fact": "安全历史"}, {"chapter": 489, "fact": "未来"}],
|
||
"cards": [{"name": "已知实体", "milestones": [{"chapter": 488, "step": "历史"}]}],
|
||
},
|
||
},
|
||
"sources": [{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
|
||
"commonContext": common,
|
||
"arms": {
|
||
"outline_only": {"cards": [], "cardSourceIds": [], "cardStrategy": "none"},
|
||
"outline_plus_cards": {"cards": [{"name": "正确卡"}], "cardSourceIds": ["card-correct-1"], "cardStrategy": "correct"},
|
||
"outline_plus_placebo_cards": {"cards": [{"name": "错配卡"}], "cardSourceIds": ["card-placebo-1"], "cardStrategy": "placebo"},
|
||
},
|
||
}
|
||
|
||
|
||
def candidate(goal="机制 smoke", *, events=True):
|
||
"""构造满足细纲闭集合同的合成候选。"""
|
||
|
||
return {
|
||
"targetChapter": 489,
|
||
"chapterGoal": goal,
|
||
"keyEvents": (
|
||
[
|
||
{
|
||
"id": "event-1",
|
||
"order": 1,
|
||
"event": "侦察敌情",
|
||
"participants": ["测试角色"],
|
||
"trigger": "收到异常信号",
|
||
"resultDirection": "确认威胁存在",
|
||
}
|
||
]
|
||
if events
|
||
else []
|
||
),
|
||
"entities": [],
|
||
"foreshadowing": [],
|
||
"stateChanges": [],
|
||
"hook": "下一步",
|
||
"unknowns": [],
|
||
"assumptions": [],
|
||
}
|
||
|
||
|
||
def write_fake_chat(directory, mode="stable"):
|
||
"""构造与 muse_llm.chat_governed 同签名的假治理调用,测试不触发真实模型。
|
||
|
||
角色识别依据派发方注入的系统提示词(planner 身份段 / detector / judge 角色文件
|
||
frontmatter name),与 run_replay 的注入合同一致。调用记录沿用 agent-calls.jsonl。
|
||
"""
|
||
|
||
directory = pathlib.Path(directory)
|
||
log_path = directory / "agent-calls.jsonl"
|
||
count_path = directory / "planner-count.txt"
|
||
run_result_path = directory / "run" / "run_result.json"
|
||
usage = {"input_tokens": 1, "output_tokens": 1}
|
||
|
||
def fake_chat(prompt, system=None, temperature=0.2, top_p=None, *,
|
||
run_id=None, caller=None, persist_call=None, **_kwargs):
|
||
system_text = system or ""
|
||
if "planner identity" in system_text:
|
||
agent = "planner"
|
||
elif "name: detector" in system_text:
|
||
agent = "detector"
|
||
elif "name: judge" in system_text:
|
||
agent = "judge"
|
||
else:
|
||
agent = "unknown"
|
||
run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None
|
||
with log_path.open("a", encoding="utf-8") as handle:
|
||
handle.write(json.dumps({"agent": agent, "prompt": prompt, "runState": run_state}, ensure_ascii=False) + "\n")
|
||
if mode == f"{agent}_timeout":
|
||
raise TimeoutError("fake timeout")
|
||
if mode == f"{agent}_exit":
|
||
raise RuntimeError("fake execution failure")
|
||
if mode == f"{agent}_invalid_json":
|
||
return "{invalid-json", usage, FIXED_OPUS_MODEL_ID
|
||
if agent == "planner":
|
||
count = int(count_path.read_text() if count_path.exists() else "0") + 1
|
||
count_path.write_text(str(count))
|
||
events = not (mode == "invalid_candidate" and count == 2)
|
||
value = candidate()
|
||
value["chapterGoal"] = f"goal-{count}"
|
||
if not events:
|
||
value["keyEvents"] = []
|
||
return json.dumps(value, ensure_ascii=False), usage, FIXED_OPUS_MODEL_ID
|
||
if agent == "detector":
|
||
request = json.loads(prompt)
|
||
findings = []
|
||
if mode == "high_detector" and request["candidate"]["chapterGoal"] == "goal-2":
|
||
findings.append({"severity": "high", "category": "entity_state", "location": "keyEvents[0]", "evidenceSummary": "冻结事实冲突"})
|
||
if mode == "invalid_detector":
|
||
findings.append({"category": "entity_state"})
|
||
return json.dumps({"protocol": "fine_outline_detector_v0", "candidateId": request["candidateId"], "findings": findings, "coverageFindings": []}, ensure_ascii=False), usage, FIXED_OPUS_MODEL_ID
|
||
if agent == "judge":
|
||
request = json.loads(prompt)
|
||
dimensions = ["structure_completeness", "direction_causality", "order_pacing", "entity_state", "foreshadowing_action", "handoff_hook"]
|
||
evaluations = []
|
||
for item in request["candidates"]:
|
||
base = {"goal-1": 3, "goal-2": 4, "goal-3": 2}[item["candidate"]["chapterGoal"]]
|
||
scores = {dimension: {"score": base, "evidence": f"{dimension}-结构化证据"} for dimension in dimensions}
|
||
if mode == "unstable" and request["judgeId"] == "judge-secondary":
|
||
scores["order_pacing"]["score"] = min(5, base + 1)
|
||
if mode == "invalid_rubric":
|
||
scores["order_pacing"]["score"] = 6
|
||
evaluations.append({"candidateId": item["candidateId"], "scores": scores, "summary": "结构化评分摘要"})
|
||
return json.dumps({"profile": "fine_outline_replay", "judgeId": request["judgeId"], "evaluations": evaluations}, ensure_ascii=False), usage, FIXED_OPUS_MODEL_ID
|
||
raise RuntimeError(f"未知角色: {agent}")
|
||
|
||
return fake_chat, log_path
|
||
|
||
|
||
def read_calls(log_path):
|
||
if not log_path.exists():
|
||
return []
|
||
return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()]
|
||
|
||
|
||
def nested_keys(value):
|
||
"""收集嵌套 JSON 的全部字段名,供盲化边界测试使用。"""
|
||
|
||
if isinstance(value, dict):
|
||
keys = set(value)
|
||
for item in value.values():
|
||
keys.update(nested_keys(item))
|
||
return keys
|
||
if isinstance(value, list):
|
||
keys = set()
|
||
for item in value:
|
||
keys.update(nested_keys(item))
|
||
return keys
|
||
return set()
|
||
|
||
|
||
class ReplayRunTest(unittest.TestCase):
|
||
def test_execute_without_signed_writer_gate_receipt_is_rejected(self):
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
result = _run_replay(
|
||
config(),
|
||
pathlib.Path(directory) / "run",
|
||
mode="execute",
|
||
)
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_writer_gate_receipt")
|
||
|
||
def test_public_planner_context_does_not_include_snapshot_cards(self):
|
||
prompt = _planner_prompt(
|
||
target=489,
|
||
as_of=488,
|
||
snapshot={"cards": [{"name": "hidden-card"}], "safe": "public"},
|
||
common_input={},
|
||
cards=[{"name": "injected-card"}],
|
||
)
|
||
self.assertNotIn("hidden-card", prompt)
|
||
self.assertIn("injected-card", prompt)
|
||
|
||
def test_dry_run_passes_and_writes_only_metadata(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
result = run_replay(config(), pathlib.Path(directory), mode="dry_run")
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "ready")
|
||
manifest = json.loads((pathlib.Path(directory) / "snapshot_manifest.json").read_text())
|
||
self.assertNotIn("payload", json.dumps(manifest, ensure_ascii=False))
|
||
self.assertFalse(list(pathlib.Path(directory).glob("planner_*.raw.json")))
|
||
|
||
def test_bad_authorization_stops_before_model(self):
|
||
bad = config()
|
||
bad["authorization"] = {"sourceStatus": "active"}
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_authorization")
|
||
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
||
|
||
def test_reused_output_dir_bad_config_overwrites_previous_completed_state(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
output_dir.mkdir()
|
||
result_path = output_dir / "run_result.json"
|
||
result_path.write_text(
|
||
json.dumps({"runId": "old-run", "status": "completed", "ok": True}),
|
||
encoding="utf-8",
|
||
)
|
||
bad = config()
|
||
bad["snapshot"]["data"]["unknownSection"] = []
|
||
|
||
with self.assertRaisesRegex(ValueError, "未登记顶层分区"):
|
||
run_replay(bad, output_dir, mode="execute")
|
||
|
||
persisted = json.loads(result_path.read_text(encoding="utf-8"))
|
||
self.assertEqual(persisted["runId"], "smoke-001")
|
||
self.assertEqual(persisted["status"], "config_invalid")
|
||
self.assertFalse(persisted["ok"])
|
||
self.assertNotEqual(persisted["status"], "completed")
|
||
|
||
def test_content_leak_stops_before_manifest(self):
|
||
bad = config()
|
||
bad["leakageAudit"]["targetFacts"]["forbiddenFacts"][0]["text"] = "安全历史"
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_snapshot")
|
||
self.assertEqual(result["leakageAudit"]["findingCount"], 1)
|
||
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
||
|
||
def test_missing_content_audit_stops_before_model(self):
|
||
bad = config()
|
||
del bad["leakageAudit"]
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "blocked_leakage_audit")
|
||
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
||
|
||
def test_arm_card_future_record_stops_before_model(self):
|
||
bad = config()
|
||
bad["arms"]["outline_plus_cards"]["cards"] = [
|
||
{"name": "未来卡", "milestones": [{"chapter": 489, "fact": "未来"}]}
|
||
]
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "invalid_snapshot")
|
||
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
||
|
||
def test_report_contains_hashes_but_not_raw_fields(self):
|
||
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
||
report = render_report(result)
|
||
self.assertIn("smoke-001", report)
|
||
self.assertIn("snapshotManifestSha256", report)
|
||
self.assertNotIn('"prompt":', report)
|
||
self.assertNotIn('"payload":', report)
|
||
|
||
def test_report_rejects_text_injection_through_identifier_fields(self):
|
||
base = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
||
injections = {
|
||
"runId": "safe-run\n完整目标细纲:第一幕到第三幕的全部事件原文",
|
||
"referenceWork": "深空之影目标章原文与完整细纲",
|
||
}
|
||
for field, injected in injections.items():
|
||
with self.subTest(field=field):
|
||
unsafe = dict(base)
|
||
unsafe[field] = injected
|
||
with self.assertRaisesRegex(ValueError, field):
|
||
render_report(unsafe)
|
||
|
||
def test_report_does_not_render_preflight_free_text(self):
|
||
bad = config()
|
||
injected = "目标章事实文本"
|
||
bad["authorization"] = {**bad["authorization"], "sourceStatus": injected}
|
||
result = run_replay(bad, pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
||
report = render_report(result)
|
||
self.assertNotIn(injected, report)
|
||
self.assertIn("授权前置门未通过", report)
|
||
|
||
def test_report_accepts_registered_target_source_failure_status(self):
|
||
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
||
result["results"] = {
|
||
"outline_only": {"status": "target_source_forbidden", "ok": False}
|
||
}
|
||
report = render_report(result)
|
||
self.assertIn("target_source_forbidden", report)
|
||
|
||
def test_execute_uses_external_planner_and_validates_each_candidate(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
fake, log_path = write_fake_chat(directory)
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
result = run_replay(config(), output_dir, mode="execute", governed_chat=fake)
|
||
first_call = read_calls(log_path)[0]
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertFalse(first_call["runState"]["ok"])
|
||
self.assertEqual(first_call["runState"]["status"], "running_planner")
|
||
self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"})
|
||
self.assertTrue(list(output_dir.glob("candidate_*.json")))
|
||
raw_receipts = list(output_dir.glob("*.raw.json"))
|
||
self.assertTrue(raw_receipts)
|
||
for path in raw_receipts:
|
||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||
self.assertEqual(
|
||
set(payload),
|
||
{"rawRequestSha256", "rawResponseSha256", "receipt"},
|
||
)
|
||
self.assertNotIn("rawRequest", payload)
|
||
self.assertNotIn("rawResponse", payload)
|
||
|
||
def test_model_channel_failure_fails_closed_and_persists_result(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
|
||
def broken_chat(*_args, **_kwargs):
|
||
raise OSError("模型通道不可用")
|
||
|
||
try:
|
||
result = run_replay(config(), output_dir, mode="execute", governed_chat=broken_chat)
|
||
except OSError as error:
|
||
self.fail(f"底座启动异常不得逃逸: {error}")
|
||
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["status"], "planner_failed")
|
||
self.assertEqual(persisted["status"], "planner_failed")
|
||
self.assertFalse(persisted["ok"])
|
||
self.assertNotIn(persisted["status"], {"ready", "completed"})
|
||
|
||
def test_planner_nonzero_exit_stops_after_first_call(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "planner_exit")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "planner_failed")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len(calls), 1)
|
||
|
||
def test_planner_invalid_json_stops_after_first_call(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "planner_invalid_json")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "planner_invalid")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len(calls), 1)
|
||
|
||
def test_planner_timeout_fails_closed_and_persists_timeout_status(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "planner_timeout")
|
||
output_dir = pathlib.Path(directory) / "run"
|
||
result = run_replay(
|
||
config(),
|
||
output_dir,
|
||
mode="execute",
|
||
governed_chat=runner,
|
||
timeout_seconds=1.0,
|
||
)
|
||
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
|
||
self.assertEqual(result["status"], "planner_timeout")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(persisted["status"], "planner_timeout")
|
||
self.assertEqual(len(read_calls(log_path)), 1)
|
||
|
||
def test_detector_nonzero_exit_stops_before_judges(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "detector_exit")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "detector_failed")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
||
|
||
def test_detector_invalid_json_stops_before_judges(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "detector_invalid_json")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "detector_invalid")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
||
|
||
def test_detector_timeout_fails_closed_before_judges(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "detector_timeout")
|
||
result = run_replay(
|
||
config(),
|
||
pathlib.Path(directory) / "run",
|
||
mode="execute",
|
||
governed_chat=runner,
|
||
timeout_seconds=1.0,
|
||
)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "detector_timeout")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
||
|
||
def test_primary_judge_nonzero_exit_does_not_call_secondary(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "judge_exit")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
||
self.assertEqual(result["status"], "judge_failed")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len(judge_calls), 1)
|
||
|
||
def test_primary_judge_invalid_json_does_not_call_secondary(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "judge_invalid_json")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
||
self.assertEqual(result["status"], "judge_invalid")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len(judge_calls), 1)
|
||
|
||
def test_primary_judge_timeout_does_not_call_secondary(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "judge_timeout")
|
||
result = run_replay(
|
||
config(),
|
||
pathlib.Path(directory) / "run",
|
||
mode="execute",
|
||
governed_chat=runner,
|
||
timeout_seconds=1.0,
|
||
)
|
||
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
||
self.assertEqual(result["status"], "judge_timeout")
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(len(judge_calls), 1)
|
||
|
||
def test_cli_accepts_subprocess_timeout_seconds(self):
|
||
argv = [
|
||
"run_replay.py",
|
||
"--config",
|
||
"/tmp/replay-config.json",
|
||
"--output-dir",
|
||
"/tmp/replay-output",
|
||
"--timeout-seconds",
|
||
"12.5",
|
||
]
|
||
with patch.object(sys, "argv", argv):
|
||
args = _parse_args()
|
||
self.assertEqual(args.timeout_seconds, 12.5)
|
||
|
||
def test_non_finite_timeout_is_persisted_as_config_invalid(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
for index, timeout_seconds in enumerate((float("nan"), float("inf"))):
|
||
with self.subTest(timeout_seconds=timeout_seconds):
|
||
output_dir = pathlib.Path(directory) / f"run-{index}"
|
||
with self.assertRaisesRegex(ValueError, "timeout_seconds"):
|
||
run_replay(
|
||
config(),
|
||
output_dir,
|
||
mode="execute",
|
||
timeout_seconds=timeout_seconds,
|
||
)
|
||
persisted = json.loads(
|
||
(output_dir / "run_result.json").read_text(encoding="utf-8")
|
||
)
|
||
self.assertEqual(persisted["status"], "config_invalid")
|
||
self.assertFalse(persisted["ok"])
|
||
|
||
def test_schema_failure_stops_before_all_detectors_and_judges(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "invalid_candidate")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "planner_invalid")
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 2)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 0)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
||
|
||
def test_high_detector_finding_blocks_group_and_never_calls_judge(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "high_detector")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
detector_calls = [call for call in calls if call["agent"] == "detector"]
|
||
self.assertEqual(result["status"], "detector_blocked")
|
||
self.assertEqual(len(detector_calls), 3)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
||
self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls))
|
||
self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls))
|
||
|
||
def test_detector_requests_cannot_distinguish_arm_specific_card_identity(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "stable")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
self.assertEqual(result["status"], "completed")
|
||
requests = [
|
||
json.loads(call["prompt"])
|
||
for call in read_calls(log_path)
|
||
if call["agent"] == "detector"
|
||
]
|
||
self.assertEqual(len(requests), 3)
|
||
for request in requests:
|
||
self.assertTrue(
|
||
nested_keys(request).isdisjoint({"arm", "cardInjection", "cardManifest"})
|
||
)
|
||
serialized = json.dumps(request, ensure_ascii=False)
|
||
self.assertNotIn("正确卡", serialized)
|
||
self.assertNotIn("错配卡", serialized)
|
||
self.assertNotIn("card-correct-1", serialized)
|
||
self.assertNotIn("card-placebo-1", serialized)
|
||
public_parts = [
|
||
{key: value for key, value in request.items() if key not in {"candidateId", "candidate"}}
|
||
for request in requests
|
||
]
|
||
self.assertEqual(public_parts[0], public_parts[1])
|
||
self.assertEqual(public_parts[1], public_parts[2])
|
||
|
||
def test_invalid_detector_report_fails_closed_before_judge(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "invalid_detector")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
calls = read_calls(log_path)
|
||
self.assertEqual(result["status"], "detector_invalid")
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
|
||
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
|
||
|
||
def test_two_judges_are_independent_blind_reversed_and_unblinded(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, log_path = write_fake_chat(directory, "stable")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
|
||
self.assertEqual(result["status"], "completed")
|
||
self.assertEqual(len(judge_calls), 2)
|
||
first = json.loads(judge_calls[0]["prompt"])
|
||
second = json.loads(judge_calls[1]["prompt"])
|
||
self.assertNotEqual(first["judgeId"], second["judgeId"])
|
||
self.assertEqual(
|
||
[item["candidateId"] for item in second["candidates"]],
|
||
list(reversed([item["candidateId"] for item in first["candidates"]])),
|
||
)
|
||
self.assertNotIn("outline_only", judge_calls[0]["prompt"])
|
||
self.assertNotIn("outline_plus_cards", judge_calls[1]["prompt"])
|
||
self.assertTrue(result["evaluation"]["stability"]["stable"])
|
||
self.assertEqual(result["evaluation"]["deltas"]["B-A"]["order_pacing"], 1.0)
|
||
self.assertEqual(result["evaluation"]["deltas"]["C-A"]["order_pacing"], -1.0)
|
||
|
||
def test_invalid_rubric_report_does_not_complete(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, _ = write_fake_chat(directory, "invalid_rubric")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
self.assertEqual(result["status"], "judge_invalid")
|
||
self.assertFalse(result["ok"])
|
||
|
||
def test_unstable_judges_have_explicit_non_completed_status(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, _ = write_fake_chat(directory, "unstable")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
self.assertEqual(result["status"], "judge_unstable")
|
||
self.assertFalse(result["ok"])
|
||
self.assertFalse(result["evaluation"]["stability"]["stable"])
|
||
self.assertNotIn("deltas", result["evaluation"])
|
||
|
||
def test_report_contains_only_aggregated_evaluation(self):
|
||
with tempfile.TemporaryDirectory() as directory:
|
||
runner, _ = write_fake_chat(directory, "stable")
|
||
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", governed_chat=runner)
|
||
report = render_report(result)
|
||
self.assertIn("B-A", report)
|
||
self.assertIn("C-A", report)
|
||
self.assertIn("评委稳定性", report)
|
||
self.assertNotIn("结构化证据", report)
|
||
self.assertNotIn("目标章未进入快照", report)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|