zizi 2666d50a8e 修复: 收紧授权用途与导入任务匹配
阻断允许用途与禁止用途冲突,并让授权快照外层与内层保持一致。
导入任务按参考作品原文件唯一匹配,避免无关成功任务误阻断回放。
2026-07-19 22:19:47 +08:00

592 lines
29 KiB
Python

#!/usr/bin/env python3
"""回放编排器和安全摘要的无网络测试。"""
from __future__ import annotations
import json
import os
import pathlib
import sys
import tempfile
import unittest
from unittest.mock import patch
sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent))
from run_replay import _parse_args, _planner_prompt, run_replay # noqa: E402
from write_report import render_report # noqa: E402
SOURCE_HASH = "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4"
SOURCE_VERSION = f"raw-file-v1:{SOURCE_HASH}"
AUTH = {
"sourceStatus": "active",
"copyrightStatus": "research_only",
"sourceHash": SOURCE_HASH,
"sourceVersion": SOURCE_VERSION,
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution"],
"authorizationSnapshot": {
"id": "auth-1",
"version": "v1",
"immutable": True,
"sourceHash": SOURCE_HASH,
"sourceVersion": SOURCE_VERSION,
"sourceStatus": "active",
"copyrightStatus": "research_only",
"authorizationBasis": "user_authorization",
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution"],
"checkedAt": "2026-07-19T00:00:00Z",
"revalidationAt": "2099-07-20T00:00:00Z",
},
}
def config():
common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}}
return {
"runId": "smoke-001",
"referenceWork": {"id": "deep-space", "version": SOURCE_VERSION},
"evaluationSetVersion": "set-v1",
"strategyVersion": "strategy-v1",
"authorization": AUTH,
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
"leakageAudit": {
"targetFacts": {
"targetChapter": 489,
"forbiddenFacts": [
{
"id": "future-fact-1",
"firstChapter": 489,
"text": "目标章未进入快照",
}
],
}
},
"targetChapter": 489,
"snapshot": {
"asOfChapter": 488,
"snapshotVersion": "next_fine_outline_replay_v0",
"data": {
"milestones": [{"chapter": 488, "fact": "安全历史"}, {"chapter": 489, "fact": "未来"}],
"cards": [{"name": "已知实体", "milestones": [{"chapter": 488, "step": "历史"}]}],
},
},
"sources": [{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
"commonContext": common,
"arms": {
"outline_only": {"cards": [], "cardSourceIds": [], "cardStrategy": "none"},
"outline_plus_cards": {"cards": [{"name": "正确卡"}], "cardSourceIds": ["card-correct-1"], "cardStrategy": "correct"},
"outline_plus_placebo_cards": {"cards": [{"name": "错配卡"}], "cardSourceIds": ["card-placebo-1"], "cardStrategy": "placebo"},
},
}
def candidate(goal="机制 smoke", *, events=True):
"""构造满足细纲闭集合同的合成候选。"""
return {
"targetChapter": 489,
"chapterGoal": goal,
"keyEvents": (
[
{
"id": "event-1",
"order": 1,
"event": "侦察敌情",
"participants": ["测试角色"],
"trigger": "收到异常信号",
"resultDirection": "确认威胁存在",
}
]
if events
else []
),
"entities": [],
"foreshadowing": [],
"stateChanges": [],
"hook": "下一步",
"unknowns": [],
"assumptions": [],
}
def write_fake_runner(directory, mode="stable"):
"""生成可记录调用输入的假模型二进制,测试不触发真实模型。"""
directory = pathlib.Path(directory)
runner = directory / "fake-agent.py"
log_path = directory / "agent-calls.jsonl"
count_path = directory / "planner-count.txt"
run_result_path = directory / "run" / "run_result.json"
runner.write_text(
"#!/usr/bin/env python3\n"
"import json, pathlib, sys, time\n"
f"mode = {mode!r}\n"
f"log_path = pathlib.Path({str(log_path)!r})\n"
f"count_path = pathlib.Path({str(count_path)!r})\n"
f"run_result_path = pathlib.Path({str(run_result_path)!r})\n"
"args = sys.argv[1:]\n"
"agent = args[args.index('--agent') + 1]\n"
"prompt = args[-1]\n"
"run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None\n"
"with log_path.open('a', encoding='utf-8') as handle:\n"
" handle.write(json.dumps({'agent': agent, 'prompt': prompt, 'runState': run_state}, ensure_ascii=False) + '\\n')\n"
"if mode == f'{agent}_timeout':\n"
" time.sleep(2)\n"
"if mode == f'{agent}_exit':\n"
" raise SystemExit(7)\n"
"if mode == f'{agent}_invalid_json':\n"
" print('{invalid-json')\n"
" raise SystemExit(0)\n"
"if agent == 'planner':\n"
" count = int(count_path.read_text() if count_path.exists() else '0') + 1\n"
" count_path.write_text(str(count))\n"
" events = not (mode == 'invalid_candidate' and count == 2)\n"
f" value = {candidate()!r}\n"
" value['chapterGoal'] = f'goal-{count}'\n"
" if not events:\n"
" value['keyEvents'] = []\n"
" print(json.dumps({'result': json.dumps(value, ensure_ascii=False)}, ensure_ascii=False))\n"
"elif agent == 'detector':\n"
" request = json.loads(prompt)\n"
" findings = []\n"
" if mode == 'high_detector' and request['candidate']['chapterGoal'] == 'goal-2':\n"
" findings.append({'severity': 'high', 'category': 'entity_state', 'location': 'keyEvents[0]', 'evidenceSummary': '冻结事实冲突'})\n"
" if mode == 'invalid_detector':\n"
" findings.append({'category': 'entity_state'})\n"
" print(json.dumps({'protocol': 'fine_outline_detector_v0', 'candidateId': request['candidateId'], 'findings': findings, 'coverageFindings': []}, ensure_ascii=False))\n"
"elif agent == 'judge':\n"
" request = json.loads(prompt)\n"
" dimensions = ['structure_completeness', 'direction_causality', 'order_pacing', 'entity_state', 'foreshadowing_action', 'handoff_hook']\n"
" evaluations = []\n"
" for item in request['candidates']:\n"
" base = {'goal-1': 3, 'goal-2': 4, 'goal-3': 2}[item['candidate']['chapterGoal']]\n"
" scores = {dimension: {'score': base, 'evidence': f'{dimension}-结构化证据'} for dimension in dimensions}\n"
" if mode == 'unstable' and request['judgeId'] == 'judge-secondary':\n"
" scores['order_pacing']['score'] = min(5, base + 1)\n"
" if mode == 'invalid_rubric':\n"
" scores['order_pacing']['score'] = 6\n"
" evaluations.append({'candidateId': item['candidateId'], 'scores': scores, 'summary': '结构化评分摘要'})\n"
" print(json.dumps({'profile': 'fine_outline_replay', 'judgeId': request['judgeId'], 'evaluations': evaluations}, ensure_ascii=False))\n",
encoding="utf-8",
)
os.chmod(runner, 0o755)
return runner, log_path
def read_calls(log_path):
if not log_path.exists():
return []
return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()]
def nested_keys(value):
"""收集嵌套 JSON 的全部字段名,供盲化边界测试使用。"""
if isinstance(value, dict):
keys = set(value)
for item in value.values():
keys.update(nested_keys(item))
return keys
if isinstance(value, list):
keys = set()
for item in value:
keys.update(nested_keys(item))
return keys
return set()
class ReplayRunTest(unittest.TestCase):
def test_public_planner_context_does_not_include_snapshot_cards(self):
prompt = _planner_prompt(
target=489,
as_of=488,
snapshot={"cards": [{"name": "hidden-card"}], "safe": "public"},
common_input={},
cards=[{"name": "injected-card"}],
)
self.assertNotIn("hidden-card", prompt)
self.assertIn("injected-card", prompt)
def test_dry_run_passes_and_writes_only_metadata(self):
with tempfile.TemporaryDirectory() as directory:
result = run_replay(config(), pathlib.Path(directory), mode="dry_run")
self.assertTrue(result["ok"])
self.assertEqual(result["status"], "ready")
manifest = json.loads((pathlib.Path(directory) / "snapshot_manifest.json").read_text())
self.assertNotIn("payload", json.dumps(manifest, ensure_ascii=False))
self.assertFalse(list(pathlib.Path(directory).glob("planner_*.raw.json")))
def test_bad_authorization_stops_before_model(self):
bad = config()
bad["authorization"] = {"sourceStatus": "active"}
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_authorization")
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_reused_output_dir_bad_config_overwrites_previous_completed_state(self):
with tempfile.TemporaryDirectory() as directory:
output_dir = pathlib.Path(directory) / "run"
output_dir.mkdir()
result_path = output_dir / "run_result.json"
result_path.write_text(
json.dumps({"runId": "old-run", "status": "completed", "ok": True}),
encoding="utf-8",
)
bad = config()
bad["snapshot"]["data"]["unknownSection"] = []
with self.assertRaisesRegex(ValueError, "未登记顶层分区"):
run_replay(bad, output_dir, mode="execute")
persisted = json.loads(result_path.read_text(encoding="utf-8"))
self.assertEqual(persisted["runId"], "smoke-001")
self.assertEqual(persisted["status"], "config_invalid")
self.assertFalse(persisted["ok"])
self.assertNotEqual(persisted["status"], "completed")
def test_content_leak_stops_before_manifest(self):
bad = config()
bad["leakageAudit"]["targetFacts"]["forbiddenFacts"][0]["text"] = "安全历史"
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_snapshot")
self.assertEqual(result["leakageAudit"]["findingCount"], 1)
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_missing_content_audit_stops_before_model(self):
bad = config()
del bad["leakageAudit"]
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "blocked_leakage_audit")
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_arm_card_future_record_stops_before_model(self):
bad = config()
bad["arms"]["outline_plus_cards"]["cards"] = [
{"name": "未来卡", "milestones": [{"chapter": 489, "fact": "未来"}]}
]
with tempfile.TemporaryDirectory() as directory:
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "invalid_snapshot")
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
def test_report_contains_hashes_but_not_raw_fields(self):
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
report = render_report(result)
self.assertIn("smoke-001", report)
self.assertIn("snapshotManifestSha256", report)
self.assertNotIn('"prompt":', report)
self.assertNotIn('"payload":', report)
def test_report_rejects_text_injection_through_identifier_fields(self):
base = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
injections = {
"runId": "safe-run\n完整目标细纲:第一幕到第三幕的全部事件原文",
"referenceWork": "深空之影目标章原文与完整细纲",
}
for field, injected in injections.items():
with self.subTest(field=field):
unsafe = dict(base)
unsafe[field] = injected
with self.assertRaisesRegex(ValueError, field):
render_report(unsafe)
def test_report_does_not_render_preflight_free_text(self):
bad = config()
injected = "目标章事实文本"
bad["authorization"] = {**bad["authorization"], "sourceStatus": injected}
result = run_replay(bad, pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
report = render_report(result)
self.assertNotIn(injected, report)
self.assertIn("授权前置门未通过", report)
def test_report_accepts_registered_target_source_failure_status(self):
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
result["results"] = {
"outline_only": {"status": "target_source_forbidden", "ok": False}
}
report = render_report(result)
self.assertIn("target_source_forbidden", report)
def test_execute_uses_external_planner_and_validates_each_candidate(self):
with tempfile.TemporaryDirectory() as directory:
fake, log_path = write_fake_runner(directory)
output_dir = pathlib.Path(directory) / "run"
result = run_replay(config(), output_dir, mode="execute", planner_bin=str(fake))
first_call = read_calls(log_path)[0]
self.assertTrue(result["ok"])
self.assertEqual(result["status"], "completed")
self.assertFalse(first_call["runState"]["ok"])
self.assertEqual(first_call["runState"]["status"], "running_planner")
self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"})
self.assertTrue(list(output_dir.glob("candidate_*.json")))
def test_missing_planner_binary_fails_closed_and_persists_result(self):
with tempfile.TemporaryDirectory() as directory:
output_dir = pathlib.Path(directory) / "run"
missing = pathlib.Path(directory) / "missing-planner"
try:
result = run_replay(config(), output_dir, mode="execute", planner_bin=str(missing))
except OSError as error:
self.fail(f"runner 启动异常不得逃逸: {error}")
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
self.assertFalse(result["ok"])
self.assertEqual(result["status"], "planner_failed")
self.assertEqual(persisted["status"], "planner_failed")
self.assertFalse(persisted["ok"])
self.assertNotIn(persisted["status"], {"ready", "completed"})
def test_planner_nonzero_exit_stops_after_first_call(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "planner_exit")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
self.assertEqual(result["status"], "planner_failed")
self.assertFalse(result["ok"])
self.assertEqual(len(calls), 1)
def test_planner_invalid_json_stops_after_first_call(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "planner_invalid_json")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
self.assertEqual(result["status"], "planner_invalid")
self.assertFalse(result["ok"])
self.assertEqual(len(calls), 1)
def test_planner_timeout_fails_closed_and_persists_timeout_status(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "planner_timeout")
output_dir = pathlib.Path(directory) / "run"
result = run_replay(
config(),
output_dir,
mode="execute",
planner_bin=str(runner),
timeout_seconds=1.0,
)
persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8"))
self.assertEqual(result["status"], "planner_timeout")
self.assertFalse(result["ok"])
self.assertEqual(persisted["status"], "planner_timeout")
self.assertEqual(len(read_calls(log_path)), 1)
def test_detector_nonzero_exit_stops_before_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "detector_exit")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_failed")
self.assertFalse(result["ok"])
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3)
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_detector_invalid_json_stops_before_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "detector_invalid_json")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_invalid")
self.assertFalse(result["ok"])
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_detector_timeout_fails_closed_before_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "detector_timeout")
result = run_replay(
config(),
pathlib.Path(directory) / "run",
mode="execute",
planner_bin=str(runner),
timeout_seconds=1.0,
)
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_timeout")
self.assertFalse(result["ok"])
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_primary_judge_nonzero_exit_does_not_call_secondary(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "judge_exit")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "judge_failed")
self.assertFalse(result["ok"])
self.assertEqual(len(judge_calls), 1)
def test_primary_judge_invalid_json_does_not_call_secondary(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "judge_invalid_json")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "judge_invalid")
self.assertFalse(result["ok"])
self.assertEqual(len(judge_calls), 1)
def test_primary_judge_timeout_does_not_call_secondary(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "judge_timeout")
result = run_replay(
config(),
pathlib.Path(directory) / "run",
mode="execute",
planner_bin=str(runner),
timeout_seconds=1.0,
)
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "judge_timeout")
self.assertFalse(result["ok"])
self.assertEqual(len(judge_calls), 1)
def test_cli_accepts_subprocess_timeout_seconds(self):
argv = [
"run_replay.py",
"--config",
"/tmp/replay-config.json",
"--output-dir",
"/tmp/replay-output",
"--timeout-seconds",
"12.5",
]
with patch.object(sys, "argv", argv):
args = _parse_args()
self.assertEqual(args.timeout_seconds, 12.5)
def test_non_finite_timeout_is_persisted_as_config_invalid(self):
with tempfile.TemporaryDirectory() as directory:
for index, timeout_seconds in enumerate((float("nan"), float("inf"))):
with self.subTest(timeout_seconds=timeout_seconds):
output_dir = pathlib.Path(directory) / f"run-{index}"
with self.assertRaisesRegex(ValueError, "timeout_seconds"):
run_replay(
config(),
output_dir,
mode="execute",
timeout_seconds=timeout_seconds,
)
persisted = json.loads(
(output_dir / "run_result.json").read_text(encoding="utf-8")
)
self.assertEqual(persisted["status"], "config_invalid")
self.assertFalse(persisted["ok"])
def test_schema_failure_stops_before_all_detectors_and_judges(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "invalid_candidate")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
self.assertEqual(result["status"], "candidate_blocked")
self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3)
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 0)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_high_detector_finding_blocks_group_and_never_calls_judge(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "high_detector")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
detector_calls = [call for call in calls if call["agent"] == "detector"]
self.assertEqual(result["status"], "detector_blocked")
self.assertEqual(len(detector_calls), 3)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls))
self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls))
def test_detector_requests_cannot_distinguish_arm_specific_card_identity(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "stable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
self.assertEqual(result["status"], "completed")
requests = [
json.loads(call["prompt"])
for call in read_calls(log_path)
if call["agent"] == "detector"
]
self.assertEqual(len(requests), 3)
for request in requests:
self.assertTrue(
nested_keys(request).isdisjoint({"arm", "cardInjection", "cardManifest"})
)
serialized = json.dumps(request, ensure_ascii=False)
self.assertNotIn("正确卡", serialized)
self.assertNotIn("错配卡", serialized)
self.assertNotIn("card-correct-1", serialized)
self.assertNotIn("card-placebo-1", serialized)
public_parts = [
{key: value for key, value in request.items() if key not in {"candidateId", "candidate"}}
for request in requests
]
self.assertEqual(public_parts[0], public_parts[1])
self.assertEqual(public_parts[1], public_parts[2])
def test_invalid_detector_report_fails_closed_before_judge(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "invalid_detector")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
calls = read_calls(log_path)
self.assertEqual(result["status"], "detector_invalid")
self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1)
self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0)
def test_two_judges_are_independent_blind_reversed_and_unblinded(self):
with tempfile.TemporaryDirectory() as directory:
runner, log_path = write_fake_runner(directory, "stable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"]
self.assertEqual(result["status"], "completed")
self.assertEqual(len(judge_calls), 2)
first = json.loads(judge_calls[0]["prompt"])
second = json.loads(judge_calls[1]["prompt"])
self.assertNotEqual(first["judgeId"], second["judgeId"])
self.assertEqual(
[item["candidateId"] for item in second["candidates"]],
list(reversed([item["candidateId"] for item in first["candidates"]])),
)
self.assertNotIn("outline_only", judge_calls[0]["prompt"])
self.assertNotIn("outline_plus_cards", judge_calls[1]["prompt"])
self.assertTrue(result["evaluation"]["stability"]["stable"])
self.assertEqual(result["evaluation"]["deltas"]["B-A"]["order_pacing"], 1.0)
self.assertEqual(result["evaluation"]["deltas"]["C-A"]["order_pacing"], -1.0)
def test_invalid_rubric_report_does_not_complete(self):
with tempfile.TemporaryDirectory() as directory:
runner, _ = write_fake_runner(directory, "invalid_rubric")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
self.assertEqual(result["status"], "judge_invalid")
self.assertFalse(result["ok"])
def test_unstable_judges_have_explicit_non_completed_status(self):
with tempfile.TemporaryDirectory() as directory:
runner, _ = write_fake_runner(directory, "unstable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
self.assertEqual(result["status"], "judge_unstable")
self.assertFalse(result["ok"])
self.assertFalse(result["evaluation"]["stability"]["stable"])
self.assertNotIn("deltas", result["evaluation"])
def test_report_contains_only_aggregated_evaluation(self):
with tempfile.TemporaryDirectory() as directory:
runner, _ = write_fake_runner(directory, "stable")
result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner))
report = render_report(result)
self.assertIn("B-A", report)
self.assertIn("C-A", report)
self.assertIn("评委稳定性", report)
self.assertNotIn("结构化证据", report)
self.assertNotIn("目标章未进入快照", report)
if __name__ == "__main__":
unittest.main()