181 lines
7.8 KiB
Python
181 lines
7.8 KiB
Python
#!/usr/bin/env python3
|
|
"""评委智能体派发桥离线测试(阶段 F 第三部分)。
|
|
|
|
固定合同:圈定授权工具白名单(实验无工具/生产预检只读)、任务包不带身份
|
|
(防泄漏走查通过、输入与盲评模型输入全等)、每位评委独立会话、派发失败
|
|
失败关闭、盲评输入组装哈希绑定经官方校验器回环。
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import pathlib
|
|
import sys
|
|
import tempfile
|
|
import unittest
|
|
from unittest import mock
|
|
|
|
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
|
SCRIPT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "judge" / "评估内容质量" / "scripts"
|
|
for path in (SCRIPT_DIR,):
|
|
if str(path) not in sys.path:
|
|
sys.path.insert(0, str(path))
|
|
|
|
import dispatch_judge_bridge as bridge # noqa: E402
|
|
import judge_via_dispatch as cli # noqa: E402
|
|
from dispatch_judge_bridge import ( # noqa: E402
|
|
DispatchJudgeError,
|
|
DispatchJudgeRunner,
|
|
judge_tool_allowlist,
|
|
)
|
|
from run_writer_blind_judge import ( # noqa: E402
|
|
FORBIDDEN_KEYS,
|
|
BlindJudgeContractError,
|
|
_walk_for_leakage,
|
|
validate_blind_judge_input,
|
|
)
|
|
|
|
MODEL_INPUT = {
|
|
"candidates": [{"blindCandidateId": "candidate-1", "candidateBody": "正文甲"}],
|
|
"rubric": {"dimensions": []},
|
|
}
|
|
|
|
|
|
class AllowlistTest(unittest.TestCase):
|
|
|
|
def test_experiment_has_no_tools(self):
|
|
self.assertEqual(judge_tool_allowlist("experiment"), ())
|
|
|
|
def test_production_preflight_read_tools(self):
|
|
self.assertEqual(judge_tool_allowlist("production_preflight"),
|
|
("read_chapter_text", "search_entities"))
|
|
|
|
def test_unknown_mode_fails_closed(self):
|
|
with self.assertRaises(DispatchJudgeError):
|
|
judge_tool_allowlist("anything-else")
|
|
|
|
|
|
class RunnerTest(unittest.TestCase):
|
|
|
|
def setUp(self):
|
|
self._tmp = tempfile.TemporaryDirectory()
|
|
self.tmp = pathlib.Path(self._tmp.name)
|
|
self.captured = []
|
|
|
|
def tearDown(self):
|
|
self._tmp.cleanup()
|
|
|
|
def _runner(self, mode="production_preflight"):
|
|
return DispatchJudgeRunner(
|
|
run_id="run-judge-test", work_id=12, repo_root=self.tmp,
|
|
provider="catproxy-anthropic", model="claude-opus-5", mode=mode,
|
|
spec_dir=self.tmp,
|
|
)
|
|
|
|
def _fake_dispatch(self, output, status="completed", code=0):
|
|
def fake(spec_file, **kwargs):
|
|
self.captured.append({"spec": json.loads(pathlib.Path(spec_file).read_text(encoding="utf-8")),
|
|
"kwargs": kwargs})
|
|
run_dir = self.tmp / kwargs["run_id"]
|
|
run_dir.mkdir(parents=True)
|
|
if status == "completed":
|
|
(run_dir / "output.json").write_text(json.dumps(output, ensure_ascii=False), encoding="utf-8")
|
|
return {"status": status, "runDir": str(run_dir), "errorCode": None if code == 0 else "TIMEOUT"}, code
|
|
return fake
|
|
|
|
def test_run_returns_structured_output_and_receipt_hash(self):
|
|
runner = self._runner()
|
|
draft = {"schemaVersion": "blind-judge-draft-v3", "candidateScores": []}
|
|
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch(draft)):
|
|
out = runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={"type": "object"})
|
|
self.assertEqual(out["structuredOutput"], draft)
|
|
self.assertTrue(out["modelReceiptSha256"].startswith("sha256:"))
|
|
self.assertEqual(runner.dispatch_run_ids, ["run-judge-test-judge-v1"])
|
|
|
|
def test_task_spec_carries_no_identity(self):
|
|
runner = self._runner()
|
|
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})):
|
|
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={"type": "object"})
|
|
spec = self.captured[0]["spec"]
|
|
self.assertEqual(spec["role"], "judge")
|
|
# 输入与盲评模型输入全等:桥不附加身份字段
|
|
self.assertEqual(spec["input"], MODEL_INPUT)
|
|
# 防泄漏走查:任务包整体不含禁看键
|
|
_walk_for_leakage(spec)
|
|
for key in FORBIDDEN_KEYS:
|
|
self.assertNotIn(key, spec)
|
|
|
|
def test_each_judge_gets_fresh_session(self):
|
|
runner = self._runner()
|
|
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})):
|
|
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
|
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
|
sessions = [c["kwargs"]["session_id"] for c in self.captured]
|
|
self.assertEqual(sessions, ["judge-run-judge-test-judge-v1", "judge-run-judge-test-judge-v2"])
|
|
self.assertEqual(len(set(sessions)), 2)
|
|
|
|
def test_mode_controls_tool_allowlist_in_spec(self):
|
|
runner = self._runner(mode="experiment")
|
|
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})):
|
|
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
|
self.assertEqual(self.captured[0]["spec"]["toolAllowlist"], [])
|
|
self.assertFalse(self.captured[0]["kwargs"]["enable_read_tools"])
|
|
|
|
def test_dispatch_failure_fails_closed(self):
|
|
runner = self._runner()
|
|
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({}, status="failed", code=1)):
|
|
with self.assertRaises(BlindJudgeContractError) as ctx:
|
|
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
|
self.assertEqual(ctx.exception.code, "BLIND_JUDGE_RUNTIME_FAILED")
|
|
|
|
def test_wrong_adapter_role_rejected(self):
|
|
runner = self._runner()
|
|
with self.assertRaises(BlindJudgeContractError):
|
|
runner.run(adapter_role="writer", model_input=MODEL_INPUT, output_schema={})
|
|
|
|
|
|
class BlindInputBuilderTest(unittest.TestCase):
|
|
|
|
def test_assembled_input_passes_official_validator(self):
|
|
bodies = {123: "甲候选正文:茧撕开舱门。", 164: "乙候选正文:林深盯着深渊。"}
|
|
outline = {
|
|
"sourceRef": {"sourceId": "outline:volume_1", "sourceVersion": "v1"},
|
|
"hardConstraints": ["硬约束甲", "硬约束乙"],
|
|
"adjustableBeats": ["拍一"],
|
|
"declaredNewFacts": [
|
|
{"factId": "fact-ch3-1", "text": "新事实", "sourceRef": {"sourceId": "outline:volume_1", "sourceVersion": "v1"}},
|
|
],
|
|
}
|
|
|
|
class _Conn:
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *exc):
|
|
return False
|
|
|
|
def execute(self, sql, params=None):
|
|
class _Rows:
|
|
def fetchall(self):
|
|
return [(cid, bodies[cid]) for cid in params[0]]
|
|
return _Rows()
|
|
|
|
with mock.patch.object(cli, "connect", lambda readonly=True: _Conn()), \
|
|
mock.patch.object(cli, "_latest_writer_context", lambda work_id, chapter: {"fineOutline": outline}):
|
|
built = cli.build_blind_input_for_candidates(
|
|
work_id=12, chapter=3, candidate_ids=[123, 164],
|
|
run_id="run-judge-test", sample_id="sample-1", scenario="battle",
|
|
authorization_snapshot_id="auth-work12-production-v1",
|
|
)
|
|
blind_input = built["blindInput"]
|
|
# 官方校验器回环:防泄漏、哈希绑定、候选哈希全部通过
|
|
validate_blind_judge_input(blind_input)
|
|
# 盲 ID 与库内候选脱钩但映射完整
|
|
self.assertEqual(sorted(built["blindMapping"].values()), [123, 164])
|
|
self.assertEqual(sorted(blind_input["candidateOrder"]), ["candidate-1", "candidate-2"])
|
|
# 目标章断言来自细纲硬约束
|
|
self.assertEqual(len(blind_input["oracleTruthPack"]["targetAssertions"]), 2)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|