diff --git a/.claude/skills/detect/SKILL.md b/.claude/skills/detect/SKILL.md index 8225d0b..fa4bbc1 100644 --- a/.claude/skills/detect/SKILL.md +++ b/.claude/skills/detect/SKILL.md @@ -50,7 +50,9 @@ schema 给字段加上 detection 用途,检查项自动+1,本 skill 与 detector | 来源引用 | `sourceRefs` 是否来自快照、是否包含目标章及以后 | 目标章/未来来源出现在候选引用中 | 来源 ID + 章号范围 | | 未知项纪律 | `unknowns` / `assumptions` 是否显式承载缺口 | 用无来源断言替代未知项 | 候选字段路径 | -报告仍然只产审查结果,不修改候选。回放中任一高严重度问题阻断该臂进入 judge;“卡里缺少目标新角色”要单列为资料覆盖发现,不冒充规划器错误。 +机器报告的类别是闭集:`findings.category` 只允许 `candidate_structure`、`causal_chain`、`entity_state`、`foreshadowing_action`、`source_reference`、`unknowns_discipline`;`coverageFindings.category` 只允许 `frozen_context_gap`。未登记类别不得用近义词或变体绕过,编排器必须失败关闭。 + +报告仍然只产审查结果,不修改候选。回放中任一高严重度问题阻断该臂进入 judge;目标章新角色是否缺卡需要目标章标准事实,detector 无权判断,归 judge/eval。 ## 红线 diff --git a/.claude/skills/replay-eval/scripts/fine_outline_detector.py b/.claude/skills/replay-eval/scripts/fine_outline_detector.py index 7a52eec..61bbcbf 100644 --- a/.claude/skills/replay-eval/scripts/fine_outline_detector.py +++ b/.claude/skills/replay-eval/scripts/fine_outline_detector.py @@ -8,13 +8,17 @@ from typing import Any, Mapping DETECTOR_PROTOCOL = "fine_outline_detector_v0" ALLOWED_SEVERITIES = frozenset({"high", "medium", "low"}) -FORBIDDEN_COVERAGE_CATEGORIES = frozenset( +ALLOWED_FINDING_CATEGORIES = frozenset( { - "missing_target_role_card", - "new_target_role_missing_card", - "target_role_card_coverage", + "candidate_structure", + "causal_chain", + "entity_state", + "foreshadowing_action", + "source_reference", + "unknowns_discipline", } ) +ALLOWED_COVERAGE_CATEGORIES = frozenset({"frozen_context_gap"}) def build_detector_request( @@ -75,17 +79,21 @@ def validate_detector_report( errors.append("coverageFindings 必须是数组") coverage_findings = [] - for section, items in (("findings", findings), ("coverageFindings", coverage_findings)): + sections = ( + ("findings", findings, ALLOWED_FINDING_CATEGORIES), + ("coverageFindings", coverage_findings, ALLOWED_COVERAGE_CATEGORIES), + ) + for section, items, allowed_categories in sections: for index, finding in enumerate(items): if not isinstance(finding, Mapping): errors.append(f"{section}[{index}] 必须是对象") continue category = finding.get("category") - if category in FORBIDDEN_COVERAGE_CATEGORIES: - errors.append(f"{section}[{index}] 越权判断目标新角色卡覆盖") + if not isinstance(category, str) or category not in allowed_categories: + errors.append(f"{section}[{index}].category 未登记") if section == "findings": severity = finding.get("severity") - if severity not in ALLOWED_SEVERITIES: + if not isinstance(severity, str) or severity not in ALLOWED_SEVERITIES: errors.append(f"findings[{index}].severity 无效") for field in ("category", "location", "evidenceSummary"): value = finding.get(field) diff --git a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py index dfc2bb4..688d839 100644 --- a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py +++ b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py @@ -115,15 +115,17 @@ def validate_report( for item in evaluations if isinstance(item, Mapping) ] - if len(candidate_ids) != len(evaluations) or any( - not isinstance(candidate_id, str) or not candidate_id + valid_candidate_ids = len(candidate_ids) == len(evaluations) and all( + isinstance(candidate_id, str) and bool(candidate_id) for candidate_id in candidate_ids - ): + ) + if not valid_candidate_ids: errors.append("每个 evaluation 必须包含非空 candidateId") - if len(candidate_ids) != len(set(candidate_ids)): - errors.append("evaluations 包含重复 candidateId") - if expected_candidate_ids is not None and set(candidate_ids) != set(expected_candidate_ids): - errors.append("evaluations 未精确覆盖盲化候选集合") + else: + if len(candidate_ids) != len(set(candidate_ids)): + errors.append("evaluations 包含重复 candidateId") + if expected_candidate_ids is not None and set(candidate_ids) != set(expected_candidate_ids): + errors.append("evaluations 未精确覆盖盲化候选集合") for index, evaluation in enumerate(evaluations): if not isinstance(evaluation, Mapping): errors.append(f"evaluations[{index}] 必须是对象") diff --git a/.claude/skills/replay-eval/scripts/run_replay.py b/.claude/skills/replay-eval/scripts/run_replay.py index e0ffdc1..12e4c43 100644 --- a/.claude/skills/replay-eval/scripts/run_replay.py +++ b/.claude/skills/replay-eval/scripts/run_replay.py @@ -36,6 +36,14 @@ class ReplayRunError(ValueError): """回放配置不符合运行边界。""" +class RunnerInvocationError(ReplayRunError): + """外部 runner 无法启动或以非零状态退出。""" + + +class RunnerOutputError(ReplayRunError): + """外部 runner 返回的内容不符合机器合同。""" + + def _read_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8")) @@ -95,7 +103,7 @@ def _extract_candidate(output: str) -> Mapping[str, Any]: outer = match.group(1) if match else outer.strip() outer = json.loads(outer) if not isinstance(outer, Mapping): - raise ReplayRunError("模型输出不是 JSON 对象") + raise RunnerOutputError("模型输出不是 JSON 对象") return outer @@ -188,10 +196,13 @@ def _invoke_planner( "本次是严格离线回放;不要调用任何工具,不要读取文件,不要输出 JSON 以外内容。", prompt, ] - completed = subprocess.run(command, text=True, capture_output=True, check=False) + try: + completed = subprocess.run(command, text=True, capture_output=True, check=False) + except OSError as error: + raise RunnerInvocationError(f"planner 启动失败: {error}") from error output_path.write_text(completed.stdout, encoding="utf-8") if completed.returncode != 0: - raise ReplayRunError(f"planner 调用失败,退出码={completed.returncode}") + raise RunnerInvocationError(f"planner 调用失败,退出码={completed.returncode}") return _extract_candidate(completed.stdout) @@ -225,10 +236,13 @@ def _invoke_structured_agent( f"独立身份={identity};只处理给定 JSON;禁止调用工具、读取文件或输出 JSON 以外内容。", _safe_json(request), ] - completed = subprocess.run(command, text=True, capture_output=True, check=False) + try: + completed = subprocess.run(command, text=True, capture_output=True, check=False) + except OSError as error: + raise RunnerInvocationError(f"{agent} 启动失败: {error}") from error output_path.write_text(completed.stdout, encoding="utf-8") if completed.returncode != 0: - raise ReplayRunError(f"{agent} 调用失败,退出码={completed.returncode}") + raise RunnerInvocationError(f"{agent} 调用失败,退出码={completed.returncode}") return _extract_candidate(completed.stdout) @@ -508,6 +522,11 @@ def run_replay( _write_result(output_dir, result) return result + result["status"] = "running_planner" + result["ok"] = False + result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"] + _write_result(output_dir, result) + candidates: dict[str, Mapping[str, Any]] = {} for name in REQUIRED_ARMS: arm = _require_mapping(arms, name) @@ -539,13 +558,26 @@ def run_replay( "candidateSha256": sha256_value(candidate), "candidatePath": str(candidate_path), } - except (ReplayRunError, json.JSONDecodeError) as error: + except RunnerInvocationError as error: result["results"][name] = { - "status": "planner_output_invalid", + "status": "planner_failed", "ok": False, "errors": [str(error)], "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, } + result["status"] = "planner_failed" + _write_result(output_dir, result) + return result + except (RunnerOutputError, json.JSONDecodeError) as error: + result["results"][name] = { + "status": "planner_invalid", + "ok": False, + "errors": [str(error)], + "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, + } + result["status"] = "planner_invalid" + _write_result(output_dir, result) + return result if not all(item["ok"] for item in result["results"].values()): result["status"] = "candidate_blocked" result["ok"] = False @@ -555,6 +587,8 @@ def run_replay( run_id = str(result["runId"]) assignments = _blind_assignments(candidates, run_id) detector_runner = detector_bin or planner_bin + result["status"] = "running_detector" + _write_result(output_dir, result) detector_invalid = False detector_blocked = False for assignment in assignments: @@ -598,8 +632,27 @@ def run_replay( "reportSha256": sha256_value(detector_report), "errors": detector_check["errors"], } - except (ReplayRunError, json.JSONDecodeError) as error: - detector_invalid = True + if not detector_check["ok"]: + result["status"] = "detector_invalid" + _write_result(output_dir, result) + return result + except RunnerInvocationError as error: + result["results"][arm]["detector"] = { + "status": "failed", + "findingCount": 0, + "coverageFindingCount": 0, + "highSeverityCount": 0, + "errors": [str(error)], + "rawOutputSha256": ( + sha256_value(detector_path.read_text(encoding="utf-8")) + if detector_path.exists() + else None + ), + } + result["status"] = "detector_failed" + _write_result(output_dir, result) + return result + except (RunnerOutputError, json.JSONDecodeError) as error: result["results"][arm]["detector"] = { "status": "invalid", "findingCount": 0, @@ -612,6 +665,9 @@ def run_replay( else None ), } + result["status"] = "detector_invalid" + _write_result(output_dir, result) + return result if detector_invalid or detector_blocked: result["status"] = "detector_invalid" if detector_invalid else "detector_blocked" @@ -632,6 +688,8 @@ def run_replay( _write_result(output_dir, result) return result + result["status"] = "running_judge" + _write_result(output_dir, result) for index, judge_id in enumerate(JUDGE_IDS): ordered = assignments if index == 0 else list(reversed(assignments)) request = _judge_request( @@ -660,11 +718,34 @@ def run_replay( expected_candidate_ids=candidate_ids, ) if judge_errors: - judge_invalid = True + result["status"] = "judge_invalid" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "invalid", + } + _write_result(output_dir, result) + return result else: judge_reports.append(judge_report) - except (ReplayRunError, json.JSONDecodeError): - judge_invalid = True + except RunnerInvocationError: + result["status"] = "judge_failed" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "failed", + } + _write_result(output_dir, result) + return result + except (RunnerOutputError, json.JSONDecodeError): + result["status"] = "judge_invalid" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "invalid", + } + _write_result(output_dir, result) + return result if judge_invalid or len(judge_reports) != 2: result["status"] = "judge_invalid" diff --git a/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py index 83ddb6f..436e73a 100644 --- a/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py +++ b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py @@ -32,6 +32,54 @@ class FineOutlineDetectorTest(unittest.TestCase): } self.assertFalse(validate_detector_report(target_role_judgment, "blind-1")["ok"]) + def test_semantic_alias_and_unknown_categories_fail_closed(self): + for section, category in ( + ("findings", "target_new_character_card_missing"), + ("coverageFindings", "new_role_without_card"), + ("findings", "invented_detector_category"), + ("coverageFindings", ["frozen_context_gap"]), + ): + report = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [], + "coverageFindings": [], + } + finding = { + "category": category, + "severity": "low", + "location": "candidate", + "evidenceSummary": "测试", + } + report[section] = [finding] + self.assertFalse(validate_detector_report(report, "blind-1")["ok"]) + + def test_registered_categories_are_accepted_and_detect_skill_assigns_target_coverage_to_eval(self): + report = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [ + { + "category": "entity_state", + "severity": "medium", + "location": "entities[0]", + "evidenceSummary": "与冻结状态不一致", + } + ], + "coverageFindings": [ + { + "category": "frozen_context_gap", + "summary": "冻结资料缺少已知实体字段", + } + ], + } + self.assertTrue(validate_detector_report(report, "blind-1")["ok"]) + + skill_path = pathlib.Path(__file__).resolve().parents[2] / "detect" / "SKILL.md" + skill = skill_path.read_text(encoding="utf-8") + self.assertNotIn("要单列为资料覆盖发现", skill) + self.assertIn("归 judge/eval", skill) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_rubric.py b/.claude/skills/replay-eval/scripts/test_rubric.py index 1d793de..189f4f4 100644 --- a/.claude/skills/replay-eval/scripts/test_rubric.py +++ b/.claude/skills/replay-eval/scripts/test_rubric.py @@ -84,6 +84,12 @@ class FineOutlineRubricTest(unittest.TestCase): ) ) + invalid_id = { + **report, + "evaluations": [{**report["evaluations"][0], "candidateId": ["blind-1"]}], + } + self.assertTrue(validate_report(invalid_id, expected_judge_id="judge-primary")) + arm_leak = {**report, "arm": "outline_only"} self.assertTrue( validate_report( diff --git a/.claude/skills/replay-eval/scripts/test_run_replay.py b/.claude/skills/replay-eval/scripts/test_run_replay.py index 120b669..18d2ecd 100644 --- a/.claude/skills/replay-eval/scripts/test_run_replay.py +++ b/.claude/skills/replay-eval/scripts/test_run_replay.py @@ -109,17 +109,25 @@ def write_fake_runner(directory, mode="stable"): runner = directory / "fake-agent.py" log_path = directory / "agent-calls.jsonl" count_path = directory / "planner-count.txt" + run_result_path = directory / "run" / "run_result.json" runner.write_text( "#!/usr/bin/env python3\n" "import json, pathlib, sys\n" f"mode = {mode!r}\n" f"log_path = pathlib.Path({str(log_path)!r})\n" f"count_path = pathlib.Path({str(count_path)!r})\n" + f"run_result_path = pathlib.Path({str(run_result_path)!r})\n" "args = sys.argv[1:]\n" "agent = args[args.index('--agent') + 1]\n" "prompt = args[-1]\n" + "run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None\n" "with log_path.open('a', encoding='utf-8') as handle:\n" - " handle.write(json.dumps({'agent': agent, 'prompt': prompt}, ensure_ascii=False) + '\\n')\n" + " handle.write(json.dumps({'agent': agent, 'prompt': prompt, 'runState': run_state}, ensure_ascii=False) + '\\n')\n" + "if mode == f'{agent}_exit':\n" + " raise SystemExit(7)\n" + "if mode == f'{agent}_invalid_json':\n" + " print('{invalid-json')\n" + " raise SystemExit(0)\n" "if agent == 'planner':\n" " count = int(count_path.read_text() if count_path.exists() else '0') + 1\n" " count_path.write_text(str(count))\n" @@ -232,14 +240,89 @@ class ReplayRunTest(unittest.TestCase): def test_execute_uses_external_planner_and_validates_each_candidate(self): with tempfile.TemporaryDirectory() as directory: - fake, _ = write_fake_runner(directory) + fake, log_path = write_fake_runner(directory) output_dir = pathlib.Path(directory) / "run" result = run_replay(config(), output_dir, mode="execute", planner_bin=str(fake)) + first_call = read_calls(log_path)[0] self.assertTrue(result["ok"]) self.assertEqual(result["status"], "completed") + self.assertFalse(first_call["runState"]["ok"]) + self.assertEqual(first_call["runState"]["status"], "running_planner") self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"}) self.assertTrue(list(output_dir.glob("candidate_*.json"))) + def test_missing_planner_binary_fails_closed_and_persists_result(self): + with tempfile.TemporaryDirectory() as directory: + output_dir = pathlib.Path(directory) / "run" + missing = pathlib.Path(directory) / "missing-planner" + try: + result = run_replay(config(), output_dir, mode="execute", planner_bin=str(missing)) + except OSError as error: + self.fail(f"runner 启动异常不得逃逸: {error}") + persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8")) + self.assertFalse(result["ok"]) + self.assertEqual(result["status"], "planner_failed") + self.assertEqual(persisted["status"], "planner_failed") + self.assertFalse(persisted["ok"]) + self.assertNotIn(persisted["status"], {"ready", "completed"}) + + def test_planner_nonzero_exit_stops_after_first_call(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "planner_exit") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "planner_failed") + self.assertFalse(result["ok"]) + self.assertEqual(len(calls), 1) + + def test_planner_invalid_json_stops_after_first_call(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "planner_invalid_json") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "planner_invalid") + self.assertFalse(result["ok"]) + self.assertEqual(len(calls), 1) + + def test_detector_nonzero_exit_stops_before_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "detector_exit") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_failed") + self.assertFalse(result["ok"]) + self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_detector_invalid_json_stops_before_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "detector_invalid_json") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_invalid") + self.assertFalse(result["ok"]) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_primary_judge_nonzero_exit_does_not_call_secondary(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "judge_exit") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "judge_failed") + self.assertFalse(result["ok"]) + self.assertEqual(len(judge_calls), 1) + + def test_primary_judge_invalid_json_does_not_call_secondary(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "judge_invalid_json") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "judge_invalid") + self.assertFalse(result["ok"]) + self.assertEqual(len(judge_calls), 1) + def test_schema_failure_stops_before_all_detectors_and_judges(self): with tempfile.TemporaryDirectory() as directory: runner, log_path = write_fake_runner(directory, "invalid_candidate") @@ -268,6 +351,7 @@ class ReplayRunTest(unittest.TestCase): result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) calls = read_calls(log_path) self.assertEqual(result["status"], "detector_invalid") + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) def test_two_judges_are_independent_blind_reversed_and_unblinded(self): diff --git a/docs/2026-07-19-回放评测-细纲首跑设计与计划.md b/docs/2026-07-19-回放评测-细纲首跑设计与计划.md index 93404a4..9954cab 100644 --- a/docs/2026-07-19-回放评测-细纲首跑设计与计划.md +++ b/docs/2026-07-19-回放评测-细纲首跑设计与计划.md @@ -11,12 +11,12 @@ ## 执行状态(2026-07-19) - 已落地并同步到 `agent-example/main`:`fbb262c`(冻结/细纲/评分合同)、`a54a3a4`(审查反馈收紧与回放编排)、`1243d9b`(外部 planner execute 链路测试)。 -- 已验证:回放脚本 38 个离线测试、fine-outline 合同 3 个测试全部通过;fake planner 已跑通三臂 execute 链路。测试只使用合成 fixture,不代表参考作品评测结果。 +- 已验证:replay-eval 58 个离线测试、fine-outline 合同 3 个测试全部通过;fake runner 已跑通三臂 planner、逐臂 detector、双 judge 反序盲评、稳定性门和去盲矩阵。测试只使用合成 fixture,不代表参考作品评测结果。 - 已新增并验证:`audit_leakage.py` 对公共快照和三臂卡注入区执行内容级事实审计;`load_reference_work.py` 通过 PostgreSQL 只读事务组装仓库外临时配置,候选卡标记为 `eval_draft`,不进入生产检索。 - 已执行真实适配 smoke:work=8、冻结 488、目标 489,组装 31 个完整历史大纲窗、6 个近章细纲摘要、正确/错配卡各 2 张;送入回放仍为 `blocked_authorization`(`example_reference_work` 没有授权快照),未生成 `snapshot_manifest`,未调用模型。 - 已执行真实数据库前置 smoke:深空之影 `work_id=8` 的 `example_reference_work` 表没有不可变授权快照/版权用途字段,结果为 `blocked_authorization`,三臂均未调用模型。 - `.claude/skills/llm/scripts/test_quota.py` 已统一到当前 `$24/6000` 契约:预算场景使用 `$24.5`,调用上限场景使用 `6000`,并增加策略常量断言。 -- 当前允许的结论是“冻结、变量控制、内容级泄露审计、候选结构门和安全摘要机制已通”;不能说细纲智能体或知识卡已通过。真实授权接入、detector/judge 双评和 430/489/550 实样仍待完成。 +- 当前允许的结论是“冻结、变量控制、内容级泄露审计、候选结构门、detector/judge 双评编排和安全摘要机制已通”;不能说细纲智能体或知识卡已通过。真实授权接入和 430/489/550 实样仍待完成。 --- @@ -214,7 +214,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} - 是否把 `unknown` 写成确定事实; - 是否出现候选自身引用未来来源的证据。 -审查只产报告,不改候选;高严重度阻断项不进入盲评。审查发现“卡里缺少目标新角色”时必须单列为设计发现,不把它误判成规划器读卡失败。 +审查只产报告,不改候选;高严重度阻断项不进入盲评。“卡里缺少目标新角色”需要目标章标准事实才能判断,由 judge/eval 侧单列,detector 不得判断或输出该类别。 ### 6.2 细纲专用评委 @@ -247,7 +247,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} | 假阴 | planner 以为卡已足够而少用公共大纲 | 三臂输入共用大纲,记录实际检索/引用卡 ID | | 假阳 | 未来事件/终态摘要泄露、错配卡只是额外文字、目标章信息参与召回 | 内容级泄露审计、placebo 臂、目标实体禁入选择器 | | 假阳 | 评委偏爱词面相似、参考细纲本身有误 | 结构化事实评分、两评委、proxy confidence、禁止正文文风维度 | -| 误归因 | 新角色无卡但某臂猜中、模型随机性、模型路由不一致 | 新角色缺卡单列;固定路由/预算;重复评审;不以 n=1 判决 | +| 误归因 | 新角色无卡但某臂猜中、模型随机性、模型路由不一致 | judge/eval 单列新角色缺卡;固定路由/预算;重复评审;不以 n=1 判决 | --- @@ -327,7 +327,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} - [x] 给 detector 增加细纲候选的检查对象、严重度和证据格式。 - [x] 给 judge 增加 `fine_outline_replay` rubric profile,明确不评正文文风和文笔。 -- [ ] 在真实 judge 编排中固定两评委顺序互换和匿名 arm;当前仅完成 rubric 稳定性校验合同。 +- [x] 在回放编排中固定两个独立 judge、匿名 arm 和第二评委候选顺序反转;fake runner 已覆盖双评、rubric 合同和稳定性失败关闭,真实样本尚未越过授权门。 - [x] 机械测试确保 rubric 不包含正文质量维度,且每个分数必须有证据字段。 ### Task 4:回放编排与最小首跑 @@ -341,7 +341,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} - [x] 先用合成 fixture 完成一个机制 dry-run,并用 fake planner 验证 execute 链路;真实样本尚未越过授权门。 - [x] 编排器支持每个目标章 A/B/C 三臂,planner 使用相同身份段、功能段、预算和输出合同;尚未运行参考作品目标章。 -- [ ] 两个独立 judge 对每章候选做顺序互换盲评;detector 阻断样本不送 judge。 +- [x] 两个独立 judge 对每章候选做顺序互换盲评;detector 阻断样本不送 judge。当前由 fake runner 离线测试验证,尚无真实样本结果。 - [x] 运行时原始候选和标准事实摘要只放仓库外临时目录;安全报告只写摘要、定位、评分入口、哈希和失败类别。 - [x] 接入真实数据库只读适配 smoke;授权缺失时维持 `blocked_authorization`,不启动 planner。 - [ ] 产出真实卡增量矩阵和样本级假阴/假阳说明;不能用机制 smoke 的合成结果代替。