diff --git a/.claude/skills/replay-eval/SKILL.md b/.claude/skills/replay-eval/SKILL.md index cb939bb..9fed84a 100644 --- a/.claude/skills/replay-eval/SKILL.md +++ b/.claude/skills/replay-eval/SKILL.md @@ -47,9 +47,10 @@ disable-model-invocation: true - `scripts/run_replay.py --mode dry_run`:只执行授权、来源、冻结和三臂 manifest 预检,不调用模型;这是首个机制 smoke 入口。 - `scripts/load_reference_work.py`:从 PostgreSQL 只读事务组装仓库外临时配置;缺授权字段仍会生成可审计配置,但送入 `run_replay` 后必须保持 `blocked_authorization`。 - `scripts/run_replay.py --mode execute`:在全部前置门通过后,依次执行三臂 planner、整组 schema、逐臂盲 detector、两个独立盲 judge、rubric 校验、稳定性门和去盲汇总;`--output-dir` 必须位于仓库外的临时目录。 -- detector 输入输出均为 JSON;输入只有匿名候选 ID、候选和冻结到 `as_of` 的规划上下文,不含 arm 名、目标章 proxy 或其他评委结果。任一 `high` 严重度发现整组标记 `detector_blocked`,judge 调用数必须为 0;报告不合约时标记 `detector_invalid`。 +- detector 输入输出均为 JSON;输入只有匿名候选 ID、候选和公共冻结到 `as_of` 的规划上下文,不含 arm 名、`cardInjection`、`cardManifest`、任何臂特有卡内容、目标章 proxy 或其他评委结果。卡注入合法性只由确定性预检负责。任一 `high` 严重度发现整组标记 `detector_blocked`,judge 调用数必须为 0;报告不合约时标记 `detector_invalid`。 - 两个 judge 使用不同 `judgeId` 和独立无会话进程。第二个 judge 的匿名候选顺序必须与第一个完全相反;任何 rubric 不合约标记 `judge_invalid`,任一同维差值大于 `0.5` 标记 `judge_unstable`,两者都不得标记 `completed`。 - 只有三臂 schema、detector、双 judge rubric 和稳定性门全部通过,才去盲生成逐维 `B-A` / `C-A` 差值矩阵并标记 `completed`。`--detector-bin`、`--judge-primary-bin`、`--judge-secondary-bin` 可分别指定本地 runner;未指定时复用 `--planner-bin`,测试只能使用 fake binary。 -- `scripts/write_report.py`:从 `run_result.json` 生成安全摘要;它不会读取候选正文,也不会把候选路径以外的原始响应写入报告。 +- planner、detector、judge 子进程统一受 `--timeout-seconds` 限制,默认 300 秒;任一超时分别落盘 `planner_timeout`、`detector_timeout`、`judge_timeout`,不得继续进入后续阶段或标记 `completed`。 +- `scripts/write_report.py`:从 `run_result.json` 生成独立严格 schema 的安全摘要,只接受受限标识符、枚举、数字、短安全摘要和 SHA-256;不会读取候选正文,也不会把候选路径以外的原始响应写入报告。 真实作品运行前必须先从权威来源取得不可变授权快照。数据库没有该字段时,使用 dry-run 证明机制并保持 `blocked_authorization`,不得用本地配置或口头许可伪造放行。 diff --git a/.claude/skills/replay-eval/scripts/fine_outline_detector.py b/.claude/skills/replay-eval/scripts/fine_outline_detector.py index 61bbcbf..24f8ee2 100644 --- a/.claude/skills/replay-eval/scripts/fine_outline_detector.py +++ b/.claude/skills/replay-eval/scripts/fine_outline_detector.py @@ -19,6 +19,8 @@ ALLOWED_FINDING_CATEGORIES = frozenset( } ) ALLOWED_COVERAGE_CATEGORIES = frozenset({"frozen_context_gap"}) +FINDING_FIELDS = frozenset({"category", "severity", "location", "evidenceSummary"}) +COVERAGE_FINDING_FIELDS = frozenset({"category", "summary"}) def build_detector_request( @@ -28,10 +30,9 @@ def build_detector_request( as_of_chapter: int, frozen_snapshot: Mapping[str, Any], common_context: Mapping[str, Any], - card_context: list[Any], sources: list[Any], ) -> dict[str, Any]: - """构造不含 arm、目标章 proxy 或卡策略标签的盲检输入。""" + """只用公共冻结事实和匿名候选构造真正不识别评测臂的盲检输入。""" return { "protocol": DETECTOR_PROTOCOL, @@ -41,7 +42,6 @@ def build_detector_request( "planningContext": { "frozenSnapshot": dict(frozen_snapshot), "commonContext": dict(common_context), - "cardInjection": list(card_context), "sources": list(sources), }, "rules": { @@ -88,6 +88,12 @@ def validate_detector_report( if not isinstance(finding, Mapping): errors.append(f"{section}[{index}] 必须是对象") continue + allowed_fields = FINDING_FIELDS if section == "findings" else COVERAGE_FINDING_FIELDS + unexpected_fields = sorted(set(finding) - allowed_fields) + if unexpected_fields: + errors.append( + f"{section}[{index}] 包含未登记字段: {','.join(unexpected_fields)}" + ) category = finding.get("category") if not isinstance(category, str) or category not in allowed_categories: errors.append(f"{section}[{index}].category 未登记") @@ -99,6 +105,10 @@ def validate_detector_report( value = finding.get(field) if not isinstance(value, str) or not value.strip(): errors.append(f"findings[{index}].{field} 必须是非空字符串") + else: + summary = finding.get("summary") + if not isinstance(summary, str) or not summary.strip(): + errors.append(f"coverageFindings[{index}].summary 必须是非空字符串") high_count = sum( 1 diff --git a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py index 688d839..e007bbd 100644 --- a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py +++ b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py @@ -21,6 +21,8 @@ DIMENSIONS = ( PROSE_DIMENSIONS = frozenset( {"style_fit", "readability", "文风一致性", "文笔", "pacing_tension", "information_density"} ) +SCORE_FIELDS = frozenset({"score", "evidence"}) +EVALUATION_FIELDS = frozenset({"candidateId", "scores", "summary"}) def validate_scores(scores: Mapping[str, Any]) -> list[str]: @@ -43,6 +45,11 @@ def validate_scores(scores: Mapping[str, Any]) -> list[str]: if not isinstance(value, Mapping): errors.append(f"维度必须包含 score/evidence 对象: {dimension}") continue + unexpected_fields = sorted(set(value) - SCORE_FIELDS) + if unexpected_fields: + errors.append( + f"维度包含未登记字段: {dimension}:{','.join(unexpected_fields)}" + ) score = value.get("score") if isinstance(score, bool) or not isinstance(score, (int, float)) or not 1 <= score <= 5: errors.append(f"分数必须在 1-5: {dimension}") @@ -130,6 +137,11 @@ def validate_report( if not isinstance(evaluation, Mapping): errors.append(f"evaluations[{index}] 必须是对象") continue + unexpected_evaluation_fields = sorted(set(evaluation) - EVALUATION_FIELDS) + if unexpected_evaluation_fields: + errors.append( + f"evaluations[{index}] 包含未登记字段: {','.join(unexpected_evaluation_fields)}" + ) errors.extend( f"evaluations[{index}]: {error}" for error in validate_scores(evaluation.get("scores", {})) diff --git a/.claude/skills/replay-eval/scripts/run_replay.py b/.claude/skills/replay-eval/scripts/run_replay.py index 12e4c43..3d3a255 100644 --- a/.claude/skills/replay-eval/scripts/run_replay.py +++ b/.claude/skills/replay-eval/scripts/run_replay.py @@ -9,12 +9,15 @@ from __future__ import annotations import argparse import json +import math +import os import re import subprocess +import tempfile from pathlib import Path from typing import Any, Mapping, Sequence -from build_snapshot import build_snapshot, normalize_chapter, sha256_value +from build_snapshot import SnapshotError, build_snapshot, normalize_chapter, sha256_value from audit_leakage import audit_snapshot from check_snapshot import ( STATUS_READY, @@ -30,6 +33,7 @@ REPO_ROOT = Path(__file__).resolve().parents[4] SKILL_PATH = REPO_ROOT / ".claude/skills/fine-outline/SKILL.md" PLANNER_PATH = REPO_ROOT / ".claude/agents/planner.md" JUDGE_IDS = ("judge-primary", "judge-secondary") +DEFAULT_TIMEOUT_SECONDS = 300.0 class ReplayRunError(ValueError): @@ -44,6 +48,10 @@ class RunnerOutputError(ReplayRunError): """外部 runner 返回的内容不符合机器合同。""" +class RunnerTimeoutError(ReplayRunError): + """外部 runner 超过允许的最长执行时间。""" + + def _read_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8")) @@ -52,6 +60,15 @@ def _safe_json(value: Any) -> str: return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")) +def _timeout_stdout(error: subprocess.TimeoutExpired) -> str: + """规范化超时前捕获的标准输出,供临时目录留痕和哈希审计。""" + + output = error.stdout or "" + if isinstance(output, bytes): + return output.decode("utf-8", errors="replace") + return output + + def _require_mapping(config: Mapping[str, Any], key: str) -> Mapping[str, Any]: value = config.get(key) if not isinstance(value, Mapping): @@ -175,6 +192,7 @@ def _invoke_planner( model: str, output_path: Path, max_budget_usd: float, + timeout_seconds: float, ) -> Mapping[str, Any]: """用无工具、无会话持久化的 Claude print 模式运行 planner。""" @@ -197,7 +215,16 @@ def _invoke_planner( prompt, ] try: - completed = subprocess.run(command, text=True, capture_output=True, check=False) + completed = subprocess.run( + command, + text=True, + capture_output=True, + check=False, + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as error: + output_path.write_text(_timeout_stdout(error), encoding="utf-8") + raise RunnerTimeoutError(f"planner 调用超时,限制={timeout_seconds:g}秒") from error except OSError as error: raise RunnerInvocationError(f"planner 启动失败: {error}") from error output_path.write_text(completed.stdout, encoding="utf-8") @@ -215,6 +242,7 @@ def _invoke_structured_agent( output_path: Path, max_budget_usd: float, identity: str, + timeout_seconds: float, ) -> Mapping[str, Any]: """以独立无会话进程调用 detector/judge,并保存仓库外原始响应。""" @@ -237,7 +265,16 @@ def _invoke_structured_agent( _safe_json(request), ] try: - completed = subprocess.run(command, text=True, capture_output=True, check=False) + completed = subprocess.run( + command, + text=True, + capture_output=True, + check=False, + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as error: + output_path.write_text(_timeout_stdout(error), encoding="utf-8") + raise RunnerTimeoutError(f"{agent} 调用超时,限制={timeout_seconds:g}秒") from error except OSError as error: raise RunnerInvocationError(f"{agent} 启动失败: {error}") from error output_path.write_text(completed.stdout, encoding="utf-8") @@ -380,7 +417,26 @@ def _aggregate_evaluation( def _write_result(output_dir: Path, result: Mapping[str, Any]) -> None: """每个阶段都覆盖写入可恢复的结构化运行状态。""" - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + result_path = output_dir / "run_result.json" + temporary_path: Path | None = None + try: + # 唯一临时文件避免同目录并发写相互覆盖;同目录 replace 保证正式状态原子切换。 + with tempfile.NamedTemporaryFile( + mode="w", + encoding="utf-8", + dir=output_dir, + prefix=".run_result.", + suffix=".tmp", + delete=False, + ) as handle: + temporary_path = Path(handle.name) + handle.write(_safe_json(result) + "\n") + handle.flush() + os.fsync(handle.fileno()) + temporary_path.replace(result_path) + finally: + if temporary_path is not None and temporary_path.exists(): + temporary_path.unlink() def run_replay( @@ -394,67 +450,98 @@ def run_replay( judge_secondary_bin: str | None = None, model: str = "opus", max_budget_usd: float = 1.0, + timeout_seconds: float = DEFAULT_TIMEOUT_SECONDS, ) -> dict[str, Any]: """执行一次单目标三臂回放;任何前置门失败都不调用模型。""" - if mode not in {"dry_run", "execute"}: - raise ReplayRunError("mode 只能是 dry_run 或 execute") output_dir = output_dir.resolve() if output_dir.is_relative_to(REPO_ROOT.resolve()): raise ReplayRunError("原始候选运行目录不得位于仓库内") output_dir.mkdir(parents=True, exist_ok=True) - - snapshot_config = _require_mapping(config, "snapshot") - as_of = normalize_chapter(snapshot_config.get("asOfChapter")) - target = normalize_chapter(config.get("targetChapter")) - snapshot_version = str(snapshot_config.get("snapshotVersion") or "") - if as_of is None or target is None: - raise ReplayRunError("as_of/target 必须是明确正整数") - if target != as_of + 1: - raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1") - reference_work = _require_mapping(config, "referenceWork") - authorization = _require_mapping(config, "authorization") - sources = config.get("sources", []) - if not isinstance(sources, list): - raise ReplayRunError("sources 必须是数组") - common_input = _require_mapping(config, "commonContext") - arms = _require_mapping(config, "arms") - if set(arms) != set(REQUIRED_ARMS): - raise ReplayRunError("生产回放必须精确配置三臂") - - arm_manifests = { - name: _build_arm_manifest( - name=name, - arm=_require_mapping(arms, name), - common_input=common_input, - as_of=as_of, - target=target, - snapshot_version=snapshot_version, - ) - for name in REQUIRED_ARMS - } - preflight = check_replay( - authorization=authorization, - as_of_chapter=as_of, - target_chapter=target, - planner_sources=sources, - arm_manifests=arm_manifests, - ) result: dict[str, Any] = { "runId": str(config.get("runId") or "unassigned"), "mode": mode, - "status": preflight["status"], - "ok": preflight["ok"], - "referenceWork": str(reference_work.get("id") or ""), - "referenceWorkVersion": str(reference_work.get("version") or ""), - "asOfChapter": as_of, - "targetChapter": target, - "snapshotVersion": snapshot_version, - "preflight": {"status": preflight["status"], "errors": preflight["errors"], "warnings": preflight["warnings"]}, - "arms": arm_manifests, + "status": "validating_config", + "ok": False, "results": {}, } _write_result(output_dir, result) + + try: + if mode not in {"dry_run", "execute"}: + raise ReplayRunError("mode 只能是 dry_run 或 execute") + if ( + isinstance(timeout_seconds, bool) + or not isinstance(timeout_seconds, (int, float)) + or not math.isfinite(float(timeout_seconds)) + or timeout_seconds <= 0 + ): + raise ReplayRunError("timeout_seconds 必须是正数") + snapshot_config = _require_mapping(config, "snapshot") + as_of = normalize_chapter(snapshot_config.get("asOfChapter")) + target = normalize_chapter(config.get("targetChapter")) + snapshot_version = str(snapshot_config.get("snapshotVersion") or "") + if as_of is None or target is None: + raise ReplayRunError("as_of/target 必须是明确正整数") + if target != as_of + 1: + raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1") + reference_work = _require_mapping(config, "referenceWork") + authorization = _require_mapping(config, "authorization") + sources = config.get("sources", []) + if not isinstance(sources, list): + raise ReplayRunError("sources 必须是数组") + common_input = _require_mapping(config, "commonContext") + arms = _require_mapping(config, "arms") + if set(arms) != set(REQUIRED_ARMS): + raise ReplayRunError("生产回放必须精确配置三臂") + + arm_manifests = { + name: _build_arm_manifest( + name=name, + arm=_require_mapping(arms, name), + common_input=common_input, + as_of=as_of, + target=target, + snapshot_version=snapshot_version, + ) + for name in REQUIRED_ARMS + } + preflight = check_replay( + authorization=authorization, + as_of_chapter=as_of, + target_chapter=target, + planner_sources=sources, + arm_manifests=arm_manifests, + ) + except ReplayRunError as error: + result["status"] = "config_invalid" + result["errors"] = [str(error)] + _write_result(output_dir, result) + raise + + result.update( + { + "referenceWork": str(reference_work.get("id") or ""), + "referenceWorkVersion": str(reference_work.get("version") or ""), + "asOfChapter": as_of, + "targetChapter": target, + "snapshotVersion": snapshot_version, + "preflight": { + "status": preflight["status"], + "errors": preflight["errors"], + "warnings": preflight["warnings"], + }, + "arms": arm_manifests, + } + ) + if preflight["ok"]: + # 前置门通过不等于快照配置有效;深层冻结完成前始终保持非成功态。 + result["status"] = "validating_config" + result["ok"] = False + else: + result["status"] = preflight["status"] + result["ok"] = False + _write_result(output_dir, result) if not preflight["ok"]: return result @@ -484,13 +571,20 @@ def run_replay( "armConfig": {"arms": list(REQUIRED_ARMS)}, } snapshot_data = snapshot_config.get("data", {}) - frozen = build_snapshot( - snapshot_data, - as_of, - snapshot_version, - target_chapter=target, - manifest_metadata=metadata, - ) + try: + frozen = build_snapshot( + snapshot_data, + as_of, + snapshot_version, + target_chapter=target, + manifest_metadata=metadata, + ) + except SnapshotError as error: + result["status"] = "config_invalid" + result["ok"] = False + result["errors"] = [str(error)] + _write_result(output_dir, result) + raise # 内容审计同时覆盖公共冻结快照和各臂卡注入区;卡不在公共区,不能因此逃过未来事实检查。 audit_payload = { @@ -546,6 +640,7 @@ def run_replay( model=model, output_path=raw_path, max_budget_usd=max_budget_usd, + timeout_seconds=float(timeout_seconds), ) candidate_path = output_dir / f"candidate_{name}.json" candidate_path.write_text(_safe_json(candidate) + "\n", encoding="utf-8") @@ -558,6 +653,16 @@ def run_replay( "candidateSha256": sha256_value(candidate), "candidatePath": str(candidate_path), } + except RunnerTimeoutError as error: + result["results"][name] = { + "status": "planner_timeout", + "ok": False, + "errors": [str(error)], + "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, + } + result["status"] = "planner_timeout" + _write_result(output_dir, result) + return result except RunnerInvocationError as error: result["results"][name] = { "status": "planner_failed", @@ -594,14 +699,12 @@ def run_replay( for assignment in assignments: arm = str(assignment["arm"]) candidate_id = str(assignment["candidateId"]) - arm_config = _require_mapping(arms, arm) request = build_detector_request( candidate_id=candidate_id, candidate=assignment["candidate"], as_of_chapter=as_of, frozen_snapshot=_public_snapshot(frozen["snapshot"]), common_context=common_input, - card_context=arm_config.get("cards", []), sources=sources, ) detector_path = output_dir / f"detector_{candidate_id}.raw.json" @@ -614,6 +717,7 @@ def run_replay( output_path=detector_path, max_budget_usd=max_budget_usd, identity="blind-detector", + timeout_seconds=float(timeout_seconds), ) detector_check = validate_detector_report(detector_report, candidate_id) detector_invalid = detector_invalid or not detector_check["ok"] @@ -636,6 +740,22 @@ def run_replay( result["status"] = "detector_invalid" _write_result(output_dir, result) return result + except RunnerTimeoutError as error: + result["results"][arm]["detector"] = { + "status": "timeout", + "findingCount": 0, + "coverageFindingCount": 0, + "highSeverityCount": 0, + "errors": [str(error)], + "rawOutputSha256": ( + sha256_value(detector_path.read_text(encoding="utf-8")) + if detector_path.exists() + else None + ), + } + result["status"] = "detector_timeout" + _write_result(output_dir, result) + return result except RunnerInvocationError as error: result["results"][arm]["detector"] = { "status": "failed", @@ -710,6 +830,7 @@ def run_replay( output_path=judge_path, max_budget_usd=max_budget_usd, identity=judge_id, + timeout_seconds=float(timeout_seconds), ) candidate_ids = tuple(str(item["candidateId"]) for item in ordered) judge_errors = validate_report( @@ -728,6 +849,15 @@ def run_replay( return result else: judge_reports.append(judge_report) + except RunnerTimeoutError: + result["status"] = "judge_timeout" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "timeout", + } + _write_result(output_dir, result) + return result except RunnerInvocationError: result["status"] = "judge_failed" result["evaluation"] = { @@ -782,6 +912,7 @@ def _parse_args() -> argparse.Namespace: parser.add_argument("--judge-secondary-bin") parser.add_argument("--model", default="opus") parser.add_argument("--max-budget-usd", type=float, default=1.0) + parser.add_argument("--timeout-seconds", type=float, default=DEFAULT_TIMEOUT_SECONDS) return parser.parse_args() @@ -797,6 +928,7 @@ def main() -> int: judge_secondary_bin=args.judge_secondary_bin, model=args.model, max_budget_usd=args.max_budget_usd, + timeout_seconds=args.timeout_seconds, ) return 0 if result["ok"] else 2 diff --git a/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py index 436e73a..abb13d3 100644 --- a/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py +++ b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py @@ -80,6 +80,38 @@ class FineOutlineDetectorTest(unittest.TestCase): self.assertNotIn("要单列为资料覆盖发现", skill) self.assertIn("归 judge/eval", skill) + def test_nested_findings_reject_arm_and_card_manifest_leakage(self): + base = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [ + { + "category": "entity_state", + "severity": "low", + "location": "entities[0]", + "evidenceSummary": "冻结状态核对", + } + ], + "coverageFindings": [ + { + "category": "frozen_context_gap", + "summary": "公共冻结事实缺字段", + } + ], + } + leaking_finding = { + **base, + "findings": [{**base["findings"][0], "arm": "outline_plus_cards"}], + } + leaking_coverage = { + **base, + "coverageFindings": [ + {**base["coverageFindings"][0], "cardManifest": {"count": 1}} + ], + } + self.assertFalse(validate_detector_report(leaking_finding, "blind-1")["ok"]) + self.assertFalse(validate_detector_report(leaking_coverage, "blind-1")["ok"]) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_rubric.py b/.claude/skills/replay-eval/scripts/test_rubric.py index 189f4f4..56a759a 100644 --- a/.claude/skills/replay-eval/scripts/test_rubric.py +++ b/.claude/skills/replay-eval/scripts/test_rubric.py @@ -99,6 +99,45 @@ class FineOutlineRubricTest(unittest.TestCase): ) ) + def test_nested_evaluation_and_score_reject_arm_card_manifest_leakage(self): + evaluation = { + "candidateId": "blind-1", + "scores": valid_scores(), + "summary": "短安全摘要", + } + report = { + "profile": RUBRIC_PROFILE, + "judgeId": "judge-primary", + "evaluations": [evaluation], + } + leaking_evaluation = { + **report, + "evaluations": [{**evaluation, "arm": "outline_only"}], + } + leaking_scores = valid_scores() + leaking_scores[DIMENSIONS[0]] = { + **leaking_scores[DIMENSIONS[0]], + "cardManifest": {"count": 1}, + } + leaking_score = { + **report, + "evaluations": [{**evaluation, "scores": leaking_scores}], + } + self.assertTrue( + validate_report( + leaking_evaluation, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1",), + ) + ) + self.assertTrue( + validate_report( + leaking_score, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1",), + ) + ) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_run_replay.py b/.claude/skills/replay-eval/scripts/test_run_replay.py index 18d2ecd..f10052c 100644 --- a/.claude/skills/replay-eval/scripts/test_run_replay.py +++ b/.claude/skills/replay-eval/scripts/test_run_replay.py @@ -9,9 +9,10 @@ import pathlib import sys import tempfile import unittest +from unittest.mock import patch sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) -from run_replay import _planner_prompt, run_replay # noqa: E402 +from run_replay import _parse_args, _planner_prompt, run_replay # noqa: E402 from write_report import render_report # noqa: E402 @@ -112,7 +113,7 @@ def write_fake_runner(directory, mode="stable"): run_result_path = directory / "run" / "run_result.json" runner.write_text( "#!/usr/bin/env python3\n" - "import json, pathlib, sys\n" + "import json, pathlib, sys, time\n" f"mode = {mode!r}\n" f"log_path = pathlib.Path({str(log_path)!r})\n" f"count_path = pathlib.Path({str(count_path)!r})\n" @@ -123,6 +124,8 @@ def write_fake_runner(directory, mode="stable"): "run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None\n" "with log_path.open('a', encoding='utf-8') as handle:\n" " handle.write(json.dumps({'agent': agent, 'prompt': prompt, 'runState': run_state}, ensure_ascii=False) + '\\n')\n" + "if mode == f'{agent}_timeout':\n" + " time.sleep(2)\n" "if mode == f'{agent}_exit':\n" " raise SystemExit(7)\n" "if mode == f'{agent}_invalid_json':\n" @@ -170,6 +173,22 @@ def read_calls(log_path): return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()] +def nested_keys(value): + """收集嵌套 JSON 的全部字段名,供盲化边界测试使用。""" + + if isinstance(value, dict): + keys = set(value) + for item in value.values(): + keys.update(nested_keys(item)) + return keys + if isinstance(value, list): + keys = set() + for item in value: + keys.update(nested_keys(item)) + return keys + return set() + + class ReplayRunTest(unittest.TestCase): def test_public_planner_context_does_not_include_snapshot_cards(self): prompt = _planner_prompt( @@ -200,6 +219,27 @@ class ReplayRunTest(unittest.TestCase): self.assertEqual(result["status"], "blocked_authorization") self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists()) + def test_reused_output_dir_bad_config_overwrites_previous_completed_state(self): + with tempfile.TemporaryDirectory() as directory: + output_dir = pathlib.Path(directory) / "run" + output_dir.mkdir() + result_path = output_dir / "run_result.json" + result_path.write_text( + json.dumps({"runId": "old-run", "status": "completed", "ok": True}), + encoding="utf-8", + ) + bad = config() + bad["snapshot"]["data"]["unknownSection"] = [] + + with self.assertRaisesRegex(ValueError, "未登记顶层分区"): + run_replay(bad, output_dir, mode="execute") + + persisted = json.loads(result_path.read_text(encoding="utf-8")) + self.assertEqual(persisted["runId"], "smoke-001") + self.assertEqual(persisted["status"], "config_invalid") + self.assertFalse(persisted["ok"]) + self.assertNotEqual(persisted["status"], "completed") + def test_content_leak_stops_before_manifest(self): bad = config() bad["leakageAudit"]["targetFacts"]["forbiddenFacts"][0]["text"] = "安全历史" @@ -238,6 +278,36 @@ class ReplayRunTest(unittest.TestCase): self.assertNotIn('"prompt":', report) self.assertNotIn('"payload":', report) + def test_report_rejects_text_injection_through_identifier_fields(self): + base = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + injections = { + "runId": "safe-run\n完整目标细纲:第一幕到第三幕的全部事件原文", + "referenceWork": "深空之影目标章原文与完整细纲", + } + for field, injected in injections.items(): + with self.subTest(field=field): + unsafe = dict(base) + unsafe[field] = injected + with self.assertRaisesRegex(ValueError, field): + render_report(unsafe) + + def test_report_does_not_render_preflight_free_text(self): + bad = config() + injected = "目标章事实文本" + bad["authorization"] = {**bad["authorization"], "sourceStatus": injected} + result = run_replay(bad, pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + report = render_report(result) + self.assertNotIn(injected, report) + self.assertIn("授权前置门未通过", report) + + def test_report_accepts_registered_target_source_failure_status(self): + result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + result["results"] = { + "outline_only": {"status": "target_source_forbidden", "ok": False} + } + report = render_report(result) + self.assertIn("target_source_forbidden", report) + def test_execute_uses_external_planner_and_validates_each_candidate(self): with tempfile.TemporaryDirectory() as directory: fake, log_path = write_fake_runner(directory) @@ -284,6 +354,23 @@ class ReplayRunTest(unittest.TestCase): self.assertFalse(result["ok"]) self.assertEqual(len(calls), 1) + def test_planner_timeout_fails_closed_and_persists_timeout_status(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "planner_timeout") + output_dir = pathlib.Path(directory) / "run" + result = run_replay( + config(), + output_dir, + mode="execute", + planner_bin=str(runner), + timeout_seconds=1.0, + ) + persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8")) + self.assertEqual(result["status"], "planner_timeout") + self.assertFalse(result["ok"]) + self.assertEqual(persisted["status"], "planner_timeout") + self.assertEqual(len(read_calls(log_path)), 1) + def test_detector_nonzero_exit_stops_before_judges(self): with tempfile.TemporaryDirectory() as directory: runner, log_path = write_fake_runner(directory, "detector_exit") @@ -305,6 +392,22 @@ class ReplayRunTest(unittest.TestCase): self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + def test_detector_timeout_fails_closed_before_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "detector_timeout") + result = run_replay( + config(), + pathlib.Path(directory) / "run", + mode="execute", + planner_bin=str(runner), + timeout_seconds=1.0, + ) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_timeout") + self.assertFalse(result["ok"]) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + def test_primary_judge_nonzero_exit_does_not_call_secondary(self): with tempfile.TemporaryDirectory() as directory: runner, log_path = write_fake_runner(directory, "judge_exit") @@ -323,6 +426,53 @@ class ReplayRunTest(unittest.TestCase): self.assertFalse(result["ok"]) self.assertEqual(len(judge_calls), 1) + def test_primary_judge_timeout_does_not_call_secondary(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "judge_timeout") + result = run_replay( + config(), + pathlib.Path(directory) / "run", + mode="execute", + planner_bin=str(runner), + timeout_seconds=1.0, + ) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "judge_timeout") + self.assertFalse(result["ok"]) + self.assertEqual(len(judge_calls), 1) + + def test_cli_accepts_subprocess_timeout_seconds(self): + argv = [ + "run_replay.py", + "--config", + "/tmp/replay-config.json", + "--output-dir", + "/tmp/replay-output", + "--timeout-seconds", + "12.5", + ] + with patch.object(sys, "argv", argv): + args = _parse_args() + self.assertEqual(args.timeout_seconds, 12.5) + + def test_non_finite_timeout_is_persisted_as_config_invalid(self): + with tempfile.TemporaryDirectory() as directory: + for index, timeout_seconds in enumerate((float("nan"), float("inf"))): + with self.subTest(timeout_seconds=timeout_seconds): + output_dir = pathlib.Path(directory) / f"run-{index}" + with self.assertRaisesRegex(ValueError, "timeout_seconds"): + run_replay( + config(), + output_dir, + mode="execute", + timeout_seconds=timeout_seconds, + ) + persisted = json.loads( + (output_dir / "run_result.json").read_text(encoding="utf-8") + ) + self.assertEqual(persisted["status"], "config_invalid") + self.assertFalse(persisted["ok"]) + def test_schema_failure_stops_before_all_detectors_and_judges(self): with tempfile.TemporaryDirectory() as directory: runner, log_path = write_fake_runner(directory, "invalid_candidate") @@ -345,6 +495,33 @@ class ReplayRunTest(unittest.TestCase): self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls)) self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls)) + def test_detector_requests_cannot_distinguish_arm_specific_card_identity(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "stable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + self.assertEqual(result["status"], "completed") + requests = [ + json.loads(call["prompt"]) + for call in read_calls(log_path) + if call["agent"] == "detector" + ] + self.assertEqual(len(requests), 3) + for request in requests: + self.assertTrue( + nested_keys(request).isdisjoint({"arm", "cardInjection", "cardManifest"}) + ) + serialized = json.dumps(request, ensure_ascii=False) + self.assertNotIn("正确卡", serialized) + self.assertNotIn("错配卡", serialized) + self.assertNotIn("card-correct-1", serialized) + self.assertNotIn("card-placebo-1", serialized) + public_parts = [ + {key: value for key, value in request.items() if key not in {"candidateId", "candidate"}} + for request in requests + ] + self.assertEqual(public_parts[0], public_parts[1]) + self.assertEqual(public_parts[1], public_parts[2]) + def test_invalid_detector_report_fails_closed_before_judge(self): with tempfile.TemporaryDirectory() as directory: runner, log_path = write_fake_runner(directory, "invalid_detector") diff --git a/.claude/skills/replay-eval/scripts/write_report.py b/.claude/skills/replay-eval/scripts/write_report.py index b1fc8c1..8dfe90b 100644 --- a/.claude/skills/replay-eval/scripts/write_report.py +++ b/.claude/skills/replay-eval/scripts/write_report.py @@ -5,6 +5,8 @@ from __future__ import annotations import argparse import json +import math +import re from pathlib import Path from typing import Any, Mapping @@ -27,6 +29,79 @@ REPORT_FORBIDDEN_KEYS = frozenset( "完整目标细纲", } ) +IDENTIFIER_PATTERN = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,63}\Z") +SHA256_PATTERN = re.compile(r"[0-9a-f]{64}\Z") +ARM_NAMES = frozenset( + {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"} +) +RUN_STATUSES = frozenset( + { + "unknown", + "validating_config", + "config_invalid", + "blocked_authorization", + "invalid_snapshot", + "invalid_arm_diff", + "target_source_forbidden", + "invalid_audit_input", + "blocked_leakage_audit", + "ready", + "running_planner", + "planner_timeout", + "planner_failed", + "planner_invalid", + "candidate_blocked", + "running_detector", + "detector_timeout", + "detector_failed", + "detector_invalid", + "detector_blocked", + "running_judge", + "judge_timeout", + "judge_failed", + "judge_invalid", + "judge_unstable", + "completed", + } +) +PREFLIGHT_STATUSES = frozenset( + {"unknown", "ready", "blocked_authorization", "invalid_snapshot", "invalid_arm_diff", "target_source_forbidden"} +) +ARM_STATUSES = frozenset( + { + "not_run", + "ready", + "schema_invalid", + "invalid_snapshot", + "target_source_forbidden", + "planner_timeout", + "planner_failed", + "planner_invalid", + } +) +DETECTOR_STATUSES = frozenset( + {"not_run", "passed", "blocked_high", "timeout", "failed", "invalid"} +) +CARD_STRATEGIES = frozenset({"unknown", "none", "correct", "placebo"}) +RUBRIC_PROFILES = frozenset({"unknown", "fine_outline_replay"}) +RUBRIC_DIMENSIONS = frozenset( + { + "structure_completeness", + "direction_causality", + "order_pacing", + "entity_state", + "foreshadowing_action", + "handoff_hook", + } +) +PREFLIGHT_SUMMARIES = { + "unknown": "前置门状态不可用", + "ready": "前置门通过", + "blocked_authorization": "授权前置门未通过", + "invalid_snapshot": "冻结快照前置门未通过", + "invalid_arm_diff": "三臂公共输入一致性未通过", + "target_source_forbidden": "发现目标章或未来来源", +} def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None: @@ -42,13 +117,165 @@ def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None: _assert_no_forbidden_keys(item, f"{path}[{index}]") -def _arm_line(name: str, arm: Mapping[str, Any], result: Mapping[str, Any]) -> str: - outcome = result.get(name) or {} - detector = outcome.get("detector") or {} +def _mapping(value: Any, path: str) -> Mapping[str, Any]: + if not isinstance(value, Mapping): + raise ValueError(f"{path} 必须是对象") + return value + + +def _identifier(value: Any, path: str, default: str = "unknown") -> str: + candidate = default if value is None or value == "" else value + if not isinstance(candidate, str) or IDENTIFIER_PATTERN.fullmatch(candidate) is None: + raise ValueError(f"{path} 必须是长度不超过 64 的安全标识符") + return candidate + + +def _enum(value: Any, allowed: frozenset[str], path: str, default: str) -> str: + candidate = default if value is None or value == "" else value + if not isinstance(candidate, str) or candidate not in allowed: + raise ValueError(f"{path} 不是已登记枚举值") + return candidate + + +def _integer(value: Any, path: str, *, minimum: int = 0) -> int | None: + if value is None: + return None + if isinstance(value, bool) or not isinstance(value, int) or value < minimum: + raise ValueError(f"{path} 必须是大于等于 {minimum} 的整数") + return value + + +def _number(value: Any, path: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{path} 必须是数字") + number = float(value) + if not math.isfinite(number): + raise ValueError(f"{path} 必须是有限数字") + return number + + +def _hash(value: Any, path: str) -> str | None: + if value is None or value == "": + return None + if not isinstance(value, str) or SHA256_PATTERN.fullmatch(value) is None: + raise ValueError(f"{path} 必须是 SHA-256") + return value + + +def _numeric_map(value: Any, path: str, allowed_keys: frozenset[str]) -> dict[str, float]: + mapping = _mapping(value, path) + unexpected = sorted(set(mapping) - allowed_keys) + if unexpected: + raise ValueError(f"{path} 包含未登记字段: {','.join(unexpected)}") + return {str(key): _number(item, f"{path}.{key}") for key, item in mapping.items()} + + +def _project_evaluation(value: Any) -> dict[str, Any]: + if value is None or value == "" or value == {}: + return {} + evaluation = _mapping(value, "evaluation") + projected: dict[str, Any] = { + "profile": _enum(evaluation.get("profile"), RUBRIC_PROFILES, "evaluation.profile", "unknown") + } + stability = evaluation.get("stability") + if stability is not None: + stability_mapping = _mapping(stability, "evaluation.stability") + stable = stability_mapping.get("stable") + if not isinstance(stable, bool): + raise ValueError("evaluation.stability.stable 必须是布尔枚举") + projected["stabilityStatus"] = "stable" if stable else "unstable" + projected["maxGaps"] = _numeric_map( + stability_mapping.get("maxGaps", {}), + "evaluation.stability.maxGaps", + RUBRIC_DIMENSIONS, + ) + arm_scores = _mapping(evaluation.get("armScores", {}), "evaluation.armScores") + unexpected_arms = sorted(set(arm_scores) - ARM_NAMES) + if unexpected_arms: + raise ValueError(f"evaluation.armScores 包含未登记臂: {','.join(unexpected_arms)}") + projected["armScores"] = { + str(arm): _numeric_map(scores, f"evaluation.armScores.{arm}", RUBRIC_DIMENSIONS) + for arm, scores in arm_scores.items() + } + deltas = _mapping(evaluation.get("deltas", {}), "evaluation.deltas") + unexpected_deltas = sorted(set(deltas) - {"B-A", "C-A"}) + if unexpected_deltas: + raise ValueError(f"evaluation.deltas 包含未登记比较: {','.join(unexpected_deltas)}") + projected["deltas"] = { + str(name): _numeric_map(scores, f"evaluation.deltas.{name}", RUBRIC_DIMENSIONS) + for name, scores in deltas.items() + } + return projected + + +def _project_report(result: Mapping[str, Any]) -> dict[str, Any]: + """从运行态提取独立安全 schema,未登记类型和值一律拒绝。""" + + _assert_no_forbidden_keys(result) + preflight = _mapping(result.get("preflight", {}), "preflight") + preflight_status = _enum( + preflight.get("status"), + PREFLIGHT_STATUSES, + "preflight.status", + "unknown", + ) + arms = _mapping(result.get("arms", {}), "arms") + outcomes = _mapping(result.get("results", {}), "results") + unexpected_arms = sorted((set(arms) | set(outcomes)) - ARM_NAMES) + if unexpected_arms: + raise ValueError(f"报告包含未登记评测臂: {','.join(unexpected_arms)}") + projected_arms: dict[str, Any] = {} + for name in sorted(arms): + arm = _mapping(arms[name], f"arms.{name}") + outcome = _mapping(outcomes.get(name, {}), f"results.{name}") + detector = _mapping(outcome.get("detector", {}), f"results.{name}.detector") + projected_arms[name] = { + "status": _enum(outcome.get("status"), ARM_STATUSES, f"results.{name}.status", "not_run"), + "candidateSha256": _hash( + outcome.get("candidateSha256") or outcome.get("rawOutputSha256"), + f"results.{name}.candidateSha256", + ), + "detectorStatus": _enum( + detector.get("status"), + DETECTOR_STATUSES, + f"results.{name}.detector.status", + "not_run", + ), + "cardStrategy": _enum( + arm.get("cardStrategy"), + CARD_STRATEGIES, + f"arms.{name}.cardStrategy", + "unknown", + ), + "cardInjectionCount": _integer( + arm.get("cardInjectionCount", 0), + f"arms.{name}.cardInjectionCount", + ), + } + return { + "runId": _identifier(result.get("runId"), "runId"), + "status": _enum(result.get("status"), RUN_STATUSES, "status", "unknown"), + "mode": _enum(result.get("mode"), frozenset({"unknown", "dry_run", "execute"}), "mode", "unknown"), + "referenceWork": _identifier(result.get("referenceWork"), "referenceWork"), + "referenceWorkVersion": _identifier(result.get("referenceWorkVersion"), "referenceWorkVersion"), + "asOfChapter": _integer(result.get("asOfChapter"), "asOfChapter", minimum=1), + "targetChapter": _integer(result.get("targetChapter"), "targetChapter", minimum=1), + "snapshotVersion": _identifier(result.get("snapshotVersion"), "snapshotVersion"), + "preflightStatus": preflight_status, + "preflightSummary": PREFLIGHT_SUMMARIES[preflight_status], + "arms": projected_arms, + "snapshotManifestSha256": _hash( + result.get("snapshotManifestSha256"), "snapshotManifestSha256" + ), + "evaluation": _project_evaluation(result.get("evaluation")), + } + + +def _arm_line(name: str, arm: Mapping[str, Any]) -> str: return ( - f"- `{name}`: {outcome.get('status', 'not_run')}," - f"候选哈希 `{outcome.get('candidateSha256') or outcome.get('rawOutputSha256') or 'n/a'}`," - f"detector `{detector.get('status', 'not_run')}`," + f"- `{name}`: {arm['status']}," + f"候选哈希 `{arm['candidateSha256'] or 'n/a'}`," + f"detector `{arm['detectorStatus']}`," f"卡策略 `{arm.get('cardStrategy', 'unknown')}`,卡数 `{arm.get('cardInjectionCount', 0)}`" ) @@ -58,15 +285,14 @@ def _render_evaluation(lines: list[str], evaluation: Mapping[str, Any]) -> None: if not evaluation: return - stability = evaluation.get("stability") or {} lines.extend( [ "", "## 双盲评分", "", f"- profile: `{evaluation.get('profile', 'unknown')}`", - f"- 评委稳定性: `{'stable' if stability.get('stable') else 'unstable'}`", - f"- 最大维度差: `{json.dumps(stability.get('maxGaps') or {}, ensure_ascii=False, sort_keys=True)}`", + f"- 评委稳定性: `{evaluation.get('stabilityStatus', 'not_available')}`", + f"- 最大维度差: `{json.dumps(evaluation.get('maxGaps') or {}, ensure_ascii=False, sort_keys=True)}`", ] ) arm_scores = evaluation.get("armScores") or {} @@ -85,32 +311,28 @@ def _render_evaluation(lines: list[str], evaluation: Mapping[str, Any]) -> None: def render_report(result: Mapping[str, Any]) -> str: """只从运行结果的白名单字段渲染摘要。""" - _assert_no_forbidden_keys(result) + safe = _project_report(result) lines = [ "# 细纲回放摘要", "", - f"- runId: `{result.get('runId', 'unknown')}`", - f"- status: `{result.get('status', 'unknown')}`", - f"- mode: `{result.get('mode', 'unknown')}`", - f"- referenceWork: `{result.get('referenceWork', 'unknown')}@{result.get('referenceWorkVersion', 'unknown')}`", - f"- freeze: 第 `{result.get('asOfChapter', 'unknown')}` 章,目标第 `{result.get('targetChapter', 'unknown')}` 章", - f"- snapshot: `{result.get('snapshotVersion', 'unknown')}`", + f"- runId: `{safe['runId']}`", + f"- status: `{safe['status']}`", + f"- mode: `{safe['mode']}`", + f"- referenceWork: `{safe['referenceWork']}@{safe['referenceWorkVersion']}`", + f"- freeze: 第 `{safe['asOfChapter'] if safe['asOfChapter'] is not None else 'unknown'}` 章,目标第 `{safe['targetChapter'] if safe['targetChapter'] is not None else 'unknown'}` 章", + f"- snapshot: `{safe['snapshotVersion']}`", "", "## 前置门", "", - f"- status: `{(result.get('preflight') or {}).get('status', 'unknown')}`", + f"- status: `{safe['preflightStatus']}`", + f"- 摘要: {safe['preflightSummary']}", ] - errors = (result.get("preflight") or {}).get("errors") or [] - for error in errors: - lines.append(f"- 阻断: {error}") lines.extend(["", "## 三臂", ""]) - arms = result.get("arms") or {} - outcomes = result.get("results") or {} - for name in sorted(arms): - lines.append(_arm_line(name, arms[name], outcomes)) - if result.get("snapshotManifestSha256"): - lines.extend(["", f"- snapshotManifestSha256: `{result['snapshotManifestSha256']}`"]) - _render_evaluation(lines, result.get("evaluation") or {}) + for name, arm in safe["arms"].items(): + lines.append(_arm_line(name, arm)) + if safe["snapshotManifestSha256"]: + lines.extend(["", f"- snapshotManifestSha256: `{safe['snapshotManifestSha256']}`"]) + _render_evaluation(lines, safe["evaluation"]) lines.extend( [ "", diff --git a/docs/2026-07-19-回放评测-细纲首跑设计与计划.md b/docs/2026-07-19-回放评测-细纲首跑设计与计划.md index 9954cab..b71d3c1 100644 --- a/docs/2026-07-19-回放评测-细纲首跑设计与计划.md +++ b/docs/2026-07-19-回放评测-细纲首跑设计与计划.md @@ -11,7 +11,7 @@ ## 执行状态(2026-07-19) - 已落地并同步到 `agent-example/main`:`fbb262c`(冻结/细纲/评分合同)、`a54a3a4`(审查反馈收紧与回放编排)、`1243d9b`(外部 planner execute 链路测试)。 -- 已验证:replay-eval 58 个离线测试、fine-outline 合同 3 个测试全部通过;fake runner 已跑通三臂 planner、逐臂 detector、双 judge 反序盲评、稳定性门和去盲矩阵。测试只使用合成 fixture,不代表参考作品评测结果。 +- 已验证:replay-eval 70 个离线测试、fine-outline 合同 3 个测试全部通过;fake runner 已跑通三臂 planner、真正不识别臂卡内容的逐候选 detector、双 judge 反序盲评、子进程超时失败关闭、稳定性门和去盲矩阵。测试只使用合成 fixture,不代表参考作品评测结果。 - 已新增并验证:`audit_leakage.py` 对公共快照和三臂卡注入区执行内容级事实审计;`load_reference_work.py` 通过 PostgreSQL 只读事务组装仓库外临时配置,候选卡标记为 `eval_draft`,不进入生产检索。 - 已执行真实适配 smoke:work=8、冻结 488、目标 489,组装 31 个完整历史大纲窗、6 个近章细纲摘要、正确/错配卡各 2 张;送入回放仍为 `blocked_authorization`(`example_reference_work` 没有授权快照),未生成 `snapshot_manifest`,未调用模型。 - 已执行真实数据库前置 smoke:深空之影 `work_id=8` 的 `example_reference_work` 表没有不可变授权快照/版权用途字段,结果为 `blocked_authorization`,三臂均未调用模型。 @@ -206,7 +206,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} ### 6.1 审查智能体 -审查使用现有 `detector`,但要新增“细纲候选”输入分支;它只看到冻结到 N 的 writer/planner 基线加检测增量,不看到目标章标准事实。检查: +审查使用现有 `detector`,但要新增“细纲候选”输入分支;它只看到匿名候选和冻结到 N 的公共 writer/planner 基线,不看到 arm、`cardInjection`、`cardManifest`、任何臂特有卡内容或目标章标准事实。卡注入合法性留在确定性预检。检查: - 细纲字段是否完整、事件顺序是否自洽; - 角色/势力/地点/能力是否违反 N 时点已知事实;