diff --git a/.claude/agents/detector.md b/.claude/agents/detector.md index 5430b62..26f3e13 100644 --- a/.claude/agents/detector.md +++ b/.claude/agents/detector.md @@ -17,6 +17,10 @@ model: opus 每条问题必须引原句、指依据卡与字段;无依据的观感问题归「建议」并标明主观;严重度(高/中/低)按"不修是否误导后续章节"定级。 +## 细纲回放机器合同 + +细纲回放时只接收 `fine_outline_detector_v0` JSON,不读取目标章 proxy,不得输出或推断 arm。响应必须是 JSON 对象:`protocol`、`candidateId`、`findings`、`coverageFindings`;每条问题包含 `severity`、`category`、`location`、`evidenceSummary`。不得判断“目标新角色缺卡”,该覆盖问题只属于 judge/eval 侧。任一 `high` 由编排器机械阻断整组三臂,detector 自身不改候选、不裁决卡效用。 + ## 禁区 -只读+写报告;不改正文/规划/知识卡;不执行 git 写操作。 +只读+写报告;不改正文/规划/知识卡;不执行 git 写操作。回放模式的报告只能写入仓库外临时运行目录。 diff --git a/.claude/agents/judge.md b/.claude/agents/judge.md index 19eebe2..3507a1c 100644 --- a/.claude/agents/judge.md +++ b/.claude/agents/judge.md @@ -21,6 +21,10 @@ model: opus - 末尾「最值得改的三点」按提升空间排序:问题→根因层猜测(prompt/上下文/设定卡)→具体改法。 - 细纲回放时,把末尾建议替换为“最值得补齐的三项结构缺口”,并标注它属于候选结构、公共大纲、卡注入、原文检索还是标准事实不确定;若两次同维分差大于 0.5,只写稳定性警告,不强行裁决。 +## 细纲回放机器合同 + +回放 judge 只接收 `fine_outline_judge_v0` JSON:不同 `judgeId` 的两个独立无会话进程分别评同组三个匿名候选,第二个输入顺序与第一个完全相反。输入不得包含 arm 名、卡 manifest 或另一评委结果。响应必须是 `profile=fine_outline_replay`、当前 `judgeId` 和 `evaluations` 数组;每个匿名候选必须精确出现一次,每维均给出 `score` 和结构化 `evidence`,并附非空 `summary`。编排器只把聚合数字、稳定性和去盲差值写入安全报告,原始 evidence 留在仓库外临时目录。 + ## 禁区 只产评分报告;不改候选;不执行 git 写操作。 diff --git a/.claude/skills/replay-eval/SKILL.md b/.claude/skills/replay-eval/SKILL.md index 707e634..cb939bb 100644 --- a/.claude/skills/replay-eval/SKILL.md +++ b/.claude/skills/replay-eval/SKILL.md @@ -46,7 +46,10 @@ disable-model-invocation: true - `scripts/run_replay.py --mode dry_run`:只执行授权、来源、冻结和三臂 manifest 预检,不调用模型;这是首个机制 smoke 入口。 - `scripts/load_reference_work.py`:从 PostgreSQL 只读事务组装仓库外临时配置;缺授权字段仍会生成可审计配置,但送入 `run_replay` 后必须保持 `blocked_authorization`。 -- `scripts/run_replay.py --mode execute`:在全部前置门通过后,使用无工具、无会话持久化的本地 planner CLI 逐臂生成候选;`--output-dir` 必须位于仓库外的临时目录。 +- `scripts/run_replay.py --mode execute`:在全部前置门通过后,依次执行三臂 planner、整组 schema、逐臂盲 detector、两个独立盲 judge、rubric 校验、稳定性门和去盲汇总;`--output-dir` 必须位于仓库外的临时目录。 +- detector 输入输出均为 JSON;输入只有匿名候选 ID、候选和冻结到 `as_of` 的规划上下文,不含 arm 名、目标章 proxy 或其他评委结果。任一 `high` 严重度发现整组标记 `detector_blocked`,judge 调用数必须为 0;报告不合约时标记 `detector_invalid`。 +- 两个 judge 使用不同 `judgeId` 和独立无会话进程。第二个 judge 的匿名候选顺序必须与第一个完全相反;任何 rubric 不合约标记 `judge_invalid`,任一同维差值大于 `0.5` 标记 `judge_unstable`,两者都不得标记 `completed`。 +- 只有三臂 schema、detector、双 judge rubric 和稳定性门全部通过,才去盲生成逐维 `B-A` / `C-A` 差值矩阵并标记 `completed`。`--detector-bin`、`--judge-primary-bin`、`--judge-secondary-bin` 可分别指定本地 runner;未指定时复用 `--planner-bin`,测试只能使用 fake binary。 - `scripts/write_report.py`:从 `run_result.json` 生成安全摘要;它不会读取候选正文,也不会把候选路径以外的原始响应写入报告。 真实作品运行前必须先从权威来源取得不可变授权快照。数据库没有该字段时,使用 dry-run 证明机制并保持 `blocked_authorization`,不得用本地配置或口头许可伪造放行。 diff --git a/.claude/skills/replay-eval/scripts/check_snapshot.py b/.claude/skills/replay-eval/scripts/check_snapshot.py index 023ba4f..56a6df9 100644 --- a/.claude/skills/replay-eval/scripts/check_snapshot.py +++ b/.claude/skills/replay-eval/scripts/check_snapshot.py @@ -339,6 +339,8 @@ def check_candidate_output( errors.append(f"字段必须是字符串: {field}") events = candidate.get("keyEvents") if isinstance(events, list): + if not events: + errors.append("keyEvents 不能为空") event_ids = [item.get("id") for item in events if isinstance(item, Mapping) and item.get("id")] if len(event_ids) != len(set(event_ids)): errors.append("keyEvents 包含重复事件 id") @@ -359,6 +361,15 @@ def check_candidate_output( errors.append(f"keyEvents[{index}].{field} 必须是整数") elif expected_type is not int and not isinstance(event[field], expected_type): errors.append(f"keyEvents[{index}].{field} 类型错误") + event_orders = [ + event.get("order") + for event in events + if isinstance(event, Mapping) + and isinstance(event.get("order"), int) + and not isinstance(event.get("order"), bool) + ] + if len(event_orders) == len(events) and event_orders != list(range(1, len(events) + 1)): + errors.append("keyEvents.order 必须唯一且从 1 开始连续严格递增") entities = candidate.get("entities") if isinstance(entities, list): for index, entity in enumerate(entities): diff --git a/.claude/skills/replay-eval/scripts/fine_outline_detector.py b/.claude/skills/replay-eval/scripts/fine_outline_detector.py new file mode 100644 index 0000000..7a52eec --- /dev/null +++ b/.claude/skills/replay-eval/scripts/fine_outline_detector.py @@ -0,0 +1,106 @@ +#!/usr/bin/env python3 +"""细纲 detector 的机器可读输入输出合同。""" + +from __future__ import annotations + +from typing import Any, Mapping + + +DETECTOR_PROTOCOL = "fine_outline_detector_v0" +ALLOWED_SEVERITIES = frozenset({"high", "medium", "low"}) +FORBIDDEN_COVERAGE_CATEGORIES = frozenset( + { + "missing_target_role_card", + "new_target_role_missing_card", + "target_role_card_coverage", + } +) + + +def build_detector_request( + *, + candidate_id: str, + candidate: Mapping[str, Any], + as_of_chapter: int, + frozen_snapshot: Mapping[str, Any], + common_context: Mapping[str, Any], + card_context: list[Any], + sources: list[Any], +) -> dict[str, Any]: + """构造不含 arm、目标章 proxy 或卡策略标签的盲检输入。""" + + return { + "protocol": DETECTOR_PROTOCOL, + "candidateId": candidate_id, + "asOfChapter": as_of_chapter, + "candidate": dict(candidate), + "planningContext": { + "frozenSnapshot": dict(frozen_snapshot), + "commonContext": dict(common_context), + "cardInjection": list(card_context), + "sources": list(sources), + }, + "rules": { + "referenceProxyVisible": False, + "mayJudgeTargetRoleCardCoverage": False, + "highSeverityBlocksGroup": True, + }, + } + + +def validate_detector_report( + report: Mapping[str, Any], expected_candidate_id: str +) -> dict[str, Any]: + """机械校验 detector JSON;合同不完整时失败关闭。""" + + errors: list[str] = [] + if not isinstance(report, Mapping): + return {"ok": False, "errors": ["detector 报告必须是对象"], "highSeverityCount": 0} + if report.get("protocol") != DETECTOR_PROTOCOL: + errors.append(f"protocol 必须是 {DETECTOR_PROTOCOL}") + if report.get("candidateId") != expected_candidate_id: + errors.append("candidateId 与盲检输入不一致") + unexpected = sorted( + set(report) - {"protocol", "candidateId", "findings", "coverageFindings"} + ) + if unexpected: + errors.append(f"detector 报告包含未登记字段: {','.join(unexpected)}") + + findings = report.get("findings") + coverage_findings = report.get("coverageFindings") + if not isinstance(findings, list): + errors.append("findings 必须是数组") + findings = [] + if not isinstance(coverage_findings, list): + errors.append("coverageFindings 必须是数组") + coverage_findings = [] + + for section, items in (("findings", findings), ("coverageFindings", coverage_findings)): + for index, finding in enumerate(items): + if not isinstance(finding, Mapping): + errors.append(f"{section}[{index}] 必须是对象") + continue + category = finding.get("category") + if category in FORBIDDEN_COVERAGE_CATEGORIES: + errors.append(f"{section}[{index}] 越权判断目标新角色卡覆盖") + if section == "findings": + severity = finding.get("severity") + if severity not in ALLOWED_SEVERITIES: + errors.append(f"findings[{index}].severity 无效") + for field in ("category", "location", "evidenceSummary"): + value = finding.get(field) + if not isinstance(value, str) or not value.strip(): + errors.append(f"findings[{index}].{field} 必须是非空字符串") + + high_count = sum( + 1 + for finding in findings + if isinstance(finding, Mapping) and finding.get("severity") == "high" + ) + return { + "ok": not errors, + "errors": errors, + "findingCount": len(findings), + "coverageFindingCount": len(coverage_findings), + "highSeverityCount": high_count, + } diff --git a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py index d79a9ca..dfc2bb4 100644 --- a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py +++ b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py @@ -59,19 +59,80 @@ def stability_warning( ) -> dict[str, Any]: """比较两次评审,返回差异和是否需要人工复核。""" + missing = [ + dimension + for dimension in DIMENSIONS + if dimension not in first or dimension not in second + ] gaps = { dimension: abs(float(first[dimension]) - float(second[dimension])) for dimension in DIMENSIONS if dimension in first and dimension in second } - return {"stable": all(gap <= threshold for gap in gaps.values()), "gaps": gaps} + return { + "stable": not missing and all(gap <= threshold for gap in gaps.values()), + "gaps": gaps, + "missingDimensions": missing, + } -def validate_report(report: Mapping[str, Any]) -> list[str]: - """校验一个评委报告的 profile 和评分结构。""" +def validate_report( + report: Mapping[str, Any], + *, + expected_judge_id: str | None = None, + expected_candidate_ids: tuple[str, ...] | None = None, +) -> list[str]: + """校验单候选或批量评委报告,批量模式必须覆盖精确候选集合。""" errors: list[str] = [] + if not isinstance(report, Mapping): + return ["judge 报告必须是对象"] if report.get("profile") != RUBRIC_PROFILE: errors.append(f"profile 必须是 {RUBRIC_PROFILE}") - errors.extend(validate_scores(report.get("scores", {}))) + if expected_judge_id is None and expected_candidate_ids is None: + unexpected = sorted(set(report) - {"profile", "scores"}) + if unexpected: + errors.append(f"judge 报告包含未登记字段: {','.join(unexpected)}") + errors.extend(validate_scores(report.get("scores", {}))) + return errors + + unexpected = sorted(set(report) - {"profile", "judgeId", "evaluations"}) + if unexpected: + errors.append(f"judge 批报告包含未登记字段: {','.join(unexpected)}") + + judge_id = report.get("judgeId") + if not isinstance(judge_id, str) or not judge_id.strip(): + errors.append("judgeId 必须是非空字符串") + elif expected_judge_id is not None and judge_id != expected_judge_id: + errors.append(f"judgeId 不一致: expected={expected_judge_id}") + + evaluations = report.get("evaluations") + if not isinstance(evaluations, list): + errors.append("evaluations 必须是数组") + return errors + candidate_ids = [ + item.get("candidateId") + for item in evaluations + if isinstance(item, Mapping) + ] + if len(candidate_ids) != len(evaluations) or any( + not isinstance(candidate_id, str) or not candidate_id + for candidate_id in candidate_ids + ): + errors.append("每个 evaluation 必须包含非空 candidateId") + if len(candidate_ids) != len(set(candidate_ids)): + errors.append("evaluations 包含重复 candidateId") + if expected_candidate_ids is not None and set(candidate_ids) != set(expected_candidate_ids): + errors.append("evaluations 未精确覆盖盲化候选集合") + for index, evaluation in enumerate(evaluations): + if not isinstance(evaluation, Mapping): + errors.append(f"evaluations[{index}] 必须是对象") + continue + errors.extend( + f"evaluations[{index}]: {error}" + for error in validate_scores(evaluation.get("scores", {})) + ) + summary = evaluation.get("summary") + if not isinstance(summary, str) or not summary.strip(): + errors.append(f"evaluations[{index}].summary 必须是非空摘要") return errors diff --git a/.claude/skills/replay-eval/scripts/run_replay.py b/.claude/skills/replay-eval/scripts/run_replay.py index c5952a2..e0ffdc1 100644 --- a/.claude/skills/replay-eval/scripts/run_replay.py +++ b/.claude/skills/replay-eval/scripts/run_replay.py @@ -12,7 +12,7 @@ import json import re import subprocess from pathlib import Path -from typing import Any, Mapping +from typing import Any, Mapping, Sequence from build_snapshot import build_snapshot, normalize_chapter, sha256_value from audit_leakage import audit_snapshot @@ -21,12 +21,15 @@ from check_snapshot import ( check_candidate_output, check_replay, ) +from fine_outline_detector import build_detector_request, validate_detector_report +from fine_outline_rubric import DIMENSIONS, RUBRIC_PROFILE, stability_warning, validate_report REQUIRED_ARMS = ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards") REPO_ROOT = Path(__file__).resolve().parents[4] SKILL_PATH = REPO_ROOT / ".claude/skills/fine-outline/SKILL.md" PLANNER_PATH = REPO_ROOT / ".claude/agents/planner.md" +JUDGE_IDS = ("judge-primary", "judge-secondary") class ReplayRunError(ValueError): @@ -92,7 +95,7 @@ def _extract_candidate(output: str) -> Mapping[str, Any]: outer = match.group(1) if match else outer.strip() outer = json.loads(outer) if not isinstance(outer, Mapping): - raise ReplayRunError("planner 输出不是 JSON 对象") + raise ReplayRunError("模型输出不是 JSON 对象") return outer @@ -147,6 +150,7 @@ def _planner_prompt( "assumptions", ], "eventFields": ["id", "order", "event", "participants", "trigger", "resultDirection"], + "eventOrder": "至少一个事件;order 必须从 1 开始连续严格递增", "entityFields": ["name", "type", "role"], "foreshadowingFields": ["action", "subject", "evidence"], "sourceRefs": "可选;只能引用冻结来源 ID", @@ -191,12 +195,189 @@ def _invoke_planner( return _extract_candidate(completed.stdout) +def _invoke_structured_agent( + *, + agent: str, + request: Mapping[str, Any], + runner_bin: str, + model: str, + output_path: Path, + max_budget_usd: float, + identity: str, +) -> Mapping[str, Any]: + """以独立无会话进程调用 detector/judge,并保存仓库外原始响应。""" + + command = [ + runner_bin, + "-p", + "--agent", + agent, + "--model", + model, + "--tools", + "", + "--no-session-persistence", + "--output-format", + "json", + "--max-budget-usd", + str(max_budget_usd), + "--append-system-prompt", + f"独立身份={identity};只处理给定 JSON;禁止调用工具、读取文件或输出 JSON 以外内容。", + _safe_json(request), + ] + completed = subprocess.run(command, text=True, capture_output=True, check=False) + output_path.write_text(completed.stdout, encoding="utf-8") + if completed.returncode != 0: + raise ReplayRunError(f"{agent} 调用失败,退出码={completed.returncode}") + return _extract_candidate(completed.stdout) + + +def _public_snapshot(snapshot: Mapping[str, Any]) -> dict[str, Any]: + """公共评测上下文不携带快照中的卡集合。""" + + return {str(key): value for key, value in snapshot.items() if str(key) != "cards"} + + +def _blind_assignments( + candidates: Mapping[str, Mapping[str, Any]], run_id: str +) -> list[dict[str, Any]]: + """按运行 ID 稳定打乱三臂,并只向审查模型暴露匿名候选 ID。""" + + ordered_arms = sorted( + candidates, + key=lambda arm: sha256_value({"runId": run_id, "purpose": "blind-order", "arm": arm}), + ) + return [ + { + "arm": arm, + "candidateId": f"candidate-{sha256_value({'runId': run_id, 'arm': arm})[:12]}", + "candidate": candidates[arm], + } + for arm in ordered_arms + ] + + +def _judge_request( + *, + judge_id: str, + assignments: Sequence[Mapping[str, Any]], + as_of: int, + frozen_snapshot: Mapping[str, Any], + common_context: Mapping[str, Any], + reference_proxy: Mapping[str, Any], +) -> dict[str, Any]: + """构造不含 arm 和卡 manifest 的盲评批输入。""" + + return { + "protocol": "fine_outline_judge_v0", + "profile": RUBRIC_PROFILE, + "judgeId": judge_id, + "asOfChapter": as_of, + "candidates": [ + {"candidateId": item["candidateId"], "candidate": item["candidate"]} + for item in assignments + ], + "frozenContext": { + "snapshot": _public_snapshot(frozen_snapshot), + "commonContext": dict(common_context), + }, + "referenceProxy": dict(reference_proxy), + "rules": { + "armIdentityVisible": False, + "cardManifestVisible": False, + "proseDimensionsForbidden": True, + "scoreEvidenceRequired": True, + }, + } + + +def _evaluation_by_candidate(report: Mapping[str, Any]) -> dict[str, Mapping[str, Any]]: + """把已校验的 judge 批报告按匿名候选 ID 建索引。""" + + return { + str(item["candidateId"]): item + for item in report["evaluations"] + if isinstance(item, Mapping) + } + + +def _aggregate_evaluation( + assignments: Sequence[Mapping[str, Any]], reports: Sequence[Mapping[str, Any]] +) -> dict[str, Any]: + """去盲汇总双评分;稳定性未通过时不计算卡增量矩阵。""" + + indexed = [_evaluation_by_candidate(report) for report in reports] + arm_scores: dict[str, dict[str, float]] = {} + stability_by_arm: dict[str, dict[str, Any]] = {} + for assignment in assignments: + arm = str(assignment["arm"]) + candidate_id = str(assignment["candidateId"]) + first_scores = { + dimension: float(indexed[0][candidate_id]["scores"][dimension]["score"]) + for dimension in DIMENSIONS + } + second_scores = { + dimension: float(indexed[1][candidate_id]["scores"][dimension]["score"]) + for dimension in DIMENSIONS + } + stability_by_arm[arm] = stability_warning(first_scores, second_scores) + arm_scores[arm] = { + dimension: round((first_scores[dimension] + second_scores[dimension]) / 2, 3) + for dimension in DIMENSIONS + } + + max_gaps = { + dimension: max( + stability_by_arm[arm]["gaps"].get(dimension, float("inf")) + for arm in REQUIRED_ARMS + ) + for dimension in DIMENSIONS + } + stable = all(item["stable"] for item in stability_by_arm.values()) + evaluation: dict[str, Any] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "armScores": arm_scores, + "stability": { + "stable": stable, + "threshold": 0.5, + "maxGaps": max_gaps, + "byArm": stability_by_arm, + }, + } + if stable: + baseline = arm_scores["outline_only"] + evaluation["deltas"] = { + "B-A": { + dimension: round(arm_scores["outline_plus_cards"][dimension] - baseline[dimension], 3) + for dimension in DIMENSIONS + }, + "C-A": { + dimension: round( + arm_scores["outline_plus_placebo_cards"][dimension] - baseline[dimension], + 3, + ) + for dimension in DIMENSIONS + }, + } + return evaluation + + +def _write_result(output_dir: Path, result: Mapping[str, Any]) -> None: + """每个阶段都覆盖写入可恢复的结构化运行状态。""" + + (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + + def run_replay( config: Mapping[str, Any], output_dir: Path, *, mode: str = "dry_run", planner_bin: str = "claude", + detector_bin: str | None = None, + judge_primary_bin: str | None = None, + judge_secondary_bin: str | None = None, model: str = "opus", max_budget_usd: float = 1.0, ) -> dict[str, Any]: @@ -259,7 +440,7 @@ def run_replay( "arms": arm_manifests, "results": {}, } - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) if not preflight["ok"]: return result @@ -276,7 +457,7 @@ def run_replay( "findingCount": 0, "findings": [], } - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) return result metadata = { @@ -315,7 +496,7 @@ def run_replay( # 审计失败的快照不生成 manifest,也不允许进入任何 planner 臂。 result["status"] = audit_result["status"] result["ok"] = False - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) return result (output_dir / "snapshot_manifest.json").write_text(_safe_json(frozen["manifest"]) + "\n", encoding="utf-8") @@ -324,9 +505,10 @@ def run_replay( result["status"] = STATUS_READY result["ok"] = True result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"] - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) return result + candidates: dict[str, Mapping[str, Any]] = {} for name in REQUIRED_ARMS: arm = _require_mapping(arms, name) cards = arm.get("cards", []) @@ -349,6 +531,7 @@ def run_replay( candidate_path = output_dir / f"candidate_{name}.json" candidate_path.write_text(_safe_json(candidate) + "\n", encoding="utf-8") schema = check_candidate_output(candidate, target, sources) + candidates[name] = candidate result["results"][name] = { "status": schema["status"], "ok": schema["ok"], @@ -363,9 +546,147 @@ def run_replay( "errors": [str(error)], "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, } - result["status"] = "completed" if all(item["ok"] for item in result["results"].values()) else "candidate_blocked" - result["ok"] = result["status"] == "completed" - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + if not all(item["ok"] for item in result["results"].values()): + result["status"] = "candidate_blocked" + result["ok"] = False + _write_result(output_dir, result) + return result + + run_id = str(result["runId"]) + assignments = _blind_assignments(candidates, run_id) + detector_runner = detector_bin or planner_bin + detector_invalid = False + detector_blocked = False + for assignment in assignments: + arm = str(assignment["arm"]) + candidate_id = str(assignment["candidateId"]) + arm_config = _require_mapping(arms, arm) + request = build_detector_request( + candidate_id=candidate_id, + candidate=assignment["candidate"], + as_of_chapter=as_of, + frozen_snapshot=_public_snapshot(frozen["snapshot"]), + common_context=common_input, + card_context=arm_config.get("cards", []), + sources=sources, + ) + detector_path = output_dir / f"detector_{candidate_id}.raw.json" + try: + detector_report = _invoke_structured_agent( + agent="detector", + request=request, + runner_bin=detector_runner, + model=model, + output_path=detector_path, + max_budget_usd=max_budget_usd, + identity="blind-detector", + ) + detector_check = validate_detector_report(detector_report, candidate_id) + detector_invalid = detector_invalid or not detector_check["ok"] + detector_blocked = detector_blocked or detector_check["highSeverityCount"] > 0 + result["results"][arm]["detector"] = { + "status": ( + "invalid" + if not detector_check["ok"] + else "blocked_high" + if detector_check["highSeverityCount"] + else "passed" + ), + "findingCount": detector_check["findingCount"], + "coverageFindingCount": detector_check["coverageFindingCount"], + "highSeverityCount": detector_check["highSeverityCount"], + "reportSha256": sha256_value(detector_report), + "errors": detector_check["errors"], + } + except (ReplayRunError, json.JSONDecodeError) as error: + detector_invalid = True + result["results"][arm]["detector"] = { + "status": "invalid", + "findingCount": 0, + "coverageFindingCount": 0, + "highSeverityCount": 0, + "errors": [str(error)], + "rawOutputSha256": ( + sha256_value(detector_path.read_text(encoding="utf-8")) + if detector_path.exists() + else None + ), + } + + if detector_invalid or detector_blocked: + result["status"] = "detector_invalid" if detector_invalid else "detector_blocked" + result["ok"] = False + _write_result(output_dir, result) + return result + + judge_runners = (judge_primary_bin or planner_bin, judge_secondary_bin or planner_bin) + if len(set(JUDGE_IDS)) != 2: + raise ReplayRunError("两个 judge 身份必须不同") + judge_reports: list[Mapping[str, Any]] = [] + judge_invalid = False + reference_proxy = leakage_audit_config.get("targetFacts") + if not isinstance(reference_proxy, Mapping): + result["status"] = "judge_invalid" + result["ok"] = False + result["evaluation"] = {"profile": RUBRIC_PROFILE, "errors": ["缺少结构化 reference proxy"]} + _write_result(output_dir, result) + return result + + for index, judge_id in enumerate(JUDGE_IDS): + ordered = assignments if index == 0 else list(reversed(assignments)) + request = _judge_request( + judge_id=judge_id, + assignments=ordered, + as_of=as_of, + frozen_snapshot=frozen["snapshot"], + common_context=common_input, + reference_proxy=reference_proxy, + ) + judge_path = output_dir / f"judge_{judge_id}.raw.json" + try: + judge_report = _invoke_structured_agent( + agent="judge", + request=request, + runner_bin=judge_runners[index], + model=model, + output_path=judge_path, + max_budget_usd=max_budget_usd, + identity=judge_id, + ) + candidate_ids = tuple(str(item["candidateId"]) for item in ordered) + judge_errors = validate_report( + judge_report, + expected_judge_id=judge_id, + expected_candidate_ids=candidate_ids, + ) + if judge_errors: + judge_invalid = True + else: + judge_reports.append(judge_report) + except (ReplayRunError, json.JSONDecodeError): + judge_invalid = True + + if judge_invalid or len(judge_reports) != 2: + result["status"] = "judge_invalid" + result["ok"] = False + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "invalid", + } + _write_result(output_dir, result) + return result + + result["evaluation"] = _aggregate_evaluation(assignments, judge_reports) + if not result["evaluation"]["stability"]["stable"]: + result["status"] = "judge_unstable" + result["ok"] = False + _write_result(output_dir, result) + return result + + result["status"] = "completed" + result["ok"] = True + _write_result(output_dir, result) return result @@ -375,6 +696,9 @@ def _parse_args() -> argparse.Namespace: parser.add_argument("--output-dir", type=Path, required=True) parser.add_argument("--mode", choices=("dry_run", "execute"), default="dry_run") parser.add_argument("--planner-bin", default="claude") + parser.add_argument("--detector-bin") + parser.add_argument("--judge-primary-bin") + parser.add_argument("--judge-secondary-bin") parser.add_argument("--model", default="opus") parser.add_argument("--max-budget-usd", type=float, default=1.0) return parser.parse_args() @@ -387,6 +711,9 @@ def main() -> int: args.output_dir, mode=args.mode, planner_bin=args.planner_bin, + detector_bin=args.detector_bin, + judge_primary_bin=args.judge_primary_bin, + judge_secondary_bin=args.judge_secondary_bin, model=args.model, max_budget_usd=args.max_budget_usd, ) diff --git a/.claude/skills/replay-eval/scripts/test_check_snapshot.py b/.claude/skills/replay-eval/scripts/test_check_snapshot.py index 66cbbd3..623cbf7 100644 --- a/.claude/skills/replay-eval/scripts/test_check_snapshot.py +++ b/.claude/skills/replay-eval/scripts/test_check_snapshot.py @@ -89,7 +89,16 @@ class CheckSnapshotTest(unittest.TestCase): candidate = { "targetChapter": 489, "chapterGoal": "突破", - "keyEvents": [], + "keyEvents": [ + { + "id": "event-1", + "order": 1, + "event": "侦察敌情", + "participants": ["苏铭"], + "trigger": "收到异常信号", + "resultDirection": "确认威胁存在", + } + ], "entities": [], "foreshadowing": [], "stateChanges": [], @@ -118,6 +127,67 @@ class CheckSnapshotTest(unittest.TestCase): STATUS_TARGET_SOURCE_FORBIDDEN, ) + def test_candidate_requires_non_empty_and_contiguous_key_events(self): + candidate = { + "targetChapter": 489, + "chapterGoal": "突破", + "keyEvents": [ + { + "id": "event-1", + "order": 1, + "event": "侦察敌情", + "participants": ["苏铭"], + "trigger": "收到异常信号", + "resultDirection": "确认威胁存在", + }, + { + "id": "event-2", + "order": 2, + "event": "布置伏击", + "participants": ["苏铭", "队友"], + "trigger": "确认威胁存在", + "resultDirection": "完成前置布防", + }, + ], + "entities": [], + "foreshadowing": [], + "stateChanges": [], + "hook": "悬念", + "unknowns": [], + "assumptions": [], + } + self.assertTrue(check_candidate_output(candidate, 489)["ok"]) + + empty_events = {**candidate, "keyEvents": []} + self.assertEqual(check_candidate_output(empty_events, 489)["status"], STATUS_SCHEMA_INVALID) + + duplicate_order = { + **candidate, + "keyEvents": [ + {**candidate["keyEvents"][0], "order": 1}, + {**candidate["keyEvents"][1], "order": 1}, + ], + } + self.assertEqual(check_candidate_output(duplicate_order, 489)["status"], STATUS_SCHEMA_INVALID) + + gap_order = { + **candidate, + "keyEvents": [ + {**candidate["keyEvents"][0], "order": 1}, + {**candidate["keyEvents"][1], "order": 3}, + ], + } + self.assertEqual(check_candidate_output(gap_order, 489)["status"], STATUS_SCHEMA_INVALID) + + bad_start = { + **candidate, + "keyEvents": [ + {**candidate["keyEvents"][0], "order": 2}, + {**candidate["keyEvents"][1], "order": 3}, + ], + } + self.assertEqual(check_candidate_output(bad_start, 489)["status"], STATUS_SCHEMA_INVALID) + def test_replay_fails_closed_before_model(self): common = {"snapshotVersion": "v0", "asOfChapter": 488} manifests = { diff --git a/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py new file mode 100644 index 0000000..83ddb6f --- /dev/null +++ b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py @@ -0,0 +1,37 @@ +#!/usr/bin/env python3 +"""细纲 detector 机器合同的离线测试。""" + +from __future__ import annotations + +import pathlib +import sys +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) +from fine_outline_detector import validate_detector_report # noqa: E402 + + +class FineOutlineDetectorTest(unittest.TestCase): + def test_report_must_not_reveal_arm_or_judge_target_role_coverage(self): + arm_leak = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "arm": "outline_only", + "findings": [], + "coverageFindings": [], + } + self.assertFalse(validate_detector_report(arm_leak, "blind-1")["ok"]) + + target_role_judgment = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [], + "coverageFindings": [ + {"category": "missing_target_role_card", "summary": "越权判断"} + ], + } + self.assertFalse(validate_detector_report(target_role_judgment, "blind-1")["ok"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_rubric.py b/.claude/skills/replay-eval/scripts/test_rubric.py index 2ebe557..1d793de 100644 --- a/.claude/skills/replay-eval/scripts/test_rubric.py +++ b/.claude/skills/replay-eval/scripts/test_rubric.py @@ -2,6 +2,7 @@ """细纲 rubric 的离线回归测试。""" import pathlib +import inspect import sys import unittest @@ -49,6 +50,49 @@ class FineOutlineRubricTest(unittest.TestCase): self.assertFalse(result["stable"]) self.assertEqual(result["gaps"][DIMENSIONS[2]], 1.0) + def test_missing_dimension_fails_stability_closed(self): + first = {dimension: 4 for dimension in DIMENSIONS} + second = {dimension: 4 for dimension in DIMENSIONS[:-1]} + result = stability_warning(first, second) + self.assertFalse(result["stable"]) + self.assertIn(DIMENSIONS[-1], result["missingDimensions"]) + + def test_batch_report_requires_exact_candidates_and_distinct_judge_identity(self): + self.assertIn("expected_judge_id", inspect.signature(validate_report).parameters) + report = { + "profile": RUBRIC_PROFILE, + "judgeId": "judge-primary", + "evaluations": [ + {"candidateId": "blind-1", "scores": valid_scores(), "summary": "摘要一"}, + {"candidateId": "blind-2", "scores": valid_scores(), "summary": "摘要二"}, + ], + } + self.assertEqual( + validate_report( + report, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1", "blind-2"), + ), + [], + ) + duplicate = {**report, "evaluations": [report["evaluations"][0], report["evaluations"][0]]} + self.assertTrue( + validate_report( + duplicate, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1", "blind-2"), + ) + ) + + arm_leak = {**report, "arm": "outline_only"} + self.assertTrue( + validate_report( + arm_leak, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1", "blind-2"), + ) + ) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_run_replay.py b/.claude/skills/replay-eval/scripts/test_run_replay.py index c23887f..120b669 100644 --- a/.claude/skills/replay-eval/scripts/test_run_replay.py +++ b/.claude/skills/replay-eval/scripts/test_run_replay.py @@ -73,6 +73,95 @@ def config(): } +def candidate(goal="机制 smoke", *, events=True): + """构造满足细纲闭集合同的合成候选。""" + + return { + "targetChapter": 489, + "chapterGoal": goal, + "keyEvents": ( + [ + { + "id": "event-1", + "order": 1, + "event": "侦察敌情", + "participants": ["测试角色"], + "trigger": "收到异常信号", + "resultDirection": "确认威胁存在", + } + ] + if events + else [] + ), + "entities": [], + "foreshadowing": [], + "stateChanges": [], + "hook": "下一步", + "unknowns": [], + "assumptions": [], + } + + +def write_fake_runner(directory, mode="stable"): + """生成可记录调用输入的假模型二进制,测试不触发真实模型。""" + + directory = pathlib.Path(directory) + runner = directory / "fake-agent.py" + log_path = directory / "agent-calls.jsonl" + count_path = directory / "planner-count.txt" + runner.write_text( + "#!/usr/bin/env python3\n" + "import json, pathlib, sys\n" + f"mode = {mode!r}\n" + f"log_path = pathlib.Path({str(log_path)!r})\n" + f"count_path = pathlib.Path({str(count_path)!r})\n" + "args = sys.argv[1:]\n" + "agent = args[args.index('--agent') + 1]\n" + "prompt = args[-1]\n" + "with log_path.open('a', encoding='utf-8') as handle:\n" + " handle.write(json.dumps({'agent': agent, 'prompt': prompt}, ensure_ascii=False) + '\\n')\n" + "if agent == 'planner':\n" + " count = int(count_path.read_text() if count_path.exists() else '0') + 1\n" + " count_path.write_text(str(count))\n" + " events = not (mode == 'invalid_candidate' and count == 2)\n" + f" value = {candidate()!r}\n" + " value['chapterGoal'] = f'goal-{count}'\n" + " if not events:\n" + " value['keyEvents'] = []\n" + " print(json.dumps({'result': json.dumps(value, ensure_ascii=False)}, ensure_ascii=False))\n" + "elif agent == 'detector':\n" + " request = json.loads(prompt)\n" + " findings = []\n" + " if mode == 'high_detector' and request['candidate']['chapterGoal'] == 'goal-2':\n" + " findings.append({'severity': 'high', 'category': 'entity_state', 'location': 'keyEvents[0]', 'evidenceSummary': '冻结事实冲突'})\n" + " if mode == 'invalid_detector':\n" + " findings.append({'category': 'entity_state'})\n" + " print(json.dumps({'protocol': 'fine_outline_detector_v0', 'candidateId': request['candidateId'], 'findings': findings, 'coverageFindings': []}, ensure_ascii=False))\n" + "elif agent == 'judge':\n" + " request = json.loads(prompt)\n" + " dimensions = ['structure_completeness', 'direction_causality', 'order_pacing', 'entity_state', 'foreshadowing_action', 'handoff_hook']\n" + " evaluations = []\n" + " for item in request['candidates']:\n" + " base = {'goal-1': 3, 'goal-2': 4, 'goal-3': 2}[item['candidate']['chapterGoal']]\n" + " scores = {dimension: {'score': base, 'evidence': f'{dimension}-结构化证据'} for dimension in dimensions}\n" + " if mode == 'unstable' and request['judgeId'] == 'judge-secondary':\n" + " scores['order_pacing']['score'] = min(5, base + 1)\n" + " if mode == 'invalid_rubric':\n" + " scores['order_pacing']['score'] = 6\n" + " evaluations.append({'candidateId': item['candidateId'], 'scores': scores, 'summary': '结构化评分摘要'})\n" + " print(json.dumps({'profile': 'fine_outline_replay', 'judgeId': request['judgeId'], 'evaluations': evaluations}, ensure_ascii=False))\n", + encoding="utf-8", + ) + os.chmod(runner, 0o755) + return runner, log_path + + +def read_calls(log_path): + if not log_path.exists(): + return [] + return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()] + + class ReplayRunTest(unittest.TestCase): def test_public_planner_context_does_not_include_snapshot_cards(self): prompt = _planner_prompt( @@ -142,26 +231,8 @@ class ReplayRunTest(unittest.TestCase): self.assertNotIn('"payload":', report) def test_execute_uses_external_planner_and_validates_each_candidate(self): - candidate = { - "targetChapter": 489, - "chapterGoal": "机制 smoke", - "keyEvents": [], - "entities": [], - "foreshadowing": [], - "stateChanges": [], - "hook": "下一步", - "unknowns": [], - "assumptions": [], - } with tempfile.TemporaryDirectory() as directory: - fake = pathlib.Path(directory) / "fake-planner.py" - fake.write_text( - "#!/usr/bin/env python3\n" - "import json\n" - f"print(json.dumps({{'result': json.dumps({candidate!r}, ensure_ascii=False)}}))\n", - encoding="utf-8", - ) - os.chmod(fake, 0o755) + fake, _ = write_fake_runner(directory) output_dir = pathlib.Path(directory) / "run" result = run_replay(config(), output_dir, mode="execute", planner_bin=str(fake)) self.assertTrue(result["ok"]) @@ -169,6 +240,83 @@ class ReplayRunTest(unittest.TestCase): self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"}) self.assertTrue(list(output_dir.glob("candidate_*.json"))) + def test_schema_failure_stops_before_all_detectors_and_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "invalid_candidate") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "candidate_blocked") + self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 0) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_high_detector_finding_blocks_group_and_never_calls_judge(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "high_detector") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + detector_calls = [call for call in calls if call["agent"] == "detector"] + self.assertEqual(result["status"], "detector_blocked") + self.assertEqual(len(detector_calls), 3) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls)) + self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls)) + + def test_invalid_detector_report_fails_closed_before_judge(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "invalid_detector") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_invalid") + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_two_judges_are_independent_blind_reversed_and_unblinded(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "stable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "completed") + self.assertEqual(len(judge_calls), 2) + first = json.loads(judge_calls[0]["prompt"]) + second = json.loads(judge_calls[1]["prompt"]) + self.assertNotEqual(first["judgeId"], second["judgeId"]) + self.assertEqual( + [item["candidateId"] for item in second["candidates"]], + list(reversed([item["candidateId"] for item in first["candidates"]])), + ) + self.assertNotIn("outline_only", judge_calls[0]["prompt"]) + self.assertNotIn("outline_plus_cards", judge_calls[1]["prompt"]) + self.assertTrue(result["evaluation"]["stability"]["stable"]) + self.assertEqual(result["evaluation"]["deltas"]["B-A"]["order_pacing"], 1.0) + self.assertEqual(result["evaluation"]["deltas"]["C-A"]["order_pacing"], -1.0) + + def test_invalid_rubric_report_does_not_complete(self): + with tempfile.TemporaryDirectory() as directory: + runner, _ = write_fake_runner(directory, "invalid_rubric") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + self.assertEqual(result["status"], "judge_invalid") + self.assertFalse(result["ok"]) + + def test_unstable_judges_have_explicit_non_completed_status(self): + with tempfile.TemporaryDirectory() as directory: + runner, _ = write_fake_runner(directory, "unstable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + self.assertEqual(result["status"], "judge_unstable") + self.assertFalse(result["ok"]) + self.assertFalse(result["evaluation"]["stability"]["stable"]) + self.assertNotIn("deltas", result["evaluation"]) + + def test_report_contains_only_aggregated_evaluation(self): + with tempfile.TemporaryDirectory() as directory: + runner, _ = write_fake_runner(directory, "stable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + report = render_report(result) + self.assertIn("B-A", report) + self.assertIn("C-A", report) + self.assertIn("评委稳定性", report) + self.assertNotIn("结构化证据", report) + self.assertNotIn("目标章未进入快照", report) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/write_report.py b/.claude/skills/replay-eval/scripts/write_report.py index 5fa9a7b..b1fc8c1 100644 --- a/.claude/skills/replay-eval/scripts/write_report.py +++ b/.claude/skills/replay-eval/scripts/write_report.py @@ -44,13 +44,44 @@ def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None: def _arm_line(name: str, arm: Mapping[str, Any], result: Mapping[str, Any]) -> str: outcome = result.get(name) or {} + detector = outcome.get("detector") or {} return ( f"- `{name}`: {outcome.get('status', 'not_run')}," f"候选哈希 `{outcome.get('candidateSha256') or outcome.get('rawOutputSha256') or 'n/a'}`," + f"detector `{detector.get('status', 'not_run')}`," f"卡策略 `{arm.get('cardStrategy', 'unknown')}`,卡数 `{arm.get('cardInjectionCount', 0)}`" ) +def _render_evaluation(lines: list[str], evaluation: Mapping[str, Any]) -> None: + """只渲染聚合分数、稳定性和差值,不带 judge 原始证据。""" + + if not evaluation: + return + stability = evaluation.get("stability") or {} + lines.extend( + [ + "", + "## 双盲评分", + "", + f"- profile: `{evaluation.get('profile', 'unknown')}`", + f"- 评委稳定性: `{'stable' if stability.get('stable') else 'unstable'}`", + f"- 最大维度差: `{json.dumps(stability.get('maxGaps') or {}, ensure_ascii=False, sort_keys=True)}`", + ] + ) + arm_scores = evaluation.get("armScores") or {} + for arm in sorted(arm_scores): + lines.append( + f"- `{arm}` 聚合分数: `{json.dumps(arm_scores[arm], ensure_ascii=False, sort_keys=True)}`" + ) + deltas = evaluation.get("deltas") or {} + for comparison in ("B-A", "C-A"): + if comparison in deltas: + lines.append( + f"- `{comparison}` 差值: `{json.dumps(deltas[comparison], ensure_ascii=False, sort_keys=True)}`" + ) + + def render_report(result: Mapping[str, Any]) -> str: """只从运行结果的白名单字段渲染摘要。""" @@ -79,6 +110,7 @@ def render_report(result: Mapping[str, Any]) -> str: lines.append(_arm_line(name, arms[name], outcomes)) if result.get("snapshotManifestSha256"): lines.extend(["", f"- snapshotManifestSha256: `{result['snapshotManifestSha256']}`"]) + _render_evaluation(lines, result.get("evaluation") or {}) lines.extend( [ "",