diff --git a/.claude/agents/detector.md b/.claude/agents/detector.md index 58b64ad..a7d1e5a 100644 --- a/.claude/agents/detector.md +++ b/.claude/agents/detector.md @@ -24,6 +24,10 @@ model: opus 每条问题必须引原句、指依据卡与字段;无依据的观感问题归「建议」并标明主观;严重度(高/中/低)按"不修是否误导后续章节"定级。 +## 细纲回放机器合同 + +细纲回放时只接收 `fine_outline_detector_v0` JSON,不读取目标章 proxy,不得输出或推断 arm。响应必须是 JSON 对象:`protocol`、`candidateId`、`findings`、`coverageFindings`;每条问题包含 `severity`、`category`、`location`、`evidenceSummary`。不得判断“目标新角色缺卡”,该覆盖问题只属于 judge/eval 侧。任一 `high` 由编排器机械阻断整组三臂,detector 自身不改候选、不裁决卡效用。 + ## 禁区 -只读+写报告;不改正文/规划/知识卡;不执行 git 写操作。 +只读+写报告;不改正文/规划/知识卡;不执行 git 写操作。回放模式的报告只能写入仓库外临时运行目录。 diff --git a/.claude/agents/judge.md b/.claude/agents/judge.md index 0a87e15..ded0488 100644 --- a/.claude/agents/judge.md +++ b/.claude/agents/judge.md @@ -23,6 +23,10 @@ model: opus - 末尾「最值得改的三点」按提升空间排序:问题→根因层猜测(prompt/上下文/设定卡)→具体改法。 - 细纲回放时,把末尾建议替换为“最值得补齐的三项结构缺口”,并标注它属于候选结构、公共大纲、卡注入、原文检索还是标准事实不确定;若两次同维分差大于 0.5,只写稳定性警告,不强行裁决。 +## 细纲回放机器合同 + +回放 judge 只接收 `fine_outline_judge_v0` JSON:不同 `judgeId` 的两个独立无会话进程分别评同组三个匿名候选,第二个输入顺序与第一个完全相反。输入不得包含 arm 名、卡 manifest 或另一评委结果。响应必须是 `profile=fine_outline_replay`、当前 `judgeId` 和 `evaluations` 数组;每个匿名候选必须精确出现一次,每维均给出 `score` 和结构化 `evidence`,并附非空 `summary`。编排器只把聚合数字、稳定性和去盲差值写入安全报告,原始 evidence 留在仓库外临时目录。 + ## 禁区 只产评分报告;不改候选;不执行 git 写操作。 diff --git a/.claude/skills/db/scripts/test_authorization_snapshot_ddl.py b/.claude/skills/db/scripts/test_authorization_snapshot_ddl.py new file mode 100644 index 0000000..ca8bca2 --- /dev/null +++ b/.claude/skills/db/scripts/test_authorization_snapshot_ddl.py @@ -0,0 +1,34 @@ +#!/usr/bin/env python3 +"""参考作品授权快照 DDL 的 append-only 静态门禁。""" + +import pathlib +import re +import unittest + + +DDL_PATH = pathlib.Path(__file__).resolve().parents[4] / "db" / "ddl" / "96-example参考作品授权快照.sql" + + +class AuthorizationSnapshotDdlTest(unittest.TestCase): + def test_authorization_snapshot_is_append_only_and_version_unique(self): + self.assertTrue(DDL_PATH.is_file(), "缺少授权快照 DDL") + ddl = DDL_PATH.read_text(encoding="utf-8") + self.assertIn("CREATE TABLE example_reference_authorization_snapshot", ddl) + self.assertRegex(ddl, r"UNIQUE\s*\(tenant_id,\s*work_id,\s*snapshot_version\)") + self.assertRegex(ddl, r"BEFORE\s+UPDATE\s+OR\s+DELETE") + self.assertIn("source_hash", ddl) + self.assertIn("source_version", ddl) + self.assertRegex(ddl, r"research_only.*public_domain.*licensed.*unauthorized") + self.assertRegex(ddl, r"jsonb_typeof\(allowed_purpose\)\s*=\s*'array'") + self.assertRegex(ddl, r"jsonb_typeof\(forbidden_purpose\)\s*=\s*'array'") + self.assertRegex( + ddl, + r"CHECK\s*\(example_jsonb_text_arrays_disjoint\(allowed_purpose,\s*forbidden_purpose\)\)", + ) + self.assertIn("authorization_basis <> 'user_authorization' OR copyright_status = 'research_only'", ddl) + self.assertIn("allowed_purpose = '[\"offline_evaluation\"]'::jsonb", ddl) + self.assertIsNotNone(re.search(r"RAISE\s+EXCEPTION", ddl, re.IGNORECASE)) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/detect/SKILL.md b/.claude/skills/detect/SKILL.md index 19eb813..0776375 100644 --- a/.claude/skills/detect/SKILL.md +++ b/.claude/skills/detect/SKILL.md @@ -67,7 +67,9 @@ schema 给字段加上 detection 用途,检查项自动+1,本 skill 与 detector | 来源引用 | `sourceRefs` 是否来自快照、是否包含目标章及以后 | 目标章/未来来源出现在候选引用中 | 来源 ID + 章号范围 | | 未知项纪律 | `unknowns` / `assumptions` 是否显式承载缺口 | 用无来源断言替代未知项 | 候选字段路径 | -报告仍然只产审查结果,不修改候选。回放中任一高严重度问题阻断该臂进入 judge;“卡里缺少目标新角色”要单列为资料覆盖发现,不冒充规划器错误。 +机器报告的类别是闭集:`findings.category` 只允许 `candidate_structure`、`causal_chain`、`entity_state`、`foreshadowing_action`、`source_reference`、`unknowns_discipline`;`coverageFindings.category` 只允许 `frozen_context_gap`。未登记类别不得用近义词或变体绕过,编排器必须失败关闭。 + +报告仍然只产审查结果,不修改候选。回放中任一高严重度问题阻断该臂进入 judge;目标章新角色是否缺卡需要目标章标准事实,detector 无权判断,归 judge/eval。 ## 红线 diff --git a/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py b/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py index 1389f64..70b99eb 100644 --- a/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py +++ b/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py @@ -362,18 +362,24 @@ class RetrieveWriterSourcesTest(unittest.TestCase): ], "authorization": { "sourceStatus": "active", - "copyrightStatus": "licensed", + "copyrightStatus": "research_only", + "sourceHash": "sha256:" + "a" * 64, "sourceVersion": SOURCE_VERSION, "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution"], "authorizationSnapshot": { "id": "auth-1", "version": "v1", "immutable": True, + "sourceHash": "sha256:" + "a" * 64, "sourceVersion": SOURCE_VERSION, "sourceStatus": "active", + "copyrightStatus": "research_only", + "authorizationBasis": "user_authorization", "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution"], "checkedAt": "2026-07-20T00:00:00Z", - "revalidationAt": "2026-07-21T00:00:00Z", + "revalidationAt": "2099-07-21T00:00:00Z", }, }, "leakageAudit": { diff --git a/.claude/skills/replay-eval/SKILL.md b/.claude/skills/replay-eval/SKILL.md index 5a4ae37..18d750d 100644 --- a/.claude/skills/replay-eval/SKILL.md +++ b/.claude/skills/replay-eval/SKILL.md @@ -8,7 +8,9 @@ disable-model-invocation: true 本 skill 只负责确定性的评测编排边界,不调用模型、不替代统一读取器,也不写正式规划或知识。细纲回放见 `docs/2026-07-19-回放评测-细纲首跑设计与计划.md`;正文回放的唯一任务 SoT 是 `docs/2026-07-20-正文智能体正式优化设计与计划.md`。 -真实参考作品配置由 `scripts/load_reference_work.py` 从实验库只读组装;它只取作品元数据、窗级大纲、章级细纲摘要和预注册卡 ID 对应的候选卡历史。候选卡必须标记为 `eval_draft`,不能当作生产知识检索结果。 +真实参考作品配置由 `scripts/load_reference_work.py` 在 `REPEATABLE READ READ ONLY` 事务中从实验库组装;它只取作品元数据、窗级大纲、章级细纲摘要和预注册卡 ID 对应的候选卡历史。候选卡必须标记为 `eval_draft`,不能当作生产知识检索结果。 + +来源版本只认原文件证明链:`example_reference_work.source_file` 必须唯一匹配成功 import task 和未软删 knowledge document,三方文件名一致,文档 `file_hash` 为 64 位小写 SHA-256,且 import `command_id` 以该 hash 前 16 位开头。`sourceHash=sha256:`,`sourceVersion=raw-file-v1:sha256:`;作品 revision 和导入章数不得改变原文件版本。 ## 输入合同 @@ -45,11 +47,15 @@ disable-model-invocation: true ## 编排入口 - `scripts/run_replay.py --mode dry_run`:只执行授权、来源、冻结和三臂 manifest 预检,不调用模型;这是首个机制 smoke 入口。 -- `scripts/load_reference_work.py`:从 PostgreSQL 只读事务组装仓库外临时配置;缺授权字段仍会生成可审计配置,但送入 `run_replay` 后必须保持 `blocked_authorization`。 -- `scripts/run_replay.py --mode execute`:在全部前置门通过后,使用无工具、无会话持久化的本地 planner CLI 逐臂生成候选;`--output-dir` 必须位于仓库外的临时目录。 -- `scripts/write_report.py`:从 `run_result.json` 生成安全摘要;它不会读取候选正文,也不会把候选路径以外的原始响应写入报告。 +- `scripts/load_reference_work.py`:从 PostgreSQL 只读事务组装仓库外临时配置;读取 `example_reference_authorization_snapshot` 当前原文件版本的最新快照,组装 authorization 外层与 snapshot。缺授权记录仍生成可审计配置,但送入 `run_replay` 后必须保持 `blocked_authorization`。 +- `scripts/run_replay.py --mode execute`:在全部前置门通过后,依次执行三臂 planner、整组 schema、逐臂盲 detector、两个独立盲 judge、rubric 校验、稳定性门和去盲汇总;`--output-dir` 必须位于仓库外的临时目录。 +- detector 输入输出均为 JSON;输入只有匿名候选 ID、候选和公共冻结到 `as_of` 的规划上下文,不含 arm 名、`cardInjection`、`cardManifest`、任何臂特有卡内容、目标章 proxy 或其他评委结果。卡注入合法性只由确定性预检负责。任一 `high` 严重度发现整组标记 `detector_blocked`,judge 调用数必须为 0;报告不合约时标记 `detector_invalid`。 +- 两个 judge 使用不同 `judgeId` 和独立无会话进程。第二个 judge 的匿名候选顺序必须与第一个完全相反;任何 rubric 不合约标记 `judge_invalid`,任一同维差值大于 `0.5` 标记 `judge_unstable`,两者都不得标记 `completed`。 +- 只有三臂 schema、detector、双 judge rubric 和稳定性门全部通过,才去盲生成逐维 `B-A` / `C-A` 差值矩阵并标记 `completed`。`--detector-bin`、`--judge-primary-bin`、`--judge-secondary-bin` 可分别指定本地 runner;未指定时复用 `--planner-bin`,测试只能使用 fake binary。 +- planner、detector、judge 子进程统一受 `--timeout-seconds` 限制,默认 300 秒;任一超时分别落盘 `planner_timeout`、`detector_timeout`、`judge_timeout`,不得继续进入后续阶段或标记 `completed`。 +- `scripts/write_report.py`:从 `run_result.json` 生成独立严格 schema 的安全摘要,只接受受限标识符、枚举、数字、短安全摘要和 SHA-256;不会读取候选正文,也不会把候选路径以外的原始响应写入报告。 -真实作品运行前必须先从权威来源取得不可变授权快照。数据库没有该字段时,使用 dry-run 证明机制并保持 `blocked_authorization`,不得用本地配置或口头许可伪造放行。 +真实作品运行前必须先从权威来源取得不可变授权快照。当前状态:DDL 已实现待 apply,真实记录未写,实跑未开始。用户授权只能登记为 `research_only` 且 `allowedPurpose=["offline_evaluation"]`,不得伪造为 `licensed`;缺记录继续保持 `blocked_authorization`。 ## 正文 A/B/C 回放 diff --git a/.claude/skills/replay-eval/configs/writer-gate-a-deep-space-v1.json b/.claude/skills/replay-eval/configs/writer-gate-a-deep-space-v1.json index 7f225a1..a0446c9 100644 --- a/.claude/skills/replay-eval/configs/writer-gate-a-deep-space-v1.json +++ b/.claude/skills/replay-eval/configs/writer-gate-a-deep-space-v1.json @@ -5,24 +5,38 @@ "referenceWork": { "id": 8, "title": "深空之影", - "version": "db-work-8-gate-a-preregistered-v1" + "version": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65" }, "authorization": { "sourceStatus": "authorized", - "copyrightStatus": "owned", - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "copyrightStatus": "research_only", + "sourceHash": "sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "allowedPurpose": [ "offline_evaluation" ], + "forbiddenPurpose": [ + "external_distribution", + "model_training", + "production_generation" + ], "authorizationSnapshot": { "id": "auth-work-8-gate-a-v1", "version": "auth-work-8-gate-a-v1", "immutable": true, - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "sourceHash": "sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "sourceStatus": "authorized", + "copyrightStatus": "research_only", + "authorizationBasis": "sanitized_contract_fixture", "allowedPurpose": [ "offline_evaluation" ], + "forbiddenPurpose": [ + "external_distribution", + "model_training", + "production_generation" + ], "checkedAt": "2026-07-20T00:00:00Z", "revalidationAt": "2026-08-19T00:00:00Z" } @@ -122,7 +136,7 @@ }, "writerContextInput": { "contentMode": "sanitized_contract_fixture", - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "authorizationSnapshot": { "snapshotId": "auth-work-8-gate-a-v1", "allowedPurpose": "evaluation", @@ -347,7 +361,7 @@ }, "writerContextInput": { "contentMode": "sanitized_contract_fixture", - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "authorizationSnapshot": { "snapshotId": "auth-work-8-gate-a-v1", "allowedPurpose": "evaluation", @@ -574,7 +588,7 @@ }, "writerContextInput": { "contentMode": "sanitized_contract_fixture", - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "authorizationSnapshot": { "snapshotId": "auth-work-8-gate-a-v1", "allowedPurpose": "evaluation", @@ -808,7 +822,7 @@ }, "writerContextInput": { "contentMode": "sanitized_contract_fixture", - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "authorizationSnapshot": { "snapshotId": "auth-work-8-gate-a-v1", "allowedPurpose": "evaluation", @@ -1042,7 +1056,7 @@ }, "writerContextInput": { "contentMode": "sanitized_contract_fixture", - "sourceVersion": "db-work-8-gate-a-preregistered-v1", + "sourceVersion": "sanitized-fixture-v1:sha256:25fad8197fc51db3a8c51352bd4f9d528b4a2b02239837bbd6cfbaf62ca4ee65", "authorizationSnapshot": { "snapshotId": "auth-work-8-gate-a-v1", "allowedPurpose": "evaluation", diff --git a/.claude/skills/replay-eval/scripts/check_snapshot.py b/.claude/skills/replay-eval/scripts/check_snapshot.py index 023ba4f..dc691c7 100644 --- a/.claude/skills/replay-eval/scripts/check_snapshot.py +++ b/.claude/skills/replay-eval/scripts/check_snapshot.py @@ -8,6 +8,8 @@ from __future__ import annotations import argparse import json +import re +from datetime import datetime, timezone from pathlib import Path from typing import Any, Mapping, Sequence @@ -25,10 +27,15 @@ FORBIDDEN_SOURCE_STATUSES = frozenset( {"revoked", "delisted", "recalled", "blocked", "owner_missing", "unauthorized"} ) ALLOWED_SOURCE_STATUSES = frozenset({"active", "approved", "authorized", "licensed"}) -ALLOWED_COPYRIGHT_STATUSES = frozenset({"active", "approved", "authorized", "licensed", "owned"}) +ALLOWED_COPYRIGHT_STATUSES = frozenset({"licensed", "public_domain", "research_only"}) FORBIDDEN_COPYRIGHT_STATUSES = frozenset( {"unauthorized", "unlicensed", "revoked", "expired", "blocked"} ) +SOURCE_HASH_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$") +SOURCE_VERSION_PATTERN = re.compile(r"^raw-file-v1:sha256:[0-9a-f]{64}$") +SANITIZED_FIXTURE_VERSION_PATTERN = re.compile( + r"^sanitized-fixture-v1:sha256:[0-9a-f]{64}$" +) CARD_KEYS = frozenset( { "arm", @@ -97,6 +104,20 @@ def _field(value: Mapping[str, Any], *keys: str) -> Any: return None +def _parse_utc_time(value: Any, field: str) -> tuple[datetime | None, str | None]: + """解析带时区的 ISO-8601 时间;格式含糊时按失败关闭处理。""" + + if not isinstance(value, str) or not value.strip(): + return None, f"授权快照 {field} 不是有效时间" + try: + parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00")) + except ValueError: + return None, f"授权快照 {field} 不是有效 ISO-8601 时间" + if parsed.tzinfo is None: + return None, f"授权快照 {field} 缺少时区" + return parsed.astimezone(timezone.utc), None + + def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, Any]: """授权信息缺失、用途不符或来源进入危险状态时关闭评测。""" @@ -110,8 +131,11 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An "id", "version", "immutable", + "sourceHash", "sourceVersion", "sourceStatus", + "copyrightStatus", + "authorizationBasis", "allowedPurpose", "checkedAt", ) @@ -126,6 +150,22 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An if not snapshot.get("expiresAt") and not snapshot.get("revalidationAt"): return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照缺少过期或重验时间"]) + now = datetime.now(timezone.utc) + checked_at, error = _parse_utc_time(snapshot.get("checkedAt"), "checkedAt") + if error: + return _result(STATUS_BLOCKED_AUTHORIZATION, [error]) + if checked_at is not None and checked_at > now: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 checkedAt 晚于当前时间"]) + for field in ("expiresAt", "revalidationAt"): + raw_time = snapshot.get(field) + if not raw_time: + continue + deadline, error = _parse_utc_time(raw_time, field) + if error: + return _result(STATUS_BLOCKED_AUTHORIZATION, [error]) + if deadline is not None and deadline <= now: + return _result(STATUS_BLOCKED_AUTHORIZATION, [f"授权快照 {field} 已到期"]) + source_status = str(_field(authorization, "sourceStatus", "source_status") or "").lower() if not source_status: return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceStatus"]) @@ -143,10 +183,33 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态禁止评测: {copyright_status}"]) if copyright_status not in ALLOWED_COPYRIGHT_STATUSES: return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态未登记,拒绝评测: {copyright_status}"]) + if str(snapshot["copyrightStatus"]).lower() != copyright_status: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 copyrightStatus 不一致"]) + source_hash = str(_field(authorization, "sourceHash", "source_hash") or "") + if not SOURCE_HASH_PATTERN.fullmatch(source_hash): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["sourceHash 必须是 sha256:<64位小写十六进制>"]) + if str(snapshot.get("sourceHash")) != source_hash: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 sourceHash 不一致"]) + + authorization_basis = str(snapshot.get("authorizationBasis") or "") source_version = str(_field(authorization, "sourceVersion", "source_version") or "") - if not source_version: - return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceVersion"]) + if authorization_basis == "sanitized_contract_fixture": + if not SANITIZED_FIXTURE_VERSION_PATTERN.fullmatch(source_version): + return _result( + STATUS_BLOCKED_AUTHORIZATION, + ["脱敏夹具 sourceVersion 必须是 sanitized-fixture-v1:sha256:<64位小写十六进制>"], + ) + if source_version != f"sanitized-fixture-v1:{source_hash}": + return _result(STATUS_BLOCKED_AUTHORIZATION, ["脱敏夹具 sourceVersion 与 sourceHash 不一致"]) + else: + if not SOURCE_VERSION_PATTERN.fullmatch(source_version): + return _result( + STATUS_BLOCKED_AUTHORIZATION, + ["sourceVersion 必须是 raw-file-v1:sha256:<64位小写十六进制>"], + ) + if source_version != f"raw-file-v1:{source_hash}": + return _result(STATUS_BLOCKED_AUTHORIZATION, ["sourceVersion 与 sourceHash 不一致"]) if str(snapshot.get("sourceVersion")) != source_version: return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 sourceVersion 不一致"]) if str(snapshot["sourceStatus"]).lower() != source_status: @@ -166,6 +229,28 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照用途不包含 offline_evaluation"]) if set(snapshot_allowed) != set(allowed): return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照用途与运行用途不一致"]) + forbidden = _field(authorization, "forbiddenPurpose", "forbidden_purpose") + if not isinstance(forbidden, list): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["forbiddenPurpose 不是用途列表"]) + snapshot_forbidden = snapshot.get("forbiddenPurpose", snapshot.get("forbidden_purpose")) + if not isinstance(snapshot_forbidden, list): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 forbiddenPurpose 不是用途列表"]) + if set(snapshot_forbidden) != set(forbidden): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照禁止用途与运行禁止用途不一致"]) + if "offline_evaluation" in forbidden: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["forbiddenPurpose 禁止 offline_evaluation"]) + if set(allowed) & set(forbidden): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 与 forbiddenPurpose 存在冲突"]) + if authorization_basis == "user_authorization": + if copyright_status != "research_only": + return _result(STATUS_BLOCKED_AUTHORIZATION, ["用户授权不能登记为 licensed"]) + if list(allowed) != ["offline_evaluation"]: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["用户授权仅允许 offline_evaluation"]) + if authorization_basis == "sanitized_contract_fixture": + if copyright_status != "research_only": + return _result(STATUS_BLOCKED_AUTHORIZATION, ["脱敏合同夹具只能登记为 research_only"]) + if list(allowed) != ["offline_evaluation"]: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["脱敏合同夹具仅允许 offline_evaluation"]) return _result(STATUS_READY) @@ -339,6 +424,8 @@ def check_candidate_output( errors.append(f"字段必须是字符串: {field}") events = candidate.get("keyEvents") if isinstance(events, list): + if not events: + errors.append("keyEvents 不能为空") event_ids = [item.get("id") for item in events if isinstance(item, Mapping) and item.get("id")] if len(event_ids) != len(set(event_ids)): errors.append("keyEvents 包含重复事件 id") @@ -359,6 +446,15 @@ def check_candidate_output( errors.append(f"keyEvents[{index}].{field} 必须是整数") elif expected_type is not int and not isinstance(event[field], expected_type): errors.append(f"keyEvents[{index}].{field} 类型错误") + event_orders = [ + event.get("order") + for event in events + if isinstance(event, Mapping) + and isinstance(event.get("order"), int) + and not isinstance(event.get("order"), bool) + ] + if len(event_orders) == len(events) and event_orders != list(range(1, len(events) + 1)): + errors.append("keyEvents.order 必须唯一且从 1 开始连续严格递增") entities = candidate.get("entities") if isinstance(entities, list): for index, entity in enumerate(entities): diff --git a/.claude/skills/replay-eval/scripts/fine_outline_detector.py b/.claude/skills/replay-eval/scripts/fine_outline_detector.py new file mode 100644 index 0000000..24f8ee2 --- /dev/null +++ b/.claude/skills/replay-eval/scripts/fine_outline_detector.py @@ -0,0 +1,124 @@ +#!/usr/bin/env python3 +"""细纲 detector 的机器可读输入输出合同。""" + +from __future__ import annotations + +from typing import Any, Mapping + + +DETECTOR_PROTOCOL = "fine_outline_detector_v0" +ALLOWED_SEVERITIES = frozenset({"high", "medium", "low"}) +ALLOWED_FINDING_CATEGORIES = frozenset( + { + "candidate_structure", + "causal_chain", + "entity_state", + "foreshadowing_action", + "source_reference", + "unknowns_discipline", + } +) +ALLOWED_COVERAGE_CATEGORIES = frozenset({"frozen_context_gap"}) +FINDING_FIELDS = frozenset({"category", "severity", "location", "evidenceSummary"}) +COVERAGE_FINDING_FIELDS = frozenset({"category", "summary"}) + + +def build_detector_request( + *, + candidate_id: str, + candidate: Mapping[str, Any], + as_of_chapter: int, + frozen_snapshot: Mapping[str, Any], + common_context: Mapping[str, Any], + sources: list[Any], +) -> dict[str, Any]: + """只用公共冻结事实和匿名候选构造真正不识别评测臂的盲检输入。""" + + return { + "protocol": DETECTOR_PROTOCOL, + "candidateId": candidate_id, + "asOfChapter": as_of_chapter, + "candidate": dict(candidate), + "planningContext": { + "frozenSnapshot": dict(frozen_snapshot), + "commonContext": dict(common_context), + "sources": list(sources), + }, + "rules": { + "referenceProxyVisible": False, + "mayJudgeTargetRoleCardCoverage": False, + "highSeverityBlocksGroup": True, + }, + } + + +def validate_detector_report( + report: Mapping[str, Any], expected_candidate_id: str +) -> dict[str, Any]: + """机械校验 detector JSON;合同不完整时失败关闭。""" + + errors: list[str] = [] + if not isinstance(report, Mapping): + return {"ok": False, "errors": ["detector 报告必须是对象"], "highSeverityCount": 0} + if report.get("protocol") != DETECTOR_PROTOCOL: + errors.append(f"protocol 必须是 {DETECTOR_PROTOCOL}") + if report.get("candidateId") != expected_candidate_id: + errors.append("candidateId 与盲检输入不一致") + unexpected = sorted( + set(report) - {"protocol", "candidateId", "findings", "coverageFindings"} + ) + if unexpected: + errors.append(f"detector 报告包含未登记字段: {','.join(unexpected)}") + + findings = report.get("findings") + coverage_findings = report.get("coverageFindings") + if not isinstance(findings, list): + errors.append("findings 必须是数组") + findings = [] + if not isinstance(coverage_findings, list): + errors.append("coverageFindings 必须是数组") + coverage_findings = [] + + sections = ( + ("findings", findings, ALLOWED_FINDING_CATEGORIES), + ("coverageFindings", coverage_findings, ALLOWED_COVERAGE_CATEGORIES), + ) + for section, items, allowed_categories in sections: + for index, finding in enumerate(items): + if not isinstance(finding, Mapping): + errors.append(f"{section}[{index}] 必须是对象") + continue + allowed_fields = FINDING_FIELDS if section == "findings" else COVERAGE_FINDING_FIELDS + unexpected_fields = sorted(set(finding) - allowed_fields) + if unexpected_fields: + errors.append( + f"{section}[{index}] 包含未登记字段: {','.join(unexpected_fields)}" + ) + category = finding.get("category") + if not isinstance(category, str) or category not in allowed_categories: + errors.append(f"{section}[{index}].category 未登记") + if section == "findings": + severity = finding.get("severity") + if not isinstance(severity, str) or severity not in ALLOWED_SEVERITIES: + errors.append(f"findings[{index}].severity 无效") + for field in ("category", "location", "evidenceSummary"): + value = finding.get(field) + if not isinstance(value, str) or not value.strip(): + errors.append(f"findings[{index}].{field} 必须是非空字符串") + else: + summary = finding.get("summary") + if not isinstance(summary, str) or not summary.strip(): + errors.append(f"coverageFindings[{index}].summary 必须是非空字符串") + + high_count = sum( + 1 + for finding in findings + if isinstance(finding, Mapping) and finding.get("severity") == "high" + ) + return { + "ok": not errors, + "errors": errors, + "findingCount": len(findings), + "coverageFindingCount": len(coverage_findings), + "highSeverityCount": high_count, + } diff --git a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py index d79a9ca..e007bbd 100644 --- a/.claude/skills/replay-eval/scripts/fine_outline_rubric.py +++ b/.claude/skills/replay-eval/scripts/fine_outline_rubric.py @@ -21,6 +21,8 @@ DIMENSIONS = ( PROSE_DIMENSIONS = frozenset( {"style_fit", "readability", "文风一致性", "文笔", "pacing_tension", "information_density"} ) +SCORE_FIELDS = frozenset({"score", "evidence"}) +EVALUATION_FIELDS = frozenset({"candidateId", "scores", "summary"}) def validate_scores(scores: Mapping[str, Any]) -> list[str]: @@ -43,6 +45,11 @@ def validate_scores(scores: Mapping[str, Any]) -> list[str]: if not isinstance(value, Mapping): errors.append(f"维度必须包含 score/evidence 对象: {dimension}") continue + unexpected_fields = sorted(set(value) - SCORE_FIELDS) + if unexpected_fields: + errors.append( + f"维度包含未登记字段: {dimension}:{','.join(unexpected_fields)}" + ) score = value.get("score") if isinstance(score, bool) or not isinstance(score, (int, float)) or not 1 <= score <= 5: errors.append(f"分数必须在 1-5: {dimension}") @@ -59,19 +66,87 @@ def stability_warning( ) -> dict[str, Any]: """比较两次评审,返回差异和是否需要人工复核。""" + missing = [ + dimension + for dimension in DIMENSIONS + if dimension not in first or dimension not in second + ] gaps = { dimension: abs(float(first[dimension]) - float(second[dimension])) for dimension in DIMENSIONS if dimension in first and dimension in second } - return {"stable": all(gap <= threshold for gap in gaps.values()), "gaps": gaps} + return { + "stable": not missing and all(gap <= threshold for gap in gaps.values()), + "gaps": gaps, + "missingDimensions": missing, + } -def validate_report(report: Mapping[str, Any]) -> list[str]: - """校验一个评委报告的 profile 和评分结构。""" +def validate_report( + report: Mapping[str, Any], + *, + expected_judge_id: str | None = None, + expected_candidate_ids: tuple[str, ...] | None = None, +) -> list[str]: + """校验单候选或批量评委报告,批量模式必须覆盖精确候选集合。""" errors: list[str] = [] + if not isinstance(report, Mapping): + return ["judge 报告必须是对象"] if report.get("profile") != RUBRIC_PROFILE: errors.append(f"profile 必须是 {RUBRIC_PROFILE}") - errors.extend(validate_scores(report.get("scores", {}))) + if expected_judge_id is None and expected_candidate_ids is None: + unexpected = sorted(set(report) - {"profile", "scores"}) + if unexpected: + errors.append(f"judge 报告包含未登记字段: {','.join(unexpected)}") + errors.extend(validate_scores(report.get("scores", {}))) + return errors + + unexpected = sorted(set(report) - {"profile", "judgeId", "evaluations"}) + if unexpected: + errors.append(f"judge 批报告包含未登记字段: {','.join(unexpected)}") + + judge_id = report.get("judgeId") + if not isinstance(judge_id, str) or not judge_id.strip(): + errors.append("judgeId 必须是非空字符串") + elif expected_judge_id is not None and judge_id != expected_judge_id: + errors.append(f"judgeId 不一致: expected={expected_judge_id}") + + evaluations = report.get("evaluations") + if not isinstance(evaluations, list): + errors.append("evaluations 必须是数组") + return errors + candidate_ids = [ + item.get("candidateId") + for item in evaluations + if isinstance(item, Mapping) + ] + valid_candidate_ids = len(candidate_ids) == len(evaluations) and all( + isinstance(candidate_id, str) and bool(candidate_id) + for candidate_id in candidate_ids + ) + if not valid_candidate_ids: + errors.append("每个 evaluation 必须包含非空 candidateId") + else: + if len(candidate_ids) != len(set(candidate_ids)): + errors.append("evaluations 包含重复 candidateId") + if expected_candidate_ids is not None and set(candidate_ids) != set(expected_candidate_ids): + errors.append("evaluations 未精确覆盖盲化候选集合") + for index, evaluation in enumerate(evaluations): + if not isinstance(evaluation, Mapping): + errors.append(f"evaluations[{index}] 必须是对象") + continue + unexpected_evaluation_fields = sorted(set(evaluation) - EVALUATION_FIELDS) + if unexpected_evaluation_fields: + errors.append( + f"evaluations[{index}] 包含未登记字段: {','.join(unexpected_evaluation_fields)}" + ) + errors.extend( + f"evaluations[{index}]: {error}" + for error in validate_scores(evaluation.get("scores", {})) + ) + summary = evaluation.get("summary") + if not isinstance(summary, str) or not summary.strip(): + errors.append(f"evaluations[{index}].summary 必须是非空摘要") return errors diff --git a/.claude/skills/replay-eval/scripts/load_reference_work.py b/.claude/skills/replay-eval/scripts/load_reference_work.py index b5bee44..6506a83 100644 --- a/.claude/skills/replay-eval/scripts/load_reference_work.py +++ b/.claude/skills/replay-eval/scripts/load_reference_work.py @@ -13,6 +13,7 @@ import copy import hashlib import json import re +from datetime import date, datetime from pathlib import Path from typing import Any, Mapping, Sequence @@ -29,6 +30,7 @@ DSN = ( TENANT_ID = 1 REPO_ROOT = Path(__file__).resolve().parents[4] DEFAULT_SNAPSHOT_VERSION = "next_fine_outline_replay_v0" +FILE_HASH_PATTERN = re.compile(r"^[0-9a-f]{64}$") class AdapterError(ValueError): @@ -50,13 +52,132 @@ def _required_chapter(value: Any, field: str) -> int: return chapter -def _reference_version(work: Mapping[str, Any], reference: Mapping[str, Any]) -> str: - """由数据库可见的修订和导入计数形成稳定来源版本。""" +def _unique_record(rows: Sequence[Mapping[str, Any]], label: str) -> Mapping[str, Any]: + """来源证明链要求唯一行;缺失或重复都不能猜测选取。""" - work_id = work.get("id") - revision = work.get("revision") or 0 - imported = reference.get("imported_chapter_count") or 0 - return f"db-work-{work_id}-rev-{revision}-imported-{imported}" + if not isinstance(rows, Sequence) or isinstance(rows, (str, bytes)) or len(rows) != 1: + count = len(rows) if isinstance(rows, Sequence) and not isinstance(rows, (str, bytes)) else 0 + raise AdapterError(f"{label} 必须唯一匹配,实际 {count} 行") + row = rows[0] + if not isinstance(row, Mapping): + raise AdapterError(f"{label} 不是对象") + if row.get("deleted") is True: + raise AdapterError(f"{label} 已软删") + return row + + +def validate_source_records( + reference_rows: Sequence[Mapping[str, Any]], + import_task_rows: Sequence[Mapping[str, Any]], + document_rows: Sequence[Mapping[str, Any]], +) -> dict[str, Any]: + """交叉核验原文件登记、成功导入任务和知识文档,生成稳定文件版本。""" + + reference = _unique_record(reference_rows, "reference.source_file") + import_task = _unique_record(import_task_rows, "成功 import task") + document = _unique_record(document_rows, "未删 knowledge document") + + source_file = str(reference.get("source_file") or "").strip() + if not source_file: + raise AdapterError("reference.source_file 为空") + if str(import_task.get("status") or "") != "succeeded": + raise AdapterError("import task 不是 succeeded") + source_snapshot = import_task.get("source_snapshot") + if not isinstance(source_snapshot, Mapping): + raise AdapterError("import task 缺少 source_snapshot") + if str(source_snapshot.get("file") or "") != source_file: + raise AdapterError("import task filename 与 reference.source_file 不一致") + if str(document.get("file_name") or "") != source_file: + raise AdapterError("knowledge document filename 与 reference.source_file 不一致") + + file_hash = str(document.get("file_hash") or "") + if not FILE_HASH_PATTERN.fullmatch(file_hash): + raise AdapterError("knowledge document file_hash 必须是 64 位小写十六进制") + expected_command_prefix = f"import-{file_hash[:16]}" + command_id = str(import_task.get("command_id") or "") + if not command_id.startswith(expected_command_prefix): + raise AdapterError("import task command_id 前缀与原文件 hash 不一致") + + source_hash = f"sha256:{file_hash}" + return { + "referenceWorkId": str(reference.get("id") or ""), + "importTaskId": str(import_task.get("id") or ""), + "documentId": str(document.get("id") or ""), + "fileName": source_file, + "sourceHash": source_hash, + "sourceVersion": f"raw-file-v1:{source_hash}", + "importCommandId": command_id, + } + + +def _iso_time(value: Any) -> str | None: + """把数据库时间统一为带时区的 ISO 字符串,空值保持为空。""" + + if value is None: + return None + if isinstance(value, (datetime, date)): + return value.isoformat() + text = str(value).strip() + return text or None + + +def project_authorization( + row: Mapping[str, Any] | None, + source: Mapping[str, Any], +) -> dict[str, Any]: + """把最新授权行投影为外层授权与不可变快照;无记录时保持阻断。""" + + source_hash = str(source.get("sourceHash") or "") + source_version = str(source.get("sourceVersion") or "") + if row is None: + return { + "sourceStatus": "missing_authorization_snapshot", + "copyrightStatus": "unknown", + "sourceHash": source_hash, + "sourceVersion": source_version, + "allowedPurpose": [], + "forbiddenPurpose": [], + "authorizationSnapshot": {}, + } + if not isinstance(row, Mapping): + raise AdapterError("授权快照行不是对象") + if str(row.get("source_hash") or "") != source_hash: + raise AdapterError("授权快照 source_hash 与原文件不一致") + if str(row.get("source_version") or "") != source_version: + raise AdapterError("授权快照 source_version 与原文件不一致") + + allowed = row.get("allowed_purpose") + forbidden = row.get("forbidden_purpose") + if not isinstance(allowed, list) or not isinstance(forbidden, list): + raise AdapterError("授权快照用途字段必须是数组") + source_status = str(row.get("source_status") or "") + copyright_status = str(row.get("copyright_status") or "") + snapshot = { + "id": str(row.get("id") or ""), + "version": str(row.get("snapshot_version") or ""), + "immutable": True, + "sourceHash": source_hash, + "sourceVersion": source_version, + "sourceStatus": source_status, + "copyrightStatus": copyright_status, + "allowedPurpose": copy.deepcopy(allowed), + "forbiddenPurpose": copy.deepcopy(forbidden), + "authorizationBasis": str(row.get("authorization_basis") or ""), + "authorizedBy": str(row.get("authorized_by") or ""), + "displaySummary": str(row.get("display_summary") or ""), + "checkedAt": _iso_time(row.get("checked_at")), + "expiresAt": _iso_time(row.get("expires_at")), + "revalidationAt": _iso_time(row.get("revalidation_at")), + } + return { + "sourceStatus": source_status, + "copyrightStatus": copyright_status, + "sourceHash": source_hash, + "sourceVersion": source_version, + "allowedPurpose": copy.deepcopy(allowed), + "forbiddenPurpose": copy.deepcopy(forbidden), + "authorizationSnapshot": snapshot, + } def _row_id(row: Mapping[str, Any]) -> str: @@ -380,6 +501,8 @@ def build_replay_config( target: int, evaluation_set_version: str, strategy_version: str, + source: Mapping[str, Any], + authorization: Mapping[str, Any], run_id: str | None = None, snapshot_version: str = DEFAULT_SNAPSHOT_VERSION, history_chapter_limit: int = 6, @@ -396,7 +519,10 @@ def build_replay_config( if target_chapter != normalized_target: raise AdapterError("target scaffold 不是目标章,拒绝混用") - source_version = _reference_version(work, reference) + source_hash = str(source.get("sourceHash") or "") + source_version = str(source.get("sourceVersion") or "") + if not source_hash.startswith("sha256:") or source_version != f"raw-file-v1:{source_hash}": + raise AdapterError("来源 hash/version 不符合原文件版本合同") kept_windows, _ = filter_outline_windows(outline_rows, normalized_as_of) projected_windows = [_project_outline(row) for row in kept_windows] @@ -439,6 +565,7 @@ def build_replay_config( { "sourceId": f"reference-work:{work.get('id')}", "sourceVersion": source_version, + "sourceHash": source_hash, "scope": "metadata", "sourceStatus": str(reference.get("parse_status") or "unknown"), } @@ -490,6 +617,7 @@ def build_replay_config( "referenceWorkId": str(work.get("id")), "outlineSourceCount": len(projected_windows), "sourceVersion": source_version, + "sourceHash": source_hash, }, "L3": { "sourceMode": "eval_draft", @@ -536,13 +664,7 @@ def build_replay_config( "cardStrategy": "placebo", }, }, - "authorization": { - "sourceStatus": "missing_authorization_snapshot", - "copyrightStatus": "unknown", - "sourceVersion": source_version, - "allowedPurpose": [], - "authorizationSnapshot": {}, - }, + "authorization": copy.deepcopy(dict(authorization)), "runPermissions": { "purpose": "offline_evaluation", "mode": "dry_run", @@ -581,18 +703,53 @@ def load_reference_rows( """, (tenant_id, work_id), ).fetchone() - reference = conn.execute( + reference_rows = conn.execute( """ SELECT id,work_id,declared_chapter_count,imported_chapter_count, - parse_scope,parse_status,source_file,notes,update_time + parse_scope,parse_status,source_file,notes,update_time,deleted FROM example_reference_work WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE - ORDER BY id DESC LIMIT 1 + ORDER BY id """, (tenant_id, work_id), - ).fetchone() - if work is None or reference is None: + ).fetchall() + if work is None: raise AdapterError("作品或参考作品登记不存在") + reference = _unique_record(reference_rows, "reference.source_file") + + import_task_rows = conn.execute( + """ + SELECT id,status,command_id,source_snapshot,deleted + FROM muse_content_import_task + WHERE tenant_id=%s AND work_id=%s AND status='succeeded' AND deleted=FALSE + AND source_snapshot->>'file' = %s + ORDER BY id + """, + (tenant_id, work_id, reference.get("source_file")), + ).fetchall() + document_rows = conn.execute( + """ + SELECT id,file_name,file_hash,deleted + FROM muse_knowledge_document + WHERE tenant_id=%s AND file_name=%s AND deleted=FALSE + ORDER BY id + """, + (tenant_id, reference.get("source_file")), + ).fetchall() + source = validate_source_records(reference_rows, import_task_rows, document_rows) + authorization_row = conn.execute( + """ + SELECT id,snapshot_version,source_hash,source_version,copyright_status,source_status, + allowed_purpose,forbidden_purpose,authorization_basis,authorized_by, + display_summary,checked_at,expires_at,revalidation_at + FROM example_reference_authorization_snapshot + WHERE tenant_id=%s AND work_id=%s AND source_version=%s + ORDER BY checked_at DESC,id DESC + LIMIT 1 + """, + (tenant_id, work_id, source["sourceVersion"]), + ).fetchone() + authorization = project_authorization(authorization_row, source) if normalized_target > int(work.get("chapter_count") or 0) + 1: raise AdapterError("目标章超出作品导入范围") @@ -645,6 +802,8 @@ def load_reference_rows( return { "work": work, "reference": reference, + "source": source, + "authorization": authorization, "outline_rows": outline_rows, "scaffold_rows": scaffold_rows, "target_scaffold": target_scaffold, @@ -711,7 +870,7 @@ def main() -> int: "correctCardCount": len(config["arms"]["outline_plus_cards"]["cards"]), "placeboCardCount": len(config["arms"]["outline_plus_placebo_cards"]["cards"]), "cardSourceMode": "eval_draft", - "authorizationStatus": "missing_authorization_snapshot", + "authorizationStatus": config["authorization"]["sourceStatus"], "configPath": str(output_dir / "config.json"), } (output_dir / "adapter_summary.json").write_text(_safe_json(summary) + "\n", encoding="utf-8") diff --git a/.claude/skills/replay-eval/scripts/run_replay.py b/.claude/skills/replay-eval/scripts/run_replay.py index c5952a2..3d3a255 100644 --- a/.claude/skills/replay-eval/scripts/run_replay.py +++ b/.claude/skills/replay-eval/scripts/run_replay.py @@ -9,30 +9,49 @@ from __future__ import annotations import argparse import json +import math +import os import re import subprocess +import tempfile from pathlib import Path -from typing import Any, Mapping +from typing import Any, Mapping, Sequence -from build_snapshot import build_snapshot, normalize_chapter, sha256_value +from build_snapshot import SnapshotError, build_snapshot, normalize_chapter, sha256_value from audit_leakage import audit_snapshot from check_snapshot import ( STATUS_READY, check_candidate_output, check_replay, ) +from fine_outline_detector import build_detector_request, validate_detector_report +from fine_outline_rubric import DIMENSIONS, RUBRIC_PROFILE, stability_warning, validate_report REQUIRED_ARMS = ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards") REPO_ROOT = Path(__file__).resolve().parents[4] SKILL_PATH = REPO_ROOT / ".claude/skills/fine-outline/SKILL.md" PLANNER_PATH = REPO_ROOT / ".claude/agents/planner.md" +JUDGE_IDS = ("judge-primary", "judge-secondary") +DEFAULT_TIMEOUT_SECONDS = 300.0 class ReplayRunError(ValueError): """回放配置不符合运行边界。""" +class RunnerInvocationError(ReplayRunError): + """外部 runner 无法启动或以非零状态退出。""" + + +class RunnerOutputError(ReplayRunError): + """外部 runner 返回的内容不符合机器合同。""" + + +class RunnerTimeoutError(ReplayRunError): + """外部 runner 超过允许的最长执行时间。""" + + def _read_json(path: Path) -> Any: return json.loads(path.read_text(encoding="utf-8")) @@ -41,6 +60,15 @@ def _safe_json(value: Any) -> str: return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")) +def _timeout_stdout(error: subprocess.TimeoutExpired) -> str: + """规范化超时前捕获的标准输出,供临时目录留痕和哈希审计。""" + + output = error.stdout or "" + if isinstance(output, bytes): + return output.decode("utf-8", errors="replace") + return output + + def _require_mapping(config: Mapping[str, Any], key: str) -> Mapping[str, Any]: value = config.get(key) if not isinstance(value, Mapping): @@ -92,7 +120,7 @@ def _extract_candidate(output: str) -> Mapping[str, Any]: outer = match.group(1) if match else outer.strip() outer = json.loads(outer) if not isinstance(outer, Mapping): - raise ReplayRunError("planner 输出不是 JSON 对象") + raise RunnerOutputError("模型输出不是 JSON 对象") return outer @@ -147,6 +175,7 @@ def _planner_prompt( "assumptions", ], "eventFields": ["id", "order", "event", "participants", "trigger", "resultDirection"], + "eventOrder": "至少一个事件;order 必须从 1 开始连续严格递增", "entityFields": ["name", "type", "role"], "foreshadowingFields": ["action", "subject", "evidence"], "sourceRefs": "可选;只能引用冻结来源 ID", @@ -163,6 +192,7 @@ def _invoke_planner( model: str, output_path: Path, max_budget_usd: float, + timeout_seconds: float, ) -> Mapping[str, Any]: """用无工具、无会话持久化的 Claude print 模式运行 planner。""" @@ -184,82 +214,334 @@ def _invoke_planner( "本次是严格离线回放;不要调用任何工具,不要读取文件,不要输出 JSON 以外内容。", prompt, ] - completed = subprocess.run(command, text=True, capture_output=True, check=False) + try: + completed = subprocess.run( + command, + text=True, + capture_output=True, + check=False, + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as error: + output_path.write_text(_timeout_stdout(error), encoding="utf-8") + raise RunnerTimeoutError(f"planner 调用超时,限制={timeout_seconds:g}秒") from error + except OSError as error: + raise RunnerInvocationError(f"planner 启动失败: {error}") from error output_path.write_text(completed.stdout, encoding="utf-8") if completed.returncode != 0: - raise ReplayRunError(f"planner 调用失败,退出码={completed.returncode}") + raise RunnerInvocationError(f"planner 调用失败,退出码={completed.returncode}") return _extract_candidate(completed.stdout) +def _invoke_structured_agent( + *, + agent: str, + request: Mapping[str, Any], + runner_bin: str, + model: str, + output_path: Path, + max_budget_usd: float, + identity: str, + timeout_seconds: float, +) -> Mapping[str, Any]: + """以独立无会话进程调用 detector/judge,并保存仓库外原始响应。""" + + command = [ + runner_bin, + "-p", + "--agent", + agent, + "--model", + model, + "--tools", + "", + "--no-session-persistence", + "--output-format", + "json", + "--max-budget-usd", + str(max_budget_usd), + "--append-system-prompt", + f"独立身份={identity};只处理给定 JSON;禁止调用工具、读取文件或输出 JSON 以外内容。", + _safe_json(request), + ] + try: + completed = subprocess.run( + command, + text=True, + capture_output=True, + check=False, + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as error: + output_path.write_text(_timeout_stdout(error), encoding="utf-8") + raise RunnerTimeoutError(f"{agent} 调用超时,限制={timeout_seconds:g}秒") from error + except OSError as error: + raise RunnerInvocationError(f"{agent} 启动失败: {error}") from error + output_path.write_text(completed.stdout, encoding="utf-8") + if completed.returncode != 0: + raise RunnerInvocationError(f"{agent} 调用失败,退出码={completed.returncode}") + return _extract_candidate(completed.stdout) + + +def _public_snapshot(snapshot: Mapping[str, Any]) -> dict[str, Any]: + """公共评测上下文不携带快照中的卡集合。""" + + return {str(key): value for key, value in snapshot.items() if str(key) != "cards"} + + +def _blind_assignments( + candidates: Mapping[str, Mapping[str, Any]], run_id: str +) -> list[dict[str, Any]]: + """按运行 ID 稳定打乱三臂,并只向审查模型暴露匿名候选 ID。""" + + ordered_arms = sorted( + candidates, + key=lambda arm: sha256_value({"runId": run_id, "purpose": "blind-order", "arm": arm}), + ) + return [ + { + "arm": arm, + "candidateId": f"candidate-{sha256_value({'runId': run_id, 'arm': arm})[:12]}", + "candidate": candidates[arm], + } + for arm in ordered_arms + ] + + +def _judge_request( + *, + judge_id: str, + assignments: Sequence[Mapping[str, Any]], + as_of: int, + frozen_snapshot: Mapping[str, Any], + common_context: Mapping[str, Any], + reference_proxy: Mapping[str, Any], +) -> dict[str, Any]: + """构造不含 arm 和卡 manifest 的盲评批输入。""" + + return { + "protocol": "fine_outline_judge_v0", + "profile": RUBRIC_PROFILE, + "judgeId": judge_id, + "asOfChapter": as_of, + "candidates": [ + {"candidateId": item["candidateId"], "candidate": item["candidate"]} + for item in assignments + ], + "frozenContext": { + "snapshot": _public_snapshot(frozen_snapshot), + "commonContext": dict(common_context), + }, + "referenceProxy": dict(reference_proxy), + "rules": { + "armIdentityVisible": False, + "cardManifestVisible": False, + "proseDimensionsForbidden": True, + "scoreEvidenceRequired": True, + }, + } + + +def _evaluation_by_candidate(report: Mapping[str, Any]) -> dict[str, Mapping[str, Any]]: + """把已校验的 judge 批报告按匿名候选 ID 建索引。""" + + return { + str(item["candidateId"]): item + for item in report["evaluations"] + if isinstance(item, Mapping) + } + + +def _aggregate_evaluation( + assignments: Sequence[Mapping[str, Any]], reports: Sequence[Mapping[str, Any]] +) -> dict[str, Any]: + """去盲汇总双评分;稳定性未通过时不计算卡增量矩阵。""" + + indexed = [_evaluation_by_candidate(report) for report in reports] + arm_scores: dict[str, dict[str, float]] = {} + stability_by_arm: dict[str, dict[str, Any]] = {} + for assignment in assignments: + arm = str(assignment["arm"]) + candidate_id = str(assignment["candidateId"]) + first_scores = { + dimension: float(indexed[0][candidate_id]["scores"][dimension]["score"]) + for dimension in DIMENSIONS + } + second_scores = { + dimension: float(indexed[1][candidate_id]["scores"][dimension]["score"]) + for dimension in DIMENSIONS + } + stability_by_arm[arm] = stability_warning(first_scores, second_scores) + arm_scores[arm] = { + dimension: round((first_scores[dimension] + second_scores[dimension]) / 2, 3) + for dimension in DIMENSIONS + } + + max_gaps = { + dimension: max( + stability_by_arm[arm]["gaps"].get(dimension, float("inf")) + for arm in REQUIRED_ARMS + ) + for dimension in DIMENSIONS + } + stable = all(item["stable"] for item in stability_by_arm.values()) + evaluation: dict[str, Any] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "armScores": arm_scores, + "stability": { + "stable": stable, + "threshold": 0.5, + "maxGaps": max_gaps, + "byArm": stability_by_arm, + }, + } + if stable: + baseline = arm_scores["outline_only"] + evaluation["deltas"] = { + "B-A": { + dimension: round(arm_scores["outline_plus_cards"][dimension] - baseline[dimension], 3) + for dimension in DIMENSIONS + }, + "C-A": { + dimension: round( + arm_scores["outline_plus_placebo_cards"][dimension] - baseline[dimension], + 3, + ) + for dimension in DIMENSIONS + }, + } + return evaluation + + +def _write_result(output_dir: Path, result: Mapping[str, Any]) -> None: + """每个阶段都覆盖写入可恢复的结构化运行状态。""" + + result_path = output_dir / "run_result.json" + temporary_path: Path | None = None + try: + # 唯一临时文件避免同目录并发写相互覆盖;同目录 replace 保证正式状态原子切换。 + with tempfile.NamedTemporaryFile( + mode="w", + encoding="utf-8", + dir=output_dir, + prefix=".run_result.", + suffix=".tmp", + delete=False, + ) as handle: + temporary_path = Path(handle.name) + handle.write(_safe_json(result) + "\n") + handle.flush() + os.fsync(handle.fileno()) + temporary_path.replace(result_path) + finally: + if temporary_path is not None and temporary_path.exists(): + temporary_path.unlink() + + def run_replay( config: Mapping[str, Any], output_dir: Path, *, mode: str = "dry_run", planner_bin: str = "claude", + detector_bin: str | None = None, + judge_primary_bin: str | None = None, + judge_secondary_bin: str | None = None, model: str = "opus", max_budget_usd: float = 1.0, + timeout_seconds: float = DEFAULT_TIMEOUT_SECONDS, ) -> dict[str, Any]: """执行一次单目标三臂回放;任何前置门失败都不调用模型。""" - if mode not in {"dry_run", "execute"}: - raise ReplayRunError("mode 只能是 dry_run 或 execute") output_dir = output_dir.resolve() if output_dir.is_relative_to(REPO_ROOT.resolve()): raise ReplayRunError("原始候选运行目录不得位于仓库内") output_dir.mkdir(parents=True, exist_ok=True) - - snapshot_config = _require_mapping(config, "snapshot") - as_of = normalize_chapter(snapshot_config.get("asOfChapter")) - target = normalize_chapter(config.get("targetChapter")) - snapshot_version = str(snapshot_config.get("snapshotVersion") or "") - if as_of is None or target is None: - raise ReplayRunError("as_of/target 必须是明确正整数") - if target != as_of + 1: - raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1") - reference_work = _require_mapping(config, "referenceWork") - authorization = _require_mapping(config, "authorization") - sources = config.get("sources", []) - if not isinstance(sources, list): - raise ReplayRunError("sources 必须是数组") - common_input = _require_mapping(config, "commonContext") - arms = _require_mapping(config, "arms") - if set(arms) != set(REQUIRED_ARMS): - raise ReplayRunError("生产回放必须精确配置三臂") - - arm_manifests = { - name: _build_arm_manifest( - name=name, - arm=_require_mapping(arms, name), - common_input=common_input, - as_of=as_of, - target=target, - snapshot_version=snapshot_version, - ) - for name in REQUIRED_ARMS - } - preflight = check_replay( - authorization=authorization, - as_of_chapter=as_of, - target_chapter=target, - planner_sources=sources, - arm_manifests=arm_manifests, - ) result: dict[str, Any] = { "runId": str(config.get("runId") or "unassigned"), "mode": mode, - "status": preflight["status"], - "ok": preflight["ok"], - "referenceWork": str(reference_work.get("id") or ""), - "referenceWorkVersion": str(reference_work.get("version") or ""), - "asOfChapter": as_of, - "targetChapter": target, - "snapshotVersion": snapshot_version, - "preflight": {"status": preflight["status"], "errors": preflight["errors"], "warnings": preflight["warnings"]}, - "arms": arm_manifests, + "status": "validating_config", + "ok": False, "results": {}, } - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) + + try: + if mode not in {"dry_run", "execute"}: + raise ReplayRunError("mode 只能是 dry_run 或 execute") + if ( + isinstance(timeout_seconds, bool) + or not isinstance(timeout_seconds, (int, float)) + or not math.isfinite(float(timeout_seconds)) + or timeout_seconds <= 0 + ): + raise ReplayRunError("timeout_seconds 必须是正数") + snapshot_config = _require_mapping(config, "snapshot") + as_of = normalize_chapter(snapshot_config.get("asOfChapter")) + target = normalize_chapter(config.get("targetChapter")) + snapshot_version = str(snapshot_config.get("snapshotVersion") or "") + if as_of is None or target is None: + raise ReplayRunError("as_of/target 必须是明确正整数") + if target != as_of + 1: + raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1") + reference_work = _require_mapping(config, "referenceWork") + authorization = _require_mapping(config, "authorization") + sources = config.get("sources", []) + if not isinstance(sources, list): + raise ReplayRunError("sources 必须是数组") + common_input = _require_mapping(config, "commonContext") + arms = _require_mapping(config, "arms") + if set(arms) != set(REQUIRED_ARMS): + raise ReplayRunError("生产回放必须精确配置三臂") + + arm_manifests = { + name: _build_arm_manifest( + name=name, + arm=_require_mapping(arms, name), + common_input=common_input, + as_of=as_of, + target=target, + snapshot_version=snapshot_version, + ) + for name in REQUIRED_ARMS + } + preflight = check_replay( + authorization=authorization, + as_of_chapter=as_of, + target_chapter=target, + planner_sources=sources, + arm_manifests=arm_manifests, + ) + except ReplayRunError as error: + result["status"] = "config_invalid" + result["errors"] = [str(error)] + _write_result(output_dir, result) + raise + + result.update( + { + "referenceWork": str(reference_work.get("id") or ""), + "referenceWorkVersion": str(reference_work.get("version") or ""), + "asOfChapter": as_of, + "targetChapter": target, + "snapshotVersion": snapshot_version, + "preflight": { + "status": preflight["status"], + "errors": preflight["errors"], + "warnings": preflight["warnings"], + }, + "arms": arm_manifests, + } + ) + if preflight["ok"]: + # 前置门通过不等于快照配置有效;深层冻结完成前始终保持非成功态。 + result["status"] = "validating_config" + result["ok"] = False + else: + result["status"] = preflight["status"] + result["ok"] = False + _write_result(output_dir, result) if not preflight["ok"]: return result @@ -276,7 +558,7 @@ def run_replay( "findingCount": 0, "findings": [], } - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) return result metadata = { @@ -289,13 +571,20 @@ def run_replay( "armConfig": {"arms": list(REQUIRED_ARMS)}, } snapshot_data = snapshot_config.get("data", {}) - frozen = build_snapshot( - snapshot_data, - as_of, - snapshot_version, - target_chapter=target, - manifest_metadata=metadata, - ) + try: + frozen = build_snapshot( + snapshot_data, + as_of, + snapshot_version, + target_chapter=target, + manifest_metadata=metadata, + ) + except SnapshotError as error: + result["status"] = "config_invalid" + result["ok"] = False + result["errors"] = [str(error)] + _write_result(output_dir, result) + raise # 内容审计同时覆盖公共冻结快照和各臂卡注入区;卡不在公共区,不能因此逃过未来事实检查。 audit_payload = { @@ -315,7 +604,7 @@ def run_replay( # 审计失败的快照不生成 manifest,也不允许进入任何 planner 臂。 result["status"] = audit_result["status"] result["ok"] = False - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) return result (output_dir / "snapshot_manifest.json").write_text(_safe_json(frozen["manifest"]) + "\n", encoding="utf-8") @@ -324,9 +613,15 @@ def run_replay( result["status"] = STATUS_READY result["ok"] = True result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"] - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + _write_result(output_dir, result) return result + result["status"] = "running_planner" + result["ok"] = False + result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"] + _write_result(output_dir, result) + + candidates: dict[str, Mapping[str, Any]] = {} for name in REQUIRED_ARMS: arm = _require_mapping(arms, name) cards = arm.get("cards", []) @@ -345,10 +640,12 @@ def run_replay( model=model, output_path=raw_path, max_budget_usd=max_budget_usd, + timeout_seconds=float(timeout_seconds), ) candidate_path = output_dir / f"candidate_{name}.json" candidate_path.write_text(_safe_json(candidate) + "\n", encoding="utf-8") schema = check_candidate_output(candidate, target, sources) + candidates[name] = candidate result["results"][name] = { "status": schema["status"], "ok": schema["ok"], @@ -356,16 +653,251 @@ def run_replay( "candidateSha256": sha256_value(candidate), "candidatePath": str(candidate_path), } - except (ReplayRunError, json.JSONDecodeError) as error: + except RunnerTimeoutError as error: result["results"][name] = { - "status": "planner_output_invalid", + "status": "planner_timeout", "ok": False, "errors": [str(error)], "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, } - result["status"] = "completed" if all(item["ok"] for item in result["results"].values()) else "candidate_blocked" - result["ok"] = result["status"] == "completed" - (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + result["status"] = "planner_timeout" + _write_result(output_dir, result) + return result + except RunnerInvocationError as error: + result["results"][name] = { + "status": "planner_failed", + "ok": False, + "errors": [str(error)], + "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, + } + result["status"] = "planner_failed" + _write_result(output_dir, result) + return result + except (RunnerOutputError, json.JSONDecodeError) as error: + result["results"][name] = { + "status": "planner_invalid", + "ok": False, + "errors": [str(error)], + "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, + } + result["status"] = "planner_invalid" + _write_result(output_dir, result) + return result + if not all(item["ok"] for item in result["results"].values()): + result["status"] = "candidate_blocked" + result["ok"] = False + _write_result(output_dir, result) + return result + + run_id = str(result["runId"]) + assignments = _blind_assignments(candidates, run_id) + detector_runner = detector_bin or planner_bin + result["status"] = "running_detector" + _write_result(output_dir, result) + detector_invalid = False + detector_blocked = False + for assignment in assignments: + arm = str(assignment["arm"]) + candidate_id = str(assignment["candidateId"]) + request = build_detector_request( + candidate_id=candidate_id, + candidate=assignment["candidate"], + as_of_chapter=as_of, + frozen_snapshot=_public_snapshot(frozen["snapshot"]), + common_context=common_input, + sources=sources, + ) + detector_path = output_dir / f"detector_{candidate_id}.raw.json" + try: + detector_report = _invoke_structured_agent( + agent="detector", + request=request, + runner_bin=detector_runner, + model=model, + output_path=detector_path, + max_budget_usd=max_budget_usd, + identity="blind-detector", + timeout_seconds=float(timeout_seconds), + ) + detector_check = validate_detector_report(detector_report, candidate_id) + detector_invalid = detector_invalid or not detector_check["ok"] + detector_blocked = detector_blocked or detector_check["highSeverityCount"] > 0 + result["results"][arm]["detector"] = { + "status": ( + "invalid" + if not detector_check["ok"] + else "blocked_high" + if detector_check["highSeverityCount"] + else "passed" + ), + "findingCount": detector_check["findingCount"], + "coverageFindingCount": detector_check["coverageFindingCount"], + "highSeverityCount": detector_check["highSeverityCount"], + "reportSha256": sha256_value(detector_report), + "errors": detector_check["errors"], + } + if not detector_check["ok"]: + result["status"] = "detector_invalid" + _write_result(output_dir, result) + return result + except RunnerTimeoutError as error: + result["results"][arm]["detector"] = { + "status": "timeout", + "findingCount": 0, + "coverageFindingCount": 0, + "highSeverityCount": 0, + "errors": [str(error)], + "rawOutputSha256": ( + sha256_value(detector_path.read_text(encoding="utf-8")) + if detector_path.exists() + else None + ), + } + result["status"] = "detector_timeout" + _write_result(output_dir, result) + return result + except RunnerInvocationError as error: + result["results"][arm]["detector"] = { + "status": "failed", + "findingCount": 0, + "coverageFindingCount": 0, + "highSeverityCount": 0, + "errors": [str(error)], + "rawOutputSha256": ( + sha256_value(detector_path.read_text(encoding="utf-8")) + if detector_path.exists() + else None + ), + } + result["status"] = "detector_failed" + _write_result(output_dir, result) + return result + except (RunnerOutputError, json.JSONDecodeError) as error: + result["results"][arm]["detector"] = { + "status": "invalid", + "findingCount": 0, + "coverageFindingCount": 0, + "highSeverityCount": 0, + "errors": [str(error)], + "rawOutputSha256": ( + sha256_value(detector_path.read_text(encoding="utf-8")) + if detector_path.exists() + else None + ), + } + result["status"] = "detector_invalid" + _write_result(output_dir, result) + return result + + if detector_invalid or detector_blocked: + result["status"] = "detector_invalid" if detector_invalid else "detector_blocked" + result["ok"] = False + _write_result(output_dir, result) + return result + + judge_runners = (judge_primary_bin or planner_bin, judge_secondary_bin or planner_bin) + if len(set(JUDGE_IDS)) != 2: + raise ReplayRunError("两个 judge 身份必须不同") + judge_reports: list[Mapping[str, Any]] = [] + judge_invalid = False + reference_proxy = leakage_audit_config.get("targetFacts") + if not isinstance(reference_proxy, Mapping): + result["status"] = "judge_invalid" + result["ok"] = False + result["evaluation"] = {"profile": RUBRIC_PROFILE, "errors": ["缺少结构化 reference proxy"]} + _write_result(output_dir, result) + return result + + result["status"] = "running_judge" + _write_result(output_dir, result) + for index, judge_id in enumerate(JUDGE_IDS): + ordered = assignments if index == 0 else list(reversed(assignments)) + request = _judge_request( + judge_id=judge_id, + assignments=ordered, + as_of=as_of, + frozen_snapshot=frozen["snapshot"], + common_context=common_input, + reference_proxy=reference_proxy, + ) + judge_path = output_dir / f"judge_{judge_id}.raw.json" + try: + judge_report = _invoke_structured_agent( + agent="judge", + request=request, + runner_bin=judge_runners[index], + model=model, + output_path=judge_path, + max_budget_usd=max_budget_usd, + identity=judge_id, + timeout_seconds=float(timeout_seconds), + ) + candidate_ids = tuple(str(item["candidateId"]) for item in ordered) + judge_errors = validate_report( + judge_report, + expected_judge_id=judge_id, + expected_candidate_ids=candidate_ids, + ) + if judge_errors: + result["status"] = "judge_invalid" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "invalid", + } + _write_result(output_dir, result) + return result + else: + judge_reports.append(judge_report) + except RunnerTimeoutError: + result["status"] = "judge_timeout" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "timeout", + } + _write_result(output_dir, result) + return result + except RunnerInvocationError: + result["status"] = "judge_failed" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "failed", + } + _write_result(output_dir, result) + return result + except (RunnerOutputError, json.JSONDecodeError): + result["status"] = "judge_invalid" + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "invalid", + } + _write_result(output_dir, result) + return result + + if judge_invalid or len(judge_reports) != 2: + result["status"] = "judge_invalid" + result["ok"] = False + result["evaluation"] = { + "profile": RUBRIC_PROFILE, + "judgeIds": list(JUDGE_IDS), + "status": "invalid", + } + _write_result(output_dir, result) + return result + + result["evaluation"] = _aggregate_evaluation(assignments, judge_reports) + if not result["evaluation"]["stability"]["stable"]: + result["status"] = "judge_unstable" + result["ok"] = False + _write_result(output_dir, result) + return result + + result["status"] = "completed" + result["ok"] = True + _write_result(output_dir, result) return result @@ -375,8 +907,12 @@ def _parse_args() -> argparse.Namespace: parser.add_argument("--output-dir", type=Path, required=True) parser.add_argument("--mode", choices=("dry_run", "execute"), default="dry_run") parser.add_argument("--planner-bin", default="claude") + parser.add_argument("--detector-bin") + parser.add_argument("--judge-primary-bin") + parser.add_argument("--judge-secondary-bin") parser.add_argument("--model", default="opus") parser.add_argument("--max-budget-usd", type=float, default=1.0) + parser.add_argument("--timeout-seconds", type=float, default=DEFAULT_TIMEOUT_SECONDS) return parser.parse_args() @@ -387,8 +923,12 @@ def main() -> int: args.output_dir, mode=args.mode, planner_bin=args.planner_bin, + detector_bin=args.detector_bin, + judge_primary_bin=args.judge_primary_bin, + judge_secondary_bin=args.judge_secondary_bin, model=args.model, max_budget_usd=args.max_budget_usd, + timeout_seconds=args.timeout_seconds, ) return 0 if result["ok"] else 2 diff --git a/.claude/skills/replay-eval/scripts/test_check_snapshot.py b/.claude/skills/replay-eval/scripts/test_check_snapshot.py index 66cbbd3..39a0b8a 100644 --- a/.claude/skills/replay-eval/scripts/test_check_snapshot.py +++ b/.claude/skills/replay-eval/scripts/test_check_snapshot.py @@ -22,18 +22,24 @@ from check_snapshot import ( # noqa: E402 AUTH = { "sourceStatus": "active", - "copyrightStatus": "licensed", - "sourceVersion": "work-v1", + "copyrightStatus": "research_only", + "sourceHash": "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4", + "sourceVersion": "raw-file-v1:sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4", "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution"], "authorizationSnapshot": { "id": "auth-1", "version": "v1", "immutable": True, - "sourceVersion": "work-v1", + "sourceHash": "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4", + "sourceVersion": "raw-file-v1:sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4", "sourceStatus": "active", + "copyrightStatus": "research_only", + "authorizationBasis": "user_authorization", "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution"], "checkedAt": "2026-07-19T00:00:00Z", - "revalidationAt": "2026-07-20T00:00:00Z", + "revalidationAt": "2099-07-20T00:00:00Z", }, } @@ -51,8 +57,125 @@ class CheckSnapshotTest(unittest.TestCase): self.assertEqual(check_authorization(denied)["status"], STATUS_BLOCKED_AUTHORIZATION) unknown_status = {**AUTH, "sourceStatus": "temporary"} self.assertEqual(check_authorization(unknown_status)["status"], STATUS_BLOCKED_AUTHORIZATION) - unlicensed = {**AUTH, "copyrightStatus": "unlicensed"} - self.assertEqual(check_authorization(unlicensed)["status"], STATUS_BLOCKED_AUTHORIZATION) + unauthorized = {**AUTH, "copyrightStatus": "unauthorized"} + self.assertEqual(check_authorization(unauthorized)["status"], STATUS_BLOCKED_AUTHORIZATION) + + def test_research_only_and_public_domain_allow_offline_evaluation(self): + self.assertTrue(check_authorization(AUTH)["ok"]) + public_domain = { + **AUTH, + "copyrightStatus": "public_domain", + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "copyrightStatus": "public_domain", + "authorizationBasis": "public_domain_record", + }, + } + self.assertTrue(check_authorization(public_domain)["ok"]) + + def test_sanitized_contract_fixture_uses_distinct_hashed_version(self): + source_hash = "sha256:" + "b" * 64 + fixture = { + **AUTH, + "sourceHash": source_hash, + "sourceVersion": f"sanitized-fixture-v1:{source_hash}", + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "sourceHash": source_hash, + "sourceVersion": f"sanitized-fixture-v1:{source_hash}", + "authorizationBasis": "sanitized_contract_fixture", + }, + } + self.assertTrue(check_authorization(fixture)["ok"]) + + forged_raw = {**fixture, "sourceVersion": f"raw-file-v1:{source_hash}"} + self.assertEqual( + check_authorization(forged_raw)["status"], STATUS_BLOCKED_AUTHORIZATION + ) + + def test_user_authorization_cannot_be_forged_as_licensed(self): + forged = { + **AUTH, + "copyrightStatus": "licensed", + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "copyrightStatus": "licensed", + }, + } + self.assertEqual(check_authorization(forged)["status"], STATUS_BLOCKED_AUTHORIZATION) + + def test_source_hash_and_version_must_match_snapshot(self): + bad_hash = {**AUTH, "sourceHash": "sha256:" + "0" * 64} + self.assertEqual(check_authorization(bad_hash)["status"], STATUS_BLOCKED_AUTHORIZATION) + bad_version = {**AUTH, "sourceVersion": "raw-file-v1:sha256:" + "0" * 64} + self.assertEqual(check_authorization(bad_version)["status"], STATUS_BLOCKED_AUTHORIZATION) + + def test_forbidden_purpose_must_be_arrays_and_match_snapshot(self): + outer_not_array = {**AUTH, "forbiddenPurpose": "external_distribution"} + self.assertEqual(check_authorization(outer_not_array)["status"], STATUS_BLOCKED_AUTHORIZATION) + + snapshot_not_array = { + **AUTH, + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "forbiddenPurpose": "external_distribution", + }, + } + self.assertEqual(check_authorization(snapshot_not_array)["status"], STATUS_BLOCKED_AUTHORIZATION) + + mismatch = { + **AUTH, + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "forbiddenPurpose": ["training"], + }, + } + self.assertEqual(check_authorization(mismatch)["status"], STATUS_BLOCKED_AUTHORIZATION) + + def test_forbidden_purpose_blocks_overlap_and_offline_evaluation(self): + overlapping = { + **AUTH, + "forbiddenPurpose": ["offline_evaluation"], + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "forbiddenPurpose": ["offline_evaluation"], + }, + } + self.assertEqual(check_authorization(overlapping)["status"], STATUS_BLOCKED_AUTHORIZATION) + + public_domain_overlap = { + **AUTH, + "copyrightStatus": "public_domain", + "allowedPurpose": ["offline_evaluation", "research"], + "forbiddenPurpose": ["research"], + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "copyrightStatus": "public_domain", + "authorizationBasis": "public_domain_record", + "allowedPurpose": ["offline_evaluation", "research"], + "forbiddenPurpose": ["research"], + }, + } + self.assertEqual(check_authorization(public_domain_overlap)["status"], STATUS_BLOCKED_AUTHORIZATION) + + def test_expired_or_due_revalidation_is_blocked(self): + expired = { + **AUTH, + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "expiresAt": "2020-01-01T00:00:00Z", + "revalidationAt": None, + }, + } + self.assertEqual(check_authorization(expired)["status"], STATUS_BLOCKED_AUTHORIZATION) + due = { + **AUTH, + "authorizationSnapshot": { + **AUTH["authorizationSnapshot"], + "revalidationAt": "2020-01-01T00:00:00Z", + }, + } + self.assertEqual(check_authorization(due)["status"], STATUS_BLOCKED_AUTHORIZATION) def test_target_source_is_blocked(self): allowed = check_target_sources( @@ -89,7 +212,16 @@ class CheckSnapshotTest(unittest.TestCase): candidate = { "targetChapter": 489, "chapterGoal": "突破", - "keyEvents": [], + "keyEvents": [ + { + "id": "event-1", + "order": 1, + "event": "侦察敌情", + "participants": ["苏铭"], + "trigger": "收到异常信号", + "resultDirection": "确认威胁存在", + } + ], "entities": [], "foreshadowing": [], "stateChanges": [], @@ -118,6 +250,67 @@ class CheckSnapshotTest(unittest.TestCase): STATUS_TARGET_SOURCE_FORBIDDEN, ) + def test_candidate_requires_non_empty_and_contiguous_key_events(self): + candidate = { + "targetChapter": 489, + "chapterGoal": "突破", + "keyEvents": [ + { + "id": "event-1", + "order": 1, + "event": "侦察敌情", + "participants": ["苏铭"], + "trigger": "收到异常信号", + "resultDirection": "确认威胁存在", + }, + { + "id": "event-2", + "order": 2, + "event": "布置伏击", + "participants": ["苏铭", "队友"], + "trigger": "确认威胁存在", + "resultDirection": "完成前置布防", + }, + ], + "entities": [], + "foreshadowing": [], + "stateChanges": [], + "hook": "悬念", + "unknowns": [], + "assumptions": [], + } + self.assertTrue(check_candidate_output(candidate, 489)["ok"]) + + empty_events = {**candidate, "keyEvents": []} + self.assertEqual(check_candidate_output(empty_events, 489)["status"], STATUS_SCHEMA_INVALID) + + duplicate_order = { + **candidate, + "keyEvents": [ + {**candidate["keyEvents"][0], "order": 1}, + {**candidate["keyEvents"][1], "order": 1}, + ], + } + self.assertEqual(check_candidate_output(duplicate_order, 489)["status"], STATUS_SCHEMA_INVALID) + + gap_order = { + **candidate, + "keyEvents": [ + {**candidate["keyEvents"][0], "order": 1}, + {**candidate["keyEvents"][1], "order": 3}, + ], + } + self.assertEqual(check_candidate_output(gap_order, 489)["status"], STATUS_SCHEMA_INVALID) + + bad_start = { + **candidate, + "keyEvents": [ + {**candidate["keyEvents"][0], "order": 2}, + {**candidate["keyEvents"][1], "order": 3}, + ], + } + self.assertEqual(check_candidate_output(bad_start, 489)["status"], STATUS_SCHEMA_INVALID) + def test_replay_fails_closed_before_model(self): common = {"snapshotVersion": "v0", "asOfChapter": 488} manifests = { diff --git a/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py new file mode 100644 index 0000000..abb13d3 --- /dev/null +++ b/.claude/skills/replay-eval/scripts/test_fine_outline_detector.py @@ -0,0 +1,117 @@ +#!/usr/bin/env python3 +"""细纲 detector 机器合同的离线测试。""" + +from __future__ import annotations + +import pathlib +import sys +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) +from fine_outline_detector import validate_detector_report # noqa: E402 + + +class FineOutlineDetectorTest(unittest.TestCase): + def test_report_must_not_reveal_arm_or_judge_target_role_coverage(self): + arm_leak = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "arm": "outline_only", + "findings": [], + "coverageFindings": [], + } + self.assertFalse(validate_detector_report(arm_leak, "blind-1")["ok"]) + + target_role_judgment = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [], + "coverageFindings": [ + {"category": "missing_target_role_card", "summary": "越权判断"} + ], + } + self.assertFalse(validate_detector_report(target_role_judgment, "blind-1")["ok"]) + + def test_semantic_alias_and_unknown_categories_fail_closed(self): + for section, category in ( + ("findings", "target_new_character_card_missing"), + ("coverageFindings", "new_role_without_card"), + ("findings", "invented_detector_category"), + ("coverageFindings", ["frozen_context_gap"]), + ): + report = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [], + "coverageFindings": [], + } + finding = { + "category": category, + "severity": "low", + "location": "candidate", + "evidenceSummary": "测试", + } + report[section] = [finding] + self.assertFalse(validate_detector_report(report, "blind-1")["ok"]) + + def test_registered_categories_are_accepted_and_detect_skill_assigns_target_coverage_to_eval(self): + report = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [ + { + "category": "entity_state", + "severity": "medium", + "location": "entities[0]", + "evidenceSummary": "与冻结状态不一致", + } + ], + "coverageFindings": [ + { + "category": "frozen_context_gap", + "summary": "冻结资料缺少已知实体字段", + } + ], + } + self.assertTrue(validate_detector_report(report, "blind-1")["ok"]) + + skill_path = pathlib.Path(__file__).resolve().parents[2] / "detect" / "SKILL.md" + skill = skill_path.read_text(encoding="utf-8") + self.assertNotIn("要单列为资料覆盖发现", skill) + self.assertIn("归 judge/eval", skill) + + def test_nested_findings_reject_arm_and_card_manifest_leakage(self): + base = { + "protocol": "fine_outline_detector_v0", + "candidateId": "blind-1", + "findings": [ + { + "category": "entity_state", + "severity": "low", + "location": "entities[0]", + "evidenceSummary": "冻结状态核对", + } + ], + "coverageFindings": [ + { + "category": "frozen_context_gap", + "summary": "公共冻结事实缺字段", + } + ], + } + leaking_finding = { + **base, + "findings": [{**base["findings"][0], "arm": "outline_plus_cards"}], + } + leaking_coverage = { + **base, + "coverageFindings": [ + {**base["coverageFindings"][0], "cardManifest": {"count": 1}} + ], + } + self.assertFalse(validate_detector_report(leaking_finding, "blind-1")["ok"]) + self.assertFalse(validate_detector_report(leaking_coverage, "blind-1")["ok"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_load_reference_work.py b/.claude/skills/replay-eval/scripts/test_load_reference_work.py index c536e8b..784d58f 100644 --- a/.claude/skills/replay-eval/scripts/test_load_reference_work.py +++ b/.claude/skills/replay-eval/scripts/test_load_reference_work.py @@ -4,11 +4,14 @@ from __future__ import annotations import json +import inspect import pathlib import sys import unittest +from unittest.mock import patch sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) +import load_reference_work as loader # noqa: E402 from load_reference_work import ( # noqa: E402 AdapterError, build_replay_config, @@ -16,6 +19,9 @@ from load_reference_work import ( # noqa: E402 ) +FILE_HASH = "02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4" +SOURCE_HASH = f"sha256:{FILE_HASH}" +SOURCE_VERSION = f"raw-file-v1:{SOURCE_HASH}" WORK = { "id": 8, "title": "深空之影", @@ -28,6 +34,37 @@ REFERENCE = { "imported_chapter_count": 594, "parse_scope": {"from": 1, "to": 594}, "parse_status": "parsing", + "source_file": "深空之影_远瞳.txt", + "deleted": False, +} +IMPORT_TASK = { + "id": 19, + "status": "succeeded", + "command_id": f"import-{FILE_HASH[:16]}", + "source_snapshot": {"file": REFERENCE["source_file"]}, + "deleted": False, +} +DOCUMENT = { + "id": 7, + "file_name": REFERENCE["source_file"], + "file_hash": FILE_HASH, + "deleted": False, +} +AUTHORIZATION_ROW = { + "id": 1, + "snapshot_version": "auth-work-8-v1", + "source_hash": SOURCE_HASH, + "source_version": SOURCE_VERSION, + "copyright_status": "research_only", + "source_status": "active", + "allowed_purpose": ["offline_evaluation"], + "forbidden_purpose": ["external_distribution"], + "authorization_basis": "user_authorization", + "authorized_by": "user:1", + "display_summary": "用户授权仅用于内部离线评测", + "checked_at": "2026-07-19T00:00:00Z", + "expires_at": None, + "revalidation_at": "2099-07-19T00:00:00Z", } @@ -58,7 +95,71 @@ def card_row(card_id=11126, name="苏铭"): } +class FakeQueryResult: + """提供 psycopg 查询结果所需的最小 fetch 接口。""" + + def __init__(self, rows): + self.rows = rows if isinstance(rows, list) else [rows] + + def fetchone(self): + return self.rows[0] if self.rows else None + + def fetchall(self): + return self.rows + + +class FakeReadOnlyConnection: + """模拟只读查询,并按 SQL 中的原文件条件过滤导入任务。""" + + def __init__(self, import_tasks): + self.import_tasks = import_tasks + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_value, traceback): + return False + + def execute(self, query, params=None): + sql = " ".join(query.split()) + if sql.startswith("SET TRANSACTION"): + return FakeQueryResult([]) + if "FROM muse_content_work" in sql: + return FakeQueryResult(WORK) + if "FROM example_reference_work" in sql: + return FakeQueryResult([REFERENCE]) + if "FROM muse_content_import_task" in sql: + rows = self.import_tasks + if "source_snapshot->>'file' = %s" in sql: + source_file = params[-1] + rows = [ + row + for row in rows + if row.get("source_snapshot", {}).get("file") == source_file + ] + return FakeQueryResult(rows) + if "FROM muse_knowledge_document" in sql: + return FakeQueryResult([DOCUMENT]) + if "FROM example_reference_authorization_snapshot" in sql: + return FakeQueryResult(AUTHORIZATION_ROW) + if "FROM example_parse_outline" in sql: + return FakeQueryResult([]) + if "FROM example_parse_scaffold" in sql and "ch.order_no<=%s" in sql: + return FakeQueryResult([]) + if "FROM example_parse_scaffold" in sql and "ch.order_no=%s" in sql: + return FakeQueryResult( + {"id": 11, "chapter": 489, "title": "目标", "outline_text": "目标事实"} + ) + if "FROM muse_knowledge_draft" in sql: + return FakeQueryResult([card_row(), card_row(11127, "赵宁")]) + raise AssertionError(f"未处理 SQL: {sql}") + + class LoadReferenceWorkTest(unittest.TestCase): + def setUp(self): + self.source = loader.validate_source_records([REFERENCE], [IMPORT_TASK], [DOCUMENT]) + self.authorization = loader.project_authorization(AUTHORIZATION_ROW, self.source) + def test_window_freeze_uses_absolute_bounds_and_excludes_target_window(self): config = build_replay_config( work=WORK, @@ -77,6 +178,8 @@ class LoadReferenceWorkTest(unittest.TestCase): target=430, evaluation_set_version="set-test", strategy_version="strategy-test", + source=self.source, + authorization=self.authorization, ) windows = config["snapshot"]["data"]["outlineWindows"] self.assertEqual([item["from_order"] for item in windows], [401]) @@ -92,7 +195,7 @@ class LoadReferenceWorkTest(unittest.TestCase): self.assertNotIn("目标章事实", public) def test_card_is_eval_draft_and_history_is_frozen(self): - result = project_card(card_row(), as_of=488, source_version="db-work-8-v12") + result = project_card(card_row(), as_of=488, source_version=SOURCE_VERSION) encoded = json.dumps(result, ensure_ascii=False) self.assertEqual(result["evaluationStatus"], "eval_draft") self.assertFalse(result["productionRetrievalEligible"]) @@ -100,13 +203,13 @@ class LoadReferenceWorkTest(unittest.TestCase): self.assertNotIn("终局摘要", encoded) self.assertNotIn("终局能力", encoded) self.assertEqual(result["source"]["sourceId"], "eval-draft:11126") - self.assertEqual(result["source"]["sourceVersion"], "db-work-8-v12") + self.assertEqual(result["source"]["sourceVersion"], SOURCE_VERSION) def test_missing_history_fails_closed_instead_of_using_static_card_fields(self): row = card_row() row["draft_payload"]["字段"].pop("演变历程") with self.assertRaises(AdapterError): - project_card(row, as_of=488, source_version="db-work-8-v12") + project_card(row, as_of=488, source_version=SOURCE_VERSION) def test_sources_have_source_id_and_version(self): config = build_replay_config( @@ -121,10 +224,135 @@ class LoadReferenceWorkTest(unittest.TestCase): target=489, evaluation_set_version="set-test", strategy_version="strategy-test", + source=self.source, + authorization=self.authorization, ) self.assertTrue(config["sources"]) self.assertTrue(all(item["sourceId"] and item["sourceVersion"] for item in config["sources"])) - self.assertEqual(config["referenceWork"]["version"], "db-work-8-rev-12-imported-594") + self.assertEqual(config["referenceWork"]["version"], SOURCE_VERSION) + self.assertEqual(config["authorization"], self.authorization) + + def test_source_records_normalize_original_file_version(self): + self.assertEqual(self.source["documentId"], "7") + self.assertEqual(self.source["fileName"], REFERENCE["source_file"]) + self.assertEqual(self.source["sourceHash"], SOURCE_HASH) + self.assertEqual(self.source["sourceVersion"], SOURCE_VERSION) + + def test_missing_source_record_fails_closed(self): + for references, tasks, documents in ( + ([], [IMPORT_TASK], [DOCUMENT]), + ([REFERENCE], [], [DOCUMENT]), + ([REFERENCE], [IMPORT_TASK], []), + ): + with self.subTest(references=len(references), tasks=len(tasks), documents=len(documents)): + with self.assertRaises(AdapterError): + loader.validate_source_records(references, tasks, documents) + + def test_duplicate_source_record_fails_closed(self): + for references, tasks, documents in ( + ([REFERENCE, REFERENCE], [IMPORT_TASK], [DOCUMENT]), + ([REFERENCE], [IMPORT_TASK, IMPORT_TASK], [DOCUMENT]), + ([REFERENCE], [IMPORT_TASK], [DOCUMENT, DOCUMENT]), + ): + with self.subTest(references=len(references), tasks=len(tasks), documents=len(documents)): + with self.assertRaises(AdapterError): + loader.validate_source_records(references, tasks, documents) + + def test_import_task_query_ignores_successful_task_for_other_source_file(self): + unrelated = { + **IMPORT_TASK, + "id": 20, + "source_snapshot": {"file": "无关作品.txt"}, + } + connection = FakeReadOnlyConnection([IMPORT_TASK, unrelated]) + with patch.object(loader.psycopg, "connect", return_value=connection): + rows = loader.load_reference_rows( + dsn="postgresql://unused", + tenant_id=1, + work_id=8, + as_of=488, + target=489, + card_selection={"correctCardIds": [11126], "placeboCardIds": [11127]}, + ) + self.assertEqual(rows["source"]["importTaskId"], "19") + + def test_import_task_query_blocks_two_tasks_for_same_source_file(self): + duplicate = {**IMPORT_TASK, "id": 20} + connection = FakeReadOnlyConnection([IMPORT_TASK, duplicate]) + with patch.object(loader.psycopg, "connect", return_value=connection): + with self.assertRaises(AdapterError): + loader.load_reference_rows( + dsn="postgresql://unused", + tenant_id=1, + work_id=8, + as_of=488, + target=489, + card_selection={"correctCardIds": [11126], "placeboCardIds": [11127]}, + ) + + def test_invalid_hash_fails_closed(self): + with self.assertRaises(AdapterError): + loader.validate_source_records([REFERENCE], [IMPORT_TASK], [{**DOCUMENT, "file_hash": "xyz"}]) + + def test_filename_mismatch_fails_closed(self): + bad_task = {**IMPORT_TASK, "source_snapshot": {"file": "别的文件.txt"}} + with self.assertRaises(AdapterError): + loader.validate_source_records([REFERENCE], [bad_task], [DOCUMENT]) + with self.assertRaises(AdapterError): + loader.validate_source_records([REFERENCE], [IMPORT_TASK], [{**DOCUMENT, "file_name": "别的文件.txt"}]) + + def test_command_prefix_mismatch_fails_closed(self): + with self.assertRaises(AdapterError): + loader.validate_source_records( + [REFERENCE], + [{**IMPORT_TASK, "command_id": "import-0000000000000000"}], + [DOCUMENT], + ) + + def test_soft_deleted_source_record_fails_closed(self): + for references, tasks, documents in ( + ([{**REFERENCE, "deleted": True}], [IMPORT_TASK], [DOCUMENT]), + ([REFERENCE], [{**IMPORT_TASK, "deleted": True}], [DOCUMENT]), + ([REFERENCE], [IMPORT_TASK], [{**DOCUMENT, "deleted": True}]), + ): + with self.subTest(references=references, tasks=tasks, documents=documents): + with self.assertRaises(AdapterError): + loader.validate_source_records(references, tasks, documents) + + def test_authorization_is_projected_or_kept_blocked(self): + self.assertEqual(self.authorization["copyrightStatus"], "research_only") + self.assertEqual(self.authorization["sourceHash"], SOURCE_HASH) + self.assertEqual(self.authorization["authorizationSnapshot"]["sourceVersion"], SOURCE_VERSION) + blocked = loader.project_authorization(None, self.source) + self.assertEqual(blocked["sourceStatus"], "missing_authorization_snapshot") + self.assertEqual(blocked["allowedPurpose"], []) + self.assertEqual(blocked["authorizationSnapshot"], {}) + + def test_work_revision_change_does_not_change_source_version(self): + changed_work = {**WORK, "revision": 999} + config = build_replay_config( + work=changed_work, + reference={**REFERENCE, "imported_chapter_count": 593}, + outline_rows=[{"id": 1, "from_order": 1, "to_order": 10, "outline_text": "历史窗"}], + scaffold_rows=[], + target_scaffold={"id": 11, "chapter": 489, "title": "目标", "outline_text": "目标事实"}, + card_rows=[card_row(), card_row(11127, "赵宁")], + card_selection={"correctCardIds": [11126], "placeboCardIds": [11127]}, + as_of=488, + target=489, + evaluation_set_version="set-test", + strategy_version="strategy-test", + source=self.source, + authorization=self.authorization, + ) + self.assertEqual(config["referenceWork"]["version"], SOURCE_VERSION) + + def test_database_transaction_is_repeatable_read_only(self): + source = inspect.getsource(loader.load_reference_rows) + helper = inspect.getsource(loader.begin_read_snapshot) + self.assertIn("begin_read_snapshot(conn)", source) + self.assertIn("SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY", helper) + self.assertIn("ORDER BY checked_at DESC,id DESC\n LIMIT 1", source) if __name__ == "__main__": diff --git a/.claude/skills/replay-eval/scripts/test_rubric.py b/.claude/skills/replay-eval/scripts/test_rubric.py index 2ebe557..56a759a 100644 --- a/.claude/skills/replay-eval/scripts/test_rubric.py +++ b/.claude/skills/replay-eval/scripts/test_rubric.py @@ -2,6 +2,7 @@ """细纲 rubric 的离线回归测试。""" import pathlib +import inspect import sys import unittest @@ -49,6 +50,94 @@ class FineOutlineRubricTest(unittest.TestCase): self.assertFalse(result["stable"]) self.assertEqual(result["gaps"][DIMENSIONS[2]], 1.0) + def test_missing_dimension_fails_stability_closed(self): + first = {dimension: 4 for dimension in DIMENSIONS} + second = {dimension: 4 for dimension in DIMENSIONS[:-1]} + result = stability_warning(first, second) + self.assertFalse(result["stable"]) + self.assertIn(DIMENSIONS[-1], result["missingDimensions"]) + + def test_batch_report_requires_exact_candidates_and_distinct_judge_identity(self): + self.assertIn("expected_judge_id", inspect.signature(validate_report).parameters) + report = { + "profile": RUBRIC_PROFILE, + "judgeId": "judge-primary", + "evaluations": [ + {"candidateId": "blind-1", "scores": valid_scores(), "summary": "摘要一"}, + {"candidateId": "blind-2", "scores": valid_scores(), "summary": "摘要二"}, + ], + } + self.assertEqual( + validate_report( + report, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1", "blind-2"), + ), + [], + ) + duplicate = {**report, "evaluations": [report["evaluations"][0], report["evaluations"][0]]} + self.assertTrue( + validate_report( + duplicate, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1", "blind-2"), + ) + ) + + invalid_id = { + **report, + "evaluations": [{**report["evaluations"][0], "candidateId": ["blind-1"]}], + } + self.assertTrue(validate_report(invalid_id, expected_judge_id="judge-primary")) + + arm_leak = {**report, "arm": "outline_only"} + self.assertTrue( + validate_report( + arm_leak, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1", "blind-2"), + ) + ) + + def test_nested_evaluation_and_score_reject_arm_card_manifest_leakage(self): + evaluation = { + "candidateId": "blind-1", + "scores": valid_scores(), + "summary": "短安全摘要", + } + report = { + "profile": RUBRIC_PROFILE, + "judgeId": "judge-primary", + "evaluations": [evaluation], + } + leaking_evaluation = { + **report, + "evaluations": [{**evaluation, "arm": "outline_only"}], + } + leaking_scores = valid_scores() + leaking_scores[DIMENSIONS[0]] = { + **leaking_scores[DIMENSIONS[0]], + "cardManifest": {"count": 1}, + } + leaking_score = { + **report, + "evaluations": [{**evaluation, "scores": leaking_scores}], + } + self.assertTrue( + validate_report( + leaking_evaluation, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1",), + ) + ) + self.assertTrue( + validate_report( + leaking_score, + expected_judge_id="judge-primary", + expected_candidate_ids=("blind-1",), + ) + ) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_run_replay.py b/.claude/skills/replay-eval/scripts/test_run_replay.py index c23887f..df8f2a9 100644 --- a/.claude/skills/replay-eval/scripts/test_run_replay.py +++ b/.claude/skills/replay-eval/scripts/test_run_replay.py @@ -9,26 +9,35 @@ import pathlib import sys import tempfile import unittest +from unittest.mock import patch sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) -from run_replay import _planner_prompt, run_replay # noqa: E402 +from run_replay import _parse_args, _planner_prompt, run_replay # noqa: E402 from write_report import render_report # noqa: E402 +SOURCE_HASH = "sha256:02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4" +SOURCE_VERSION = f"raw-file-v1:{SOURCE_HASH}" AUTH = { "sourceStatus": "active", - "copyrightStatus": "licensed", - "sourceVersion": "work-v1", + "copyrightStatus": "research_only", + "sourceHash": SOURCE_HASH, + "sourceVersion": SOURCE_VERSION, "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution"], "authorizationSnapshot": { "id": "auth-1", "version": "v1", "immutable": True, - "sourceVersion": "work-v1", + "sourceHash": SOURCE_HASH, + "sourceVersion": SOURCE_VERSION, "sourceStatus": "active", + "copyrightStatus": "research_only", + "authorizationBasis": "user_authorization", "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution"], "checkedAt": "2026-07-19T00:00:00Z", - "revalidationAt": "2026-07-20T00:00:00Z", + "revalidationAt": "2099-07-20T00:00:00Z", }, } @@ -37,7 +46,7 @@ def config(): common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}} return { "runId": "smoke-001", - "referenceWork": {"id": "deep-space", "version": "work-v1"}, + "referenceWork": {"id": "deep-space", "version": SOURCE_VERSION}, "evaluationSetVersion": "set-v1", "strategyVersion": "strategy-v1", "authorization": AUTH, @@ -73,6 +82,121 @@ def config(): } +def candidate(goal="机制 smoke", *, events=True): + """构造满足细纲闭集合同的合成候选。""" + + return { + "targetChapter": 489, + "chapterGoal": goal, + "keyEvents": ( + [ + { + "id": "event-1", + "order": 1, + "event": "侦察敌情", + "participants": ["测试角色"], + "trigger": "收到异常信号", + "resultDirection": "确认威胁存在", + } + ] + if events + else [] + ), + "entities": [], + "foreshadowing": [], + "stateChanges": [], + "hook": "下一步", + "unknowns": [], + "assumptions": [], + } + + +def write_fake_runner(directory, mode="stable"): + """生成可记录调用输入的假模型二进制,测试不触发真实模型。""" + + directory = pathlib.Path(directory) + runner = directory / "fake-agent.py" + log_path = directory / "agent-calls.jsonl" + count_path = directory / "planner-count.txt" + run_result_path = directory / "run" / "run_result.json" + runner.write_text( + "#!/usr/bin/env python3\n" + "import json, pathlib, sys, time\n" + f"mode = {mode!r}\n" + f"log_path = pathlib.Path({str(log_path)!r})\n" + f"count_path = pathlib.Path({str(count_path)!r})\n" + f"run_result_path = pathlib.Path({str(run_result_path)!r})\n" + "args = sys.argv[1:]\n" + "agent = args[args.index('--agent') + 1]\n" + "prompt = args[-1]\n" + "run_state = json.loads(run_result_path.read_text()) if run_result_path.exists() else None\n" + "with log_path.open('a', encoding='utf-8') as handle:\n" + " handle.write(json.dumps({'agent': agent, 'prompt': prompt, 'runState': run_state}, ensure_ascii=False) + '\\n')\n" + "if mode == f'{agent}_timeout':\n" + " time.sleep(2)\n" + "if mode == f'{agent}_exit':\n" + " raise SystemExit(7)\n" + "if mode == f'{agent}_invalid_json':\n" + " print('{invalid-json')\n" + " raise SystemExit(0)\n" + "if agent == 'planner':\n" + " count = int(count_path.read_text() if count_path.exists() else '0') + 1\n" + " count_path.write_text(str(count))\n" + " events = not (mode == 'invalid_candidate' and count == 2)\n" + f" value = {candidate()!r}\n" + " value['chapterGoal'] = f'goal-{count}'\n" + " if not events:\n" + " value['keyEvents'] = []\n" + " print(json.dumps({'result': json.dumps(value, ensure_ascii=False)}, ensure_ascii=False))\n" + "elif agent == 'detector':\n" + " request = json.loads(prompt)\n" + " findings = []\n" + " if mode == 'high_detector' and request['candidate']['chapterGoal'] == 'goal-2':\n" + " findings.append({'severity': 'high', 'category': 'entity_state', 'location': 'keyEvents[0]', 'evidenceSummary': '冻结事实冲突'})\n" + " if mode == 'invalid_detector':\n" + " findings.append({'category': 'entity_state'})\n" + " print(json.dumps({'protocol': 'fine_outline_detector_v0', 'candidateId': request['candidateId'], 'findings': findings, 'coverageFindings': []}, ensure_ascii=False))\n" + "elif agent == 'judge':\n" + " request = json.loads(prompt)\n" + " dimensions = ['structure_completeness', 'direction_causality', 'order_pacing', 'entity_state', 'foreshadowing_action', 'handoff_hook']\n" + " evaluations = []\n" + " for item in request['candidates']:\n" + " base = {'goal-1': 3, 'goal-2': 4, 'goal-3': 2}[item['candidate']['chapterGoal']]\n" + " scores = {dimension: {'score': base, 'evidence': f'{dimension}-结构化证据'} for dimension in dimensions}\n" + " if mode == 'unstable' and request['judgeId'] == 'judge-secondary':\n" + " scores['order_pacing']['score'] = min(5, base + 1)\n" + " if mode == 'invalid_rubric':\n" + " scores['order_pacing']['score'] = 6\n" + " evaluations.append({'candidateId': item['candidateId'], 'scores': scores, 'summary': '结构化评分摘要'})\n" + " print(json.dumps({'profile': 'fine_outline_replay', 'judgeId': request['judgeId'], 'evaluations': evaluations}, ensure_ascii=False))\n", + encoding="utf-8", + ) + os.chmod(runner, 0o755) + return runner, log_path + + +def read_calls(log_path): + if not log_path.exists(): + return [] + return [json.loads(line) for line in log_path.read_text(encoding="utf-8").splitlines()] + + +def nested_keys(value): + """收集嵌套 JSON 的全部字段名,供盲化边界测试使用。""" + + if isinstance(value, dict): + keys = set(value) + for item in value.values(): + keys.update(nested_keys(item)) + return keys + if isinstance(value, list): + keys = set() + for item in value: + keys.update(nested_keys(item)) + return keys + return set() + + class ReplayRunTest(unittest.TestCase): def test_public_planner_context_does_not_include_snapshot_cards(self): prompt = _planner_prompt( @@ -103,6 +227,27 @@ class ReplayRunTest(unittest.TestCase): self.assertEqual(result["status"], "blocked_authorization") self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists()) + def test_reused_output_dir_bad_config_overwrites_previous_completed_state(self): + with tempfile.TemporaryDirectory() as directory: + output_dir = pathlib.Path(directory) / "run" + output_dir.mkdir() + result_path = output_dir / "run_result.json" + result_path.write_text( + json.dumps({"runId": "old-run", "status": "completed", "ok": True}), + encoding="utf-8", + ) + bad = config() + bad["snapshot"]["data"]["unknownSection"] = [] + + with self.assertRaisesRegex(ValueError, "未登记顶层分区"): + run_replay(bad, output_dir, mode="execute") + + persisted = json.loads(result_path.read_text(encoding="utf-8")) + self.assertEqual(persisted["runId"], "smoke-001") + self.assertEqual(persisted["status"], "config_invalid") + self.assertFalse(persisted["ok"]) + self.assertNotEqual(persisted["status"], "completed") + def test_content_leak_stops_before_manifest(self): bad = config() bad["leakageAudit"]["targetFacts"]["forbiddenFacts"][0]["text"] = "安全历史" @@ -141,34 +286,306 @@ class ReplayRunTest(unittest.TestCase): self.assertNotIn('"prompt":', report) self.assertNotIn('"payload":', report) - def test_execute_uses_external_planner_and_validates_each_candidate(self): - candidate = { - "targetChapter": 489, - "chapterGoal": "机制 smoke", - "keyEvents": [], - "entities": [], - "foreshadowing": [], - "stateChanges": [], - "hook": "下一步", - "unknowns": [], - "assumptions": [], + def test_report_rejects_text_injection_through_identifier_fields(self): + base = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + injections = { + "runId": "safe-run\n完整目标细纲:第一幕到第三幕的全部事件原文", + "referenceWork": "深空之影目标章原文与完整细纲", } + for field, injected in injections.items(): + with self.subTest(field=field): + unsafe = dict(base) + unsafe[field] = injected + with self.assertRaisesRegex(ValueError, field): + render_report(unsafe) + + def test_report_does_not_render_preflight_free_text(self): + bad = config() + injected = "目标章事实文本" + bad["authorization"] = {**bad["authorization"], "sourceStatus": injected} + result = run_replay(bad, pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + report = render_report(result) + self.assertNotIn(injected, report) + self.assertIn("授权前置门未通过", report) + + def test_report_accepts_registered_target_source_failure_status(self): + result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + result["results"] = { + "outline_only": {"status": "target_source_forbidden", "ok": False} + } + report = render_report(result) + self.assertIn("target_source_forbidden", report) + + def test_execute_uses_external_planner_and_validates_each_candidate(self): with tempfile.TemporaryDirectory() as directory: - fake = pathlib.Path(directory) / "fake-planner.py" - fake.write_text( - "#!/usr/bin/env python3\n" - "import json\n" - f"print(json.dumps({{'result': json.dumps({candidate!r}, ensure_ascii=False)}}))\n", - encoding="utf-8", - ) - os.chmod(fake, 0o755) + fake, log_path = write_fake_runner(directory) output_dir = pathlib.Path(directory) / "run" result = run_replay(config(), output_dir, mode="execute", planner_bin=str(fake)) + first_call = read_calls(log_path)[0] self.assertTrue(result["ok"]) self.assertEqual(result["status"], "completed") + self.assertFalse(first_call["runState"]["ok"]) + self.assertEqual(first_call["runState"]["status"], "running_planner") self.assertEqual(set(result["results"]), {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"}) self.assertTrue(list(output_dir.glob("candidate_*.json"))) + def test_missing_planner_binary_fails_closed_and_persists_result(self): + with tempfile.TemporaryDirectory() as directory: + output_dir = pathlib.Path(directory) / "run" + missing = pathlib.Path(directory) / "missing-planner" + try: + result = run_replay(config(), output_dir, mode="execute", planner_bin=str(missing)) + except OSError as error: + self.fail(f"runner 启动异常不得逃逸: {error}") + persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8")) + self.assertFalse(result["ok"]) + self.assertEqual(result["status"], "planner_failed") + self.assertEqual(persisted["status"], "planner_failed") + self.assertFalse(persisted["ok"]) + self.assertNotIn(persisted["status"], {"ready", "completed"}) + + def test_planner_nonzero_exit_stops_after_first_call(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "planner_exit") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "planner_failed") + self.assertFalse(result["ok"]) + self.assertEqual(len(calls), 1) + + def test_planner_invalid_json_stops_after_first_call(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "planner_invalid_json") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "planner_invalid") + self.assertFalse(result["ok"]) + self.assertEqual(len(calls), 1) + + def test_planner_timeout_fails_closed_and_persists_timeout_status(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "planner_timeout") + output_dir = pathlib.Path(directory) / "run" + result = run_replay( + config(), + output_dir, + mode="execute", + planner_bin=str(runner), + timeout_seconds=1.0, + ) + persisted = json.loads((output_dir / "run_result.json").read_text(encoding="utf-8")) + self.assertEqual(result["status"], "planner_timeout") + self.assertFalse(result["ok"]) + self.assertEqual(persisted["status"], "planner_timeout") + self.assertEqual(len(read_calls(log_path)), 1) + + def test_detector_nonzero_exit_stops_before_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "detector_exit") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_failed") + self.assertFalse(result["ok"]) + self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_detector_invalid_json_stops_before_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "detector_invalid_json") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_invalid") + self.assertFalse(result["ok"]) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_detector_timeout_fails_closed_before_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "detector_timeout") + result = run_replay( + config(), + pathlib.Path(directory) / "run", + mode="execute", + planner_bin=str(runner), + timeout_seconds=1.0, + ) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_timeout") + self.assertFalse(result["ok"]) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_primary_judge_nonzero_exit_does_not_call_secondary(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "judge_exit") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "judge_failed") + self.assertFalse(result["ok"]) + self.assertEqual(len(judge_calls), 1) + + def test_primary_judge_invalid_json_does_not_call_secondary(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "judge_invalid_json") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "judge_invalid") + self.assertFalse(result["ok"]) + self.assertEqual(len(judge_calls), 1) + + def test_primary_judge_timeout_does_not_call_secondary(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "judge_timeout") + result = run_replay( + config(), + pathlib.Path(directory) / "run", + mode="execute", + planner_bin=str(runner), + timeout_seconds=1.0, + ) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "judge_timeout") + self.assertFalse(result["ok"]) + self.assertEqual(len(judge_calls), 1) + + def test_cli_accepts_subprocess_timeout_seconds(self): + argv = [ + "run_replay.py", + "--config", + "/tmp/replay-config.json", + "--output-dir", + "/tmp/replay-output", + "--timeout-seconds", + "12.5", + ] + with patch.object(sys, "argv", argv): + args = _parse_args() + self.assertEqual(args.timeout_seconds, 12.5) + + def test_non_finite_timeout_is_persisted_as_config_invalid(self): + with tempfile.TemporaryDirectory() as directory: + for index, timeout_seconds in enumerate((float("nan"), float("inf"))): + with self.subTest(timeout_seconds=timeout_seconds): + output_dir = pathlib.Path(directory) / f"run-{index}" + with self.assertRaisesRegex(ValueError, "timeout_seconds"): + run_replay( + config(), + output_dir, + mode="execute", + timeout_seconds=timeout_seconds, + ) + persisted = json.loads( + (output_dir / "run_result.json").read_text(encoding="utf-8") + ) + self.assertEqual(persisted["status"], "config_invalid") + self.assertFalse(persisted["ok"]) + + def test_schema_failure_stops_before_all_detectors_and_judges(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "invalid_candidate") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "candidate_blocked") + self.assertEqual(len([call for call in calls if call["agent"] == "planner"]), 3) + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 0) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_high_detector_finding_blocks_group_and_never_calls_judge(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "high_detector") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + detector_calls = [call for call in calls if call["agent"] == "detector"] + self.assertEqual(result["status"], "detector_blocked") + self.assertEqual(len(detector_calls), 3) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + self.assertTrue(all("outline_plus" not in call["prompt"] for call in detector_calls)) + self.assertTrue(all("targetFacts" not in call["prompt"] for call in detector_calls)) + + def test_detector_requests_cannot_distinguish_arm_specific_card_identity(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "stable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + self.assertEqual(result["status"], "completed") + requests = [ + json.loads(call["prompt"]) + for call in read_calls(log_path) + if call["agent"] == "detector" + ] + self.assertEqual(len(requests), 3) + for request in requests: + self.assertTrue( + nested_keys(request).isdisjoint({"arm", "cardInjection", "cardManifest"}) + ) + serialized = json.dumps(request, ensure_ascii=False) + self.assertNotIn("正确卡", serialized) + self.assertNotIn("错配卡", serialized) + self.assertNotIn("card-correct-1", serialized) + self.assertNotIn("card-placebo-1", serialized) + public_parts = [ + {key: value for key, value in request.items() if key not in {"candidateId", "candidate"}} + for request in requests + ] + self.assertEqual(public_parts[0], public_parts[1]) + self.assertEqual(public_parts[1], public_parts[2]) + + def test_invalid_detector_report_fails_closed_before_judge(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "invalid_detector") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + calls = read_calls(log_path) + self.assertEqual(result["status"], "detector_invalid") + self.assertEqual(len([call for call in calls if call["agent"] == "detector"]), 1) + self.assertEqual(len([call for call in calls if call["agent"] == "judge"]), 0) + + def test_two_judges_are_independent_blind_reversed_and_unblinded(self): + with tempfile.TemporaryDirectory() as directory: + runner, log_path = write_fake_runner(directory, "stable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + judge_calls = [call for call in read_calls(log_path) if call["agent"] == "judge"] + self.assertEqual(result["status"], "completed") + self.assertEqual(len(judge_calls), 2) + first = json.loads(judge_calls[0]["prompt"]) + second = json.loads(judge_calls[1]["prompt"]) + self.assertNotEqual(first["judgeId"], second["judgeId"]) + self.assertEqual( + [item["candidateId"] for item in second["candidates"]], + list(reversed([item["candidateId"] for item in first["candidates"]])), + ) + self.assertNotIn("outline_only", judge_calls[0]["prompt"]) + self.assertNotIn("outline_plus_cards", judge_calls[1]["prompt"]) + self.assertTrue(result["evaluation"]["stability"]["stable"]) + self.assertEqual(result["evaluation"]["deltas"]["B-A"]["order_pacing"], 1.0) + self.assertEqual(result["evaluation"]["deltas"]["C-A"]["order_pacing"], -1.0) + + def test_invalid_rubric_report_does_not_complete(self): + with tempfile.TemporaryDirectory() as directory: + runner, _ = write_fake_runner(directory, "invalid_rubric") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + self.assertEqual(result["status"], "judge_invalid") + self.assertFalse(result["ok"]) + + def test_unstable_judges_have_explicit_non_completed_status(self): + with tempfile.TemporaryDirectory() as directory: + runner, _ = write_fake_runner(directory, "unstable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + self.assertEqual(result["status"], "judge_unstable") + self.assertFalse(result["ok"]) + self.assertFalse(result["evaluation"]["stability"]["stable"]) + self.assertNotIn("deltas", result["evaluation"]) + + def test_report_contains_only_aggregated_evaluation(self): + with tempfile.TemporaryDirectory() as directory: + runner, _ = write_fake_runner(directory, "stable") + result = run_replay(config(), pathlib.Path(directory) / "run", mode="execute", planner_bin=str(runner)) + report = render_report(result) + self.assertIn("B-A", report) + self.assertIn("C-A", report) + self.assertIn("评委稳定性", report) + self.assertNotIn("结构化证据", report) + self.assertNotIn("目标章未进入快照", report) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_run_writer_replay.py b/.claude/skills/replay-eval/scripts/test_run_writer_replay.py index 862c9a1..00e3277 100644 --- a/.claude/skills/replay-eval/scripts/test_run_writer_replay.py +++ b/.claude/skills/replay-eval/scripts/test_run_writer_replay.py @@ -29,16 +29,22 @@ from writer_rubric import DIMENSIONS, RUBRIC_PROFILE # noqa: E402 AUTHORIZATION = { "sourceStatus": "authorized", - "copyrightStatus": "owned", - "sourceVersion": "raw-file-v1:sha256:" + "a" * 64, + "copyrightStatus": "research_only", + "sourceHash": "sha256:" + "a" * 64, + "sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64, "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution", "model_training", "production_generation"], "authorizationSnapshot": { "id": "auth-work-8", "version": "auth-work-8-v1", "immutable": True, - "sourceVersion": "raw-file-v1:sha256:" + "a" * 64, + "sourceHash": "sha256:" + "a" * 64, + "sourceVersion": "sanitized-fixture-v1:sha256:" + "a" * 64, "sourceStatus": "authorized", + "copyrightStatus": "research_only", + "authorizationBasis": "sanitized_contract_fixture", "allowedPurpose": ["offline_evaluation"], + "forbiddenPurpose": ["external_distribution", "model_training", "production_generation"], "checkedAt": "2026-07-20T00:00:00Z", "revalidationAt": "2026-08-19T00:00:00Z", }, diff --git a/.claude/skills/replay-eval/scripts/write_report.py b/.claude/skills/replay-eval/scripts/write_report.py index 5fa9a7b..7da6091 100644 --- a/.claude/skills/replay-eval/scripts/write_report.py +++ b/.claude/skills/replay-eval/scripts/write_report.py @@ -5,6 +5,8 @@ from __future__ import annotations import argparse import json +import math +import re from pathlib import Path from typing import Any, Mapping @@ -27,6 +29,80 @@ REPORT_FORBIDDEN_KEYS = frozenset( "完整目标细纲", } ) +IDENTIFIER_PATTERN = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,63}\Z") +SOURCE_VERSION_PATTERN = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,127}\Z") +SHA256_PATTERN = re.compile(r"[0-9a-f]{64}\Z") +ARM_NAMES = frozenset( + {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"} +) +RUN_STATUSES = frozenset( + { + "unknown", + "validating_config", + "config_invalid", + "blocked_authorization", + "invalid_snapshot", + "invalid_arm_diff", + "target_source_forbidden", + "invalid_audit_input", + "blocked_leakage_audit", + "ready", + "running_planner", + "planner_timeout", + "planner_failed", + "planner_invalid", + "candidate_blocked", + "running_detector", + "detector_timeout", + "detector_failed", + "detector_invalid", + "detector_blocked", + "running_judge", + "judge_timeout", + "judge_failed", + "judge_invalid", + "judge_unstable", + "completed", + } +) +PREFLIGHT_STATUSES = frozenset( + {"unknown", "ready", "blocked_authorization", "invalid_snapshot", "invalid_arm_diff", "target_source_forbidden"} +) +ARM_STATUSES = frozenset( + { + "not_run", + "ready", + "schema_invalid", + "invalid_snapshot", + "target_source_forbidden", + "planner_timeout", + "planner_failed", + "planner_invalid", + } +) +DETECTOR_STATUSES = frozenset( + {"not_run", "passed", "blocked_high", "timeout", "failed", "invalid"} +) +CARD_STRATEGIES = frozenset({"unknown", "none", "correct", "placebo"}) +RUBRIC_PROFILES = frozenset({"unknown", "fine_outline_replay"}) +RUBRIC_DIMENSIONS = frozenset( + { + "structure_completeness", + "direction_causality", + "order_pacing", + "entity_state", + "foreshadowing_action", + "handoff_hook", + } +) +PREFLIGHT_SUMMARIES = { + "unknown": "前置门状态不可用", + "ready": "前置门通过", + "blocked_authorization": "授权前置门未通过", + "invalid_snapshot": "冻结快照前置门未通过", + "invalid_arm_diff": "三臂公共输入一致性未通过", + "target_source_forbidden": "发现目标章或未来来源", +} def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None: @@ -42,43 +118,234 @@ def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None: _assert_no_forbidden_keys(item, f"{path}[{index}]") -def _arm_line(name: str, arm: Mapping[str, Any], result: Mapping[str, Any]) -> str: - outcome = result.get(name) or {} +def _mapping(value: Any, path: str) -> Mapping[str, Any]: + if not isinstance(value, Mapping): + raise ValueError(f"{path} 必须是对象") + return value + + +def _identifier(value: Any, path: str, default: str = "unknown") -> str: + candidate = default if value is None or value == "" else value + if not isinstance(candidate, str) or IDENTIFIER_PATTERN.fullmatch(candidate) is None: + raise ValueError(f"{path} 必须是长度不超过 64 的安全标识符") + return candidate + + +def _source_version(value: Any, path: str) -> str: + """来源版本允许容纳带算法前缀的完整 SHA-256,仍只接受安全标识字符。""" + + candidate = "unknown" if value is None or value == "" else value + if not isinstance(candidate, str) or SOURCE_VERSION_PATTERN.fullmatch(candidate) is None: + raise ValueError(f"{path} 必须是长度不超过 128 的安全来源版本") + return candidate + + +def _enum(value: Any, allowed: frozenset[str], path: str, default: str) -> str: + candidate = default if value is None or value == "" else value + if not isinstance(candidate, str) or candidate not in allowed: + raise ValueError(f"{path} 不是已登记枚举值") + return candidate + + +def _integer(value: Any, path: str, *, minimum: int = 0) -> int | None: + if value is None: + return None + if isinstance(value, bool) or not isinstance(value, int) or value < minimum: + raise ValueError(f"{path} 必须是大于等于 {minimum} 的整数") + return value + + +def _number(value: Any, path: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{path} 必须是数字") + number = float(value) + if not math.isfinite(number): + raise ValueError(f"{path} 必须是有限数字") + return number + + +def _hash(value: Any, path: str) -> str | None: + if value is None or value == "": + return None + if not isinstance(value, str) or SHA256_PATTERN.fullmatch(value) is None: + raise ValueError(f"{path} 必须是 SHA-256") + return value + + +def _numeric_map(value: Any, path: str, allowed_keys: frozenset[str]) -> dict[str, float]: + mapping = _mapping(value, path) + unexpected = sorted(set(mapping) - allowed_keys) + if unexpected: + raise ValueError(f"{path} 包含未登记字段: {','.join(unexpected)}") + return {str(key): _number(item, f"{path}.{key}") for key, item in mapping.items()} + + +def _project_evaluation(value: Any) -> dict[str, Any]: + if value is None or value == "" or value == {}: + return {} + evaluation = _mapping(value, "evaluation") + projected: dict[str, Any] = { + "profile": _enum(evaluation.get("profile"), RUBRIC_PROFILES, "evaluation.profile", "unknown") + } + stability = evaluation.get("stability") + if stability is not None: + stability_mapping = _mapping(stability, "evaluation.stability") + stable = stability_mapping.get("stable") + if not isinstance(stable, bool): + raise ValueError("evaluation.stability.stable 必须是布尔枚举") + projected["stabilityStatus"] = "stable" if stable else "unstable" + projected["maxGaps"] = _numeric_map( + stability_mapping.get("maxGaps", {}), + "evaluation.stability.maxGaps", + RUBRIC_DIMENSIONS, + ) + arm_scores = _mapping(evaluation.get("armScores", {}), "evaluation.armScores") + unexpected_arms = sorted(set(arm_scores) - ARM_NAMES) + if unexpected_arms: + raise ValueError(f"evaluation.armScores 包含未登记臂: {','.join(unexpected_arms)}") + projected["armScores"] = { + str(arm): _numeric_map(scores, f"evaluation.armScores.{arm}", RUBRIC_DIMENSIONS) + for arm, scores in arm_scores.items() + } + deltas = _mapping(evaluation.get("deltas", {}), "evaluation.deltas") + unexpected_deltas = sorted(set(deltas) - {"B-A", "C-A"}) + if unexpected_deltas: + raise ValueError(f"evaluation.deltas 包含未登记比较: {','.join(unexpected_deltas)}") + projected["deltas"] = { + str(name): _numeric_map(scores, f"evaluation.deltas.{name}", RUBRIC_DIMENSIONS) + for name, scores in deltas.items() + } + return projected + + +def _project_report(result: Mapping[str, Any]) -> dict[str, Any]: + """从运行态提取独立安全 schema,未登记类型和值一律拒绝。""" + + _assert_no_forbidden_keys(result) + preflight = _mapping(result.get("preflight", {}), "preflight") + preflight_status = _enum( + preflight.get("status"), + PREFLIGHT_STATUSES, + "preflight.status", + "unknown", + ) + arms = _mapping(result.get("arms", {}), "arms") + outcomes = _mapping(result.get("results", {}), "results") + unexpected_arms = sorted((set(arms) | set(outcomes)) - ARM_NAMES) + if unexpected_arms: + raise ValueError(f"报告包含未登记评测臂: {','.join(unexpected_arms)}") + projected_arms: dict[str, Any] = {} + for name in sorted(arms): + arm = _mapping(arms[name], f"arms.{name}") + outcome = _mapping(outcomes.get(name, {}), f"results.{name}") + detector = _mapping(outcome.get("detector", {}), f"results.{name}.detector") + projected_arms[name] = { + "status": _enum(outcome.get("status"), ARM_STATUSES, f"results.{name}.status", "not_run"), + "candidateSha256": _hash( + outcome.get("candidateSha256") or outcome.get("rawOutputSha256"), + f"results.{name}.candidateSha256", + ), + "detectorStatus": _enum( + detector.get("status"), + DETECTOR_STATUSES, + f"results.{name}.detector.status", + "not_run", + ), + "cardStrategy": _enum( + arm.get("cardStrategy"), + CARD_STRATEGIES, + f"arms.{name}.cardStrategy", + "unknown", + ), + "cardInjectionCount": _integer( + arm.get("cardInjectionCount", 0), + f"arms.{name}.cardInjectionCount", + ), + } + return { + "runId": _identifier(result.get("runId"), "runId"), + "status": _enum(result.get("status"), RUN_STATUSES, "status", "unknown"), + "mode": _enum(result.get("mode"), frozenset({"unknown", "dry_run", "execute"}), "mode", "unknown"), + "referenceWork": _identifier(result.get("referenceWork"), "referenceWork"), + "referenceWorkVersion": _source_version( + result.get("referenceWorkVersion"), + "referenceWorkVersion", + ), + "asOfChapter": _integer(result.get("asOfChapter"), "asOfChapter", minimum=1), + "targetChapter": _integer(result.get("targetChapter"), "targetChapter", minimum=1), + "snapshotVersion": _identifier(result.get("snapshotVersion"), "snapshotVersion"), + "preflightStatus": preflight_status, + "preflightSummary": PREFLIGHT_SUMMARIES[preflight_status], + "arms": projected_arms, + "snapshotManifestSha256": _hash( + result.get("snapshotManifestSha256"), "snapshotManifestSha256" + ), + "evaluation": _project_evaluation(result.get("evaluation")), + } + + +def _arm_line(name: str, arm: Mapping[str, Any]) -> str: return ( - f"- `{name}`: {outcome.get('status', 'not_run')}," - f"候选哈希 `{outcome.get('candidateSha256') or outcome.get('rawOutputSha256') or 'n/a'}`," + f"- `{name}`: {arm['status']}," + f"候选哈希 `{arm['candidateSha256'] or 'n/a'}`," + f"detector `{arm['detectorStatus']}`," f"卡策略 `{arm.get('cardStrategy', 'unknown')}`,卡数 `{arm.get('cardInjectionCount', 0)}`" ) +def _render_evaluation(lines: list[str], evaluation: Mapping[str, Any]) -> None: + """只渲染聚合分数、稳定性和差值,不带 judge 原始证据。""" + + if not evaluation: + return + lines.extend( + [ + "", + "## 双盲评分", + "", + f"- profile: `{evaluation.get('profile', 'unknown')}`", + f"- 评委稳定性: `{evaluation.get('stabilityStatus', 'not_available')}`", + f"- 最大维度差: `{json.dumps(evaluation.get('maxGaps') or {}, ensure_ascii=False, sort_keys=True)}`", + ] + ) + arm_scores = evaluation.get("armScores") or {} + for arm in sorted(arm_scores): + lines.append( + f"- `{arm}` 聚合分数: `{json.dumps(arm_scores[arm], ensure_ascii=False, sort_keys=True)}`" + ) + deltas = evaluation.get("deltas") or {} + for comparison in ("B-A", "C-A"): + if comparison in deltas: + lines.append( + f"- `{comparison}` 差值: `{json.dumps(deltas[comparison], ensure_ascii=False, sort_keys=True)}`" + ) + + def render_report(result: Mapping[str, Any]) -> str: """只从运行结果的白名单字段渲染摘要。""" - _assert_no_forbidden_keys(result) + safe = _project_report(result) lines = [ "# 细纲回放摘要", "", - f"- runId: `{result.get('runId', 'unknown')}`", - f"- status: `{result.get('status', 'unknown')}`", - f"- mode: `{result.get('mode', 'unknown')}`", - f"- referenceWork: `{result.get('referenceWork', 'unknown')}@{result.get('referenceWorkVersion', 'unknown')}`", - f"- freeze: 第 `{result.get('asOfChapter', 'unknown')}` 章,目标第 `{result.get('targetChapter', 'unknown')}` 章", - f"- snapshot: `{result.get('snapshotVersion', 'unknown')}`", + f"- runId: `{safe['runId']}`", + f"- status: `{safe['status']}`", + f"- mode: `{safe['mode']}`", + f"- referenceWork: `{safe['referenceWork']}@{safe['referenceWorkVersion']}`", + f"- freeze: 第 `{safe['asOfChapter'] if safe['asOfChapter'] is not None else 'unknown'}` 章,目标第 `{safe['targetChapter'] if safe['targetChapter'] is not None else 'unknown'}` 章", + f"- snapshot: `{safe['snapshotVersion']}`", "", "## 前置门", "", - f"- status: `{(result.get('preflight') or {}).get('status', 'unknown')}`", + f"- status: `{safe['preflightStatus']}`", + f"- 摘要: {safe['preflightSummary']}", ] - errors = (result.get("preflight") or {}).get("errors") or [] - for error in errors: - lines.append(f"- 阻断: {error}") lines.extend(["", "## 三臂", ""]) - arms = result.get("arms") or {} - outcomes = result.get("results") or {} - for name in sorted(arms): - lines.append(_arm_line(name, arms[name], outcomes)) - if result.get("snapshotManifestSha256"): - lines.extend(["", f"- snapshotManifestSha256: `{result['snapshotManifestSha256']}`"]) + for name, arm in safe["arms"].items(): + lines.append(_arm_line(name, arm)) + if safe["snapshotManifestSha256"]: + lines.extend(["", f"- snapshotManifestSha256: `{safe['snapshotManifestSha256']}`"]) + _render_evaluation(lines, safe["evaluation"]) lines.extend( [ "", diff --git a/db/ddl/96-example参考作品授权快照.sql b/db/ddl/96-example参考作品授权快照.sql new file mode 100644 index 0000000..0c3b2e1 --- /dev/null +++ b/db/ddl/96-example参考作品授权快照.sql @@ -0,0 +1,98 @@ +-- example_reference_authorization_snapshot:参考作品用途授权的不可变快照。 +-- 本表只允许 INSERT;授权变化通过新增版本表达,禁止覆盖或删除历史证据。 +CREATE FUNCTION example_jsonb_text_arrays_disjoint(left_values JSONB, right_values JSONB) +RETURNS BOOLEAN +LANGUAGE sql +IMMUTABLE +STRICT +AS $$ + SELECT NOT EXISTS ( + SELECT 1 + FROM jsonb_array_elements_text(left_values) AS left_value(value) + JOIN jsonb_array_elements_text(right_values) AS right_value(value) USING (value) + ); +$$; + +CREATE TABLE example_reference_authorization_snapshot ( + id BIGINT GENERATED ALWAYS AS IDENTITY PRIMARY KEY, + reference_work_id BIGINT NOT NULL, -- → example_reference_work.id(软引用) + work_id BIGINT NOT NULL, -- → muse_content_work.id(软引用) + knowledge_document_id BIGINT NOT NULL, -- → muse_knowledge_document.id(软引用) + import_task_id BIGINT NOT NULL, -- → muse_content_import_task.id(软引用) + source_file VARCHAR(500) NOT NULL, -- 授权对应的原文件名 + source_hash VARCHAR(71) NOT NULL, -- sha256:<64位小写十六进制> + source_version VARCHAR(96) NOT NULL, -- raw-file-v1:sha256:<64位小写十六进制> + snapshot_version VARCHAR(160) NOT NULL, -- 授权快照业务版本,不使用物理主键充当合同版本 + copyright_status VARCHAR(30) NOT NULL, -- licensed/public_domain/research_only/unauthorized + source_status VARCHAR(30) NOT NULL DEFAULT 'active', + allowed_purpose JSONB NOT NULL DEFAULT '[]', -- 允许用途数组,例如 ["offline_evaluation"] + forbidden_purpose JSONB NOT NULL DEFAULT '[]', -- 明确禁止用途数组,例如 ["external_distribution"] + authorization_basis VARCHAR(40) NOT NULL, -- user_authorization/public_domain_record/license_contract + authorized_by VARCHAR(128) NOT NULL, -- 授权主体或登记责任人 + authorization_evidence JSONB NOT NULL, -- 授权原文摘要、时间和证据定位,不存敏感全文 + display_summary VARCHAR(500) NOT NULL, -- 面向审计面的脱敏摘要 + checked_at TIMESTAMPTZ NOT NULL, -- 本次授权核验时间 + expires_at TIMESTAMPTZ, -- 到期即阻断 + revalidation_at TIMESTAMPTZ, -- 到点必须重验 + creator VARCHAR(64) NOT NULL DEFAULT '', + create_time TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP, + tenant_id BIGINT NOT NULL DEFAULT 0, + CONSTRAINT uk_example_reference_auth_work_version UNIQUE (tenant_id, work_id, snapshot_version), + CONSTRAINT chk_example_reference_auth_source_hash + CHECK (source_hash ~ '^sha256:[0-9a-f]{64}$'), + CONSTRAINT chk_example_reference_auth_source_version + CHECK (source_version = 'raw-file-v1:' || source_hash), + CONSTRAINT chk_example_reference_auth_copyright + CHECK (copyright_status IN ('research_only','public_domain','licensed','unauthorized')), + CONSTRAINT chk_example_reference_auth_source_status + CHECK (source_status IN ('active','stale','revoked','delisted','recalled','blocked','owner_missing','unauthorized')), + CONSTRAINT chk_example_reference_auth_allowed_purpose + CHECK (jsonb_typeof(allowed_purpose) = 'array'), + CONSTRAINT chk_example_reference_auth_forbidden_purpose + CHECK (jsonb_typeof(forbidden_purpose) = 'array'), + CONSTRAINT chk_example_reference_auth_purpose_disjoint + CHECK (example_jsonb_text_arrays_disjoint(allowed_purpose, forbidden_purpose)), + CONSTRAINT chk_example_reference_auth_basis + CHECK (authorization_basis IN ('user_authorization','public_domain_record','license_contract')), + CONSTRAINT chk_example_reference_auth_basis_copyright + CHECK ( + (authorization_basis <> 'user_authorization' OR copyright_status = 'research_only') + AND (authorization_basis <> 'public_domain_record' OR copyright_status = 'public_domain') + AND (authorization_basis <> 'license_contract' OR copyright_status = 'licensed') + ), + CONSTRAINT chk_example_reference_auth_user_purpose + CHECK ( + authorization_basis <> 'user_authorization' + OR allowed_purpose = '["offline_evaluation"]'::jsonb + ), + CONSTRAINT chk_example_reference_auth_evidence + CHECK (jsonb_typeof(authorization_evidence) = 'object'), + CONSTRAINT chk_example_reference_auth_purpose + CHECK ( + (copyright_status = 'unauthorized' AND jsonb_array_length(allowed_purpose) = 0) + OR (copyright_status <> 'unauthorized' AND jsonb_array_length(allowed_purpose) > 0) + ), + CONSTRAINT chk_example_reference_auth_recheck + CHECK (expires_at IS NOT NULL OR revalidation_at IS NOT NULL), + CONSTRAINT chk_example_reference_auth_expiry_order + CHECK (expires_at IS NULL OR expires_at > checked_at), + CONSTRAINT chk_example_reference_auth_revalidation_order + CHECK (revalidation_at IS NULL OR revalidation_at > checked_at) +); + +CREATE INDEX idx_example_reference_auth_latest + ON example_reference_authorization_snapshot(tenant_id, work_id, checked_at DESC, id DESC); + +-- 授权快照是审计证据;撤销、到期或用途变化都必须插入新版本,不能改写旧记录。 +CREATE FUNCTION reject_example_reference_authorization_snapshot_mutation() +RETURNS TRIGGER +LANGUAGE plpgsql +AS $$ +BEGIN + RAISE EXCEPTION 'example_reference_authorization_snapshot 是 append-only 表,禁止 UPDATE/DELETE'; +END; +$$; + +CREATE TRIGGER trg_example_reference_auth_append_only +BEFORE UPDATE OR DELETE ON example_reference_authorization_snapshot +FOR EACH ROW EXECUTE FUNCTION reject_example_reference_authorization_snapshot_mutation(); diff --git a/db/表映射.md b/db/表映射.md index a631ae4..87b5ee4 100644 --- a/db/表映射.md +++ b/db/表映射.md @@ -2,7 +2,8 @@ > 口径(创始人拍板③ 2026-07-10):主仓表**原样不改列**;实验私货全进 `example_*` 前缀。 > 建表方式:`db/ddl/` 下文件经 db skill `apply`,主仓部分为 `muse-cloud/sql/muse/` 原文拷贝或逐字摘录。 -> 应用顺序:V1 → V3 → V5 → 90-ALTER摘录 → V26 → 91-example。已于 2026-07-13 应用,共 **24 张表**。 +> 已应用顺序:V1 → V3 → V5 → 90-ALTER摘录 → V26 → 91-example。已于 2026-07-13 应用,共 **24 张表**。 +> 待应用:`96-example参考作品授权快照.sql` 已实现但尚未通过 db skill `apply`,库内尚无该表和真实授权记录。 ## 主仓一致表(20 张) @@ -57,3 +58,9 @@ - 软删照主仓:`deleted=TRUE`,不物理删。**例外**:meta 种子行(`muse_meta_field`)幂等重跑=删旧插新(种子演练场景,豁免软删)。 - meta 字段行 sort_order 段位约定:1–9 基础字段,11 起特有字段。 - 双轨对应:B2 产出全落 `muse_knowledge_draft(status='pending')`;B5 管理员确认(仅创始人触发)= draft 翻 `confirmed` + 落 `muse_knowledge_entity(status='active')`;丢弃= draft 翻 `ignored`。 + +## 待应用实验表 + +| DDL | 表 | 用途 | 当前状态 | +|---|---|---|---| +| 96 | example_reference_authorization_snapshot | 参考作品原文件版本对应的不可变用途授权快照;同作品快照版本唯一,禁止 UPDATE/DELETE | DDL 已实现待 apply;真实记录未写 | diff --git a/docs/2026-07-19-回放评测-细纲首跑设计与计划.md b/docs/2026-07-19-回放评测-细纲首跑设计与计划.md index 93404a4..7aa6bed 100644 --- a/docs/2026-07-19-回放评测-细纲首跑设计与计划.md +++ b/docs/2026-07-19-回放评测-细纲首跑设计与计划.md @@ -11,12 +11,14 @@ ## 执行状态(2026-07-19) - 已落地并同步到 `agent-example/main`:`fbb262c`(冻结/细纲/评分合同)、`a54a3a4`(审查反馈收紧与回放编排)、`1243d9b`(外部 planner execute 链路测试)。 -- 已验证:回放脚本 38 个离线测试、fine-outline 合同 3 个测试全部通过;fake planner 已跑通三臂 execute 链路。测试只使用合成 fixture,不代表参考作品评测结果。 +- 已验证:replay-eval 70 个离线测试、fine-outline 合同 3 个测试全部通过;fake runner 已跑通三臂 planner、真正不识别臂卡内容的逐候选 detector、双 judge 反序盲评、子进程超时失败关闭、稳定性门和去盲矩阵。测试只使用合成 fixture,不代表参考作品评测结果。 - 已新增并验证:`audit_leakage.py` 对公共快照和三臂卡注入区执行内容级事实审计;`load_reference_work.py` 通过 PostgreSQL 只读事务组装仓库外临时配置,候选卡标记为 `eval_draft`,不进入生产检索。 - 已执行真实适配 smoke:work=8、冻结 488、目标 489,组装 31 个完整历史大纲窗、6 个近章细纲摘要、正确/错配卡各 2 张;送入回放仍为 `blocked_authorization`(`example_reference_work` 没有授权快照),未生成 `snapshot_manifest`,未调用模型。 - 已执行真实数据库前置 smoke:深空之影 `work_id=8` 的 `example_reference_work` 表没有不可变授权快照/版权用途字段,结果为 `blocked_authorization`,三臂均未调用模型。 +- 任务 A 已实现原文件版本与 append-only 授权快照 DDL:work=8 的权威原文件 `document_id=7`、SHA-256=`02cf1f8c1ca03c26e0b839d88fe536e83c0af20fd8972235b7aedca6a33becf4`,来源版本固定为 `raw-file-v1:sha256:`,不再使用作品 revision+导入章数。 +- 当前状态:**DDL 已实现待 apply,真实记录未写,实跑未开始**。用户授权只能登记为 `research_only + offline_evaluation`,不得伪造 `licensed`;apply 与 INSERT 均不在本任务执行范围。 - `.claude/skills/llm/scripts/test_quota.py` 已统一到当前 `$24/6000` 契约:预算场景使用 `$24.5`,调用上限场景使用 `6000`,并增加策略常量断言。 -- 当前允许的结论是“冻结、变量控制、内容级泄露审计、候选结构门和安全摘要机制已通”;不能说细纲智能体或知识卡已通过。真实授权接入、detector/judge 双评和 430/489/550 实样仍待完成。 +- 当前允许的结论是“冻结、变量控制、内容级泄露审计、候选结构门、detector/judge 双评编排、安全摘要机制和授权快照代码合同已通”;不能说数据库表已应用、真实授权已写入、细纲智能体或知识卡已通过。真实授权接入和 430/489/550 实样仍待完成。 --- @@ -206,7 +208,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} ### 6.1 审查智能体 -审查使用现有 `detector`,但要新增“细纲候选”输入分支;它只看到冻结到 N 的 writer/planner 基线加检测增量,不看到目标章标准事实。检查: +审查使用现有 `detector`,但要新增“细纲候选”输入分支;它只看到匿名候选和冻结到 N 的公共 writer/planner 基线,不看到 arm、`cardInjection`、`cardManifest`、任何臂特有卡内容或目标章标准事实。卡注入合法性留在确定性预检。检查: - 细纲字段是否完整、事件顺序是否自洽; - 角色/势力/地点/能力是否违反 N 时点已知事实; @@ -214,7 +216,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} - 是否把 `unknown` 写成确定事实; - 是否出现候选自身引用未来来源的证据。 -审查只产报告,不改候选;高严重度阻断项不进入盲评。审查发现“卡里缺少目标新角色”时必须单列为设计发现,不把它误判成规划器读卡失败。 +审查只产报告,不改候选;高严重度阻断项不进入盲评。“卡里缺少目标新角色”需要目标章标准事实才能判断,由 judge/eval 侧单列,detector 不得判断或输出该类别。 ### 6.2 细纲专用评委 @@ -247,7 +249,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} | 假阴 | planner 以为卡已足够而少用公共大纲 | 三臂输入共用大纲,记录实际检索/引用卡 ID | | 假阳 | 未来事件/终态摘要泄露、错配卡只是额外文字、目标章信息参与召回 | 内容级泄露审计、placebo 臂、目标实体禁入选择器 | | 假阳 | 评委偏爱词面相似、参考细纲本身有误 | 结构化事实评分、两评委、proxy confidence、禁止正文文风维度 | -| 误归因 | 新角色无卡但某臂猜中、模型随机性、模型路由不一致 | 新角色缺卡单列;固定路由/预算;重复评审;不以 n=1 判决 | +| 误归因 | 新角色无卡但某臂猜中、模型随机性、模型路由不一致 | judge/eval 单列新角色缺卡;固定路由/预算;重复评审;不以 n=1 判决 | --- @@ -327,7 +329,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} - [x] 给 detector 增加细纲候选的检查对象、严重度和证据格式。 - [x] 给 judge 增加 `fine_outline_replay` rubric profile,明确不评正文文风和文笔。 -- [ ] 在真实 judge 编排中固定两评委顺序互换和匿名 arm;当前仅完成 rubric 稳定性校验合同。 +- [x] 在回放编排中固定两个独立 judge、匿名 arm 和第二评委候选顺序反转;fake runner 已覆盖双评、rubric 合同和稳定性失败关闭,真实样本尚未越过授权门。 - [x] 机械测试确保 rubric 不包含正文质量维度,且每个分数必须有证据字段。 ### Task 4:回放编排与最小首跑 @@ -341,7 +343,7 @@ arm ∈ {outline_only, outline_plus_cards, outline_plus_placebo_cards} - [x] 先用合成 fixture 完成一个机制 dry-run,并用 fake planner 验证 execute 链路;真实样本尚未越过授权门。 - [x] 编排器支持每个目标章 A/B/C 三臂,planner 使用相同身份段、功能段、预算和输出合同;尚未运行参考作品目标章。 -- [ ] 两个独立 judge 对每章候选做顺序互换盲评;detector 阻断样本不送 judge。 +- [x] 两个独立 judge 对每章候选做顺序互换盲评;detector 阻断样本不送 judge。当前由 fake runner 离线测试验证,尚无真实样本结果。 - [x] 运行时原始候选和标准事实摘要只放仓库外临时目录;安全报告只写摘要、定位、评分入口、哈希和失败类别。 - [x] 接入真实数据库只读适配 smoke;授权缺失时维持 `blocked_authorization`,不启动 planner。 - [ ] 产出真实卡增量矩阵和样本级假阴/假阳说明;不能用机制 smoke 的合成结果代替。