From a54a3a4e397dd818a62f3afd89e3b599fe75f6ea Mon Sep 17 00:00:00 2001 From: zizi Date: Sun, 19 Jul 2026 13:52:35 +0800 Subject: [PATCH] =?UTF-8?q?=E6=A1=86=E6=9E=B6:=20=E6=94=B6=E7=B4=A7?= =?UTF-8?q?=E5=9B=9E=E6=94=BE=E8=AF=84=E6=B5=8B=E5=89=8D=E7=BD=AE=E9=97=A8?= =?UTF-8?q?=E5=B9=B6=E8=A1=A5=E9=BD=90=E7=BC=96=E6=8E=92?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .claude/skills/replay-eval/SKILL.md | 8 + .../replay-eval/scripts/build_snapshot.py | 290 +++++++++++++- .../replay-eval/scripts/check_snapshot.py | 215 ++++++++++- .../skills/replay-eval/scripts/run_replay.py | 353 ++++++++++++++++++ .../scripts/test_build_snapshot.py | 82 +++- .../scripts/test_check_snapshot.py | 76 +++- .../replay-eval/scripts/test_run_replay.py | 92 +++++ .../replay-eval/scripts/write_report.py | 107 ++++++ 8 files changed, 1191 insertions(+), 32 deletions(-) create mode 100644 .claude/skills/replay-eval/scripts/run_replay.py create mode 100644 .claude/skills/replay-eval/scripts/test_run_replay.py create mode 100644 .claude/skills/replay-eval/scripts/write_report.py diff --git a/.claude/skills/replay-eval/SKILL.md b/.claude/skills/replay-eval/SKILL.md index 79475ad..e1a4402 100644 --- a/.claude/skills/replay-eval/SKILL.md +++ b/.claude/skills/replay-eval/SKILL.md @@ -36,3 +36,11 @@ disable-model-invocation: true - 原始候选、标准事实摘要和完整输入只能留在临时运行目录或 `/tmp`。 - 最终报告只允许评分、摘要、章节定位、失败类别和哈希。 - 禁止写入原书正文、完整目标细纲、完整 Prompt/Response、供应商原始响应、token、密钥或未脱敏授权资料。 + +## 编排入口 + +- `scripts/run_replay.py --mode dry_run`:只执行授权、来源、冻结和三臂 manifest 预检,不调用模型;这是首个机制 smoke 入口。 +- `scripts/run_replay.py --mode execute`:在全部前置门通过后,使用无工具、无会话持久化的本地 planner CLI 逐臂生成候选;`--output-dir` 必须位于仓库外的临时目录。 +- `scripts/write_report.py`:从 `run_result.json` 生成安全摘要;它不会读取候选正文,也不会把候选路径以外的原始响应写入报告。 + +真实作品运行前必须先从权威来源取得不可变授权快照。数据库没有该字段时,使用 dry-run 证明机制并保持 `blocked_authorization`,不得用本地配置或口头许可伪造放行。 diff --git a/.claude/skills/replay-eval/scripts/build_snapshot.py b/.claude/skills/replay-eval/scripts/build_snapshot.py index a11ad33..8880a96 100644 --- a/.claude/skills/replay-eval/scripts/build_snapshot.py +++ b/.claude/skills/replay-eval/scripts/build_snapshot.py @@ -25,21 +25,91 @@ TERMINAL_FIELDS = frozenset( "final_summary", "terminal_summary", "final_state", + "finalState", "current_state", + "currentState", "future_arc", + "futureArc", "future_plan", + "futurePlan", + "event_result", + "eventResult", + "relationship_plan", + "relationshipPlan", + "growth_arc", + "growthArc", + "terminalResult", "成长弧线", "当前态", "终态摘要", "终局状态", "未来弧线", "未来计划", + "事件结果", + "关系计划", + } +) + +SNAPSHOT_TOP_LEVEL_KEYS = frozenset( + { + "milestones", + "outlineWindows", + "cards", + "chapters", + "events", + "relations", + "relationships", + "stateRecords", + "state_records", + "facts", + } +) +CHAPTER_SCOPED_COLLECTIONS = frozenset( + { + "milestones", + "outlineWindows", + "chapters", + "events", + "relations", + "relationships", + "stateRecords", + "state_records", + "facts", + } +) + +FINAL_REPORT_ALLOWED_FIELDS = frozenset( + { + "status", + "runId", + "referenceWork", + "referenceWorkVersion", + "targetChapter", + "asOfChapter", + "snapshotVersion", + "evaluationSetVersion", + "strategyVersion", + "profile", + "arm", + "armComparison", + "candidateCount", + "scores", + "summary", + "chapterRefs", + "failureClass", + "failureReason", + "hashes", + "warnings", + "stability", + "proxyConfidence", + "goldUncertain", } ) FINAL_REPORT_FORBIDDEN_FIELDS = frozenset( { "raw", + "payload", "raw_text", "body", "content", @@ -52,6 +122,8 @@ FINAL_REPORT_FORBIDDEN_FIELDS = frozenset( "target_chapter_text", "prompt", "response", + "fullPrompt", + "fullResponse", } ) @@ -59,6 +131,15 @@ _INTEGER_RE = re.compile(r"^\s*(\d+)\s*$") _RANGE_RE = re.compile(r"^\s*(?:第\s*)?(\d+)\s*(?:-|–|—|~|至|到)\s*(?:第\s*)?(\d+)\s*(?:章)?\s*$") +def _normalized_key(key: Any) -> str: + """统一英文终态字段的大小写、下划线和短横线写法。""" + + return re.sub(r"[_-]", "", str(key)).lower() + + +TERMINAL_FIELD_NAMES = frozenset(_normalized_key(field) for field in TERMINAL_FIELDS) + + def normalize_chapter(value: Any) -> int | None: """只接受明确的正整数章号;不把 bool、浮点或模糊文本猜成章号。""" @@ -181,33 +262,85 @@ def filter_outline_windows( return kept, omitted -def remove_terminal_fields(value: Any) -> Any: - """递归删除不能直接作为 as_of 事实使用的终态/未来字段。""" +def remove_terminal_fields( + value: Any, + omitted_fields: list[dict[str, Any]] | None = None, + path: str = "$", +) -> Any: + """递归删除终态字段,并记录裁剪路径供审计。""" if isinstance(value, list): - return [remove_terminal_fields(item) for item in value] + return [ + remove_terminal_fields(item, omitted_fields, f"{path}[{index}]") + for index, item in enumerate(value) + ] if not isinstance(value, Mapping): return copy.deepcopy(value) projected: dict[str, Any] = {} for key, item in value.items(): - if str(key) in TERMINAL_FIELDS: + key_text = str(key) + if _normalized_key(key_text) in TERMINAL_FIELD_NAMES or key_text in TERMINAL_FIELDS: + if omitted_fields is not None: + omitted_fields.append({"path": f"{path}.{key_text}", "reason": "terminal_field_removed"}) continue - projected[str(key)] = remove_terminal_fields(item) + projected[key_text] = remove_terminal_fields(item, omitted_fields, f"{path}.{key_text}") return projected -def _validate_final_report(value: Any, path: str = "finalReport") -> None: - """禁止最终报告携带原书全文、完整响应或完整 prompt。""" +_DROP = object() + + +def _freeze_nested_chapter_records( + value: Any, + as_of: int, + omitted_sources: list[dict[str, Any]], + path: str = "$", +) -> Any: + """递归冻结所有带绝对章号的嵌套记录,未标明章号的显式记录不猜测。""" + + if isinstance(value, Mapping): + chapter_keys = {"chapter", "chapter_no", "order_no", "章", "章号", "from_order", "to_order"} + if chapter_keys.intersection(value): + bounds = normalize_chapter_range(_chapter_value(value)) + if bounds is None: + omitted_sources.append({"path": path, "reason": "missing_or_invalid_chapter", "recordHash": sha256_value(value)}) + return _DROP + if bounds[1] > as_of: + omitted_sources.append({"path": path, "reason": "future_or_crosses_as_of", "recordHash": sha256_value(value)}) + return _DROP + projected: dict[str, Any] = {} + for key, item in value.items(): + frozen = _freeze_nested_chapter_records(item, as_of, omitted_sources, f"{path}.{key}") + if frozen is not _DROP: + projected[str(key)] = frozen + return projected + + if isinstance(value, list): + projected_list: list[Any] = [] + for index, item in enumerate(value): + frozen = _freeze_nested_chapter_records(item, as_of, omitted_sources, f"{path}[{index}]") + if frozen is not _DROP: + projected_list.append(frozen) + return projected_list + + return copy.deepcopy(value) + + +def _validate_final_report(value: Any, path: str = "finalReport", top_level: bool = True) -> None: + """用顶层白名单和递归禁用字段保护最终报告边界。""" if isinstance(value, Mapping): for key, item in value.items(): - if str(key) in FINAL_REPORT_FORBIDDEN_FIELDS: - raise SnapshotError(f"{path}.{key} 不得进入最终报告") - _validate_final_report(item, f"{path}.{key}") + key_text = str(key) + if key_text in FINAL_REPORT_FORBIDDEN_FIELDS: + raise SnapshotError(f"{path}.{key_text} 不得进入最终报告") + if top_level and key_text not in FINAL_REPORT_ALLOWED_FIELDS: + raise SnapshotError(f"{path}.{key_text} 不是允许的最终报告字段") + _validate_final_report(item, f"{path}.{key_text}", False) elif isinstance(value, list): for index, item in enumerate(value): - _validate_final_report(item, f"{path}[{index}]") + _validate_final_report(item, f"{path}[{index}]", False) def _source_record(section: str, value: Any) -> dict[str, Any]: @@ -215,16 +348,15 @@ def _source_record(section: str, value: Any) -> dict[str, Any]: if isinstance(value, Mapping) and "payload" in value: source_id = str(value.get("sourceId") or section) - source_version = str(value.get("sourceVersion") or "unknown") + source_version = str(value.get("sourceVersion") or "") + if not source_id or not source_version: + raise SnapshotError(f"来源 {section} 缺少 sourceId/sourceVersion") payload = value["payload"] omitted_fields = list(value.get("omittedFields") or []) omitted_sources = list(value.get("omittedSources") or []) else: source_id = section - source_version = "unknown" - payload = value - omitted_fields = [] - omitted_sources = [] + raise SnapshotError(f"来源 {section} 必须显式提供 sourceId/sourceVersion/payload") serialized = _safe_json(payload) return { @@ -241,7 +373,14 @@ def _source_record(section: str, value: Any) -> dict[str, Any]: def build_snapshot_manifest( *, as_of: int, + target_chapter: int, snapshot_version: str, + reference_work: Mapping[str, Any], + evaluation_set_version: str, + strategy_version: str, + authorization_snapshot: Mapping[str, Any], + run_permissions: Mapping[str, Any], + arm_config: Mapping[str, Any], sections: Mapping[str, Any], omitted_fields: Sequence[Any] | None = None, omitted_sources: Sequence[Any] | None = None, @@ -252,14 +391,71 @@ def build_snapshot_manifest( normalized_as_of = normalize_chapter(as_of) if normalized_as_of is None: raise SnapshotError("as_of 必须是正整数章号") + normalized_target = normalize_chapter(target_chapter) + if normalized_target != normalized_as_of + 1: + raise SnapshotError("target_chapter 必须等于 as_of+1") if not snapshot_version.strip(): raise SnapshotError("snapshot_version 不能为空") + if not isinstance(reference_work, Mapping) or not reference_work.get("id") or not reference_work.get("version"): + raise SnapshotError("reference_work 必须包含 id/version") + if not evaluation_set_version.strip() or not strategy_version.strip(): + raise SnapshotError("evaluation_set_version/strategy_version 不能为空") + if not isinstance(authorization_snapshot, Mapping): + raise SnapshotError("authorization_snapshot 必须是对象") + authorization_required = ( + "id", + "version", + "immutable", + "sourceVersion", + "sourceStatus", + "allowedPurpose", + "checkedAt", + ) + if any(not authorization_snapshot.get(field) for field in authorization_required): + raise SnapshotError("authorization_snapshot 缺少必填字段") + if authorization_snapshot.get("immutable") is not True: + raise SnapshotError("authorization_snapshot 必须是不可变快照") + if not authorization_snapshot.get("expiresAt") and not authorization_snapshot.get("revalidationAt"): + raise SnapshotError("authorization_snapshot 缺少过期或重验时间") + if not isinstance(run_permissions, Mapping) or not run_permissions.get("purpose"): + raise SnapshotError("run_permissions 必须包含 purpose") + if not isinstance(arm_config, Mapping) or not arm_config.get("arms"): + raise SnapshotError("arm_config 必须包含 arms") + if set(arm_config["arms"]) != { + "outline_only", + "outline_plus_cards", + "outline_plus_placebo_cards", + }: + raise SnapshotError("arm_config 必须是完整三臂") if final_report is not None: _validate_final_report(final_report) + auth_safe = { + "id": str(authorization_snapshot["id"]), + "version": str(authorization_snapshot.get("version") or ""), + "sha256": sha256_value(authorization_snapshot), + "checkedAt": authorization_snapshot.get("checkedAt"), + } + if not auth_safe["version"]: + raise SnapshotError("authorization_snapshot 必须包含 version") + manifest: dict[str, Any] = { "snapshotVersion": snapshot_version, "asOfChapter": normalized_as_of, + "targetChapter": normalized_target, + "referenceWork": {"id": str(reference_work["id"]), "version": str(reference_work["version"])}, + "evaluationSetVersion": evaluation_set_version, + "strategyVersion": strategy_version, + "authorization": auth_safe, + "runPermissions": { + "purpose": str(run_permissions["purpose"]), + "mode": str(run_permissions.get("mode") or "offline"), + "sha256": sha256_value(run_permissions), + }, + "armConfig": { + "arms": sorted(str(arm) for arm in arm_config["arms"]), + "sha256": sha256_value(arm_config), + }, "sections": { str(section): _source_record(str(section), value) for section, value in sorted(sections.items(), key=lambda pair: str(pair[0])) @@ -272,11 +468,34 @@ def build_snapshot_manifest( return manifest -def build_snapshot(data: Mapping[str, Any], as_of: int, snapshot_version: str) -> dict[str, Any]: +def build_snapshot( + data: Mapping[str, Any], + as_of: int, + snapshot_version: str, + *, + target_chapter: int | None = None, + manifest_metadata: Mapping[str, Any] | None = None, +) -> dict[str, Any]: """从最小 JSON 输入生成冻结后的结构化快照和 manifest。""" + if not isinstance(data, Mapping): + raise SnapshotError("快照输入必须是对象") + unknown_sections = sorted(set(data) - SNAPSHOT_TOP_LEVEL_KEYS) + if unknown_sections: + raise SnapshotError(f"快照包含未登记顶层分区: {','.join(str(item) for item in unknown_sections)}") safe = copy.deepcopy(dict(data)) all_omitted: list[dict[str, Any]] = [] + metadata = dict(manifest_metadata or {}) + normalized_as_of = normalize_chapter(as_of) + if normalized_as_of is None: + raise SnapshotError("as_of 必须是正整数章号") + normalized_target = normalize_chapter(target_chapter or metadata.get("targetChapter")) + if normalized_target != normalized_as_of + 1: + raise SnapshotError("target_chapter 必须等于 as_of+1") + + for section in CHAPTER_SCOPED_COLLECTIONS: + if section in safe and not isinstance(safe[section], list): + raise SnapshotError(f"分区 {section} 必须是数组") if isinstance(safe.get("milestones"), list): milestones, omitted = filter_milestones(safe["milestones"], as_of) @@ -309,11 +528,33 @@ def build_snapshot(data: Mapping[str, Any], as_of: int, snapshot_version: str) - frozen_cards.append(projected) safe["cards"] = frozen_cards - safe = remove_terminal_fields(safe) + safe = _freeze_nested_chapter_records(safe, normalized_as_of, all_omitted) + if safe is _DROP: + raise SnapshotError("快照根对象不能被冻结过滤") + omitted_fields: list[dict[str, Any]] = [] + safe = remove_terminal_fields(safe, omitted_fields) + required_metadata = { + "referenceWork": metadata.get("referenceWork"), + "evaluationSetVersion": metadata.get("evaluationSetVersion"), + "strategyVersion": metadata.get("strategyVersion"), + "authorizationSnapshot": metadata.get("authorizationSnapshot"), + "runPermissions": metadata.get("runPermissions"), + "armConfig": metadata.get("armConfig"), + } + if any(value is None for value in required_metadata.values()): + raise SnapshotError("缺少完整回放 manifest 元数据") manifest = build_snapshot_manifest( - as_of=as_of, + as_of=normalized_as_of, + target_chapter=normalized_target, snapshot_version=snapshot_version, + reference_work=required_metadata["referenceWork"], + evaluation_set_version=required_metadata["evaluationSetVersion"], + strategy_version=required_metadata["strategyVersion"], + authorization_snapshot=required_metadata["authorizationSnapshot"], + run_permissions=required_metadata["runPermissions"], + arm_config=required_metadata["armConfig"], sections={"snapshot": {"sourceId": "frozen-snapshot", "sourceVersion": snapshot_version, "payload": safe}}, + omitted_fields=omitted_fields, omitted_sources=all_omitted, ) return {"snapshot": safe, "manifest": manifest} @@ -324,6 +565,8 @@ def _parse_args() -> argparse.Namespace: parser.add_argument("--input", type=Path, required=True, help="结构化 JSON 输入") parser.add_argument("--output", type=Path, required=True, help="输出 JSON 路径") parser.add_argument("--as-of", type=int, required=True, dest="as_of") + parser.add_argument("--target-chapter", type=int, required=True, dest="target_chapter") + parser.add_argument("--metadata", type=Path, required=True, help="回放 manifest 元数据 JSON") parser.add_argument("--snapshot-version", default="next_fine_outline_replay_v0") return parser.parse_args() @@ -331,7 +574,14 @@ def _parse_args() -> argparse.Namespace: def main() -> int: args = _parse_args() data = json.loads(args.input.read_text(encoding="utf-8")) - result = build_snapshot(data, args.as_of, args.snapshot_version) + metadata = json.loads(args.metadata.read_text(encoding="utf-8")) + result = build_snapshot( + data, + args.as_of, + args.snapshot_version, + target_chapter=args.target_chapter, + manifest_metadata=metadata, + ) args.output.write_text(_safe_json(result) + "\n", encoding="utf-8") return 0 diff --git a/.claude/skills/replay-eval/scripts/check_snapshot.py b/.claude/skills/replay-eval/scripts/check_snapshot.py index fd3b0e7..023ba4f 100644 --- a/.claude/skills/replay-eval/scripts/check_snapshot.py +++ b/.claude/skills/replay-eval/scripts/check_snapshot.py @@ -25,6 +25,7 @@ FORBIDDEN_SOURCE_STATUSES = frozenset( {"revoked", "delisted", "recalled", "blocked", "owner_missing", "unauthorized"} ) ALLOWED_SOURCE_STATUSES = frozenset({"active", "approved", "authorized", "licensed"}) +ALLOWED_COPYRIGHT_STATUSES = frozenset({"active", "approved", "authorized", "licensed", "owned"}) FORBIDDEN_COPYRIGHT_STATUSES = frozenset( {"unauthorized", "unlicensed", "revoked", "expired", "blocked"} ) @@ -34,6 +35,8 @@ CARD_KEYS = frozenset( "card", "cards", "cardInjection", + "cardInjectionSha256", + "cardInjectionCount", "cardSection", "cardSections", "cardManifest", @@ -43,6 +46,8 @@ CARD_KEYS = frozenset( "l2-card", "l2-placebo", "card_injection", + "card_injection_sha256", + "card_injection_count", "card_section", "card_sections", "card_manifest", @@ -65,6 +70,17 @@ REQUIRED_CANDIDATE_FIELDS = ( "unknowns", "assumptions", ) +ALLOWED_CANDIDATE_FIELDS = frozenset((*REQUIRED_CANDIDATE_FIELDS, "sourceRefs")) +EVENT_FIELDS = { + "id": str, + "order": int, + "event": str, + "participants": list, + "trigger": str, + "resultDirection": str, +} +ENTITY_FIELDS = {"name": str, "type": str, "role": str} +FORESHADOWING_FIELDS = {"action": str, "subject": str, "evidence": str} FORBIDDEN_CANDIDATE_FIELDS = frozenset( {"body", "raw", "rawText", "content", "正文", "原文", "正文全文", "完整目标细纲"} ) @@ -90,6 +106,26 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An if not isinstance(snapshot, Mapping) or not snapshot: return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照为空"]) + snapshot_required = ( + "id", + "version", + "immutable", + "sourceVersion", + "sourceStatus", + "allowedPurpose", + "checkedAt", + ) + missing_snapshot = [field for field in snapshot_required if not snapshot.get(field)] + if missing_snapshot: + return _result( + STATUS_BLOCKED_AUTHORIZATION, + [f"授权快照缺少字段: {','.join(missing_snapshot)}"], + ) + if snapshot.get("immutable") is not True: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照不是不可变快照"]) + if not snapshot.get("expiresAt") and not snapshot.get("revalidationAt"): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照缺少过期或重验时间"]) + source_status = str(_field(authorization, "sourceStatus", "source_status") or "").lower() if not source_status: return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceStatus"]) @@ -101,8 +137,20 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An copyright_status = str( _field(authorization, "copyrightStatus", "copyright_status") or "" ).lower() + if not copyright_status: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 copyrightStatus"]) if copyright_status in FORBIDDEN_COPYRIGHT_STATUSES: return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态禁止评测: {copyright_status}"]) + if copyright_status not in ALLOWED_COPYRIGHT_STATUSES: + return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态未登记,拒绝评测: {copyright_status}"]) + + source_version = str(_field(authorization, "sourceVersion", "source_version") or "") + if not source_version: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceVersion"]) + if str(snapshot.get("sourceVersion")) != source_version: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 sourceVersion 不一致"]) + if str(snapshot["sourceStatus"]).lower() != source_status: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 sourceStatus 不一致"]) allowed = _field(authorization, "allowedPurpose", "allowed_purpose") if isinstance(allowed, str): @@ -111,6 +159,13 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 不是用途列表"]) if "offline_evaluation" not in allowed: return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 不包含 offline_evaluation"]) + snapshot_allowed = snapshot.get("allowedPurpose", snapshot.get("allowed_purpose")) + if not isinstance(snapshot_allowed, Sequence) or isinstance(snapshot_allowed, (str, bytes)): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照缺少 allowedPurpose"]) + if "offline_evaluation" not in snapshot_allowed: + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照用途不包含 offline_evaluation"]) + if set(snapshot_allowed) != set(allowed): + return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照用途与运行用途不一致"]) return _result(STATUS_READY) @@ -120,11 +175,16 @@ def check_target_sources(target_chapter: int, sources: Sequence[Mapping[str, Any target = normalize_chapter(target_chapter) if target is None: return _result(STATUS_INVALID_SNAPSHOT, ["目标章号无效"]) + if not isinstance(sources, Sequence) or isinstance(sources, (str, bytes)) or not sources: + return _result(STATUS_INVALID_SNAPSHOT, ["缺少规划来源"]) errors: list[str] = [] for index, source in enumerate(sources): if not isinstance(source, Mapping): errors.append(f"source[{index}] 不是对象") continue + if not source.get("sourceId") or not source.get("sourceVersion"): + errors.append(f"source[{index}] 缺少 sourceId/sourceVersion") + continue value = _field(source, "chapter", "chapterNo", "chapter_no", "章", "章号") bounds = normalize_chapter_range(value) if bounds is None: @@ -139,6 +199,11 @@ def check_target_sources(target_chapter: int, sources: Sequence[Mapping[str, Any if bounds is None and "from_order" in source: value = f"{source.get('from_order')}-{source.get('to_order')}" bounds = normalize_chapter_range(value) + if bounds is None: + scope = str(source.get("scope") or "chapter").lower() + if scope not in {"work", "authorization", "metadata"}: + errors.append(f"source[{index}] 缺少可验证绝对章号或完整区间") + continue if bounds is not None and bounds[1] >= target: errors.append(f"source[{index}] 包含目标章或未来章: {bounds[0]}-{bounds[1]}") return _result(STATUS_TARGET_SOURCE_FORBIDDEN if errors else STATUS_READY, errors) @@ -160,13 +225,50 @@ def _without_card_fields(value: Any) -> Any: return result +def _canonical_hash(value: Any) -> str: + """用 JSON 类型保真的规范序列化比较公共区。""" + + import hashlib + + encoded = json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8") + return hashlib.sha256(encoded).hexdigest() + + +def _check_arm_semantics(name: str, manifest: Mapping[str, Any]) -> list[str]: + """校验生产三臂的无卡/正确卡/placebo 语义。""" + + count = manifest.get("cardInjectionCount", 0) + source_ids = manifest.get("cardSourceIds", []) + strategy = str(manifest.get("cardStrategy") or "none") + has_payload = bool(manifest.get("cardInjection")) + if name == "outline_only": + if has_payload or count not in (0, None) or source_ids or strategy not in {"none", ""}: + return ["outline_only 必须不含卡注入"] + return [] + if name not in {"outline_plus_cards", "outline_plus_placebo_cards"}: + return [f"未登记评测臂: {name}"] + if has_payload or not isinstance(count, int) or count <= 0 or not source_ids: + return [f"{name} 必须包含带来源的卡注入元数据"] + expected_strategy = "correct" if name == "outline_plus_cards" else "placebo" + if strategy != expected_strategy: + return [f"{name} 的 cardStrategy 必须是 {expected_strategy}"] + return [] + + def check_arm_manifests( manifests: Mapping[str, Mapping[str, Any]], required_arms: Sequence[str] | None = None, + *, + as_of_chapter: int | None = None, + target_chapter: int | None = None, + enforce_semantics: bool = False, + smoke: bool = False, ) -> dict[str, Any]: """确保三臂除卡注入区外完全一致。""" required = set(required_arms or ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards")) + if not smoke and required != {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"}: + return _result(STATUS_INVALID_ARM_DIFF, ["生产回放必须使用完整三臂;双臂只允许 smoke"]) actual = set(manifests) missing = sorted(required - actual) if missing: @@ -177,9 +279,34 @@ def check_arm_manifests( if any(not isinstance(manifest, Mapping) for manifest in manifests.values()): return _result(STATUS_INVALID_ARM_DIFF, ["评测臂 manifest 必须是对象"]) + if as_of_chapter is not None or target_chapter is not None: + expected_as_of = normalize_chapter(as_of_chapter) + expected_target = normalize_chapter(target_chapter) + if expected_as_of is None or expected_target != expected_as_of + 1: + return _result(STATUS_INVALID_SNAPSHOT, ["as_of/target 不是连续章节"]) + for name in sorted(required): + manifest = manifests[name] + if normalize_chapter(manifest.get("asOfChapter")) != expected_as_of: + return _result(STATUS_INVALID_SNAPSHOT, [f"{name} 的 asOfChapter 不一致"]) + if normalize_chapter(manifest.get("targetChapter")) != expected_target: + return _result(STATUS_INVALID_SNAPSHOT, [f"{name} 的 targetChapter 不一致"]) + + if enforce_semantics: + semantic_errors = [ + f"{name}: {error}" + for name in sorted(required) + for error in _check_arm_semantics(name, manifests[name]) + ] + if semantic_errors: + return _result(STATUS_INVALID_ARM_DIFF, semantic_errors) + names = sorted(required) baseline = _without_card_fields(manifests[names[0]]) - differences = [name for name in names[1:] if _without_card_fields(manifests[name]) != baseline] + baseline_hash = _canonical_hash(baseline) + differences = [ + name for name in names[1:] + if _canonical_hash(_without_card_fields(manifests[name])) != baseline_hash + ] if differences: return _result(STATUS_INVALID_ARM_DIFF, [f"公共输入区不一致: {','.join(differences)}"]) return _result(STATUS_READY) @@ -188,7 +315,7 @@ def check_arm_manifests( def check_candidate_output( candidate: Mapping[str, Any], target_chapter: int, - source_catalog: Sequence[Mapping[str, Any]] = (), + source_catalog: Sequence[Mapping[str, Any]] | None = None, ) -> dict[str, Any]: """校验细纲候选的结构边界,不判断内容是否命中标准答案。""" @@ -198,6 +325,9 @@ def check_candidate_output( if expected_target is None: return _result(STATUS_INVALID_SNAPSHOT, ["目标章号无效"]) errors = [f"缺少字段: {field}" for field in REQUIRED_CANDIDATE_FIELDS if field not in candidate] + unexpected = sorted(set(candidate) - ALLOWED_CANDIDATE_FIELDS) + if unexpected: + errors.append(f"候选包含未登记字段: {','.join(unexpected)}") forbidden = sorted(set(candidate) & FORBIDDEN_CANDIDATE_FIELDS) if forbidden: errors.append(f"候选包含正文/原文字段: {','.join(forbidden)}") @@ -216,22 +346,73 @@ def check_candidate_output( errors.append("keyEvents 每项必须是对象") if any(isinstance(item, Mapping) and not item.get("id") for item in events): errors.append("keyEvents 每项必须有 id") + for index, event in enumerate(events): + if not isinstance(event, Mapping): + continue + unexpected_event_fields = sorted(set(event) - set(EVENT_FIELDS)) + if unexpected_event_fields: + errors.append(f"keyEvents[{index}] 包含未登记字段: {','.join(unexpected_event_fields)}") + for field, expected_type in EVENT_FIELDS.items(): + if field not in event: + errors.append(f"keyEvents[{index}] 缺少字段: {field}") + elif expected_type is int and (isinstance(event[field], bool) or not isinstance(event[field], int)): + errors.append(f"keyEvents[{index}].{field} 必须是整数") + elif expected_type is not int and not isinstance(event[field], expected_type): + errors.append(f"keyEvents[{index}].{field} 类型错误") + entities = candidate.get("entities") + if isinstance(entities, list): + for index, entity in enumerate(entities): + if not isinstance(entity, Mapping): + errors.append(f"entities[{index}] 必须是对象") + continue + for field, expected_type in ENTITY_FIELDS.items(): + if field not in entity: + errors.append(f"entities[{index}] 缺少字段: {field}") + elif not isinstance(entity[field], expected_type): + errors.append(f"entities[{index}].{field} 类型错误") + unexpected_entity_fields = sorted(set(entity) - set(ENTITY_FIELDS)) + if unexpected_entity_fields: + errors.append(f"entities[{index}] 包含未登记字段: {','.join(unexpected_entity_fields)}") + foreshadowing = candidate.get("foreshadowing") + if isinstance(foreshadowing, list): + for index, item in enumerate(foreshadowing): + if not isinstance(item, Mapping): + errors.append(f"foreshadowing[{index}] 必须是对象") + continue + for field, expected_type in FORESHADOWING_FIELDS.items(): + if field not in item: + errors.append(f"foreshadowing[{index}] 缺少字段: {field}") + elif not isinstance(item[field], expected_type): + errors.append(f"foreshadowing[{index}].{field} 类型错误") + unexpected_foreshadowing_fields = sorted(set(item) - set(FORESHADOWING_FIELDS)) + if unexpected_foreshadowing_fields: + errors.append( + f"foreshadowing[{index}] 包含未登记字段: {','.join(unexpected_foreshadowing_fields)}" + ) + for field in ("stateChanges", "unknowns", "assumptions"): + values = candidate.get(field) + if isinstance(values, list) and any(not isinstance(item, str) for item in values): + errors.append(f"{field} 每项必须是字符串") source_errors: list[str] = [] source_refs = candidate.get("sourceRefs") if source_refs is not None: if not isinstance(source_refs, list): errors.append("sourceRefs 必须是数组") else: + if source_catalog is None: + return _result(STATUS_SCHEMA_INVALID, ["sourceRefs 校验缺少冻结来源目录"]) catalog = { str(source.get("sourceId")): source for source in source_catalog if isinstance(source, Mapping) and source.get("sourceId") } for index, reference in enumerate(source_refs): - source = reference if isinstance(reference, Mapping) else catalog.get(str(reference)) + if not isinstance(reference, str): + errors.append(f"sourceRefs[{index}] 必须是 sourceId 字符串") + continue + source = catalog.get(reference) if source is None: - if source_catalog: - errors.append(f"sourceRefs[{index}] 未登记来源") + errors.append(f"sourceRefs[{index}] 未登记来源") continue source_value = _field( source, @@ -247,7 +428,11 @@ def check_candidate_output( bounds = normalize_chapter_range(source_value) if bounds is None and "from_order" in source: bounds = normalize_chapter_range(f"{source.get('from_order')}-{source.get('to_order')}") - if bounds is not None and bounds[1] >= expected_target: + if bounds is None: + scope = str(source.get("scope") or "chapter").lower() + if scope not in {"work", "authorization", "metadata"}: + source_errors.append(f"sourceRefs[{index}] 缺少可验证绝对章号或完整区间") + elif bounds[1] >= expected_target: source_errors.append( f"sourceRefs[{index}] 包含目标章或未来章: {bounds[0]}-{bounds[1]}" ) @@ -264,17 +449,31 @@ def check_candidate_output( def check_replay( *, authorization: Mapping[str, Any] | None, + as_of_chapter: int, target_chapter: int, planner_sources: Sequence[Mapping[str, Any]], arm_manifests: Mapping[str, Mapping[str, Any]], required_arms: Sequence[str] | None = None, + smoke: bool = False, ) -> dict[str, Any]: """执行回放前置门,任何一项失败都不允许进入模型调用。""" + as_of = normalize_chapter(as_of_chapter) + target = normalize_chapter(target_chapter) + if as_of is None or target != as_of + 1: + return _result(STATUS_INVALID_SNAPSHOT, ["as_of/target 不是连续章节"]) + checks = [ check_authorization(authorization), check_target_sources(target_chapter, planner_sources), - check_arm_manifests(arm_manifests, required_arms), + check_arm_manifests( + arm_manifests, + required_arms, + as_of_chapter=as_of, + target_chapter=target, + enforce_semantics=True, + smoke=smoke, + ), ] failures = [check for check in checks if not check["ok"]] if failures: @@ -291,6 +490,7 @@ def _parse_args() -> argparse.Namespace: parser.add_argument("--authorization", type=Path, required=True) parser.add_argument("--sources", type=Path, required=True) parser.add_argument("--arms", type=Path, required=True) + parser.add_argument("--as-of-chapter", type=int, required=True) parser.add_argument("--target-chapter", type=int, required=True) parser.add_argument("--output", type=Path, required=True) return parser.parse_args() @@ -303,6 +503,7 @@ def main() -> int: arms = json.loads(args.arms.read_text(encoding="utf-8")) result = check_replay( authorization=authorization, + as_of_chapter=args.as_of_chapter, target_chapter=args.target_chapter, planner_sources=sources, arm_manifests=arms, diff --git a/.claude/skills/replay-eval/scripts/run_replay.py b/.claude/skills/replay-eval/scripts/run_replay.py new file mode 100644 index 0000000..ec23eb1 --- /dev/null +++ b/.claude/skills/replay-eval/scripts/run_replay.py @@ -0,0 +1,353 @@ +#!/usr/bin/env python3 +"""细纲回放编排器。 + +默认只做 dry-run。真实模式显式调用本地 Claude CLI,并将原始候选限定在运行目录; +最终结果只返回状态、哈希、评分入口和失败类别,不把原始 prompt/response 带出运行目录。 +""" + +from __future__ import annotations + +import argparse +import json +import re +import subprocess +from pathlib import Path +from typing import Any, Mapping + +from build_snapshot import build_snapshot, normalize_chapter, sha256_value +from check_snapshot import ( + STATUS_READY, + check_candidate_output, + check_replay, +) + + +REQUIRED_ARMS = ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards") +REPO_ROOT = Path(__file__).resolve().parents[4] +SKILL_PATH = REPO_ROOT / ".claude/skills/fine-outline/SKILL.md" +PLANNER_PATH = REPO_ROOT / ".claude/agents/planner.md" + + +class ReplayRunError(ValueError): + """回放配置不符合运行边界。""" + + +def _read_json(path: Path) -> Any: + return json.loads(path.read_text(encoding="utf-8")) + + +def _safe_json(value: Any) -> str: + return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")) + + +def _require_mapping(config: Mapping[str, Any], key: str) -> Mapping[str, Any]: + value = config.get(key) + if not isinstance(value, Mapping): + raise ReplayRunError(f"配置缺少对象字段: {key}") + return value + + +def _build_arm_manifest( + *, + name: str, + arm: Mapping[str, Any], + common_input: Mapping[str, Any], + as_of: int, + target: int, + snapshot_version: str, +) -> dict[str, Any]: + cards = arm.get("cards", []) + if not isinstance(cards, list): + raise ReplayRunError(f"{name}.cards 必须是数组") + source_ids = arm.get("cardSourceIds", []) + if not isinstance(source_ids, list): + raise ReplayRunError(f"{name}.cardSourceIds 必须是数组") + strategy = str(arm.get("cardStrategy") or ("none" if name == "outline_only" else "correct")) + return { + "arm": name, + "snapshotVersion": snapshot_version, + "asOfChapter": as_of, + "targetChapter": target, + "commonInputSha256": sha256_value(common_input), + "cardInjectionSha256": sha256_value(cards), + "cardInjectionCount": len(cards), + "cardSourceIds": [str(item) for item in source_ids], + "cardStrategy": strategy, + } + + +def _extract_candidate(output: str) -> Mapping[str, Any]: + """兼容 Claude JSON 外壳和代码围栏,但不把原始文本返回调用方。""" + + outer: Any = output + try: + outer = json.loads(output) + except json.JSONDecodeError: + pass + if isinstance(outer, Mapping) and isinstance(outer.get("result"), str): + outer = outer["result"] + if isinstance(outer, str): + match = re.search(r"```(?:json)?\s*(\{.*\})\s*```", outer, re.DOTALL) + outer = match.group(1) if match else outer.strip() + outer = json.loads(outer) + if not isinstance(outer, Mapping): + raise ReplayRunError("planner 输出不是 JSON 对象") + return outer + + +def _planner_prompt( + *, + target: int, + as_of: int, + snapshot: Mapping[str, Any], + common_input: Mapping[str, Any], + cards: list[Any], +) -> str: + """只把冻结后的结构化资料和功能合同送入 planner。""" + + skill = SKILL_PATH.read_text(encoding="utf-8") + identity = PLANNER_PATH.read_text(encoding="utf-8") + context = { + "targetChapter": target, + "asOfChapter": as_of, + "frozenSnapshot": snapshot, + "commonContext": common_input, + "cardInjection": cards, + } + return "\n".join( + [ + "这是 next_fine_outline_replay_v0 的离线规划任务。只输出一个 JSON 对象,不要 Markdown、正文或解释。", + "你不能调用工具,也不能读取仓库、目标章或任何未列入下面 JSON 的资料。", + "严格执行 fine-outline 合同;目标章号必须保持不变,未知内容写入 unknowns/assumptions。", + "规划上下文冻结到 as_of;卡只是事实索引和补充,不得替代公共大纲与叙事现在时。", + "--- planner identity ---", + identity, + "--- fine-outline skill ---", + skill, + "--- frozen input ---", + _safe_json(context), + "--- output contract ---", + _safe_json( + { + "targetChapter": target, + "requiredFields": [ + "chapterGoal", + "keyEvents", + "entities", + "foreshadowing", + "stateChanges", + "hook", + "unknowns", + "assumptions", + ], + "eventFields": ["id", "order", "event", "participants", "trigger", "resultDirection"], + "entityFields": ["name", "type", "role"], + "foreshadowingFields": ["action", "subject", "evidence"], + "sourceRefs": "可选;只能引用冻结来源 ID", + } + ), + ] + ) + + +def _invoke_planner( + *, + prompt: str, + planner_bin: str, + model: str, + output_path: Path, + max_budget_usd: float, +) -> Mapping[str, Any]: + """用无工具、无会话持久化的 Claude print 模式运行 planner。""" + + command = [ + planner_bin, + "-p", + "--agent", + "planner", + "--model", + model, + "--tools", + "", + "--no-session-persistence", + "--output-format", + "json", + "--max-budget-usd", + str(max_budget_usd), + "--append-system-prompt", + "本次是严格离线回放;不要调用任何工具,不要读取文件,不要输出 JSON 以外内容。", + prompt, + ] + completed = subprocess.run(command, text=True, capture_output=True, check=False) + output_path.write_text(completed.stdout, encoding="utf-8") + if completed.returncode != 0: + raise ReplayRunError(f"planner 调用失败,退出码={completed.returncode}") + return _extract_candidate(completed.stdout) + + +def run_replay( + config: Mapping[str, Any], + output_dir: Path, + *, + mode: str = "dry_run", + planner_bin: str = "claude", + model: str = "opus", + max_budget_usd: float = 1.0, +) -> dict[str, Any]: + """执行一次单目标三臂回放;任何前置门失败都不调用模型。""" + + if mode not in {"dry_run", "execute"}: + raise ReplayRunError("mode 只能是 dry_run 或 execute") + output_dir = output_dir.resolve() + if output_dir.is_relative_to(REPO_ROOT.resolve()): + raise ReplayRunError("原始候选运行目录不得位于仓库内") + output_dir.mkdir(parents=True, exist_ok=True) + + snapshot_config = _require_mapping(config, "snapshot") + as_of = normalize_chapter(snapshot_config.get("asOfChapter")) + target = normalize_chapter(config.get("targetChapter")) + snapshot_version = str(snapshot_config.get("snapshotVersion") or "") + if as_of is None or target is None: + raise ReplayRunError("as_of/target 必须是明确正整数") + if target != as_of + 1: + raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1") + reference_work = _require_mapping(config, "referenceWork") + authorization = _require_mapping(config, "authorization") + sources = config.get("sources", []) + if not isinstance(sources, list): + raise ReplayRunError("sources 必须是数组") + common_input = _require_mapping(config, "commonContext") + arms = _require_mapping(config, "arms") + if set(arms) != set(REQUIRED_ARMS): + raise ReplayRunError("生产回放必须精确配置三臂") + + arm_manifests = { + name: _build_arm_manifest( + name=name, + arm=_require_mapping(arms, name), + common_input=common_input, + as_of=as_of, + target=target, + snapshot_version=snapshot_version, + ) + for name in REQUIRED_ARMS + } + preflight = check_replay( + authorization=authorization, + as_of_chapter=as_of, + target_chapter=target, + planner_sources=sources, + arm_manifests=arm_manifests, + ) + result: dict[str, Any] = { + "runId": str(config.get("runId") or "unassigned"), + "mode": mode, + "status": preflight["status"], + "ok": preflight["ok"], + "referenceWork": str(reference_work.get("id") or ""), + "referenceWorkVersion": str(reference_work.get("version") or ""), + "asOfChapter": as_of, + "targetChapter": target, + "snapshotVersion": snapshot_version, + "preflight": {"status": preflight["status"], "errors": preflight["errors"], "warnings": preflight["warnings"]}, + "arms": arm_manifests, + "results": {}, + } + (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + if not preflight["ok"]: + return result + + metadata = { + "targetChapter": target, + "referenceWork": reference_work, + "evaluationSetVersion": config.get("evaluationSetVersion", "unregistered"), + "strategyVersion": config.get("strategyVersion", "unregistered"), + "authorizationSnapshot": authorization["authorizationSnapshot"], + "runPermissions": config.get("runPermissions", {"purpose": "offline_evaluation", "mode": mode}), + "armConfig": {"arms": list(REQUIRED_ARMS)}, + } + snapshot_data = snapshot_config.get("data", {}) + frozen = build_snapshot( + snapshot_data, + as_of, + snapshot_version, + target_chapter=target, + manifest_metadata=metadata, + ) + (output_dir / "snapshot_manifest.json").write_text(_safe_json(frozen["manifest"]) + "\n", encoding="utf-8") + + if mode == "dry_run": + result["status"] = STATUS_READY + result["ok"] = True + result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"] + (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + return result + + for name in REQUIRED_ARMS: + arm = _require_mapping(arms, name) + cards = arm.get("cards", []) + raw_path = output_dir / f"planner_{name}.raw.json" + prompt = _planner_prompt( + target=target, + as_of=as_of, + snapshot=frozen["snapshot"], + common_input=common_input, + cards=cards, + ) + try: + candidate = _invoke_planner( + prompt=prompt, + planner_bin=planner_bin, + model=model, + output_path=raw_path, + max_budget_usd=max_budget_usd, + ) + candidate_path = output_dir / f"candidate_{name}.json" + candidate_path.write_text(_safe_json(candidate) + "\n", encoding="utf-8") + schema = check_candidate_output(candidate, target, sources) + result["results"][name] = { + "status": schema["status"], + "ok": schema["ok"], + "errors": schema["errors"], + "candidateSha256": sha256_value(candidate), + "candidatePath": str(candidate_path), + } + except (ReplayRunError, json.JSONDecodeError) as error: + result["results"][name] = { + "status": "planner_output_invalid", + "ok": False, + "errors": [str(error)], + "rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None, + } + result["status"] = "completed" if all(item["ok"] for item in result["results"].values()) else "candidate_blocked" + result["ok"] = result["status"] == "completed" + (output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8") + return result + + +def _parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="运行 next_fine_outline_replay_v0") + parser.add_argument("--config", type=Path, required=True) + parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--mode", choices=("dry_run", "execute"), default="dry_run") + parser.add_argument("--planner-bin", default="claude") + parser.add_argument("--model", default="opus") + parser.add_argument("--max-budget-usd", type=float, default=1.0) + return parser.parse_args() + + +def main() -> int: + args = _parse_args() + result = run_replay( + _read_json(args.config), + args.output_dir, + mode=args.mode, + planner_bin=args.planner_bin, + model=args.model, + max_budget_usd=args.max_budget_usd, + ) + return 0 if result["ok"] else 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.claude/skills/replay-eval/scripts/test_build_snapshot.py b/.claude/skills/replay-eval/scripts/test_build_snapshot.py index 03e855e..04182c5 100644 --- a/.claude/skills/replay-eval/scripts/test_build_snapshot.py +++ b/.claude/skills/replay-eval/scripts/test_build_snapshot.py @@ -17,6 +17,26 @@ from build_snapshot import ( # noqa: E402 ) +META = { + "targetChapter": 489, + "referenceWork": {"id": "deep-space", "version": "v1"}, + "evaluationSetVersion": "set-v1", + "strategyVersion": "strategy-v1", + "authorizationSnapshot": { + "id": "auth-1", + "version": "v1", + "immutable": True, + "sourceVersion": "work-v1", + "sourceStatus": "active", + "allowedPurpose": ["offline_evaluation"], + "checkedAt": "2026-07-19T00:00:00Z", + "revalidationAt": "2026-07-20T00:00:00Z", + }, + "runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"}, + "armConfig": {"arms": ["outline_only", "outline_plus_cards", "outline_plus_placebo_cards"]}, +} + + class BuildSnapshotTest(unittest.TestCase): def test_normalize_chapter_rejects_guessing(self): self.assertEqual(normalize_chapter(488), 488) @@ -80,6 +100,8 @@ class BuildSnapshotTest(unittest.TestCase): }, 488, "v0", + target_chapter=489, + manifest_metadata=META, ) card = result["snapshot"]["cards"][0] self.assertEqual(card["milestones"], [{"chapter": 488, "step": "可见"}]) @@ -89,9 +111,16 @@ class BuildSnapshotTest(unittest.TestCase): def test_manifest_is_stable_and_does_not_store_payload(self): kwargs = { "as_of": 488, + "target_chapter": 489, "snapshot_version": "v0", + "reference_work": META["referenceWork"], + "evaluation_set_version": META["evaluationSetVersion"], + "strategy_version": META["strategyVersion"], + "authorization_snapshot": META["authorizationSnapshot"], + "run_permissions": META["runPermissions"], + "arm_config": META["armConfig"], "sections": {"l0": {"sourceId": "task", "sourceVersion": "1", "payload": {"target": 489}}}, - "final_report": {"status": "ready", "sourceHash": "abc"}, + "final_report": {"status": "ready", "hashes": {"sourceHash": "abc"}}, } first = build_snapshot_manifest(**kwargs) second = build_snapshot_manifest(**kwargs) @@ -104,11 +133,62 @@ class BuildSnapshotTest(unittest.TestCase): with self.assertRaises(SnapshotError): build_snapshot_manifest( as_of=488, + target_chapter=489, snapshot_version="v0", + reference_work=META["referenceWork"], + evaluation_set_version=META["evaluationSetVersion"], + strategy_version=META["strategyVersion"], + authorization_snapshot=META["authorizationSnapshot"], + run_permissions=META["runPermissions"], + arm_config=META["armConfig"], sections={}, final_report={"raw_text": "原书全文"}, ) + def test_unknown_chapter_collection_is_filtered_and_terminal_fields_are_audited(self): + result = build_snapshot( + { + "chapters": [{"chapter": 489, "fact": "未来"}, {"chapter": 488, "fact": "已知"}], + "cards": [{"name": "事件", "event_result": "未来结果", "relationshipPlan": "未来计划"}], + }, + 488, + "v0", + target_chapter=489, + manifest_metadata=META, + ) + self.assertEqual(result["snapshot"]["chapters"], [{"chapter": 488, "fact": "已知"}]) + self.assertNotIn("event_result", result["snapshot"]["cards"][0]) + self.assertNotIn("relationshipPlan", result["snapshot"]["cards"][0]) + reasons = [item["reason"] for item in result["manifest"]["omittedSources"]] + self.assertIn("future_or_crosses_as_of", reasons) + self.assertTrue(any(item["reason"] == "terminal_field_removed" for item in result["manifest"]["omittedFields"])) + + def test_unknown_top_level_section_fails_closed(self): + with self.assertRaises(SnapshotError): + build_snapshot( + {"chapters": [], "futurePayload": "不得透传"}, + 488, + "v0", + target_chapter=489, + manifest_metadata=META, + ) + + def test_final_report_is_closed_set(self): + with self.assertRaises(SnapshotError): + build_snapshot_manifest( + target_chapter=489, + as_of=488, + snapshot_version="v0", + reference_work=META["referenceWork"], + evaluation_set_version=META["evaluationSetVersion"], + strategy_version=META["strategyVersion"], + authorization_snapshot=META["authorizationSnapshot"], + run_permissions=META["runPermissions"], + arm_config=META["armConfig"], + sections={}, + final_report={"payload": "原书全文"}, + ) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_check_snapshot.py b/.claude/skills/replay-eval/scripts/test_check_snapshot.py index 1fcff72..66cbbd3 100644 --- a/.claude/skills/replay-eval/scripts/test_check_snapshot.py +++ b/.claude/skills/replay-eval/scripts/test_check_snapshot.py @@ -22,8 +22,19 @@ from check_snapshot import ( # noqa: E402 AUTH = { "sourceStatus": "active", + "copyrightStatus": "licensed", + "sourceVersion": "work-v1", "allowedPurpose": ["offline_evaluation"], - "authorizationSnapshot": {"id": "auth-1", "version": "v1"}, + "authorizationSnapshot": { + "id": "auth-1", + "version": "v1", + "immutable": True, + "sourceVersion": "work-v1", + "sourceStatus": "active", + "allowedPurpose": ["offline_evaluation"], + "checkedAt": "2026-07-19T00:00:00Z", + "revalidationAt": "2026-07-20T00:00:00Z", + }, } @@ -35,6 +46,7 @@ class CheckSnapshotTest(unittest.TestCase): def test_authorization_is_fail_closed(self): self.assertEqual(check_authorization(None)["status"], STATUS_BLOCKED_AUTHORIZATION) self.assertTrue(check_authorization(AUTH)["ok"]) + self.assertEqual(check_authorization({"sourceStatus": "active"})["status"], STATUS_BLOCKED_AUTHORIZATION) denied = {**AUTH, "allowedPurpose": ["read"]} self.assertEqual(check_authorization(denied)["status"], STATUS_BLOCKED_AUTHORIZATION) unknown_status = {**AUTH, "sourceStatus": "temporary"} @@ -43,9 +55,15 @@ class CheckSnapshotTest(unittest.TestCase): self.assertEqual(check_authorization(unlicensed)["status"], STATUS_BLOCKED_AUTHORIZATION) def test_target_source_is_blocked(self): - allowed = check_target_sources(489, [{"chapter": 488}, {"from_order": 450, "to_order": 482}]) + allowed = check_target_sources( + 489, + [ + {"sourceId": "chapter-488", "sourceVersion": "v1", "chapter": 488}, + {"sourceId": "outline-450-482", "sourceVersion": "v1", "from_order": 450, "to_order": 482}, + ], + ) self.assertEqual(allowed["status"], STATUS_READY) - blocked = check_target_sources(489, [{"chapter": 489}]) + blocked = check_target_sources(489, [{"sourceId": "chapter-489", "sourceVersion": "v1", "chapter": 489}]) self.assertEqual(blocked["status"], STATUS_TARGET_SOURCE_FORBIDDEN) def test_arm_common_input_must_match(self): @@ -65,7 +83,7 @@ class CheckSnapshotTest(unittest.TestCase): def test_two_arm_smoke_can_be_explicit(self): common = {"snapshotVersion": "v0", "asOfChapter": 488} manifests = {"outline_only": arm(common, []), "outline_plus_cards": arm(common, [{"id": 1}])} - self.assertTrue(check_arm_manifests(manifests, ["outline_only", "outline_plus_cards"])["ok"]) + self.assertTrue(check_arm_manifests(manifests, ["outline_only", "outline_plus_cards"], smoke=True)["ok"]) def test_candidate_contract_is_structural(self): candidate = { @@ -88,6 +106,8 @@ class CheckSnapshotTest(unittest.TestCase): self.assertEqual(check_candidate_output(duplicate, 489)["status"], STATUS_SCHEMA_INVALID) missing_id = {**candidate, "keyEvents": [{"event": "没有 id"}]} self.assertEqual(check_candidate_output(missing_id, 489)["status"], STATUS_SCHEMA_INVALID) + unknown_field = {**candidate, "futureSources": ["target-scaffold"]} + self.assertEqual(check_candidate_output(unknown_field, 489)["status"], STATUS_SCHEMA_INVALID) future_ref = {**candidate, "sourceRefs": ["target-scaffold"]} self.assertEqual( check_candidate_output( @@ -107,6 +127,7 @@ class CheckSnapshotTest(unittest.TestCase): } result = check_replay( authorization=AUTH, + as_of_chapter=488, target_chapter=489, planner_sources=[{"chapter": 489}], arm_manifests=manifests, @@ -114,6 +135,53 @@ class CheckSnapshotTest(unittest.TestCase): self.assertFalse(result["ok"]) self.assertEqual(result["status"], STATUS_TARGET_SOURCE_FORBIDDEN) + def test_production_arm_semantics_and_chapter_binding(self): + common = {"snapshotVersion": "v0", "asOfChapter": 488, "targetChapter": 489} + manifests = { + "outline_only": { + **common, + "cardInjectionCount": 0, + "cardSourceIds": [], + "cardStrategy": "none", + }, + "outline_plus_cards": { + **common, + "cardInjectionCount": 1, + "cardSourceIds": ["correct-1"], + "cardStrategy": "correct", + }, + "outline_plus_placebo_cards": { + **common, + "cardInjectionCount": 1, + "cardSourceIds": ["placebo-1"], + "cardStrategy": "placebo", + }, + } + self.assertTrue( + check_replay( + authorization=AUTH, + as_of_chapter=488, + target_chapter=489, + planner_sources=[{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}], + arm_manifests=manifests, + )["ok"] + ) + wrong = {**manifests, "outline_only": {**manifests["outline_only"], "cardInjectionCount": 1}} + self.assertEqual( + check_replay( + authorization=AUTH, + as_of_chapter=488, + target_chapter=489, + planner_sources=[{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}], + arm_manifests=wrong, + )["status"], + STATUS_INVALID_ARM_DIFF, + ) + + def test_missing_source_chapter_is_blocked(self): + result = check_target_sources(489, [{"sourceId": "unknown", "sourceVersion": "v1", "payload": "future"}]) + self.assertEqual(result["status"], STATUS_TARGET_SOURCE_FORBIDDEN) + if __name__ == "__main__": unittest.main() diff --git a/.claude/skills/replay-eval/scripts/test_run_replay.py b/.claude/skills/replay-eval/scripts/test_run_replay.py new file mode 100644 index 0000000..003cdeb --- /dev/null +++ b/.claude/skills/replay-eval/scripts/test_run_replay.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""回放编排器和安全摘要的无网络测试。""" + +from __future__ import annotations + +import json +import pathlib +import sys +import tempfile +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) +from run_replay import run_replay # noqa: E402 +from write_report import render_report # noqa: E402 + + +AUTH = { + "sourceStatus": "active", + "copyrightStatus": "licensed", + "sourceVersion": "work-v1", + "allowedPurpose": ["offline_evaluation"], + "authorizationSnapshot": { + "id": "auth-1", + "version": "v1", + "immutable": True, + "sourceVersion": "work-v1", + "sourceStatus": "active", + "allowedPurpose": ["offline_evaluation"], + "checkedAt": "2026-07-19T00:00:00Z", + "revalidationAt": "2026-07-20T00:00:00Z", + }, +} + + +def config(): + common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}} + return { + "runId": "smoke-001", + "referenceWork": {"id": "deep-space", "version": "work-v1"}, + "evaluationSetVersion": "set-v1", + "strategyVersion": "strategy-v1", + "authorization": AUTH, + "runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"}, + "targetChapter": 489, + "snapshot": { + "asOfChapter": 488, + "snapshotVersion": "next_fine_outline_replay_v0", + "data": { + "milestones": [{"chapter": 488, "fact": "安全历史"}, {"chapter": 489, "fact": "未来"}], + "cards": [{"name": "已知实体", "milestones": [{"chapter": 488, "step": "历史"}]}], + }, + }, + "sources": [{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}], + "commonContext": common, + "arms": { + "outline_only": {"cards": [], "cardSourceIds": [], "cardStrategy": "none"}, + "outline_plus_cards": {"cards": [{"name": "正确卡"}], "cardSourceIds": ["card-correct-1"], "cardStrategy": "correct"}, + "outline_plus_placebo_cards": {"cards": [{"name": "错配卡"}], "cardSourceIds": ["card-placebo-1"], "cardStrategy": "placebo"}, + }, + } + + +class ReplayRunTest(unittest.TestCase): + def test_dry_run_passes_and_writes_only_metadata(self): + with tempfile.TemporaryDirectory() as directory: + result = run_replay(config(), pathlib.Path(directory), mode="dry_run") + self.assertTrue(result["ok"]) + self.assertEqual(result["status"], "ready") + manifest = json.loads((pathlib.Path(directory) / "snapshot_manifest.json").read_text()) + self.assertNotIn("payload", json.dumps(manifest, ensure_ascii=False)) + self.assertFalse(list(pathlib.Path(directory).glob("planner_*.raw.json"))) + + def test_bad_authorization_stops_before_model(self): + bad = config() + bad["authorization"] = {"sourceStatus": "active"} + with tempfile.TemporaryDirectory() as directory: + result = run_replay(bad, pathlib.Path(directory), mode="dry_run") + self.assertFalse(result["ok"]) + self.assertEqual(result["status"], "blocked_authorization") + self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists()) + + def test_report_contains_hashes_but_not_raw_fields(self): + result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run") + report = render_report(result) + self.assertIn("smoke-001", report) + self.assertIn("snapshotManifestSha256", report) + self.assertNotIn('"prompt":', report) + self.assertNotIn('"payload":', report) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/replay-eval/scripts/write_report.py b/.claude/skills/replay-eval/scripts/write_report.py new file mode 100644 index 0000000..5fa9a7b --- /dev/null +++ b/.claude/skills/replay-eval/scripts/write_report.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""把回放运行结果压缩成不含正文和完整响应的 Markdown 摘要。""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from typing import Any, Mapping + + +REPORT_FORBIDDEN_KEYS = frozenset( + { + "raw", + "rawText", + "raw_output", + "body", + "content", + "payload", + "prompt", + "response", + "fullPrompt", + "fullResponse", + "正文", + "原文", + "正文全文", + "完整目标细纲", + } +) + + +def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None: + """报告输入也不接受原始内容字段,避免误把运行中间件带进最终摘要。""" + + if isinstance(value, Mapping): + for key, item in value.items(): + if str(key) in REPORT_FORBIDDEN_KEYS: + raise ValueError(f"{path}.{key} 不得进入最终报告") + _assert_no_forbidden_keys(item, f"{path}.{key}") + elif isinstance(value, list): + for index, item in enumerate(value): + _assert_no_forbidden_keys(item, f"{path}[{index}]") + + +def _arm_line(name: str, arm: Mapping[str, Any], result: Mapping[str, Any]) -> str: + outcome = result.get(name) or {} + return ( + f"- `{name}`: {outcome.get('status', 'not_run')}," + f"候选哈希 `{outcome.get('candidateSha256') or outcome.get('rawOutputSha256') or 'n/a'}`," + f"卡策略 `{arm.get('cardStrategy', 'unknown')}`,卡数 `{arm.get('cardInjectionCount', 0)}`" + ) + + +def render_report(result: Mapping[str, Any]) -> str: + """只从运行结果的白名单字段渲染摘要。""" + + _assert_no_forbidden_keys(result) + lines = [ + "# 细纲回放摘要", + "", + f"- runId: `{result.get('runId', 'unknown')}`", + f"- status: `{result.get('status', 'unknown')}`", + f"- mode: `{result.get('mode', 'unknown')}`", + f"- referenceWork: `{result.get('referenceWork', 'unknown')}@{result.get('referenceWorkVersion', 'unknown')}`", + f"- freeze: 第 `{result.get('asOfChapter', 'unknown')}` 章,目标第 `{result.get('targetChapter', 'unknown')}` 章", + f"- snapshot: `{result.get('snapshotVersion', 'unknown')}`", + "", + "## 前置门", + "", + f"- status: `{(result.get('preflight') or {}).get('status', 'unknown')}`", + ] + errors = (result.get("preflight") or {}).get("errors") or [] + for error in errors: + lines.append(f"- 阻断: {error}") + lines.extend(["", "## 三臂", ""]) + arms = result.get("arms") or {} + outcomes = result.get("results") or {} + for name in sorted(arms): + lines.append(_arm_line(name, arms[name], outcomes)) + if result.get("snapshotManifestSha256"): + lines.extend(["", f"- snapshotManifestSha256: `{result['snapshotManifestSha256']}`"]) + lines.extend( + [ + "", + "> 本摘要不保存原书正文、目标章全文、完整 prompt/response 或供应商原始响应;原始候选仅存在临时运行目录。", + "", + ] + ) + return "\n".join(lines) + + +def _parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="写出细纲回放安全摘要") + parser.add_argument("--input", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + return parser.parse_args() + + +def main() -> int: + args = _parse_args() + result = json.loads(args.input.read_text(encoding="utf-8")) + args.output.write_text(render_report(result), encoding="utf-8") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())