框架: 收紧回放评测前置门并补齐编排
This commit is contained in:
parent
fbb262cb57
commit
a54a3a4e39
@ -36,3 +36,11 @@ disable-model-invocation: true
|
|||||||
- 原始候选、标准事实摘要和完整输入只能留在临时运行目录或 `/tmp`。
|
- 原始候选、标准事实摘要和完整输入只能留在临时运行目录或 `/tmp`。
|
||||||
- 最终报告只允许评分、摘要、章节定位、失败类别和哈希。
|
- 最终报告只允许评分、摘要、章节定位、失败类别和哈希。
|
||||||
- 禁止写入原书正文、完整目标细纲、完整 Prompt/Response、供应商原始响应、token、密钥或未脱敏授权资料。
|
- 禁止写入原书正文、完整目标细纲、完整 Prompt/Response、供应商原始响应、token、密钥或未脱敏授权资料。
|
||||||
|
|
||||||
|
## 编排入口
|
||||||
|
|
||||||
|
- `scripts/run_replay.py --mode dry_run`:只执行授权、来源、冻结和三臂 manifest 预检,不调用模型;这是首个机制 smoke 入口。
|
||||||
|
- `scripts/run_replay.py --mode execute`:在全部前置门通过后,使用无工具、无会话持久化的本地 planner CLI 逐臂生成候选;`--output-dir` 必须位于仓库外的临时目录。
|
||||||
|
- `scripts/write_report.py`:从 `run_result.json` 生成安全摘要;它不会读取候选正文,也不会把候选路径以外的原始响应写入报告。
|
||||||
|
|
||||||
|
真实作品运行前必须先从权威来源取得不可变授权快照。数据库没有该字段时,使用 dry-run 证明机制并保持 `blocked_authorization`,不得用本地配置或口头许可伪造放行。
|
||||||
|
|||||||
@ -25,21 +25,91 @@ TERMINAL_FIELDS = frozenset(
|
|||||||
"final_summary",
|
"final_summary",
|
||||||
"terminal_summary",
|
"terminal_summary",
|
||||||
"final_state",
|
"final_state",
|
||||||
|
"finalState",
|
||||||
"current_state",
|
"current_state",
|
||||||
|
"currentState",
|
||||||
"future_arc",
|
"future_arc",
|
||||||
|
"futureArc",
|
||||||
"future_plan",
|
"future_plan",
|
||||||
|
"futurePlan",
|
||||||
|
"event_result",
|
||||||
|
"eventResult",
|
||||||
|
"relationship_plan",
|
||||||
|
"relationshipPlan",
|
||||||
|
"growth_arc",
|
||||||
|
"growthArc",
|
||||||
|
"terminalResult",
|
||||||
"成长弧线",
|
"成长弧线",
|
||||||
"当前态",
|
"当前态",
|
||||||
"终态摘要",
|
"终态摘要",
|
||||||
"终局状态",
|
"终局状态",
|
||||||
"未来弧线",
|
"未来弧线",
|
||||||
"未来计划",
|
"未来计划",
|
||||||
|
"事件结果",
|
||||||
|
"关系计划",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
SNAPSHOT_TOP_LEVEL_KEYS = frozenset(
|
||||||
|
{
|
||||||
|
"milestones",
|
||||||
|
"outlineWindows",
|
||||||
|
"cards",
|
||||||
|
"chapters",
|
||||||
|
"events",
|
||||||
|
"relations",
|
||||||
|
"relationships",
|
||||||
|
"stateRecords",
|
||||||
|
"state_records",
|
||||||
|
"facts",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
CHAPTER_SCOPED_COLLECTIONS = frozenset(
|
||||||
|
{
|
||||||
|
"milestones",
|
||||||
|
"outlineWindows",
|
||||||
|
"chapters",
|
||||||
|
"events",
|
||||||
|
"relations",
|
||||||
|
"relationships",
|
||||||
|
"stateRecords",
|
||||||
|
"state_records",
|
||||||
|
"facts",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
FINAL_REPORT_ALLOWED_FIELDS = frozenset(
|
||||||
|
{
|
||||||
|
"status",
|
||||||
|
"runId",
|
||||||
|
"referenceWork",
|
||||||
|
"referenceWorkVersion",
|
||||||
|
"targetChapter",
|
||||||
|
"asOfChapter",
|
||||||
|
"snapshotVersion",
|
||||||
|
"evaluationSetVersion",
|
||||||
|
"strategyVersion",
|
||||||
|
"profile",
|
||||||
|
"arm",
|
||||||
|
"armComparison",
|
||||||
|
"candidateCount",
|
||||||
|
"scores",
|
||||||
|
"summary",
|
||||||
|
"chapterRefs",
|
||||||
|
"failureClass",
|
||||||
|
"failureReason",
|
||||||
|
"hashes",
|
||||||
|
"warnings",
|
||||||
|
"stability",
|
||||||
|
"proxyConfidence",
|
||||||
|
"goldUncertain",
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
FINAL_REPORT_FORBIDDEN_FIELDS = frozenset(
|
FINAL_REPORT_FORBIDDEN_FIELDS = frozenset(
|
||||||
{
|
{
|
||||||
"raw",
|
"raw",
|
||||||
|
"payload",
|
||||||
"raw_text",
|
"raw_text",
|
||||||
"body",
|
"body",
|
||||||
"content",
|
"content",
|
||||||
@ -52,6 +122,8 @@ FINAL_REPORT_FORBIDDEN_FIELDS = frozenset(
|
|||||||
"target_chapter_text",
|
"target_chapter_text",
|
||||||
"prompt",
|
"prompt",
|
||||||
"response",
|
"response",
|
||||||
|
"fullPrompt",
|
||||||
|
"fullResponse",
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
@ -59,6 +131,15 @@ _INTEGER_RE = re.compile(r"^\s*(\d+)\s*$")
|
|||||||
_RANGE_RE = re.compile(r"^\s*(?:第\s*)?(\d+)\s*(?:-|–|—|~|至|到)\s*(?:第\s*)?(\d+)\s*(?:章)?\s*$")
|
_RANGE_RE = re.compile(r"^\s*(?:第\s*)?(\d+)\s*(?:-|–|—|~|至|到)\s*(?:第\s*)?(\d+)\s*(?:章)?\s*$")
|
||||||
|
|
||||||
|
|
||||||
|
def _normalized_key(key: Any) -> str:
|
||||||
|
"""统一英文终态字段的大小写、下划线和短横线写法。"""
|
||||||
|
|
||||||
|
return re.sub(r"[_-]", "", str(key)).lower()
|
||||||
|
|
||||||
|
|
||||||
|
TERMINAL_FIELD_NAMES = frozenset(_normalized_key(field) for field in TERMINAL_FIELDS)
|
||||||
|
|
||||||
|
|
||||||
def normalize_chapter(value: Any) -> int | None:
|
def normalize_chapter(value: Any) -> int | None:
|
||||||
"""只接受明确的正整数章号;不把 bool、浮点或模糊文本猜成章号。"""
|
"""只接受明确的正整数章号;不把 bool、浮点或模糊文本猜成章号。"""
|
||||||
|
|
||||||
@ -181,33 +262,85 @@ def filter_outline_windows(
|
|||||||
return kept, omitted
|
return kept, omitted
|
||||||
|
|
||||||
|
|
||||||
def remove_terminal_fields(value: Any) -> Any:
|
def remove_terminal_fields(
|
||||||
"""递归删除不能直接作为 as_of 事实使用的终态/未来字段。"""
|
value: Any,
|
||||||
|
omitted_fields: list[dict[str, Any]] | None = None,
|
||||||
|
path: str = "$",
|
||||||
|
) -> Any:
|
||||||
|
"""递归删除终态字段,并记录裁剪路径供审计。"""
|
||||||
|
|
||||||
if isinstance(value, list):
|
if isinstance(value, list):
|
||||||
return [remove_terminal_fields(item) for item in value]
|
return [
|
||||||
|
remove_terminal_fields(item, omitted_fields, f"{path}[{index}]")
|
||||||
|
for index, item in enumerate(value)
|
||||||
|
]
|
||||||
if not isinstance(value, Mapping):
|
if not isinstance(value, Mapping):
|
||||||
return copy.deepcopy(value)
|
return copy.deepcopy(value)
|
||||||
|
|
||||||
projected: dict[str, Any] = {}
|
projected: dict[str, Any] = {}
|
||||||
for key, item in value.items():
|
for key, item in value.items():
|
||||||
if str(key) in TERMINAL_FIELDS:
|
key_text = str(key)
|
||||||
|
if _normalized_key(key_text) in TERMINAL_FIELD_NAMES or key_text in TERMINAL_FIELDS:
|
||||||
|
if omitted_fields is not None:
|
||||||
|
omitted_fields.append({"path": f"{path}.{key_text}", "reason": "terminal_field_removed"})
|
||||||
continue
|
continue
|
||||||
projected[str(key)] = remove_terminal_fields(item)
|
projected[key_text] = remove_terminal_fields(item, omitted_fields, f"{path}.{key_text}")
|
||||||
return projected
|
return projected
|
||||||
|
|
||||||
|
|
||||||
def _validate_final_report(value: Any, path: str = "finalReport") -> None:
|
_DROP = object()
|
||||||
"""禁止最终报告携带原书全文、完整响应或完整 prompt。"""
|
|
||||||
|
|
||||||
|
def _freeze_nested_chapter_records(
|
||||||
|
value: Any,
|
||||||
|
as_of: int,
|
||||||
|
omitted_sources: list[dict[str, Any]],
|
||||||
|
path: str = "$",
|
||||||
|
) -> Any:
|
||||||
|
"""递归冻结所有带绝对章号的嵌套记录,未标明章号的显式记录不猜测。"""
|
||||||
|
|
||||||
|
if isinstance(value, Mapping):
|
||||||
|
chapter_keys = {"chapter", "chapter_no", "order_no", "章", "章号", "from_order", "to_order"}
|
||||||
|
if chapter_keys.intersection(value):
|
||||||
|
bounds = normalize_chapter_range(_chapter_value(value))
|
||||||
|
if bounds is None:
|
||||||
|
omitted_sources.append({"path": path, "reason": "missing_or_invalid_chapter", "recordHash": sha256_value(value)})
|
||||||
|
return _DROP
|
||||||
|
if bounds[1] > as_of:
|
||||||
|
omitted_sources.append({"path": path, "reason": "future_or_crosses_as_of", "recordHash": sha256_value(value)})
|
||||||
|
return _DROP
|
||||||
|
projected: dict[str, Any] = {}
|
||||||
|
for key, item in value.items():
|
||||||
|
frozen = _freeze_nested_chapter_records(item, as_of, omitted_sources, f"{path}.{key}")
|
||||||
|
if frozen is not _DROP:
|
||||||
|
projected[str(key)] = frozen
|
||||||
|
return projected
|
||||||
|
|
||||||
|
if isinstance(value, list):
|
||||||
|
projected_list: list[Any] = []
|
||||||
|
for index, item in enumerate(value):
|
||||||
|
frozen = _freeze_nested_chapter_records(item, as_of, omitted_sources, f"{path}[{index}]")
|
||||||
|
if frozen is not _DROP:
|
||||||
|
projected_list.append(frozen)
|
||||||
|
return projected_list
|
||||||
|
|
||||||
|
return copy.deepcopy(value)
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_final_report(value: Any, path: str = "finalReport", top_level: bool = True) -> None:
|
||||||
|
"""用顶层白名单和递归禁用字段保护最终报告边界。"""
|
||||||
|
|
||||||
if isinstance(value, Mapping):
|
if isinstance(value, Mapping):
|
||||||
for key, item in value.items():
|
for key, item in value.items():
|
||||||
if str(key) in FINAL_REPORT_FORBIDDEN_FIELDS:
|
key_text = str(key)
|
||||||
raise SnapshotError(f"{path}.{key} 不得进入最终报告")
|
if key_text in FINAL_REPORT_FORBIDDEN_FIELDS:
|
||||||
_validate_final_report(item, f"{path}.{key}")
|
raise SnapshotError(f"{path}.{key_text} 不得进入最终报告")
|
||||||
|
if top_level and key_text not in FINAL_REPORT_ALLOWED_FIELDS:
|
||||||
|
raise SnapshotError(f"{path}.{key_text} 不是允许的最终报告字段")
|
||||||
|
_validate_final_report(item, f"{path}.{key_text}", False)
|
||||||
elif isinstance(value, list):
|
elif isinstance(value, list):
|
||||||
for index, item in enumerate(value):
|
for index, item in enumerate(value):
|
||||||
_validate_final_report(item, f"{path}[{index}]")
|
_validate_final_report(item, f"{path}[{index}]", False)
|
||||||
|
|
||||||
|
|
||||||
def _source_record(section: str, value: Any) -> dict[str, Any]:
|
def _source_record(section: str, value: Any) -> dict[str, Any]:
|
||||||
@ -215,16 +348,15 @@ def _source_record(section: str, value: Any) -> dict[str, Any]:
|
|||||||
|
|
||||||
if isinstance(value, Mapping) and "payload" in value:
|
if isinstance(value, Mapping) and "payload" in value:
|
||||||
source_id = str(value.get("sourceId") or section)
|
source_id = str(value.get("sourceId") or section)
|
||||||
source_version = str(value.get("sourceVersion") or "unknown")
|
source_version = str(value.get("sourceVersion") or "")
|
||||||
|
if not source_id or not source_version:
|
||||||
|
raise SnapshotError(f"来源 {section} 缺少 sourceId/sourceVersion")
|
||||||
payload = value["payload"]
|
payload = value["payload"]
|
||||||
omitted_fields = list(value.get("omittedFields") or [])
|
omitted_fields = list(value.get("omittedFields") or [])
|
||||||
omitted_sources = list(value.get("omittedSources") or [])
|
omitted_sources = list(value.get("omittedSources") or [])
|
||||||
else:
|
else:
|
||||||
source_id = section
|
source_id = section
|
||||||
source_version = "unknown"
|
raise SnapshotError(f"来源 {section} 必须显式提供 sourceId/sourceVersion/payload")
|
||||||
payload = value
|
|
||||||
omitted_fields = []
|
|
||||||
omitted_sources = []
|
|
||||||
|
|
||||||
serialized = _safe_json(payload)
|
serialized = _safe_json(payload)
|
||||||
return {
|
return {
|
||||||
@ -241,7 +373,14 @@ def _source_record(section: str, value: Any) -> dict[str, Any]:
|
|||||||
def build_snapshot_manifest(
|
def build_snapshot_manifest(
|
||||||
*,
|
*,
|
||||||
as_of: int,
|
as_of: int,
|
||||||
|
target_chapter: int,
|
||||||
snapshot_version: str,
|
snapshot_version: str,
|
||||||
|
reference_work: Mapping[str, Any],
|
||||||
|
evaluation_set_version: str,
|
||||||
|
strategy_version: str,
|
||||||
|
authorization_snapshot: Mapping[str, Any],
|
||||||
|
run_permissions: Mapping[str, Any],
|
||||||
|
arm_config: Mapping[str, Any],
|
||||||
sections: Mapping[str, Any],
|
sections: Mapping[str, Any],
|
||||||
omitted_fields: Sequence[Any] | None = None,
|
omitted_fields: Sequence[Any] | None = None,
|
||||||
omitted_sources: Sequence[Any] | None = None,
|
omitted_sources: Sequence[Any] | None = None,
|
||||||
@ -252,14 +391,71 @@ def build_snapshot_manifest(
|
|||||||
normalized_as_of = normalize_chapter(as_of)
|
normalized_as_of = normalize_chapter(as_of)
|
||||||
if normalized_as_of is None:
|
if normalized_as_of is None:
|
||||||
raise SnapshotError("as_of 必须是正整数章号")
|
raise SnapshotError("as_of 必须是正整数章号")
|
||||||
|
normalized_target = normalize_chapter(target_chapter)
|
||||||
|
if normalized_target != normalized_as_of + 1:
|
||||||
|
raise SnapshotError("target_chapter 必须等于 as_of+1")
|
||||||
if not snapshot_version.strip():
|
if not snapshot_version.strip():
|
||||||
raise SnapshotError("snapshot_version 不能为空")
|
raise SnapshotError("snapshot_version 不能为空")
|
||||||
|
if not isinstance(reference_work, Mapping) or not reference_work.get("id") or not reference_work.get("version"):
|
||||||
|
raise SnapshotError("reference_work 必须包含 id/version")
|
||||||
|
if not evaluation_set_version.strip() or not strategy_version.strip():
|
||||||
|
raise SnapshotError("evaluation_set_version/strategy_version 不能为空")
|
||||||
|
if not isinstance(authorization_snapshot, Mapping):
|
||||||
|
raise SnapshotError("authorization_snapshot 必须是对象")
|
||||||
|
authorization_required = (
|
||||||
|
"id",
|
||||||
|
"version",
|
||||||
|
"immutable",
|
||||||
|
"sourceVersion",
|
||||||
|
"sourceStatus",
|
||||||
|
"allowedPurpose",
|
||||||
|
"checkedAt",
|
||||||
|
)
|
||||||
|
if any(not authorization_snapshot.get(field) for field in authorization_required):
|
||||||
|
raise SnapshotError("authorization_snapshot 缺少必填字段")
|
||||||
|
if authorization_snapshot.get("immutable") is not True:
|
||||||
|
raise SnapshotError("authorization_snapshot 必须是不可变快照")
|
||||||
|
if not authorization_snapshot.get("expiresAt") and not authorization_snapshot.get("revalidationAt"):
|
||||||
|
raise SnapshotError("authorization_snapshot 缺少过期或重验时间")
|
||||||
|
if not isinstance(run_permissions, Mapping) or not run_permissions.get("purpose"):
|
||||||
|
raise SnapshotError("run_permissions 必须包含 purpose")
|
||||||
|
if not isinstance(arm_config, Mapping) or not arm_config.get("arms"):
|
||||||
|
raise SnapshotError("arm_config 必须包含 arms")
|
||||||
|
if set(arm_config["arms"]) != {
|
||||||
|
"outline_only",
|
||||||
|
"outline_plus_cards",
|
||||||
|
"outline_plus_placebo_cards",
|
||||||
|
}:
|
||||||
|
raise SnapshotError("arm_config 必须是完整三臂")
|
||||||
if final_report is not None:
|
if final_report is not None:
|
||||||
_validate_final_report(final_report)
|
_validate_final_report(final_report)
|
||||||
|
|
||||||
|
auth_safe = {
|
||||||
|
"id": str(authorization_snapshot["id"]),
|
||||||
|
"version": str(authorization_snapshot.get("version") or ""),
|
||||||
|
"sha256": sha256_value(authorization_snapshot),
|
||||||
|
"checkedAt": authorization_snapshot.get("checkedAt"),
|
||||||
|
}
|
||||||
|
if not auth_safe["version"]:
|
||||||
|
raise SnapshotError("authorization_snapshot 必须包含 version")
|
||||||
|
|
||||||
manifest: dict[str, Any] = {
|
manifest: dict[str, Any] = {
|
||||||
"snapshotVersion": snapshot_version,
|
"snapshotVersion": snapshot_version,
|
||||||
"asOfChapter": normalized_as_of,
|
"asOfChapter": normalized_as_of,
|
||||||
|
"targetChapter": normalized_target,
|
||||||
|
"referenceWork": {"id": str(reference_work["id"]), "version": str(reference_work["version"])},
|
||||||
|
"evaluationSetVersion": evaluation_set_version,
|
||||||
|
"strategyVersion": strategy_version,
|
||||||
|
"authorization": auth_safe,
|
||||||
|
"runPermissions": {
|
||||||
|
"purpose": str(run_permissions["purpose"]),
|
||||||
|
"mode": str(run_permissions.get("mode") or "offline"),
|
||||||
|
"sha256": sha256_value(run_permissions),
|
||||||
|
},
|
||||||
|
"armConfig": {
|
||||||
|
"arms": sorted(str(arm) for arm in arm_config["arms"]),
|
||||||
|
"sha256": sha256_value(arm_config),
|
||||||
|
},
|
||||||
"sections": {
|
"sections": {
|
||||||
str(section): _source_record(str(section), value)
|
str(section): _source_record(str(section), value)
|
||||||
for section, value in sorted(sections.items(), key=lambda pair: str(pair[0]))
|
for section, value in sorted(sections.items(), key=lambda pair: str(pair[0]))
|
||||||
@ -272,11 +468,34 @@ def build_snapshot_manifest(
|
|||||||
return manifest
|
return manifest
|
||||||
|
|
||||||
|
|
||||||
def build_snapshot(data: Mapping[str, Any], as_of: int, snapshot_version: str) -> dict[str, Any]:
|
def build_snapshot(
|
||||||
|
data: Mapping[str, Any],
|
||||||
|
as_of: int,
|
||||||
|
snapshot_version: str,
|
||||||
|
*,
|
||||||
|
target_chapter: int | None = None,
|
||||||
|
manifest_metadata: Mapping[str, Any] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
"""从最小 JSON 输入生成冻结后的结构化快照和 manifest。"""
|
"""从最小 JSON 输入生成冻结后的结构化快照和 manifest。"""
|
||||||
|
|
||||||
|
if not isinstance(data, Mapping):
|
||||||
|
raise SnapshotError("快照输入必须是对象")
|
||||||
|
unknown_sections = sorted(set(data) - SNAPSHOT_TOP_LEVEL_KEYS)
|
||||||
|
if unknown_sections:
|
||||||
|
raise SnapshotError(f"快照包含未登记顶层分区: {','.join(str(item) for item in unknown_sections)}")
|
||||||
safe = copy.deepcopy(dict(data))
|
safe = copy.deepcopy(dict(data))
|
||||||
all_omitted: list[dict[str, Any]] = []
|
all_omitted: list[dict[str, Any]] = []
|
||||||
|
metadata = dict(manifest_metadata or {})
|
||||||
|
normalized_as_of = normalize_chapter(as_of)
|
||||||
|
if normalized_as_of is None:
|
||||||
|
raise SnapshotError("as_of 必须是正整数章号")
|
||||||
|
normalized_target = normalize_chapter(target_chapter or metadata.get("targetChapter"))
|
||||||
|
if normalized_target != normalized_as_of + 1:
|
||||||
|
raise SnapshotError("target_chapter 必须等于 as_of+1")
|
||||||
|
|
||||||
|
for section in CHAPTER_SCOPED_COLLECTIONS:
|
||||||
|
if section in safe and not isinstance(safe[section], list):
|
||||||
|
raise SnapshotError(f"分区 {section} 必须是数组")
|
||||||
|
|
||||||
if isinstance(safe.get("milestones"), list):
|
if isinstance(safe.get("milestones"), list):
|
||||||
milestones, omitted = filter_milestones(safe["milestones"], as_of)
|
milestones, omitted = filter_milestones(safe["milestones"], as_of)
|
||||||
@ -309,11 +528,33 @@ def build_snapshot(data: Mapping[str, Any], as_of: int, snapshot_version: str) -
|
|||||||
frozen_cards.append(projected)
|
frozen_cards.append(projected)
|
||||||
safe["cards"] = frozen_cards
|
safe["cards"] = frozen_cards
|
||||||
|
|
||||||
safe = remove_terminal_fields(safe)
|
safe = _freeze_nested_chapter_records(safe, normalized_as_of, all_omitted)
|
||||||
|
if safe is _DROP:
|
||||||
|
raise SnapshotError("快照根对象不能被冻结过滤")
|
||||||
|
omitted_fields: list[dict[str, Any]] = []
|
||||||
|
safe = remove_terminal_fields(safe, omitted_fields)
|
||||||
|
required_metadata = {
|
||||||
|
"referenceWork": metadata.get("referenceWork"),
|
||||||
|
"evaluationSetVersion": metadata.get("evaluationSetVersion"),
|
||||||
|
"strategyVersion": metadata.get("strategyVersion"),
|
||||||
|
"authorizationSnapshot": metadata.get("authorizationSnapshot"),
|
||||||
|
"runPermissions": metadata.get("runPermissions"),
|
||||||
|
"armConfig": metadata.get("armConfig"),
|
||||||
|
}
|
||||||
|
if any(value is None for value in required_metadata.values()):
|
||||||
|
raise SnapshotError("缺少完整回放 manifest 元数据")
|
||||||
manifest = build_snapshot_manifest(
|
manifest = build_snapshot_manifest(
|
||||||
as_of=as_of,
|
as_of=normalized_as_of,
|
||||||
|
target_chapter=normalized_target,
|
||||||
snapshot_version=snapshot_version,
|
snapshot_version=snapshot_version,
|
||||||
|
reference_work=required_metadata["referenceWork"],
|
||||||
|
evaluation_set_version=required_metadata["evaluationSetVersion"],
|
||||||
|
strategy_version=required_metadata["strategyVersion"],
|
||||||
|
authorization_snapshot=required_metadata["authorizationSnapshot"],
|
||||||
|
run_permissions=required_metadata["runPermissions"],
|
||||||
|
arm_config=required_metadata["armConfig"],
|
||||||
sections={"snapshot": {"sourceId": "frozen-snapshot", "sourceVersion": snapshot_version, "payload": safe}},
|
sections={"snapshot": {"sourceId": "frozen-snapshot", "sourceVersion": snapshot_version, "payload": safe}},
|
||||||
|
omitted_fields=omitted_fields,
|
||||||
omitted_sources=all_omitted,
|
omitted_sources=all_omitted,
|
||||||
)
|
)
|
||||||
return {"snapshot": safe, "manifest": manifest}
|
return {"snapshot": safe, "manifest": manifest}
|
||||||
@ -324,6 +565,8 @@ def _parse_args() -> argparse.Namespace:
|
|||||||
parser.add_argument("--input", type=Path, required=True, help="结构化 JSON 输入")
|
parser.add_argument("--input", type=Path, required=True, help="结构化 JSON 输入")
|
||||||
parser.add_argument("--output", type=Path, required=True, help="输出 JSON 路径")
|
parser.add_argument("--output", type=Path, required=True, help="输出 JSON 路径")
|
||||||
parser.add_argument("--as-of", type=int, required=True, dest="as_of")
|
parser.add_argument("--as-of", type=int, required=True, dest="as_of")
|
||||||
|
parser.add_argument("--target-chapter", type=int, required=True, dest="target_chapter")
|
||||||
|
parser.add_argument("--metadata", type=Path, required=True, help="回放 manifest 元数据 JSON")
|
||||||
parser.add_argument("--snapshot-version", default="next_fine_outline_replay_v0")
|
parser.add_argument("--snapshot-version", default="next_fine_outline_replay_v0")
|
||||||
return parser.parse_args()
|
return parser.parse_args()
|
||||||
|
|
||||||
@ -331,7 +574,14 @@ def _parse_args() -> argparse.Namespace:
|
|||||||
def main() -> int:
|
def main() -> int:
|
||||||
args = _parse_args()
|
args = _parse_args()
|
||||||
data = json.loads(args.input.read_text(encoding="utf-8"))
|
data = json.loads(args.input.read_text(encoding="utf-8"))
|
||||||
result = build_snapshot(data, args.as_of, args.snapshot_version)
|
metadata = json.loads(args.metadata.read_text(encoding="utf-8"))
|
||||||
|
result = build_snapshot(
|
||||||
|
data,
|
||||||
|
args.as_of,
|
||||||
|
args.snapshot_version,
|
||||||
|
target_chapter=args.target_chapter,
|
||||||
|
manifest_metadata=metadata,
|
||||||
|
)
|
||||||
args.output.write_text(_safe_json(result) + "\n", encoding="utf-8")
|
args.output.write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|||||||
@ -25,6 +25,7 @@ FORBIDDEN_SOURCE_STATUSES = frozenset(
|
|||||||
{"revoked", "delisted", "recalled", "blocked", "owner_missing", "unauthorized"}
|
{"revoked", "delisted", "recalled", "blocked", "owner_missing", "unauthorized"}
|
||||||
)
|
)
|
||||||
ALLOWED_SOURCE_STATUSES = frozenset({"active", "approved", "authorized", "licensed"})
|
ALLOWED_SOURCE_STATUSES = frozenset({"active", "approved", "authorized", "licensed"})
|
||||||
|
ALLOWED_COPYRIGHT_STATUSES = frozenset({"active", "approved", "authorized", "licensed", "owned"})
|
||||||
FORBIDDEN_COPYRIGHT_STATUSES = frozenset(
|
FORBIDDEN_COPYRIGHT_STATUSES = frozenset(
|
||||||
{"unauthorized", "unlicensed", "revoked", "expired", "blocked"}
|
{"unauthorized", "unlicensed", "revoked", "expired", "blocked"}
|
||||||
)
|
)
|
||||||
@ -34,6 +35,8 @@ CARD_KEYS = frozenset(
|
|||||||
"card",
|
"card",
|
||||||
"cards",
|
"cards",
|
||||||
"cardInjection",
|
"cardInjection",
|
||||||
|
"cardInjectionSha256",
|
||||||
|
"cardInjectionCount",
|
||||||
"cardSection",
|
"cardSection",
|
||||||
"cardSections",
|
"cardSections",
|
||||||
"cardManifest",
|
"cardManifest",
|
||||||
@ -43,6 +46,8 @@ CARD_KEYS = frozenset(
|
|||||||
"l2-card",
|
"l2-card",
|
||||||
"l2-placebo",
|
"l2-placebo",
|
||||||
"card_injection",
|
"card_injection",
|
||||||
|
"card_injection_sha256",
|
||||||
|
"card_injection_count",
|
||||||
"card_section",
|
"card_section",
|
||||||
"card_sections",
|
"card_sections",
|
||||||
"card_manifest",
|
"card_manifest",
|
||||||
@ -65,6 +70,17 @@ REQUIRED_CANDIDATE_FIELDS = (
|
|||||||
"unknowns",
|
"unknowns",
|
||||||
"assumptions",
|
"assumptions",
|
||||||
)
|
)
|
||||||
|
ALLOWED_CANDIDATE_FIELDS = frozenset((*REQUIRED_CANDIDATE_FIELDS, "sourceRefs"))
|
||||||
|
EVENT_FIELDS = {
|
||||||
|
"id": str,
|
||||||
|
"order": int,
|
||||||
|
"event": str,
|
||||||
|
"participants": list,
|
||||||
|
"trigger": str,
|
||||||
|
"resultDirection": str,
|
||||||
|
}
|
||||||
|
ENTITY_FIELDS = {"name": str, "type": str, "role": str}
|
||||||
|
FORESHADOWING_FIELDS = {"action": str, "subject": str, "evidence": str}
|
||||||
FORBIDDEN_CANDIDATE_FIELDS = frozenset(
|
FORBIDDEN_CANDIDATE_FIELDS = frozenset(
|
||||||
{"body", "raw", "rawText", "content", "正文", "原文", "正文全文", "完整目标细纲"}
|
{"body", "raw", "rawText", "content", "正文", "原文", "正文全文", "完整目标细纲"}
|
||||||
)
|
)
|
||||||
@ -90,6 +106,26 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An
|
|||||||
if not isinstance(snapshot, Mapping) or not snapshot:
|
if not isinstance(snapshot, Mapping) or not snapshot:
|
||||||
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照为空"])
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照为空"])
|
||||||
|
|
||||||
|
snapshot_required = (
|
||||||
|
"id",
|
||||||
|
"version",
|
||||||
|
"immutable",
|
||||||
|
"sourceVersion",
|
||||||
|
"sourceStatus",
|
||||||
|
"allowedPurpose",
|
||||||
|
"checkedAt",
|
||||||
|
)
|
||||||
|
missing_snapshot = [field for field in snapshot_required if not snapshot.get(field)]
|
||||||
|
if missing_snapshot:
|
||||||
|
return _result(
|
||||||
|
STATUS_BLOCKED_AUTHORIZATION,
|
||||||
|
[f"授权快照缺少字段: {','.join(missing_snapshot)}"],
|
||||||
|
)
|
||||||
|
if snapshot.get("immutable") is not True:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照不是不可变快照"])
|
||||||
|
if not snapshot.get("expiresAt") and not snapshot.get("revalidationAt"):
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照缺少过期或重验时间"])
|
||||||
|
|
||||||
source_status = str(_field(authorization, "sourceStatus", "source_status") or "").lower()
|
source_status = str(_field(authorization, "sourceStatus", "source_status") or "").lower()
|
||||||
if not source_status:
|
if not source_status:
|
||||||
return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceStatus"])
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceStatus"])
|
||||||
@ -101,8 +137,20 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An
|
|||||||
copyright_status = str(
|
copyright_status = str(
|
||||||
_field(authorization, "copyrightStatus", "copyright_status") or ""
|
_field(authorization, "copyrightStatus", "copyright_status") or ""
|
||||||
).lower()
|
).lower()
|
||||||
|
if not copyright_status:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 copyrightStatus"])
|
||||||
if copyright_status in FORBIDDEN_COPYRIGHT_STATUSES:
|
if copyright_status in FORBIDDEN_COPYRIGHT_STATUSES:
|
||||||
return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态禁止评测: {copyright_status}"])
|
return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态禁止评测: {copyright_status}"])
|
||||||
|
if copyright_status not in ALLOWED_COPYRIGHT_STATUSES:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, [f"版权状态未登记,拒绝评测: {copyright_status}"])
|
||||||
|
|
||||||
|
source_version = str(_field(authorization, "sourceVersion", "source_version") or "")
|
||||||
|
if not source_version:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["缺少 sourceVersion"])
|
||||||
|
if str(snapshot.get("sourceVersion")) != source_version:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 sourceVersion 不一致"])
|
||||||
|
if str(snapshot["sourceStatus"]).lower() != source_status:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照 sourceStatus 不一致"])
|
||||||
|
|
||||||
allowed = _field(authorization, "allowedPurpose", "allowed_purpose")
|
allowed = _field(authorization, "allowedPurpose", "allowed_purpose")
|
||||||
if isinstance(allowed, str):
|
if isinstance(allowed, str):
|
||||||
@ -111,6 +159,13 @@ def check_authorization(authorization: Mapping[str, Any] | None) -> dict[str, An
|
|||||||
return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 不是用途列表"])
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 不是用途列表"])
|
||||||
if "offline_evaluation" not in allowed:
|
if "offline_evaluation" not in allowed:
|
||||||
return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 不包含 offline_evaluation"])
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["allowedPurpose 不包含 offline_evaluation"])
|
||||||
|
snapshot_allowed = snapshot.get("allowedPurpose", snapshot.get("allowed_purpose"))
|
||||||
|
if not isinstance(snapshot_allowed, Sequence) or isinstance(snapshot_allowed, (str, bytes)):
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照缺少 allowedPurpose"])
|
||||||
|
if "offline_evaluation" not in snapshot_allowed:
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照用途不包含 offline_evaluation"])
|
||||||
|
if set(snapshot_allowed) != set(allowed):
|
||||||
|
return _result(STATUS_BLOCKED_AUTHORIZATION, ["授权快照用途与运行用途不一致"])
|
||||||
return _result(STATUS_READY)
|
return _result(STATUS_READY)
|
||||||
|
|
||||||
|
|
||||||
@ -120,11 +175,16 @@ def check_target_sources(target_chapter: int, sources: Sequence[Mapping[str, Any
|
|||||||
target = normalize_chapter(target_chapter)
|
target = normalize_chapter(target_chapter)
|
||||||
if target is None:
|
if target is None:
|
||||||
return _result(STATUS_INVALID_SNAPSHOT, ["目标章号无效"])
|
return _result(STATUS_INVALID_SNAPSHOT, ["目标章号无效"])
|
||||||
|
if not isinstance(sources, Sequence) or isinstance(sources, (str, bytes)) or not sources:
|
||||||
|
return _result(STATUS_INVALID_SNAPSHOT, ["缺少规划来源"])
|
||||||
errors: list[str] = []
|
errors: list[str] = []
|
||||||
for index, source in enumerate(sources):
|
for index, source in enumerate(sources):
|
||||||
if not isinstance(source, Mapping):
|
if not isinstance(source, Mapping):
|
||||||
errors.append(f"source[{index}] 不是对象")
|
errors.append(f"source[{index}] 不是对象")
|
||||||
continue
|
continue
|
||||||
|
if not source.get("sourceId") or not source.get("sourceVersion"):
|
||||||
|
errors.append(f"source[{index}] 缺少 sourceId/sourceVersion")
|
||||||
|
continue
|
||||||
value = _field(source, "chapter", "chapterNo", "chapter_no", "章", "章号")
|
value = _field(source, "chapter", "chapterNo", "chapter_no", "章", "章号")
|
||||||
bounds = normalize_chapter_range(value)
|
bounds = normalize_chapter_range(value)
|
||||||
if bounds is None:
|
if bounds is None:
|
||||||
@ -139,6 +199,11 @@ def check_target_sources(target_chapter: int, sources: Sequence[Mapping[str, Any
|
|||||||
if bounds is None and "from_order" in source:
|
if bounds is None and "from_order" in source:
|
||||||
value = f"{source.get('from_order')}-{source.get('to_order')}"
|
value = f"{source.get('from_order')}-{source.get('to_order')}"
|
||||||
bounds = normalize_chapter_range(value)
|
bounds = normalize_chapter_range(value)
|
||||||
|
if bounds is None:
|
||||||
|
scope = str(source.get("scope") or "chapter").lower()
|
||||||
|
if scope not in {"work", "authorization", "metadata"}:
|
||||||
|
errors.append(f"source[{index}] 缺少可验证绝对章号或完整区间")
|
||||||
|
continue
|
||||||
if bounds is not None and bounds[1] >= target:
|
if bounds is not None and bounds[1] >= target:
|
||||||
errors.append(f"source[{index}] 包含目标章或未来章: {bounds[0]}-{bounds[1]}")
|
errors.append(f"source[{index}] 包含目标章或未来章: {bounds[0]}-{bounds[1]}")
|
||||||
return _result(STATUS_TARGET_SOURCE_FORBIDDEN if errors else STATUS_READY, errors)
|
return _result(STATUS_TARGET_SOURCE_FORBIDDEN if errors else STATUS_READY, errors)
|
||||||
@ -160,13 +225,50 @@ def _without_card_fields(value: Any) -> Any:
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _canonical_hash(value: Any) -> str:
|
||||||
|
"""用 JSON 类型保真的规范序列化比较公共区。"""
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
|
||||||
|
encoded = json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
||||||
|
return hashlib.sha256(encoded).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def _check_arm_semantics(name: str, manifest: Mapping[str, Any]) -> list[str]:
|
||||||
|
"""校验生产三臂的无卡/正确卡/placebo 语义。"""
|
||||||
|
|
||||||
|
count = manifest.get("cardInjectionCount", 0)
|
||||||
|
source_ids = manifest.get("cardSourceIds", [])
|
||||||
|
strategy = str(manifest.get("cardStrategy") or "none")
|
||||||
|
has_payload = bool(manifest.get("cardInjection"))
|
||||||
|
if name == "outline_only":
|
||||||
|
if has_payload or count not in (0, None) or source_ids or strategy not in {"none", ""}:
|
||||||
|
return ["outline_only 必须不含卡注入"]
|
||||||
|
return []
|
||||||
|
if name not in {"outline_plus_cards", "outline_plus_placebo_cards"}:
|
||||||
|
return [f"未登记评测臂: {name}"]
|
||||||
|
if has_payload or not isinstance(count, int) or count <= 0 or not source_ids:
|
||||||
|
return [f"{name} 必须包含带来源的卡注入元数据"]
|
||||||
|
expected_strategy = "correct" if name == "outline_plus_cards" else "placebo"
|
||||||
|
if strategy != expected_strategy:
|
||||||
|
return [f"{name} 的 cardStrategy 必须是 {expected_strategy}"]
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
def check_arm_manifests(
|
def check_arm_manifests(
|
||||||
manifests: Mapping[str, Mapping[str, Any]],
|
manifests: Mapping[str, Mapping[str, Any]],
|
||||||
required_arms: Sequence[str] | None = None,
|
required_arms: Sequence[str] | None = None,
|
||||||
|
*,
|
||||||
|
as_of_chapter: int | None = None,
|
||||||
|
target_chapter: int | None = None,
|
||||||
|
enforce_semantics: bool = False,
|
||||||
|
smoke: bool = False,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""确保三臂除卡注入区外完全一致。"""
|
"""确保三臂除卡注入区外完全一致。"""
|
||||||
|
|
||||||
required = set(required_arms or ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards"))
|
required = set(required_arms or ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards"))
|
||||||
|
if not smoke and required != {"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"}:
|
||||||
|
return _result(STATUS_INVALID_ARM_DIFF, ["生产回放必须使用完整三臂;双臂只允许 smoke"])
|
||||||
actual = set(manifests)
|
actual = set(manifests)
|
||||||
missing = sorted(required - actual)
|
missing = sorted(required - actual)
|
||||||
if missing:
|
if missing:
|
||||||
@ -177,9 +279,34 @@ def check_arm_manifests(
|
|||||||
if any(not isinstance(manifest, Mapping) for manifest in manifests.values()):
|
if any(not isinstance(manifest, Mapping) for manifest in manifests.values()):
|
||||||
return _result(STATUS_INVALID_ARM_DIFF, ["评测臂 manifest 必须是对象"])
|
return _result(STATUS_INVALID_ARM_DIFF, ["评测臂 manifest 必须是对象"])
|
||||||
|
|
||||||
|
if as_of_chapter is not None or target_chapter is not None:
|
||||||
|
expected_as_of = normalize_chapter(as_of_chapter)
|
||||||
|
expected_target = normalize_chapter(target_chapter)
|
||||||
|
if expected_as_of is None or expected_target != expected_as_of + 1:
|
||||||
|
return _result(STATUS_INVALID_SNAPSHOT, ["as_of/target 不是连续章节"])
|
||||||
|
for name in sorted(required):
|
||||||
|
manifest = manifests[name]
|
||||||
|
if normalize_chapter(manifest.get("asOfChapter")) != expected_as_of:
|
||||||
|
return _result(STATUS_INVALID_SNAPSHOT, [f"{name} 的 asOfChapter 不一致"])
|
||||||
|
if normalize_chapter(manifest.get("targetChapter")) != expected_target:
|
||||||
|
return _result(STATUS_INVALID_SNAPSHOT, [f"{name} 的 targetChapter 不一致"])
|
||||||
|
|
||||||
|
if enforce_semantics:
|
||||||
|
semantic_errors = [
|
||||||
|
f"{name}: {error}"
|
||||||
|
for name in sorted(required)
|
||||||
|
for error in _check_arm_semantics(name, manifests[name])
|
||||||
|
]
|
||||||
|
if semantic_errors:
|
||||||
|
return _result(STATUS_INVALID_ARM_DIFF, semantic_errors)
|
||||||
|
|
||||||
names = sorted(required)
|
names = sorted(required)
|
||||||
baseline = _without_card_fields(manifests[names[0]])
|
baseline = _without_card_fields(manifests[names[0]])
|
||||||
differences = [name for name in names[1:] if _without_card_fields(manifests[name]) != baseline]
|
baseline_hash = _canonical_hash(baseline)
|
||||||
|
differences = [
|
||||||
|
name for name in names[1:]
|
||||||
|
if _canonical_hash(_without_card_fields(manifests[name])) != baseline_hash
|
||||||
|
]
|
||||||
if differences:
|
if differences:
|
||||||
return _result(STATUS_INVALID_ARM_DIFF, [f"公共输入区不一致: {','.join(differences)}"])
|
return _result(STATUS_INVALID_ARM_DIFF, [f"公共输入区不一致: {','.join(differences)}"])
|
||||||
return _result(STATUS_READY)
|
return _result(STATUS_READY)
|
||||||
@ -188,7 +315,7 @@ def check_arm_manifests(
|
|||||||
def check_candidate_output(
|
def check_candidate_output(
|
||||||
candidate: Mapping[str, Any],
|
candidate: Mapping[str, Any],
|
||||||
target_chapter: int,
|
target_chapter: int,
|
||||||
source_catalog: Sequence[Mapping[str, Any]] = (),
|
source_catalog: Sequence[Mapping[str, Any]] | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""校验细纲候选的结构边界,不判断内容是否命中标准答案。"""
|
"""校验细纲候选的结构边界,不判断内容是否命中标准答案。"""
|
||||||
|
|
||||||
@ -198,6 +325,9 @@ def check_candidate_output(
|
|||||||
if expected_target is None:
|
if expected_target is None:
|
||||||
return _result(STATUS_INVALID_SNAPSHOT, ["目标章号无效"])
|
return _result(STATUS_INVALID_SNAPSHOT, ["目标章号无效"])
|
||||||
errors = [f"缺少字段: {field}" for field in REQUIRED_CANDIDATE_FIELDS if field not in candidate]
|
errors = [f"缺少字段: {field}" for field in REQUIRED_CANDIDATE_FIELDS if field not in candidate]
|
||||||
|
unexpected = sorted(set(candidate) - ALLOWED_CANDIDATE_FIELDS)
|
||||||
|
if unexpected:
|
||||||
|
errors.append(f"候选包含未登记字段: {','.join(unexpected)}")
|
||||||
forbidden = sorted(set(candidate) & FORBIDDEN_CANDIDATE_FIELDS)
|
forbidden = sorted(set(candidate) & FORBIDDEN_CANDIDATE_FIELDS)
|
||||||
if forbidden:
|
if forbidden:
|
||||||
errors.append(f"候选包含正文/原文字段: {','.join(forbidden)}")
|
errors.append(f"候选包含正文/原文字段: {','.join(forbidden)}")
|
||||||
@ -216,22 +346,73 @@ def check_candidate_output(
|
|||||||
errors.append("keyEvents 每项必须是对象")
|
errors.append("keyEvents 每项必须是对象")
|
||||||
if any(isinstance(item, Mapping) and not item.get("id") for item in events):
|
if any(isinstance(item, Mapping) and not item.get("id") for item in events):
|
||||||
errors.append("keyEvents 每项必须有 id")
|
errors.append("keyEvents 每项必须有 id")
|
||||||
|
for index, event in enumerate(events):
|
||||||
|
if not isinstance(event, Mapping):
|
||||||
|
continue
|
||||||
|
unexpected_event_fields = sorted(set(event) - set(EVENT_FIELDS))
|
||||||
|
if unexpected_event_fields:
|
||||||
|
errors.append(f"keyEvents[{index}] 包含未登记字段: {','.join(unexpected_event_fields)}")
|
||||||
|
for field, expected_type in EVENT_FIELDS.items():
|
||||||
|
if field not in event:
|
||||||
|
errors.append(f"keyEvents[{index}] 缺少字段: {field}")
|
||||||
|
elif expected_type is int and (isinstance(event[field], bool) or not isinstance(event[field], int)):
|
||||||
|
errors.append(f"keyEvents[{index}].{field} 必须是整数")
|
||||||
|
elif expected_type is not int and not isinstance(event[field], expected_type):
|
||||||
|
errors.append(f"keyEvents[{index}].{field} 类型错误")
|
||||||
|
entities = candidate.get("entities")
|
||||||
|
if isinstance(entities, list):
|
||||||
|
for index, entity in enumerate(entities):
|
||||||
|
if not isinstance(entity, Mapping):
|
||||||
|
errors.append(f"entities[{index}] 必须是对象")
|
||||||
|
continue
|
||||||
|
for field, expected_type in ENTITY_FIELDS.items():
|
||||||
|
if field not in entity:
|
||||||
|
errors.append(f"entities[{index}] 缺少字段: {field}")
|
||||||
|
elif not isinstance(entity[field], expected_type):
|
||||||
|
errors.append(f"entities[{index}].{field} 类型错误")
|
||||||
|
unexpected_entity_fields = sorted(set(entity) - set(ENTITY_FIELDS))
|
||||||
|
if unexpected_entity_fields:
|
||||||
|
errors.append(f"entities[{index}] 包含未登记字段: {','.join(unexpected_entity_fields)}")
|
||||||
|
foreshadowing = candidate.get("foreshadowing")
|
||||||
|
if isinstance(foreshadowing, list):
|
||||||
|
for index, item in enumerate(foreshadowing):
|
||||||
|
if not isinstance(item, Mapping):
|
||||||
|
errors.append(f"foreshadowing[{index}] 必须是对象")
|
||||||
|
continue
|
||||||
|
for field, expected_type in FORESHADOWING_FIELDS.items():
|
||||||
|
if field not in item:
|
||||||
|
errors.append(f"foreshadowing[{index}] 缺少字段: {field}")
|
||||||
|
elif not isinstance(item[field], expected_type):
|
||||||
|
errors.append(f"foreshadowing[{index}].{field} 类型错误")
|
||||||
|
unexpected_foreshadowing_fields = sorted(set(item) - set(FORESHADOWING_FIELDS))
|
||||||
|
if unexpected_foreshadowing_fields:
|
||||||
|
errors.append(
|
||||||
|
f"foreshadowing[{index}] 包含未登记字段: {','.join(unexpected_foreshadowing_fields)}"
|
||||||
|
)
|
||||||
|
for field in ("stateChanges", "unknowns", "assumptions"):
|
||||||
|
values = candidate.get(field)
|
||||||
|
if isinstance(values, list) and any(not isinstance(item, str) for item in values):
|
||||||
|
errors.append(f"{field} 每项必须是字符串")
|
||||||
source_errors: list[str] = []
|
source_errors: list[str] = []
|
||||||
source_refs = candidate.get("sourceRefs")
|
source_refs = candidate.get("sourceRefs")
|
||||||
if source_refs is not None:
|
if source_refs is not None:
|
||||||
if not isinstance(source_refs, list):
|
if not isinstance(source_refs, list):
|
||||||
errors.append("sourceRefs 必须是数组")
|
errors.append("sourceRefs 必须是数组")
|
||||||
else:
|
else:
|
||||||
|
if source_catalog is None:
|
||||||
|
return _result(STATUS_SCHEMA_INVALID, ["sourceRefs 校验缺少冻结来源目录"])
|
||||||
catalog = {
|
catalog = {
|
||||||
str(source.get("sourceId")): source
|
str(source.get("sourceId")): source
|
||||||
for source in source_catalog
|
for source in source_catalog
|
||||||
if isinstance(source, Mapping) and source.get("sourceId")
|
if isinstance(source, Mapping) and source.get("sourceId")
|
||||||
}
|
}
|
||||||
for index, reference in enumerate(source_refs):
|
for index, reference in enumerate(source_refs):
|
||||||
source = reference if isinstance(reference, Mapping) else catalog.get(str(reference))
|
if not isinstance(reference, str):
|
||||||
|
errors.append(f"sourceRefs[{index}] 必须是 sourceId 字符串")
|
||||||
|
continue
|
||||||
|
source = catalog.get(reference)
|
||||||
if source is None:
|
if source is None:
|
||||||
if source_catalog:
|
errors.append(f"sourceRefs[{index}] 未登记来源")
|
||||||
errors.append(f"sourceRefs[{index}] 未登记来源")
|
|
||||||
continue
|
continue
|
||||||
source_value = _field(
|
source_value = _field(
|
||||||
source,
|
source,
|
||||||
@ -247,7 +428,11 @@ def check_candidate_output(
|
|||||||
bounds = normalize_chapter_range(source_value)
|
bounds = normalize_chapter_range(source_value)
|
||||||
if bounds is None and "from_order" in source:
|
if bounds is None and "from_order" in source:
|
||||||
bounds = normalize_chapter_range(f"{source.get('from_order')}-{source.get('to_order')}")
|
bounds = normalize_chapter_range(f"{source.get('from_order')}-{source.get('to_order')}")
|
||||||
if bounds is not None and bounds[1] >= expected_target:
|
if bounds is None:
|
||||||
|
scope = str(source.get("scope") or "chapter").lower()
|
||||||
|
if scope not in {"work", "authorization", "metadata"}:
|
||||||
|
source_errors.append(f"sourceRefs[{index}] 缺少可验证绝对章号或完整区间")
|
||||||
|
elif bounds[1] >= expected_target:
|
||||||
source_errors.append(
|
source_errors.append(
|
||||||
f"sourceRefs[{index}] 包含目标章或未来章: {bounds[0]}-{bounds[1]}"
|
f"sourceRefs[{index}] 包含目标章或未来章: {bounds[0]}-{bounds[1]}"
|
||||||
)
|
)
|
||||||
@ -264,17 +449,31 @@ def check_candidate_output(
|
|||||||
def check_replay(
|
def check_replay(
|
||||||
*,
|
*,
|
||||||
authorization: Mapping[str, Any] | None,
|
authorization: Mapping[str, Any] | None,
|
||||||
|
as_of_chapter: int,
|
||||||
target_chapter: int,
|
target_chapter: int,
|
||||||
planner_sources: Sequence[Mapping[str, Any]],
|
planner_sources: Sequence[Mapping[str, Any]],
|
||||||
arm_manifests: Mapping[str, Mapping[str, Any]],
|
arm_manifests: Mapping[str, Mapping[str, Any]],
|
||||||
required_arms: Sequence[str] | None = None,
|
required_arms: Sequence[str] | None = None,
|
||||||
|
smoke: bool = False,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""执行回放前置门,任何一项失败都不允许进入模型调用。"""
|
"""执行回放前置门,任何一项失败都不允许进入模型调用。"""
|
||||||
|
|
||||||
|
as_of = normalize_chapter(as_of_chapter)
|
||||||
|
target = normalize_chapter(target_chapter)
|
||||||
|
if as_of is None or target != as_of + 1:
|
||||||
|
return _result(STATUS_INVALID_SNAPSHOT, ["as_of/target 不是连续章节"])
|
||||||
|
|
||||||
checks = [
|
checks = [
|
||||||
check_authorization(authorization),
|
check_authorization(authorization),
|
||||||
check_target_sources(target_chapter, planner_sources),
|
check_target_sources(target_chapter, planner_sources),
|
||||||
check_arm_manifests(arm_manifests, required_arms),
|
check_arm_manifests(
|
||||||
|
arm_manifests,
|
||||||
|
required_arms,
|
||||||
|
as_of_chapter=as_of,
|
||||||
|
target_chapter=target,
|
||||||
|
enforce_semantics=True,
|
||||||
|
smoke=smoke,
|
||||||
|
),
|
||||||
]
|
]
|
||||||
failures = [check for check in checks if not check["ok"]]
|
failures = [check for check in checks if not check["ok"]]
|
||||||
if failures:
|
if failures:
|
||||||
@ -291,6 +490,7 @@ def _parse_args() -> argparse.Namespace:
|
|||||||
parser.add_argument("--authorization", type=Path, required=True)
|
parser.add_argument("--authorization", type=Path, required=True)
|
||||||
parser.add_argument("--sources", type=Path, required=True)
|
parser.add_argument("--sources", type=Path, required=True)
|
||||||
parser.add_argument("--arms", type=Path, required=True)
|
parser.add_argument("--arms", type=Path, required=True)
|
||||||
|
parser.add_argument("--as-of-chapter", type=int, required=True)
|
||||||
parser.add_argument("--target-chapter", type=int, required=True)
|
parser.add_argument("--target-chapter", type=int, required=True)
|
||||||
parser.add_argument("--output", type=Path, required=True)
|
parser.add_argument("--output", type=Path, required=True)
|
||||||
return parser.parse_args()
|
return parser.parse_args()
|
||||||
@ -303,6 +503,7 @@ def main() -> int:
|
|||||||
arms = json.loads(args.arms.read_text(encoding="utf-8"))
|
arms = json.loads(args.arms.read_text(encoding="utf-8"))
|
||||||
result = check_replay(
|
result = check_replay(
|
||||||
authorization=authorization,
|
authorization=authorization,
|
||||||
|
as_of_chapter=args.as_of_chapter,
|
||||||
target_chapter=args.target_chapter,
|
target_chapter=args.target_chapter,
|
||||||
planner_sources=sources,
|
planner_sources=sources,
|
||||||
arm_manifests=arms,
|
arm_manifests=arms,
|
||||||
|
|||||||
353
.claude/skills/replay-eval/scripts/run_replay.py
Normal file
353
.claude/skills/replay-eval/scripts/run_replay.py
Normal file
@ -0,0 +1,353 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""细纲回放编排器。
|
||||||
|
|
||||||
|
默认只做 dry-run。真实模式显式调用本地 Claude CLI,并将原始候选限定在运行目录;
|
||||||
|
最终结果只返回状态、哈希、评分入口和失败类别,不把原始 prompt/response 带出运行目录。
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
from build_snapshot import build_snapshot, normalize_chapter, sha256_value
|
||||||
|
from check_snapshot import (
|
||||||
|
STATUS_READY,
|
||||||
|
check_candidate_output,
|
||||||
|
check_replay,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
REQUIRED_ARMS = ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards")
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[4]
|
||||||
|
SKILL_PATH = REPO_ROOT / ".claude/skills/fine-outline/SKILL.md"
|
||||||
|
PLANNER_PATH = REPO_ROOT / ".claude/agents/planner.md"
|
||||||
|
|
||||||
|
|
||||||
|
class ReplayRunError(ValueError):
|
||||||
|
"""回放配置不符合运行边界。"""
|
||||||
|
|
||||||
|
|
||||||
|
def _read_json(path: Path) -> Any:
|
||||||
|
return json.loads(path.read_text(encoding="utf-8"))
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_json(value: Any) -> str:
|
||||||
|
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
||||||
|
|
||||||
|
|
||||||
|
def _require_mapping(config: Mapping[str, Any], key: str) -> Mapping[str, Any]:
|
||||||
|
value = config.get(key)
|
||||||
|
if not isinstance(value, Mapping):
|
||||||
|
raise ReplayRunError(f"配置缺少对象字段: {key}")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _build_arm_manifest(
|
||||||
|
*,
|
||||||
|
name: str,
|
||||||
|
arm: Mapping[str, Any],
|
||||||
|
common_input: Mapping[str, Any],
|
||||||
|
as_of: int,
|
||||||
|
target: int,
|
||||||
|
snapshot_version: str,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
cards = arm.get("cards", [])
|
||||||
|
if not isinstance(cards, list):
|
||||||
|
raise ReplayRunError(f"{name}.cards 必须是数组")
|
||||||
|
source_ids = arm.get("cardSourceIds", [])
|
||||||
|
if not isinstance(source_ids, list):
|
||||||
|
raise ReplayRunError(f"{name}.cardSourceIds 必须是数组")
|
||||||
|
strategy = str(arm.get("cardStrategy") or ("none" if name == "outline_only" else "correct"))
|
||||||
|
return {
|
||||||
|
"arm": name,
|
||||||
|
"snapshotVersion": snapshot_version,
|
||||||
|
"asOfChapter": as_of,
|
||||||
|
"targetChapter": target,
|
||||||
|
"commonInputSha256": sha256_value(common_input),
|
||||||
|
"cardInjectionSha256": sha256_value(cards),
|
||||||
|
"cardInjectionCount": len(cards),
|
||||||
|
"cardSourceIds": [str(item) for item in source_ids],
|
||||||
|
"cardStrategy": strategy,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_candidate(output: str) -> Mapping[str, Any]:
|
||||||
|
"""兼容 Claude JSON 外壳和代码围栏,但不把原始文本返回调用方。"""
|
||||||
|
|
||||||
|
outer: Any = output
|
||||||
|
try:
|
||||||
|
outer = json.loads(output)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
pass
|
||||||
|
if isinstance(outer, Mapping) and isinstance(outer.get("result"), str):
|
||||||
|
outer = outer["result"]
|
||||||
|
if isinstance(outer, str):
|
||||||
|
match = re.search(r"```(?:json)?\s*(\{.*\})\s*```", outer, re.DOTALL)
|
||||||
|
outer = match.group(1) if match else outer.strip()
|
||||||
|
outer = json.loads(outer)
|
||||||
|
if not isinstance(outer, Mapping):
|
||||||
|
raise ReplayRunError("planner 输出不是 JSON 对象")
|
||||||
|
return outer
|
||||||
|
|
||||||
|
|
||||||
|
def _planner_prompt(
|
||||||
|
*,
|
||||||
|
target: int,
|
||||||
|
as_of: int,
|
||||||
|
snapshot: Mapping[str, Any],
|
||||||
|
common_input: Mapping[str, Any],
|
||||||
|
cards: list[Any],
|
||||||
|
) -> str:
|
||||||
|
"""只把冻结后的结构化资料和功能合同送入 planner。"""
|
||||||
|
|
||||||
|
skill = SKILL_PATH.read_text(encoding="utf-8")
|
||||||
|
identity = PLANNER_PATH.read_text(encoding="utf-8")
|
||||||
|
context = {
|
||||||
|
"targetChapter": target,
|
||||||
|
"asOfChapter": as_of,
|
||||||
|
"frozenSnapshot": snapshot,
|
||||||
|
"commonContext": common_input,
|
||||||
|
"cardInjection": cards,
|
||||||
|
}
|
||||||
|
return "\n".join(
|
||||||
|
[
|
||||||
|
"这是 next_fine_outline_replay_v0 的离线规划任务。只输出一个 JSON 对象,不要 Markdown、正文或解释。",
|
||||||
|
"你不能调用工具,也不能读取仓库、目标章或任何未列入下面 JSON 的资料。",
|
||||||
|
"严格执行 fine-outline 合同;目标章号必须保持不变,未知内容写入 unknowns/assumptions。",
|
||||||
|
"规划上下文冻结到 as_of;卡只是事实索引和补充,不得替代公共大纲与叙事现在时。",
|
||||||
|
"--- planner identity ---",
|
||||||
|
identity,
|
||||||
|
"--- fine-outline skill ---",
|
||||||
|
skill,
|
||||||
|
"--- frozen input ---",
|
||||||
|
_safe_json(context),
|
||||||
|
"--- output contract ---",
|
||||||
|
_safe_json(
|
||||||
|
{
|
||||||
|
"targetChapter": target,
|
||||||
|
"requiredFields": [
|
||||||
|
"chapterGoal",
|
||||||
|
"keyEvents",
|
||||||
|
"entities",
|
||||||
|
"foreshadowing",
|
||||||
|
"stateChanges",
|
||||||
|
"hook",
|
||||||
|
"unknowns",
|
||||||
|
"assumptions",
|
||||||
|
],
|
||||||
|
"eventFields": ["id", "order", "event", "participants", "trigger", "resultDirection"],
|
||||||
|
"entityFields": ["name", "type", "role"],
|
||||||
|
"foreshadowingFields": ["action", "subject", "evidence"],
|
||||||
|
"sourceRefs": "可选;只能引用冻结来源 ID",
|
||||||
|
}
|
||||||
|
),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _invoke_planner(
|
||||||
|
*,
|
||||||
|
prompt: str,
|
||||||
|
planner_bin: str,
|
||||||
|
model: str,
|
||||||
|
output_path: Path,
|
||||||
|
max_budget_usd: float,
|
||||||
|
) -> Mapping[str, Any]:
|
||||||
|
"""用无工具、无会话持久化的 Claude print 模式运行 planner。"""
|
||||||
|
|
||||||
|
command = [
|
||||||
|
planner_bin,
|
||||||
|
"-p",
|
||||||
|
"--agent",
|
||||||
|
"planner",
|
||||||
|
"--model",
|
||||||
|
model,
|
||||||
|
"--tools",
|
||||||
|
"",
|
||||||
|
"--no-session-persistence",
|
||||||
|
"--output-format",
|
||||||
|
"json",
|
||||||
|
"--max-budget-usd",
|
||||||
|
str(max_budget_usd),
|
||||||
|
"--append-system-prompt",
|
||||||
|
"本次是严格离线回放;不要调用任何工具,不要读取文件,不要输出 JSON 以外内容。",
|
||||||
|
prompt,
|
||||||
|
]
|
||||||
|
completed = subprocess.run(command, text=True, capture_output=True, check=False)
|
||||||
|
output_path.write_text(completed.stdout, encoding="utf-8")
|
||||||
|
if completed.returncode != 0:
|
||||||
|
raise ReplayRunError(f"planner 调用失败,退出码={completed.returncode}")
|
||||||
|
return _extract_candidate(completed.stdout)
|
||||||
|
|
||||||
|
|
||||||
|
def run_replay(
|
||||||
|
config: Mapping[str, Any],
|
||||||
|
output_dir: Path,
|
||||||
|
*,
|
||||||
|
mode: str = "dry_run",
|
||||||
|
planner_bin: str = "claude",
|
||||||
|
model: str = "opus",
|
||||||
|
max_budget_usd: float = 1.0,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""执行一次单目标三臂回放;任何前置门失败都不调用模型。"""
|
||||||
|
|
||||||
|
if mode not in {"dry_run", "execute"}:
|
||||||
|
raise ReplayRunError("mode 只能是 dry_run 或 execute")
|
||||||
|
output_dir = output_dir.resolve()
|
||||||
|
if output_dir.is_relative_to(REPO_ROOT.resolve()):
|
||||||
|
raise ReplayRunError("原始候选运行目录不得位于仓库内")
|
||||||
|
output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
snapshot_config = _require_mapping(config, "snapshot")
|
||||||
|
as_of = normalize_chapter(snapshot_config.get("asOfChapter"))
|
||||||
|
target = normalize_chapter(config.get("targetChapter"))
|
||||||
|
snapshot_version = str(snapshot_config.get("snapshotVersion") or "")
|
||||||
|
if as_of is None or target is None:
|
||||||
|
raise ReplayRunError("as_of/target 必须是明确正整数")
|
||||||
|
if target != as_of + 1:
|
||||||
|
raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1")
|
||||||
|
reference_work = _require_mapping(config, "referenceWork")
|
||||||
|
authorization = _require_mapping(config, "authorization")
|
||||||
|
sources = config.get("sources", [])
|
||||||
|
if not isinstance(sources, list):
|
||||||
|
raise ReplayRunError("sources 必须是数组")
|
||||||
|
common_input = _require_mapping(config, "commonContext")
|
||||||
|
arms = _require_mapping(config, "arms")
|
||||||
|
if set(arms) != set(REQUIRED_ARMS):
|
||||||
|
raise ReplayRunError("生产回放必须精确配置三臂")
|
||||||
|
|
||||||
|
arm_manifests = {
|
||||||
|
name: _build_arm_manifest(
|
||||||
|
name=name,
|
||||||
|
arm=_require_mapping(arms, name),
|
||||||
|
common_input=common_input,
|
||||||
|
as_of=as_of,
|
||||||
|
target=target,
|
||||||
|
snapshot_version=snapshot_version,
|
||||||
|
)
|
||||||
|
for name in REQUIRED_ARMS
|
||||||
|
}
|
||||||
|
preflight = check_replay(
|
||||||
|
authorization=authorization,
|
||||||
|
as_of_chapter=as_of,
|
||||||
|
target_chapter=target,
|
||||||
|
planner_sources=sources,
|
||||||
|
arm_manifests=arm_manifests,
|
||||||
|
)
|
||||||
|
result: dict[str, Any] = {
|
||||||
|
"runId": str(config.get("runId") or "unassigned"),
|
||||||
|
"mode": mode,
|
||||||
|
"status": preflight["status"],
|
||||||
|
"ok": preflight["ok"],
|
||||||
|
"referenceWork": str(reference_work.get("id") or ""),
|
||||||
|
"referenceWorkVersion": str(reference_work.get("version") or ""),
|
||||||
|
"asOfChapter": as_of,
|
||||||
|
"targetChapter": target,
|
||||||
|
"snapshotVersion": snapshot_version,
|
||||||
|
"preflight": {"status": preflight["status"], "errors": preflight["errors"], "warnings": preflight["warnings"]},
|
||||||
|
"arms": arm_manifests,
|
||||||
|
"results": {},
|
||||||
|
}
|
||||||
|
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||||||
|
if not preflight["ok"]:
|
||||||
|
return result
|
||||||
|
|
||||||
|
metadata = {
|
||||||
|
"targetChapter": target,
|
||||||
|
"referenceWork": reference_work,
|
||||||
|
"evaluationSetVersion": config.get("evaluationSetVersion", "unregistered"),
|
||||||
|
"strategyVersion": config.get("strategyVersion", "unregistered"),
|
||||||
|
"authorizationSnapshot": authorization["authorizationSnapshot"],
|
||||||
|
"runPermissions": config.get("runPermissions", {"purpose": "offline_evaluation", "mode": mode}),
|
||||||
|
"armConfig": {"arms": list(REQUIRED_ARMS)},
|
||||||
|
}
|
||||||
|
snapshot_data = snapshot_config.get("data", {})
|
||||||
|
frozen = build_snapshot(
|
||||||
|
snapshot_data,
|
||||||
|
as_of,
|
||||||
|
snapshot_version,
|
||||||
|
target_chapter=target,
|
||||||
|
manifest_metadata=metadata,
|
||||||
|
)
|
||||||
|
(output_dir / "snapshot_manifest.json").write_text(_safe_json(frozen["manifest"]) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
if mode == "dry_run":
|
||||||
|
result["status"] = STATUS_READY
|
||||||
|
result["ok"] = True
|
||||||
|
result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"]
|
||||||
|
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||||||
|
return result
|
||||||
|
|
||||||
|
for name in REQUIRED_ARMS:
|
||||||
|
arm = _require_mapping(arms, name)
|
||||||
|
cards = arm.get("cards", [])
|
||||||
|
raw_path = output_dir / f"planner_{name}.raw.json"
|
||||||
|
prompt = _planner_prompt(
|
||||||
|
target=target,
|
||||||
|
as_of=as_of,
|
||||||
|
snapshot=frozen["snapshot"],
|
||||||
|
common_input=common_input,
|
||||||
|
cards=cards,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
candidate = _invoke_planner(
|
||||||
|
prompt=prompt,
|
||||||
|
planner_bin=planner_bin,
|
||||||
|
model=model,
|
||||||
|
output_path=raw_path,
|
||||||
|
max_budget_usd=max_budget_usd,
|
||||||
|
)
|
||||||
|
candidate_path = output_dir / f"candidate_{name}.json"
|
||||||
|
candidate_path.write_text(_safe_json(candidate) + "\n", encoding="utf-8")
|
||||||
|
schema = check_candidate_output(candidate, target, sources)
|
||||||
|
result["results"][name] = {
|
||||||
|
"status": schema["status"],
|
||||||
|
"ok": schema["ok"],
|
||||||
|
"errors": schema["errors"],
|
||||||
|
"candidateSha256": sha256_value(candidate),
|
||||||
|
"candidatePath": str(candidate_path),
|
||||||
|
}
|
||||||
|
except (ReplayRunError, json.JSONDecodeError) as error:
|
||||||
|
result["results"][name] = {
|
||||||
|
"status": "planner_output_invalid",
|
||||||
|
"ok": False,
|
||||||
|
"errors": [str(error)],
|
||||||
|
"rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None,
|
||||||
|
}
|
||||||
|
result["status"] = "completed" if all(item["ok"] for item in result["results"].values()) else "candidate_blocked"
|
||||||
|
result["ok"] = result["status"] == "completed"
|
||||||
|
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(description="运行 next_fine_outline_replay_v0")
|
||||||
|
parser.add_argument("--config", type=Path, required=True)
|
||||||
|
parser.add_argument("--output-dir", type=Path, required=True)
|
||||||
|
parser.add_argument("--mode", choices=("dry_run", "execute"), default="dry_run")
|
||||||
|
parser.add_argument("--planner-bin", default="claude")
|
||||||
|
parser.add_argument("--model", default="opus")
|
||||||
|
parser.add_argument("--max-budget-usd", type=float, default=1.0)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
args = _parse_args()
|
||||||
|
result = run_replay(
|
||||||
|
_read_json(args.config),
|
||||||
|
args.output_dir,
|
||||||
|
mode=args.mode,
|
||||||
|
planner_bin=args.planner_bin,
|
||||||
|
model=args.model,
|
||||||
|
max_budget_usd=args.max_budget_usd,
|
||||||
|
)
|
||||||
|
return 0 if result["ok"] else 2
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@ -17,6 +17,26 @@ from build_snapshot import ( # noqa: E402
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
META = {
|
||||||
|
"targetChapter": 489,
|
||||||
|
"referenceWork": {"id": "deep-space", "version": "v1"},
|
||||||
|
"evaluationSetVersion": "set-v1",
|
||||||
|
"strategyVersion": "strategy-v1",
|
||||||
|
"authorizationSnapshot": {
|
||||||
|
"id": "auth-1",
|
||||||
|
"version": "v1",
|
||||||
|
"immutable": True,
|
||||||
|
"sourceVersion": "work-v1",
|
||||||
|
"sourceStatus": "active",
|
||||||
|
"allowedPurpose": ["offline_evaluation"],
|
||||||
|
"checkedAt": "2026-07-19T00:00:00Z",
|
||||||
|
"revalidationAt": "2026-07-20T00:00:00Z",
|
||||||
|
},
|
||||||
|
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
|
||||||
|
"armConfig": {"arms": ["outline_only", "outline_plus_cards", "outline_plus_placebo_cards"]},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class BuildSnapshotTest(unittest.TestCase):
|
class BuildSnapshotTest(unittest.TestCase):
|
||||||
def test_normalize_chapter_rejects_guessing(self):
|
def test_normalize_chapter_rejects_guessing(self):
|
||||||
self.assertEqual(normalize_chapter(488), 488)
|
self.assertEqual(normalize_chapter(488), 488)
|
||||||
@ -80,6 +100,8 @@ class BuildSnapshotTest(unittest.TestCase):
|
|||||||
},
|
},
|
||||||
488,
|
488,
|
||||||
"v0",
|
"v0",
|
||||||
|
target_chapter=489,
|
||||||
|
manifest_metadata=META,
|
||||||
)
|
)
|
||||||
card = result["snapshot"]["cards"][0]
|
card = result["snapshot"]["cards"][0]
|
||||||
self.assertEqual(card["milestones"], [{"chapter": 488, "step": "可见"}])
|
self.assertEqual(card["milestones"], [{"chapter": 488, "step": "可见"}])
|
||||||
@ -89,9 +111,16 @@ class BuildSnapshotTest(unittest.TestCase):
|
|||||||
def test_manifest_is_stable_and_does_not_store_payload(self):
|
def test_manifest_is_stable_and_does_not_store_payload(self):
|
||||||
kwargs = {
|
kwargs = {
|
||||||
"as_of": 488,
|
"as_of": 488,
|
||||||
|
"target_chapter": 489,
|
||||||
"snapshot_version": "v0",
|
"snapshot_version": "v0",
|
||||||
|
"reference_work": META["referenceWork"],
|
||||||
|
"evaluation_set_version": META["evaluationSetVersion"],
|
||||||
|
"strategy_version": META["strategyVersion"],
|
||||||
|
"authorization_snapshot": META["authorizationSnapshot"],
|
||||||
|
"run_permissions": META["runPermissions"],
|
||||||
|
"arm_config": META["armConfig"],
|
||||||
"sections": {"l0": {"sourceId": "task", "sourceVersion": "1", "payload": {"target": 489}}},
|
"sections": {"l0": {"sourceId": "task", "sourceVersion": "1", "payload": {"target": 489}}},
|
||||||
"final_report": {"status": "ready", "sourceHash": "abc"},
|
"final_report": {"status": "ready", "hashes": {"sourceHash": "abc"}},
|
||||||
}
|
}
|
||||||
first = build_snapshot_manifest(**kwargs)
|
first = build_snapshot_manifest(**kwargs)
|
||||||
second = build_snapshot_manifest(**kwargs)
|
second = build_snapshot_manifest(**kwargs)
|
||||||
@ -104,11 +133,62 @@ class BuildSnapshotTest(unittest.TestCase):
|
|||||||
with self.assertRaises(SnapshotError):
|
with self.assertRaises(SnapshotError):
|
||||||
build_snapshot_manifest(
|
build_snapshot_manifest(
|
||||||
as_of=488,
|
as_of=488,
|
||||||
|
target_chapter=489,
|
||||||
snapshot_version="v0",
|
snapshot_version="v0",
|
||||||
|
reference_work=META["referenceWork"],
|
||||||
|
evaluation_set_version=META["evaluationSetVersion"],
|
||||||
|
strategy_version=META["strategyVersion"],
|
||||||
|
authorization_snapshot=META["authorizationSnapshot"],
|
||||||
|
run_permissions=META["runPermissions"],
|
||||||
|
arm_config=META["armConfig"],
|
||||||
sections={},
|
sections={},
|
||||||
final_report={"raw_text": "原书全文"},
|
final_report={"raw_text": "原书全文"},
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_unknown_chapter_collection_is_filtered_and_terminal_fields_are_audited(self):
|
||||||
|
result = build_snapshot(
|
||||||
|
{
|
||||||
|
"chapters": [{"chapter": 489, "fact": "未来"}, {"chapter": 488, "fact": "已知"}],
|
||||||
|
"cards": [{"name": "事件", "event_result": "未来结果", "relationshipPlan": "未来计划"}],
|
||||||
|
},
|
||||||
|
488,
|
||||||
|
"v0",
|
||||||
|
target_chapter=489,
|
||||||
|
manifest_metadata=META,
|
||||||
|
)
|
||||||
|
self.assertEqual(result["snapshot"]["chapters"], [{"chapter": 488, "fact": "已知"}])
|
||||||
|
self.assertNotIn("event_result", result["snapshot"]["cards"][0])
|
||||||
|
self.assertNotIn("relationshipPlan", result["snapshot"]["cards"][0])
|
||||||
|
reasons = [item["reason"] for item in result["manifest"]["omittedSources"]]
|
||||||
|
self.assertIn("future_or_crosses_as_of", reasons)
|
||||||
|
self.assertTrue(any(item["reason"] == "terminal_field_removed" for item in result["manifest"]["omittedFields"]))
|
||||||
|
|
||||||
|
def test_unknown_top_level_section_fails_closed(self):
|
||||||
|
with self.assertRaises(SnapshotError):
|
||||||
|
build_snapshot(
|
||||||
|
{"chapters": [], "futurePayload": "不得透传"},
|
||||||
|
488,
|
||||||
|
"v0",
|
||||||
|
target_chapter=489,
|
||||||
|
manifest_metadata=META,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_final_report_is_closed_set(self):
|
||||||
|
with self.assertRaises(SnapshotError):
|
||||||
|
build_snapshot_manifest(
|
||||||
|
target_chapter=489,
|
||||||
|
as_of=488,
|
||||||
|
snapshot_version="v0",
|
||||||
|
reference_work=META["referenceWork"],
|
||||||
|
evaluation_set_version=META["evaluationSetVersion"],
|
||||||
|
strategy_version=META["strategyVersion"],
|
||||||
|
authorization_snapshot=META["authorizationSnapshot"],
|
||||||
|
run_permissions=META["runPermissions"],
|
||||||
|
arm_config=META["armConfig"],
|
||||||
|
sections={},
|
||||||
|
final_report={"payload": "原书全文"},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
@ -22,8 +22,19 @@ from check_snapshot import ( # noqa: E402
|
|||||||
|
|
||||||
AUTH = {
|
AUTH = {
|
||||||
"sourceStatus": "active",
|
"sourceStatus": "active",
|
||||||
|
"copyrightStatus": "licensed",
|
||||||
|
"sourceVersion": "work-v1",
|
||||||
"allowedPurpose": ["offline_evaluation"],
|
"allowedPurpose": ["offline_evaluation"],
|
||||||
"authorizationSnapshot": {"id": "auth-1", "version": "v1"},
|
"authorizationSnapshot": {
|
||||||
|
"id": "auth-1",
|
||||||
|
"version": "v1",
|
||||||
|
"immutable": True,
|
||||||
|
"sourceVersion": "work-v1",
|
||||||
|
"sourceStatus": "active",
|
||||||
|
"allowedPurpose": ["offline_evaluation"],
|
||||||
|
"checkedAt": "2026-07-19T00:00:00Z",
|
||||||
|
"revalidationAt": "2026-07-20T00:00:00Z",
|
||||||
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@ -35,6 +46,7 @@ class CheckSnapshotTest(unittest.TestCase):
|
|||||||
def test_authorization_is_fail_closed(self):
|
def test_authorization_is_fail_closed(self):
|
||||||
self.assertEqual(check_authorization(None)["status"], STATUS_BLOCKED_AUTHORIZATION)
|
self.assertEqual(check_authorization(None)["status"], STATUS_BLOCKED_AUTHORIZATION)
|
||||||
self.assertTrue(check_authorization(AUTH)["ok"])
|
self.assertTrue(check_authorization(AUTH)["ok"])
|
||||||
|
self.assertEqual(check_authorization({"sourceStatus": "active"})["status"], STATUS_BLOCKED_AUTHORIZATION)
|
||||||
denied = {**AUTH, "allowedPurpose": ["read"]}
|
denied = {**AUTH, "allowedPurpose": ["read"]}
|
||||||
self.assertEqual(check_authorization(denied)["status"], STATUS_BLOCKED_AUTHORIZATION)
|
self.assertEqual(check_authorization(denied)["status"], STATUS_BLOCKED_AUTHORIZATION)
|
||||||
unknown_status = {**AUTH, "sourceStatus": "temporary"}
|
unknown_status = {**AUTH, "sourceStatus": "temporary"}
|
||||||
@ -43,9 +55,15 @@ class CheckSnapshotTest(unittest.TestCase):
|
|||||||
self.assertEqual(check_authorization(unlicensed)["status"], STATUS_BLOCKED_AUTHORIZATION)
|
self.assertEqual(check_authorization(unlicensed)["status"], STATUS_BLOCKED_AUTHORIZATION)
|
||||||
|
|
||||||
def test_target_source_is_blocked(self):
|
def test_target_source_is_blocked(self):
|
||||||
allowed = check_target_sources(489, [{"chapter": 488}, {"from_order": 450, "to_order": 482}])
|
allowed = check_target_sources(
|
||||||
|
489,
|
||||||
|
[
|
||||||
|
{"sourceId": "chapter-488", "sourceVersion": "v1", "chapter": 488},
|
||||||
|
{"sourceId": "outline-450-482", "sourceVersion": "v1", "from_order": 450, "to_order": 482},
|
||||||
|
],
|
||||||
|
)
|
||||||
self.assertEqual(allowed["status"], STATUS_READY)
|
self.assertEqual(allowed["status"], STATUS_READY)
|
||||||
blocked = check_target_sources(489, [{"chapter": 489}])
|
blocked = check_target_sources(489, [{"sourceId": "chapter-489", "sourceVersion": "v1", "chapter": 489}])
|
||||||
self.assertEqual(blocked["status"], STATUS_TARGET_SOURCE_FORBIDDEN)
|
self.assertEqual(blocked["status"], STATUS_TARGET_SOURCE_FORBIDDEN)
|
||||||
|
|
||||||
def test_arm_common_input_must_match(self):
|
def test_arm_common_input_must_match(self):
|
||||||
@ -65,7 +83,7 @@ class CheckSnapshotTest(unittest.TestCase):
|
|||||||
def test_two_arm_smoke_can_be_explicit(self):
|
def test_two_arm_smoke_can_be_explicit(self):
|
||||||
common = {"snapshotVersion": "v0", "asOfChapter": 488}
|
common = {"snapshotVersion": "v0", "asOfChapter": 488}
|
||||||
manifests = {"outline_only": arm(common, []), "outline_plus_cards": arm(common, [{"id": 1}])}
|
manifests = {"outline_only": arm(common, []), "outline_plus_cards": arm(common, [{"id": 1}])}
|
||||||
self.assertTrue(check_arm_manifests(manifests, ["outline_only", "outline_plus_cards"])["ok"])
|
self.assertTrue(check_arm_manifests(manifests, ["outline_only", "outline_plus_cards"], smoke=True)["ok"])
|
||||||
|
|
||||||
def test_candidate_contract_is_structural(self):
|
def test_candidate_contract_is_structural(self):
|
||||||
candidate = {
|
candidate = {
|
||||||
@ -88,6 +106,8 @@ class CheckSnapshotTest(unittest.TestCase):
|
|||||||
self.assertEqual(check_candidate_output(duplicate, 489)["status"], STATUS_SCHEMA_INVALID)
|
self.assertEqual(check_candidate_output(duplicate, 489)["status"], STATUS_SCHEMA_INVALID)
|
||||||
missing_id = {**candidate, "keyEvents": [{"event": "没有 id"}]}
|
missing_id = {**candidate, "keyEvents": [{"event": "没有 id"}]}
|
||||||
self.assertEqual(check_candidate_output(missing_id, 489)["status"], STATUS_SCHEMA_INVALID)
|
self.assertEqual(check_candidate_output(missing_id, 489)["status"], STATUS_SCHEMA_INVALID)
|
||||||
|
unknown_field = {**candidate, "futureSources": ["target-scaffold"]}
|
||||||
|
self.assertEqual(check_candidate_output(unknown_field, 489)["status"], STATUS_SCHEMA_INVALID)
|
||||||
future_ref = {**candidate, "sourceRefs": ["target-scaffold"]}
|
future_ref = {**candidate, "sourceRefs": ["target-scaffold"]}
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
check_candidate_output(
|
check_candidate_output(
|
||||||
@ -107,6 +127,7 @@ class CheckSnapshotTest(unittest.TestCase):
|
|||||||
}
|
}
|
||||||
result = check_replay(
|
result = check_replay(
|
||||||
authorization=AUTH,
|
authorization=AUTH,
|
||||||
|
as_of_chapter=488,
|
||||||
target_chapter=489,
|
target_chapter=489,
|
||||||
planner_sources=[{"chapter": 489}],
|
planner_sources=[{"chapter": 489}],
|
||||||
arm_manifests=manifests,
|
arm_manifests=manifests,
|
||||||
@ -114,6 +135,53 @@ class CheckSnapshotTest(unittest.TestCase):
|
|||||||
self.assertFalse(result["ok"])
|
self.assertFalse(result["ok"])
|
||||||
self.assertEqual(result["status"], STATUS_TARGET_SOURCE_FORBIDDEN)
|
self.assertEqual(result["status"], STATUS_TARGET_SOURCE_FORBIDDEN)
|
||||||
|
|
||||||
|
def test_production_arm_semantics_and_chapter_binding(self):
|
||||||
|
common = {"snapshotVersion": "v0", "asOfChapter": 488, "targetChapter": 489}
|
||||||
|
manifests = {
|
||||||
|
"outline_only": {
|
||||||
|
**common,
|
||||||
|
"cardInjectionCount": 0,
|
||||||
|
"cardSourceIds": [],
|
||||||
|
"cardStrategy": "none",
|
||||||
|
},
|
||||||
|
"outline_plus_cards": {
|
||||||
|
**common,
|
||||||
|
"cardInjectionCount": 1,
|
||||||
|
"cardSourceIds": ["correct-1"],
|
||||||
|
"cardStrategy": "correct",
|
||||||
|
},
|
||||||
|
"outline_plus_placebo_cards": {
|
||||||
|
**common,
|
||||||
|
"cardInjectionCount": 1,
|
||||||
|
"cardSourceIds": ["placebo-1"],
|
||||||
|
"cardStrategy": "placebo",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
self.assertTrue(
|
||||||
|
check_replay(
|
||||||
|
authorization=AUTH,
|
||||||
|
as_of_chapter=488,
|
||||||
|
target_chapter=489,
|
||||||
|
planner_sources=[{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
|
||||||
|
arm_manifests=manifests,
|
||||||
|
)["ok"]
|
||||||
|
)
|
||||||
|
wrong = {**manifests, "outline_only": {**manifests["outline_only"], "cardInjectionCount": 1}}
|
||||||
|
self.assertEqual(
|
||||||
|
check_replay(
|
||||||
|
authorization=AUTH,
|
||||||
|
as_of_chapter=488,
|
||||||
|
target_chapter=489,
|
||||||
|
planner_sources=[{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
|
||||||
|
arm_manifests=wrong,
|
||||||
|
)["status"],
|
||||||
|
STATUS_INVALID_ARM_DIFF,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_missing_source_chapter_is_blocked(self):
|
||||||
|
result = check_target_sources(489, [{"sourceId": "unknown", "sourceVersion": "v1", "payload": "future"}])
|
||||||
|
self.assertEqual(result["status"], STATUS_TARGET_SOURCE_FORBIDDEN)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
92
.claude/skills/replay-eval/scripts/test_run_replay.py
Normal file
92
.claude/skills/replay-eval/scripts/test_run_replay.py
Normal file
@ -0,0 +1,92 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""回放编排器和安全摘要的无网络测试。"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import pathlib
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent))
|
||||||
|
from run_replay import run_replay # noqa: E402
|
||||||
|
from write_report import render_report # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
AUTH = {
|
||||||
|
"sourceStatus": "active",
|
||||||
|
"copyrightStatus": "licensed",
|
||||||
|
"sourceVersion": "work-v1",
|
||||||
|
"allowedPurpose": ["offline_evaluation"],
|
||||||
|
"authorizationSnapshot": {
|
||||||
|
"id": "auth-1",
|
||||||
|
"version": "v1",
|
||||||
|
"immutable": True,
|
||||||
|
"sourceVersion": "work-v1",
|
||||||
|
"sourceStatus": "active",
|
||||||
|
"allowedPurpose": ["offline_evaluation"],
|
||||||
|
"checkedAt": "2026-07-19T00:00:00Z",
|
||||||
|
"revalidationAt": "2026-07-20T00:00:00Z",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def config():
|
||||||
|
common = {"l0": {"targetChapter": 489}, "l1": {"asOfChapter": 488}, "l2": {"mainline": "安全公共输入"}}
|
||||||
|
return {
|
||||||
|
"runId": "smoke-001",
|
||||||
|
"referenceWork": {"id": "deep-space", "version": "work-v1"},
|
||||||
|
"evaluationSetVersion": "set-v1",
|
||||||
|
"strategyVersion": "strategy-v1",
|
||||||
|
"authorization": AUTH,
|
||||||
|
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
|
||||||
|
"targetChapter": 489,
|
||||||
|
"snapshot": {
|
||||||
|
"asOfChapter": 488,
|
||||||
|
"snapshotVersion": "next_fine_outline_replay_v0",
|
||||||
|
"data": {
|
||||||
|
"milestones": [{"chapter": 488, "fact": "安全历史"}, {"chapter": 489, "fact": "未来"}],
|
||||||
|
"cards": [{"name": "已知实体", "milestones": [{"chapter": 488, "step": "历史"}]}],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"sources": [{"sourceId": "history-488", "sourceVersion": "v1", "chapter": 488}],
|
||||||
|
"commonContext": common,
|
||||||
|
"arms": {
|
||||||
|
"outline_only": {"cards": [], "cardSourceIds": [], "cardStrategy": "none"},
|
||||||
|
"outline_plus_cards": {"cards": [{"name": "正确卡"}], "cardSourceIds": ["card-correct-1"], "cardStrategy": "correct"},
|
||||||
|
"outline_plus_placebo_cards": {"cards": [{"name": "错配卡"}], "cardSourceIds": ["card-placebo-1"], "cardStrategy": "placebo"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class ReplayRunTest(unittest.TestCase):
|
||||||
|
def test_dry_run_passes_and_writes_only_metadata(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
result = run_replay(config(), pathlib.Path(directory), mode="dry_run")
|
||||||
|
self.assertTrue(result["ok"])
|
||||||
|
self.assertEqual(result["status"], "ready")
|
||||||
|
manifest = json.loads((pathlib.Path(directory) / "snapshot_manifest.json").read_text())
|
||||||
|
self.assertNotIn("payload", json.dumps(manifest, ensure_ascii=False))
|
||||||
|
self.assertFalse(list(pathlib.Path(directory).glob("planner_*.raw.json")))
|
||||||
|
|
||||||
|
def test_bad_authorization_stops_before_model(self):
|
||||||
|
bad = config()
|
||||||
|
bad["authorization"] = {"sourceStatus": "active"}
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
result = run_replay(bad, pathlib.Path(directory), mode="dry_run")
|
||||||
|
self.assertFalse(result["ok"])
|
||||||
|
self.assertEqual(result["status"], "blocked_authorization")
|
||||||
|
self.assertFalse((pathlib.Path(directory) / "snapshot_manifest.json").exists())
|
||||||
|
|
||||||
|
def test_report_contains_hashes_but_not_raw_fields(self):
|
||||||
|
result = run_replay(config(), pathlib.Path(tempfile.mkdtemp()), mode="dry_run")
|
||||||
|
report = render_report(result)
|
||||||
|
self.assertIn("smoke-001", report)
|
||||||
|
self.assertIn("snapshotManifestSha256", report)
|
||||||
|
self.assertNotIn('"prompt":', report)
|
||||||
|
self.assertNotIn('"payload":', report)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
107
.claude/skills/replay-eval/scripts/write_report.py
Normal file
107
.claude/skills/replay-eval/scripts/write_report.py
Normal file
@ -0,0 +1,107 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""把回放运行结果压缩成不含正文和完整响应的 Markdown 摘要。"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
|
||||||
|
REPORT_FORBIDDEN_KEYS = frozenset(
|
||||||
|
{
|
||||||
|
"raw",
|
||||||
|
"rawText",
|
||||||
|
"raw_output",
|
||||||
|
"body",
|
||||||
|
"content",
|
||||||
|
"payload",
|
||||||
|
"prompt",
|
||||||
|
"response",
|
||||||
|
"fullPrompt",
|
||||||
|
"fullResponse",
|
||||||
|
"正文",
|
||||||
|
"原文",
|
||||||
|
"正文全文",
|
||||||
|
"完整目标细纲",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None:
|
||||||
|
"""报告输入也不接受原始内容字段,避免误把运行中间件带进最终摘要。"""
|
||||||
|
|
||||||
|
if isinstance(value, Mapping):
|
||||||
|
for key, item in value.items():
|
||||||
|
if str(key) in REPORT_FORBIDDEN_KEYS:
|
||||||
|
raise ValueError(f"{path}.{key} 不得进入最终报告")
|
||||||
|
_assert_no_forbidden_keys(item, f"{path}.{key}")
|
||||||
|
elif isinstance(value, list):
|
||||||
|
for index, item in enumerate(value):
|
||||||
|
_assert_no_forbidden_keys(item, f"{path}[{index}]")
|
||||||
|
|
||||||
|
|
||||||
|
def _arm_line(name: str, arm: Mapping[str, Any], result: Mapping[str, Any]) -> str:
|
||||||
|
outcome = result.get(name) or {}
|
||||||
|
return (
|
||||||
|
f"- `{name}`: {outcome.get('status', 'not_run')},"
|
||||||
|
f"候选哈希 `{outcome.get('candidateSha256') or outcome.get('rawOutputSha256') or 'n/a'}`,"
|
||||||
|
f"卡策略 `{arm.get('cardStrategy', 'unknown')}`,卡数 `{arm.get('cardInjectionCount', 0)}`"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def render_report(result: Mapping[str, Any]) -> str:
|
||||||
|
"""只从运行结果的白名单字段渲染摘要。"""
|
||||||
|
|
||||||
|
_assert_no_forbidden_keys(result)
|
||||||
|
lines = [
|
||||||
|
"# 细纲回放摘要",
|
||||||
|
"",
|
||||||
|
f"- runId: `{result.get('runId', 'unknown')}`",
|
||||||
|
f"- status: `{result.get('status', 'unknown')}`",
|
||||||
|
f"- mode: `{result.get('mode', 'unknown')}`",
|
||||||
|
f"- referenceWork: `{result.get('referenceWork', 'unknown')}@{result.get('referenceWorkVersion', 'unknown')}`",
|
||||||
|
f"- freeze: 第 `{result.get('asOfChapter', 'unknown')}` 章,目标第 `{result.get('targetChapter', 'unknown')}` 章",
|
||||||
|
f"- snapshot: `{result.get('snapshotVersion', 'unknown')}`",
|
||||||
|
"",
|
||||||
|
"## 前置门",
|
||||||
|
"",
|
||||||
|
f"- status: `{(result.get('preflight') or {}).get('status', 'unknown')}`",
|
||||||
|
]
|
||||||
|
errors = (result.get("preflight") or {}).get("errors") or []
|
||||||
|
for error in errors:
|
||||||
|
lines.append(f"- 阻断: {error}")
|
||||||
|
lines.extend(["", "## 三臂", ""])
|
||||||
|
arms = result.get("arms") or {}
|
||||||
|
outcomes = result.get("results") or {}
|
||||||
|
for name in sorted(arms):
|
||||||
|
lines.append(_arm_line(name, arms[name], outcomes))
|
||||||
|
if result.get("snapshotManifestSha256"):
|
||||||
|
lines.extend(["", f"- snapshotManifestSha256: `{result['snapshotManifestSha256']}`"])
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
"> 本摘要不保存原书正文、目标章全文、完整 prompt/response 或供应商原始响应;原始候选仅存在临时运行目录。",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(description="写出细纲回放安全摘要")
|
||||||
|
parser.add_argument("--input", type=Path, required=True)
|
||||||
|
parser.add_argument("--output", type=Path, required=True)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
args = _parse_args()
|
||||||
|
result = json.loads(args.input.read_text(encoding="utf-8"))
|
||||||
|
args.output.write_text(render_report(result), encoding="utf-8")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
Loading…
x
Reference in New Issue
Block a user