362 lines
13 KiB
Python
362 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
"""把回放运行结果压缩成不含正文和完整响应的 Markdown 摘要。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import math
|
||
import re
|
||
from pathlib import Path
|
||
from typing import Any, Mapping
|
||
|
||
|
||
REPORT_FORBIDDEN_KEYS = frozenset(
|
||
{
|
||
"raw",
|
||
"rawText",
|
||
"raw_output",
|
||
"body",
|
||
"content",
|
||
"payload",
|
||
"prompt",
|
||
"response",
|
||
"fullPrompt",
|
||
"fullResponse",
|
||
"正文",
|
||
"原文",
|
||
"正文全文",
|
||
"完整目标细纲",
|
||
}
|
||
)
|
||
IDENTIFIER_PATTERN = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,63}\Z")
|
||
SHA256_PATTERN = re.compile(r"[0-9a-f]{64}\Z")
|
||
ARM_NAMES = frozenset(
|
||
{"outline_only", "outline_plus_cards", "outline_plus_placebo_cards"}
|
||
)
|
||
RUN_STATUSES = frozenset(
|
||
{
|
||
"unknown",
|
||
"validating_config",
|
||
"config_invalid",
|
||
"blocked_authorization",
|
||
"invalid_snapshot",
|
||
"invalid_arm_diff",
|
||
"target_source_forbidden",
|
||
"invalid_audit_input",
|
||
"blocked_leakage_audit",
|
||
"ready",
|
||
"running_planner",
|
||
"planner_timeout",
|
||
"planner_failed",
|
||
"planner_invalid",
|
||
"candidate_blocked",
|
||
"running_detector",
|
||
"detector_timeout",
|
||
"detector_failed",
|
||
"detector_invalid",
|
||
"detector_blocked",
|
||
"running_judge",
|
||
"judge_timeout",
|
||
"judge_failed",
|
||
"judge_invalid",
|
||
"judge_unstable",
|
||
"completed",
|
||
}
|
||
)
|
||
PREFLIGHT_STATUSES = frozenset(
|
||
{"unknown", "ready", "blocked_authorization", "invalid_snapshot", "invalid_arm_diff", "target_source_forbidden"}
|
||
)
|
||
ARM_STATUSES = frozenset(
|
||
{
|
||
"not_run",
|
||
"ready",
|
||
"schema_invalid",
|
||
"invalid_snapshot",
|
||
"target_source_forbidden",
|
||
"planner_timeout",
|
||
"planner_failed",
|
||
"planner_invalid",
|
||
}
|
||
)
|
||
DETECTOR_STATUSES = frozenset(
|
||
{"not_run", "passed", "blocked_high", "timeout", "failed", "invalid"}
|
||
)
|
||
CARD_STRATEGIES = frozenset({"unknown", "none", "correct", "placebo"})
|
||
RUBRIC_PROFILES = frozenset({"unknown", "fine_outline_replay"})
|
||
RUBRIC_DIMENSIONS = frozenset(
|
||
{
|
||
"structure_completeness",
|
||
"direction_causality",
|
||
"order_pacing",
|
||
"entity_state",
|
||
"foreshadowing_action",
|
||
"handoff_hook",
|
||
}
|
||
)
|
||
PREFLIGHT_SUMMARIES = {
|
||
"unknown": "前置门状态不可用",
|
||
"ready": "前置门通过",
|
||
"blocked_authorization": "授权前置门未通过",
|
||
"invalid_snapshot": "冻结快照前置门未通过",
|
||
"invalid_arm_diff": "三臂公共输入一致性未通过",
|
||
"target_source_forbidden": "发现目标章或未来来源",
|
||
}
|
||
|
||
|
||
def _assert_no_forbidden_keys(value: Any, path: str = "result") -> None:
|
||
"""报告输入也不接受原始内容字段,避免误把运行中间件带进最终摘要。"""
|
||
|
||
if isinstance(value, Mapping):
|
||
for key, item in value.items():
|
||
if str(key) in REPORT_FORBIDDEN_KEYS:
|
||
raise ValueError(f"{path}.{key} 不得进入最终报告")
|
||
_assert_no_forbidden_keys(item, f"{path}.{key}")
|
||
elif isinstance(value, list):
|
||
for index, item in enumerate(value):
|
||
_assert_no_forbidden_keys(item, f"{path}[{index}]")
|
||
|
||
|
||
def _mapping(value: Any, path: str) -> Mapping[str, Any]:
|
||
if not isinstance(value, Mapping):
|
||
raise ValueError(f"{path} 必须是对象")
|
||
return value
|
||
|
||
|
||
def _identifier(value: Any, path: str, default: str = "unknown") -> str:
|
||
candidate = default if value is None or value == "" else value
|
||
if not isinstance(candidate, str) or IDENTIFIER_PATTERN.fullmatch(candidate) is None:
|
||
raise ValueError(f"{path} 必须是长度不超过 64 的安全标识符")
|
||
return candidate
|
||
|
||
|
||
def _enum(value: Any, allowed: frozenset[str], path: str, default: str) -> str:
|
||
candidate = default if value is None or value == "" else value
|
||
if not isinstance(candidate, str) or candidate not in allowed:
|
||
raise ValueError(f"{path} 不是已登记枚举值")
|
||
return candidate
|
||
|
||
|
||
def _integer(value: Any, path: str, *, minimum: int = 0) -> int | None:
|
||
if value is None:
|
||
return None
|
||
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
|
||
raise ValueError(f"{path} 必须是大于等于 {minimum} 的整数")
|
||
return value
|
||
|
||
|
||
def _number(value: Any, path: str) -> float:
|
||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||
raise ValueError(f"{path} 必须是数字")
|
||
number = float(value)
|
||
if not math.isfinite(number):
|
||
raise ValueError(f"{path} 必须是有限数字")
|
||
return number
|
||
|
||
|
||
def _hash(value: Any, path: str) -> str | None:
|
||
if value is None or value == "":
|
||
return None
|
||
if not isinstance(value, str) or SHA256_PATTERN.fullmatch(value) is None:
|
||
raise ValueError(f"{path} 必须是 SHA-256")
|
||
return value
|
||
|
||
|
||
def _numeric_map(value: Any, path: str, allowed_keys: frozenset[str]) -> dict[str, float]:
|
||
mapping = _mapping(value, path)
|
||
unexpected = sorted(set(mapping) - allowed_keys)
|
||
if unexpected:
|
||
raise ValueError(f"{path} 包含未登记字段: {','.join(unexpected)}")
|
||
return {str(key): _number(item, f"{path}.{key}") for key, item in mapping.items()}
|
||
|
||
|
||
def _project_evaluation(value: Any) -> dict[str, Any]:
|
||
if value is None or value == "" or value == {}:
|
||
return {}
|
||
evaluation = _mapping(value, "evaluation")
|
||
projected: dict[str, Any] = {
|
||
"profile": _enum(evaluation.get("profile"), RUBRIC_PROFILES, "evaluation.profile", "unknown")
|
||
}
|
||
stability = evaluation.get("stability")
|
||
if stability is not None:
|
||
stability_mapping = _mapping(stability, "evaluation.stability")
|
||
stable = stability_mapping.get("stable")
|
||
if not isinstance(stable, bool):
|
||
raise ValueError("evaluation.stability.stable 必须是布尔枚举")
|
||
projected["stabilityStatus"] = "stable" if stable else "unstable"
|
||
projected["maxGaps"] = _numeric_map(
|
||
stability_mapping.get("maxGaps", {}),
|
||
"evaluation.stability.maxGaps",
|
||
RUBRIC_DIMENSIONS,
|
||
)
|
||
arm_scores = _mapping(evaluation.get("armScores", {}), "evaluation.armScores")
|
||
unexpected_arms = sorted(set(arm_scores) - ARM_NAMES)
|
||
if unexpected_arms:
|
||
raise ValueError(f"evaluation.armScores 包含未登记臂: {','.join(unexpected_arms)}")
|
||
projected["armScores"] = {
|
||
str(arm): _numeric_map(scores, f"evaluation.armScores.{arm}", RUBRIC_DIMENSIONS)
|
||
for arm, scores in arm_scores.items()
|
||
}
|
||
deltas = _mapping(evaluation.get("deltas", {}), "evaluation.deltas")
|
||
unexpected_deltas = sorted(set(deltas) - {"B-A", "C-A"})
|
||
if unexpected_deltas:
|
||
raise ValueError(f"evaluation.deltas 包含未登记比较: {','.join(unexpected_deltas)}")
|
||
projected["deltas"] = {
|
||
str(name): _numeric_map(scores, f"evaluation.deltas.{name}", RUBRIC_DIMENSIONS)
|
||
for name, scores in deltas.items()
|
||
}
|
||
return projected
|
||
|
||
|
||
def _project_report(result: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""从运行态提取独立安全 schema,未登记类型和值一律拒绝。"""
|
||
|
||
_assert_no_forbidden_keys(result)
|
||
preflight = _mapping(result.get("preflight", {}), "preflight")
|
||
preflight_status = _enum(
|
||
preflight.get("status"),
|
||
PREFLIGHT_STATUSES,
|
||
"preflight.status",
|
||
"unknown",
|
||
)
|
||
arms = _mapping(result.get("arms", {}), "arms")
|
||
outcomes = _mapping(result.get("results", {}), "results")
|
||
unexpected_arms = sorted((set(arms) | set(outcomes)) - ARM_NAMES)
|
||
if unexpected_arms:
|
||
raise ValueError(f"报告包含未登记评测臂: {','.join(unexpected_arms)}")
|
||
projected_arms: dict[str, Any] = {}
|
||
for name in sorted(arms):
|
||
arm = _mapping(arms[name], f"arms.{name}")
|
||
outcome = _mapping(outcomes.get(name, {}), f"results.{name}")
|
||
detector = _mapping(outcome.get("detector", {}), f"results.{name}.detector")
|
||
projected_arms[name] = {
|
||
"status": _enum(outcome.get("status"), ARM_STATUSES, f"results.{name}.status", "not_run"),
|
||
"candidateSha256": _hash(
|
||
outcome.get("candidateSha256") or outcome.get("rawOutputSha256"),
|
||
f"results.{name}.candidateSha256",
|
||
),
|
||
"detectorStatus": _enum(
|
||
detector.get("status"),
|
||
DETECTOR_STATUSES,
|
||
f"results.{name}.detector.status",
|
||
"not_run",
|
||
),
|
||
"cardStrategy": _enum(
|
||
arm.get("cardStrategy"),
|
||
CARD_STRATEGIES,
|
||
f"arms.{name}.cardStrategy",
|
||
"unknown",
|
||
),
|
||
"cardInjectionCount": _integer(
|
||
arm.get("cardInjectionCount", 0),
|
||
f"arms.{name}.cardInjectionCount",
|
||
),
|
||
}
|
||
return {
|
||
"runId": _identifier(result.get("runId"), "runId"),
|
||
"status": _enum(result.get("status"), RUN_STATUSES, "status", "unknown"),
|
||
"mode": _enum(result.get("mode"), frozenset({"unknown", "dry_run", "execute"}), "mode", "unknown"),
|
||
"referenceWork": _identifier(result.get("referenceWork"), "referenceWork"),
|
||
"referenceWorkVersion": _identifier(result.get("referenceWorkVersion"), "referenceWorkVersion"),
|
||
"asOfChapter": _integer(result.get("asOfChapter"), "asOfChapter", minimum=1),
|
||
"targetChapter": _integer(result.get("targetChapter"), "targetChapter", minimum=1),
|
||
"snapshotVersion": _identifier(result.get("snapshotVersion"), "snapshotVersion"),
|
||
"preflightStatus": preflight_status,
|
||
"preflightSummary": PREFLIGHT_SUMMARIES[preflight_status],
|
||
"arms": projected_arms,
|
||
"snapshotManifestSha256": _hash(
|
||
result.get("snapshotManifestSha256"), "snapshotManifestSha256"
|
||
),
|
||
"evaluation": _project_evaluation(result.get("evaluation")),
|
||
}
|
||
|
||
|
||
def _arm_line(name: str, arm: Mapping[str, Any]) -> str:
|
||
return (
|
||
f"- `{name}`: {arm['status']},"
|
||
f"候选哈希 `{arm['candidateSha256'] or 'n/a'}`,"
|
||
f"detector `{arm['detectorStatus']}`,"
|
||
f"卡策略 `{arm.get('cardStrategy', 'unknown')}`,卡数 `{arm.get('cardInjectionCount', 0)}`"
|
||
)
|
||
|
||
|
||
def _render_evaluation(lines: list[str], evaluation: Mapping[str, Any]) -> None:
|
||
"""只渲染聚合分数、稳定性和差值,不带 judge 原始证据。"""
|
||
|
||
if not evaluation:
|
||
return
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"## 双盲评分",
|
||
"",
|
||
f"- profile: `{evaluation.get('profile', 'unknown')}`",
|
||
f"- 评委稳定性: `{evaluation.get('stabilityStatus', 'not_available')}`",
|
||
f"- 最大维度差: `{json.dumps(evaluation.get('maxGaps') or {}, ensure_ascii=False, sort_keys=True)}`",
|
||
]
|
||
)
|
||
arm_scores = evaluation.get("armScores") or {}
|
||
for arm in sorted(arm_scores):
|
||
lines.append(
|
||
f"- `{arm}` 聚合分数: `{json.dumps(arm_scores[arm], ensure_ascii=False, sort_keys=True)}`"
|
||
)
|
||
deltas = evaluation.get("deltas") or {}
|
||
for comparison in ("B-A", "C-A"):
|
||
if comparison in deltas:
|
||
lines.append(
|
||
f"- `{comparison}` 差值: `{json.dumps(deltas[comparison], ensure_ascii=False, sort_keys=True)}`"
|
||
)
|
||
|
||
|
||
def render_report(result: Mapping[str, Any]) -> str:
|
||
"""只从运行结果的白名单字段渲染摘要。"""
|
||
|
||
safe = _project_report(result)
|
||
lines = [
|
||
"# 细纲回放摘要",
|
||
"",
|
||
f"- runId: `{safe['runId']}`",
|
||
f"- status: `{safe['status']}`",
|
||
f"- mode: `{safe['mode']}`",
|
||
f"- referenceWork: `{safe['referenceWork']}@{safe['referenceWorkVersion']}`",
|
||
f"- freeze: 第 `{safe['asOfChapter'] if safe['asOfChapter'] is not None else 'unknown'}` 章,目标第 `{safe['targetChapter'] if safe['targetChapter'] is not None else 'unknown'}` 章",
|
||
f"- snapshot: `{safe['snapshotVersion']}`",
|
||
"",
|
||
"## 前置门",
|
||
"",
|
||
f"- status: `{safe['preflightStatus']}`",
|
||
f"- 摘要: {safe['preflightSummary']}",
|
||
]
|
||
lines.extend(["", "## 三臂", ""])
|
||
for name, arm in safe["arms"].items():
|
||
lines.append(_arm_line(name, arm))
|
||
if safe["snapshotManifestSha256"]:
|
||
lines.extend(["", f"- snapshotManifestSha256: `{safe['snapshotManifestSha256']}`"])
|
||
_render_evaluation(lines, safe["evaluation"])
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"> 本摘要不保存原书正文、目标章全文、完整 prompt/response 或供应商原始响应;原始候选仅存在临时运行目录。",
|
||
"",
|
||
]
|
||
)
|
||
return "\n".join(lines)
|
||
|
||
|
||
def _parse_args() -> argparse.Namespace:
|
||
parser = argparse.ArgumentParser(description="写出细纲回放安全摘要")
|
||
parser.add_argument("--input", type=Path, required=True)
|
||
parser.add_argument("--output", type=Path, required=True)
|
||
return parser.parse_args()
|
||
|
||
|
||
def main() -> int:
|
||
args = _parse_args()
|
||
result = json.loads(args.input.read_text(encoding="utf-8"))
|
||
args.output.write_text(render_report(result), encoding="utf-8")
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|