398 lines
14 KiB
Python
398 lines
14 KiB
Python
#!/usr/bin/env python3
|
||
"""细纲回放编排器。
|
||
|
||
默认只做 dry-run。真实模式显式调用本地 Claude CLI,并将原始候选限定在运行目录;
|
||
最终结果只返回状态、哈希、评分入口和失败类别,不把原始 prompt/response 带出运行目录。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import re
|
||
import subprocess
|
||
from pathlib import Path
|
||
from typing import Any, Mapping
|
||
|
||
from build_snapshot import build_snapshot, normalize_chapter, sha256_value
|
||
from audit_leakage import audit_snapshot
|
||
from check_snapshot import (
|
||
STATUS_READY,
|
||
check_candidate_output,
|
||
check_replay,
|
||
)
|
||
|
||
|
||
REQUIRED_ARMS = ("outline_only", "outline_plus_cards", "outline_plus_placebo_cards")
|
||
REPO_ROOT = Path(__file__).resolve().parents[4]
|
||
SKILL_PATH = REPO_ROOT / ".claude/skills/fine-outline/SKILL.md"
|
||
PLANNER_PATH = REPO_ROOT / ".claude/agents/planner.md"
|
||
|
||
|
||
class ReplayRunError(ValueError):
|
||
"""回放配置不符合运行边界。"""
|
||
|
||
|
||
def _read_json(path: Path) -> Any:
|
||
return json.loads(path.read_text(encoding="utf-8"))
|
||
|
||
|
||
def _safe_json(value: Any) -> str:
|
||
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
||
|
||
|
||
def _require_mapping(config: Mapping[str, Any], key: str) -> Mapping[str, Any]:
|
||
value = config.get(key)
|
||
if not isinstance(value, Mapping):
|
||
raise ReplayRunError(f"配置缺少对象字段: {key}")
|
||
return value
|
||
|
||
|
||
def _build_arm_manifest(
|
||
*,
|
||
name: str,
|
||
arm: Mapping[str, Any],
|
||
common_input: Mapping[str, Any],
|
||
as_of: int,
|
||
target: int,
|
||
snapshot_version: str,
|
||
) -> dict[str, Any]:
|
||
cards = arm.get("cards", [])
|
||
if not isinstance(cards, list):
|
||
raise ReplayRunError(f"{name}.cards 必须是数组")
|
||
source_ids = arm.get("cardSourceIds", [])
|
||
if not isinstance(source_ids, list):
|
||
raise ReplayRunError(f"{name}.cardSourceIds 必须是数组")
|
||
strategy = str(arm.get("cardStrategy") or ("none" if name == "outline_only" else "correct"))
|
||
return {
|
||
"arm": name,
|
||
"snapshotVersion": snapshot_version,
|
||
"asOfChapter": as_of,
|
||
"targetChapter": target,
|
||
"commonInputSha256": sha256_value(common_input),
|
||
"cardInjectionSha256": sha256_value(cards),
|
||
"cardInjectionCount": len(cards),
|
||
"cardSourceIds": [str(item) for item in source_ids],
|
||
"cardStrategy": strategy,
|
||
}
|
||
|
||
|
||
def _extract_candidate(output: str) -> Mapping[str, Any]:
|
||
"""兼容 Claude JSON 外壳和代码围栏,但不把原始文本返回调用方。"""
|
||
|
||
outer: Any = output
|
||
try:
|
||
outer = json.loads(output)
|
||
except json.JSONDecodeError:
|
||
pass
|
||
if isinstance(outer, Mapping) and isinstance(outer.get("result"), str):
|
||
outer = outer["result"]
|
||
if isinstance(outer, str):
|
||
match = re.search(r"```(?:json)?\s*(\{.*\})\s*```", outer, re.DOTALL)
|
||
outer = match.group(1) if match else outer.strip()
|
||
outer = json.loads(outer)
|
||
if not isinstance(outer, Mapping):
|
||
raise ReplayRunError("planner 输出不是 JSON 对象")
|
||
return outer
|
||
|
||
|
||
def _planner_prompt(
|
||
*,
|
||
target: int,
|
||
as_of: int,
|
||
snapshot: Mapping[str, Any],
|
||
common_input: Mapping[str, Any],
|
||
cards: list[Any],
|
||
) -> str:
|
||
"""只把冻结后的结构化资料和功能合同送入 planner。"""
|
||
|
||
skill = SKILL_PATH.read_text(encoding="utf-8")
|
||
identity = PLANNER_PATH.read_text(encoding="utf-8")
|
||
# 公共快照不能携带知识卡;卡只能通过当前评测臂的独立注入区进入上下文,
|
||
# 否则 outline_only 臂会在“无卡”名义下偷看到全局卡片。
|
||
public_snapshot = {
|
||
str(key): value for key, value in snapshot.items() if str(key) != "cards"
|
||
}
|
||
context = {
|
||
"targetChapter": target,
|
||
"asOfChapter": as_of,
|
||
"frozenSnapshot": public_snapshot,
|
||
"commonContext": common_input,
|
||
"cardInjection": cards,
|
||
}
|
||
return "\n".join(
|
||
[
|
||
"这是 next_fine_outline_replay_v0 的离线规划任务。只输出一个 JSON 对象,不要 Markdown、正文或解释。",
|
||
"你不能调用工具,也不能读取仓库、目标章或任何未列入下面 JSON 的资料。",
|
||
"严格执行 fine-outline 合同;目标章号必须保持不变,未知内容写入 unknowns/assumptions。",
|
||
"规划上下文冻结到 as_of;卡只是事实索引和补充,不得替代公共大纲与叙事现在时。",
|
||
"--- planner identity ---",
|
||
identity,
|
||
"--- fine-outline skill ---",
|
||
skill,
|
||
"--- frozen input ---",
|
||
_safe_json(context),
|
||
"--- output contract ---",
|
||
_safe_json(
|
||
{
|
||
"targetChapter": target,
|
||
"requiredFields": [
|
||
"chapterGoal",
|
||
"keyEvents",
|
||
"entities",
|
||
"foreshadowing",
|
||
"stateChanges",
|
||
"hook",
|
||
"unknowns",
|
||
"assumptions",
|
||
],
|
||
"eventFields": ["id", "order", "event", "participants", "trigger", "resultDirection"],
|
||
"entityFields": ["name", "type", "role"],
|
||
"foreshadowingFields": ["action", "subject", "evidence"],
|
||
"sourceRefs": "可选;只能引用冻结来源 ID",
|
||
}
|
||
),
|
||
]
|
||
)
|
||
|
||
|
||
def _invoke_planner(
|
||
*,
|
||
prompt: str,
|
||
planner_bin: str,
|
||
model: str,
|
||
output_path: Path,
|
||
max_budget_usd: float,
|
||
) -> Mapping[str, Any]:
|
||
"""用无工具、无会话持久化的 Claude print 模式运行 planner。"""
|
||
|
||
command = [
|
||
planner_bin,
|
||
"-p",
|
||
"--agent",
|
||
"planner",
|
||
"--model",
|
||
model,
|
||
"--tools",
|
||
"",
|
||
"--no-session-persistence",
|
||
"--output-format",
|
||
"json",
|
||
"--max-budget-usd",
|
||
str(max_budget_usd),
|
||
"--append-system-prompt",
|
||
"本次是严格离线回放;不要调用任何工具,不要读取文件,不要输出 JSON 以外内容。",
|
||
prompt,
|
||
]
|
||
completed = subprocess.run(command, text=True, capture_output=True, check=False)
|
||
output_path.write_text(completed.stdout, encoding="utf-8")
|
||
if completed.returncode != 0:
|
||
raise ReplayRunError(f"planner 调用失败,退出码={completed.returncode}")
|
||
return _extract_candidate(completed.stdout)
|
||
|
||
|
||
def run_replay(
|
||
config: Mapping[str, Any],
|
||
output_dir: Path,
|
||
*,
|
||
mode: str = "dry_run",
|
||
planner_bin: str = "claude",
|
||
model: str = "opus",
|
||
max_budget_usd: float = 1.0,
|
||
) -> dict[str, Any]:
|
||
"""执行一次单目标三臂回放;任何前置门失败都不调用模型。"""
|
||
|
||
if mode not in {"dry_run", "execute"}:
|
||
raise ReplayRunError("mode 只能是 dry_run 或 execute")
|
||
output_dir = output_dir.resolve()
|
||
if output_dir.is_relative_to(REPO_ROOT.resolve()):
|
||
raise ReplayRunError("原始候选运行目录不得位于仓库内")
|
||
output_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
snapshot_config = _require_mapping(config, "snapshot")
|
||
as_of = normalize_chapter(snapshot_config.get("asOfChapter"))
|
||
target = normalize_chapter(config.get("targetChapter"))
|
||
snapshot_version = str(snapshot_config.get("snapshotVersion") or "")
|
||
if as_of is None or target is None:
|
||
raise ReplayRunError("as_of/target 必须是明确正整数")
|
||
if target != as_of + 1:
|
||
raise ReplayRunError("targetChapter 必须等于 snapshot.asOfChapter+1")
|
||
reference_work = _require_mapping(config, "referenceWork")
|
||
authorization = _require_mapping(config, "authorization")
|
||
sources = config.get("sources", [])
|
||
if not isinstance(sources, list):
|
||
raise ReplayRunError("sources 必须是数组")
|
||
common_input = _require_mapping(config, "commonContext")
|
||
arms = _require_mapping(config, "arms")
|
||
if set(arms) != set(REQUIRED_ARMS):
|
||
raise ReplayRunError("生产回放必须精确配置三臂")
|
||
|
||
arm_manifests = {
|
||
name: _build_arm_manifest(
|
||
name=name,
|
||
arm=_require_mapping(arms, name),
|
||
common_input=common_input,
|
||
as_of=as_of,
|
||
target=target,
|
||
snapshot_version=snapshot_version,
|
||
)
|
||
for name in REQUIRED_ARMS
|
||
}
|
||
preflight = check_replay(
|
||
authorization=authorization,
|
||
as_of_chapter=as_of,
|
||
target_chapter=target,
|
||
planner_sources=sources,
|
||
arm_manifests=arm_manifests,
|
||
)
|
||
result: dict[str, Any] = {
|
||
"runId": str(config.get("runId") or "unassigned"),
|
||
"mode": mode,
|
||
"status": preflight["status"],
|
||
"ok": preflight["ok"],
|
||
"referenceWork": str(reference_work.get("id") or ""),
|
||
"referenceWorkVersion": str(reference_work.get("version") or ""),
|
||
"asOfChapter": as_of,
|
||
"targetChapter": target,
|
||
"snapshotVersion": snapshot_version,
|
||
"preflight": {"status": preflight["status"], "errors": preflight["errors"], "warnings": preflight["warnings"]},
|
||
"arms": arm_manifests,
|
||
"results": {},
|
||
}
|
||
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||
if not preflight["ok"]:
|
||
return result
|
||
|
||
leakage_audit_config = config.get("leakageAudit")
|
||
if not isinstance(leakage_audit_config, Mapping):
|
||
# 没有目标事实审计就不能把 dry-run 说成可运行,避免结构冻结掩盖内容泄露。
|
||
result["status"] = "blocked_leakage_audit"
|
||
result["ok"] = False
|
||
result["leakageAudit"] = {
|
||
"status": "not_configured",
|
||
"ok": False,
|
||
"errors": ["缺少 leakageAudit 配置"],
|
||
"warnings": [],
|
||
"findingCount": 0,
|
||
"findings": [],
|
||
}
|
||
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||
return result
|
||
|
||
metadata = {
|
||
"targetChapter": target,
|
||
"referenceWork": reference_work,
|
||
"evaluationSetVersion": config.get("evaluationSetVersion", "unregistered"),
|
||
"strategyVersion": config.get("strategyVersion", "unregistered"),
|
||
"authorizationSnapshot": authorization["authorizationSnapshot"],
|
||
"runPermissions": config.get("runPermissions", {"purpose": "offline_evaluation", "mode": mode}),
|
||
"armConfig": {"arms": list(REQUIRED_ARMS)},
|
||
}
|
||
snapshot_data = snapshot_config.get("data", {})
|
||
frozen = build_snapshot(
|
||
snapshot_data,
|
||
as_of,
|
||
snapshot_version,
|
||
target_chapter=target,
|
||
manifest_metadata=metadata,
|
||
)
|
||
|
||
# 内容审计同时覆盖公共冻结快照和各臂卡注入区;卡不在公共区,不能因此逃过未来事实检查。
|
||
audit_payload = {
|
||
"snapshot": frozen["snapshot"],
|
||
"armCardInjections": {
|
||
name: _require_mapping(arms, name).get("cards", []) for name in REQUIRED_ARMS
|
||
},
|
||
}
|
||
audit_result = audit_snapshot(
|
||
audit_payload,
|
||
leakage_audit_config.get("targetFacts"),
|
||
as_of=as_of,
|
||
target=target,
|
||
)
|
||
result["leakageAudit"] = audit_result
|
||
if not audit_result["ok"]:
|
||
# 审计失败的快照不生成 manifest,也不允许进入任何 planner 臂。
|
||
result["status"] = audit_result["status"]
|
||
result["ok"] = False
|
||
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||
return result
|
||
|
||
(output_dir / "snapshot_manifest.json").write_text(_safe_json(frozen["manifest"]) + "\n", encoding="utf-8")
|
||
|
||
if mode == "dry_run":
|
||
result["status"] = STATUS_READY
|
||
result["ok"] = True
|
||
result["snapshotManifestSha256"] = frozen["manifest"]["manifestSha256"]
|
||
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||
return result
|
||
|
||
for name in REQUIRED_ARMS:
|
||
arm = _require_mapping(arms, name)
|
||
cards = arm.get("cards", [])
|
||
raw_path = output_dir / f"planner_{name}.raw.json"
|
||
prompt = _planner_prompt(
|
||
target=target,
|
||
as_of=as_of,
|
||
snapshot=frozen["snapshot"],
|
||
common_input=common_input,
|
||
cards=cards,
|
||
)
|
||
try:
|
||
candidate = _invoke_planner(
|
||
prompt=prompt,
|
||
planner_bin=planner_bin,
|
||
model=model,
|
||
output_path=raw_path,
|
||
max_budget_usd=max_budget_usd,
|
||
)
|
||
candidate_path = output_dir / f"candidate_{name}.json"
|
||
candidate_path.write_text(_safe_json(candidate) + "\n", encoding="utf-8")
|
||
schema = check_candidate_output(candidate, target, sources)
|
||
result["results"][name] = {
|
||
"status": schema["status"],
|
||
"ok": schema["ok"],
|
||
"errors": schema["errors"],
|
||
"candidateSha256": sha256_value(candidate),
|
||
"candidatePath": str(candidate_path),
|
||
}
|
||
except (ReplayRunError, json.JSONDecodeError) as error:
|
||
result["results"][name] = {
|
||
"status": "planner_output_invalid",
|
||
"ok": False,
|
||
"errors": [str(error)],
|
||
"rawOutputSha256": sha256_value(raw_path.read_text(encoding="utf-8")) if raw_path.exists() else None,
|
||
}
|
||
result["status"] = "completed" if all(item["ok"] for item in result["results"].values()) else "candidate_blocked"
|
||
result["ok"] = result["status"] == "completed"
|
||
(output_dir / "run_result.json").write_text(_safe_json(result) + "\n", encoding="utf-8")
|
||
return result
|
||
|
||
|
||
def _parse_args() -> argparse.Namespace:
|
||
parser = argparse.ArgumentParser(description="运行 next_fine_outline_replay_v0")
|
||
parser.add_argument("--config", type=Path, required=True)
|
||
parser.add_argument("--output-dir", type=Path, required=True)
|
||
parser.add_argument("--mode", choices=("dry_run", "execute"), default="dry_run")
|
||
parser.add_argument("--planner-bin", default="claude")
|
||
parser.add_argument("--model", default="opus")
|
||
parser.add_argument("--max-budget-usd", type=float, default=1.0)
|
||
return parser.parse_args()
|
||
|
||
|
||
def main() -> int:
|
||
args = _parse_args()
|
||
result = run_replay(
|
||
_read_json(args.config),
|
||
args.output_dir,
|
||
mode=args.mode,
|
||
planner_bin=args.planner_bin,
|
||
model=args.model,
|
||
max_budget_usd=args.max_budget_usd,
|
||
)
|
||
return 0 if result["ok"] else 2
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|