From befb4714c0f868a7284c16b26e8bc455533b4077 Mon Sep 17 00:00:00 2001 From: zizi Date: Sun, 23 Aug 2026 16:13:59 +0800 Subject: [PATCH] =?UTF-8?q?=E9=98=B6=E6=AE=B5F=E7=AC=AC=E4=B8=89=E9=83=A8?= =?UTF-8?q?=E5=88=86:=20=E8=AF=84=E6=B5=8B=E9=93=BE=E6=B4=BE=E5=8F=91?= =?UTF-8?q?=E2=80=94=E2=80=94=E8=AF=84=E5=A7=94=E6=99=BA=E8=83=BD=E4=BD=93?= =?UTF-8?q?=E7=BB=8FModelRunner=E5=8D=8F=E8=AE=AE=E6=8E=A5=E5=85=A5?= =?UTF-8?q?=E6=A1=86=E6=9E=B6=E6=B4=BE=E5=8F=91=EF=BC=88=E7=A6=BB=E7=BA=BF?= =?UTF-8?q?10=E9=A1=B9=E7=BB=BF+=E5=80=99=E9=80=89123vs164=E7=9C=9F?= =?UTF-8?q?=E5=AE=9E=E7=9B=B2=E8=AF=84=E9=80=9A=E8=BF=87=EF=BC=8C164?= =?UTF-8?q?=E5=9B=9B=E7=BB=B4=E9=A2=86=E5=85=88=E4=B8=8E=E4=BA=BA=E9=87=87?= =?UTF-8?q?=E7=BA=B3=E6=96=B9=E5=90=91=E4=B8=80=E8=87=B4=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../scripts/dispatch_judge_bridge.py | 179 +++++++++++++ .../scripts/judge_via_dispatch.py | 246 ++++++++++++++++++ .../2026-08-22-agent-example整体收敛总plan.md | 1 + ...22-阶段F-第三部分-评测链派发-评委智能体.md | 51 ++++ .../test_dispatch_judge_bridge.py | 180 +++++++++++++ 5 files changed, 657 insertions(+) create mode 100644 .agent/skills/score-content-quality/scripts/dispatch_judge_bridge.py create mode 100644 .agent/skills/score-content-quality/scripts/judge_via_dispatch.py create mode 100644 docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md create mode 100644 tests/skills/score-content-quality/test_dispatch_judge_bridge.py diff --git a/.agent/skills/score-content-quality/scripts/dispatch_judge_bridge.py b/.agent/skills/score-content-quality/scripts/dispatch_judge_bridge.py new file mode 100644 index 0000000..df9168a --- /dev/null +++ b/.agent/skills/score-content-quality/scripts/dispatch_judge_bridge.py @@ -0,0 +1,179 @@ +#!/usr/bin/env python3 +"""评委智能体的框架派发 runner(阶段 F 第三部分)。 + +实现 run_writer_blind_judge 的 ModelRunner 协议:盲评编排(输入防泄漏校验、 +草稿校验、有界纠错、报告绑定)全部复用既有可信适配层,本桥只负责把模型 +调用切到智能体框架派发。 + +圈定授权隔离(边界合同): +- 任务包 = 盲评模型输入本体(已过 _walk_for_leakage 防泄漏校验),桥不附加 + 任何身份字段; +- 实验场景(回放盲评):工具白名单为空,只用冻结快照; +- 生产预检场景:开放只读工具检索正典事实,用于对照设定一致性; +- 每位评委独立新会话,不共享其他评委的上下文。 +""" +from __future__ import annotations + +import json +import sys +from pathlib import Path +from typing import Any, Callable, Mapping + +SCRIPT_DIR = Path(__file__).resolve().parent +DISPATCH_SCRIPTS = SCRIPT_DIR.parents[1] / "dispatch-agent-task" / "scripts" +if str(DISPATCH_SCRIPTS) not in sys.path: + sys.path.insert(0, str(DISPATCH_SCRIPTS)) + +from run_writer_blind_judge import BlindJudgeContractError, canonical_sha256 # noqa: E402 +from dispatch_agent_task import run_dispatch # noqa: E402 +from pi_runner import ExecutionPolicy # noqa: E402 +from read_tools import TOOL_REGISTRY # noqa: E402 + +MODE_EXPERIMENT = "experiment" +MODE_PRODUCTION_PREFLIGHT = "production_preflight" + +# 圈定授权:实验场景只用冻结快照(无工具);生产预检开放正典对照检索。 +MODE_TOOL_ALLOWLIST: dict[str, tuple[str, ...]] = { + MODE_EXPERIMENT: (), + MODE_PRODUCTION_PREFLIGHT: ("read_chapter_text", "search_entities"), +} + +JUDGE_MAX_DURATION_SECONDS = 1800 +SESSION_ROOT = Path("/tmp/muse-agent-runs/judge-sessions") + + +class DispatchJudgeError(RuntimeError): + """评委派发失败:携带稳定错误码,编排方按失败关闭处理。""" + + def __init__(self, code: str, message: str, *, details: Mapping[str, Any] | None = None): + super().__init__(message) + self.code = code + self.details = dict(details or {}) + + +def judge_tool_allowlist(mode: str) -> tuple[str, ...]: + """按场景取圈定工具白名单;未登记场景失败关闭。""" + + allowlist = MODE_TOOL_ALLOWLIST.get(mode) + if allowlist is None: + raise DispatchJudgeError("JUDGE_MODE_UNKNOWN", f"评委场景未登记:{mode}") + for name in allowlist: + if name not in TOOL_REGISTRY: + raise DispatchJudgeError("JUDGE_TOOL_UNREGISTERED", f"评委白名单工具未登记:{name}") + return allowlist + + +class DispatchJudgeRunner: + """ModelRunner 协议的框架派发实现:每次 run 派发一位独立评委。""" + + def __init__( + self, + *, + run_id: str, + work_id: int, + repo_root: str | Path, + provider: str, + model: str, + mode: str = MODE_PRODUCTION_PREFLIGHT, + thinking: str | None = None, + spec_dir: str | Path | None = None, + launcher: Callable[..., Any] | None = None, + connect_factory: Callable[..., Any] | None = None, + ) -> None: + self.run_id = run_id + self.work_id = work_id + self.repo_root = repo_root + self.provider = provider + self.model = model + self.mode = mode + self.thinking = thinking + self.spec_dir = Path(spec_dir) if spec_dir is not None else SCRIPT_DIR + self.launcher = launcher + self.connect_factory = connect_factory + self.allowlist = judge_tool_allowlist(mode) + self._attempt = 0 + self.dispatch_run_ids: list[str] = [] + + def run(self, *, adapter_role: str, model_input: Mapping[str, Any], output_schema: Mapping[str, Any]) -> Mapping[str, Any]: + if adapter_role != "blind_judge": + raise BlindJudgeContractError("BLIND_JUDGE_RUNNER_INVALID", f"派发桥只承接 blind_judge,收到 {adapter_role}") + self._attempt += 1 + dispatch_run_id = f"{self.run_id}-judge-v{self._attempt}" + self.dispatch_run_ids.append(dispatch_run_id) + + spec = { + "specVersion": "agent-task-v1", + "role": "judge", + "taskPrompt": ( + "你是质量评委,执行一次独立盲评:只按盲评输入中的量表与证据逐维打分," + "每维分数必须给可定位引文;输入不包含任何候选身份,不要推测候选归属。" + "evidenceRefs 每项严格为 {sourceType, sourceId},sourceId 必须从下列既有 ID 中逐字选用," + "不得拼接、改写或附加描述词:" + "sourceType=candidate 时 sourceId=被打分候选的 blindCandidateId 本身;" + "sourceType=fine_outline 时 sourceId=fineOutline.hardConstraints 中的某个 constraintId;" + "sourceType=oracle_assertion 时 sourceId=oracleTruthPack 中的某个 assertionId;" + "sourceType=judge_inference 时 sourceId=当前打分维度名。" + "引文必须是候选正文中的逐字连续片段,不得改写、省略或加省略号。" + "按 outputContract 的完整集合一次返回全部对象,只输出一个草稿 JSON。" + ), + # 盲评模型输入本体即任务输入:已通过防泄漏校验,桥不附加身份字段。 + "input": dict(model_input), + "outputSchema": dict(output_schema), + "outputSchemaId": "blind-judge-draft-v3", + "toolAllowlist": list(self.allowlist), + "maxDurationSeconds": JUDGE_MAX_DURATION_SECONDS, + } + self.spec_dir.mkdir(parents=True, exist_ok=True) + spec_file = self.spec_dir / f"{dispatch_run_id}-task.json" + spec_file.write_text(json.dumps(spec, ensure_ascii=False, indent=1), encoding="utf-8") + + # 每位评委独立新会话:盲评独立性要求不共享其他评委上下文。 + session_id = f"judge-{dispatch_run_id}" + session_dir = SESSION_ROOT / session_id + session_dir.mkdir(parents=True, mode=0o700, exist_ok=True) + + policy = ExecutionPolicy(provider=self.provider, model=self.model, thinking=self.thinking) + receipt, code = run_dispatch( + spec_file, + repo_root=self.repo_root, + policy=policy, + run_id=dispatch_run_id, + trigger_source="user", + trigger_detail={"stage": "judge-dispatch", "evaluationRunId": self.run_id, + "mode": self.mode, "attempt": self._attempt}, + session_id=session_id, + session_dir=session_dir, + enable_read_tools=bool(self.allowlist), + launcher=self.launcher, + connect_factory=self.connect_factory, + ) + if code != 0 or receipt.get("status") != "completed": + raise BlindJudgeContractError( + "BLIND_JUDGE_RUNTIME_FAILED", + f"评委派发未成功:{receipt.get('error') or receipt.get('errorCode')}", + ) + + output_file = Path(str(receipt.get("runDir") or "")) / "output.json" + try: + structured = json.loads(output_file.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + raise BlindJudgeContractError( + "BLIND_JUDGE_RUNTIME_FAILED", f"评委派发未产出可用草稿:{exc}" + ) from exc + if not isinstance(structured, Mapping): + raise BlindJudgeContractError("BLIND_JUDGE_RUNTIME_FAILED", "评委草稿必须是 JSON 对象") + + # 回执哈希绑定派发回执本体(规范化 JSON 哈希),报告经它回指本次模型调用。 + receipt_hash = canonical_sha256(json.loads(json.dumps(receipt, ensure_ascii=False, default=str))) + return {"structuredOutput": dict(structured), "modelReceiptSha256": receipt_hash} + + +__all__ = [ + "JUDGE_MAX_DURATION_SECONDS", + "MODE_EXPERIMENT", + "MODE_PRODUCTION_PREFLIGHT", + "MODE_TOOL_ALLOWLIST", + "DispatchJudgeError", + "DispatchJudgeRunner", + "judge_tool_allowlist", +] diff --git a/.agent/skills/score-content-quality/scripts/judge_via_dispatch.py b/.agent/skills/score-content-quality/scripts/judge_via_dispatch.py new file mode 100644 index 0000000..6f3c0f3 --- /dev/null +++ b/.agent/skills/score-content-quality/scripts/judge_via_dispatch.py @@ -0,0 +1,246 @@ +#!/usr/bin/env python3 +"""评委智能体派发入口(阶段 F 第三部分):生产预检盲评。 + +对同一章的两个(及以上)候选做匿名对比盲评:组装合法盲评输入 +(blind-judge v4,哈希全绑定、防泄漏校验),经框架派发评委,报告落 +example_quality_result(judge_kind=scoring,枚举以库级 CHECK 约束为准)。 + +真实模型调用必须显式授权并显式给出 provider/model: +.venv/bin/python .agent/skills/score-content-quality/scripts/judge_via_dispatch.py \ + 12 3 --candidate-ids 123,164 --provider catproxy-anthropic --model claude-opus-5 --thinking medium +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import random +import sys +from datetime import datetime, timezone +from pathlib import Path + +SCRIPT_DIR = Path(__file__).resolve().parent +REPO_ROOT = SCRIPT_DIR.parents[3] +for _path in (SCRIPT_DIR, + REPO_ROOT / ".agent" / "skills" / "record-run-evidence" / "scripts", + REPO_ROOT / ".agent" / "skills" / "access-database" / "scripts"): + if str(_path) not in sys.path: + sys.path.insert(0, str(_path)) + +from dispatch_judge_bridge import ( # noqa: E402 + MODE_PRODUCTION_PREFLIGHT, + DispatchJudgeRunner, +) +from run_writer_blind_judge import ( # noqa: E402 + INPUT_VERSION, + ORACLE_VERSION, + canonical_sha256, + run_writer_blind_judge, + validate_blind_judge_input, +) +from writer_rubric import RUBRIC_POLICY_VERSION # noqa: E402 +from muse_db import connect # noqa: E402 +from run_registry import finish_run, new_run_id, start_run # noqa: E402 + +ARTIFACTS = REPO_ROOT / "docs" / "write-chapter" / "artifacts" +CREATOR = "judge-dispatch" + + +def _sha256_text(text: str) -> str: + return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _latest_writer_context(work_id: int, chapter: int) -> dict: + """取目标章最近一次生产冻结上下文:细纲形状已按写手合同校验过。""" + + candidates = sorted( + ARTIFACTS.glob(f"run-prod-work{work_id}-ch{chapter}-*-writer-context.json"), + key=lambda p: p.stat().st_mtime, + ) + if not candidates: + raise SystemExit(f"未找到作品{work_id}第{chapter}章的冻结上下文工件,先跑生产链") + return json.loads(candidates[-1].read_text(encoding="utf-8")) + + +def build_blind_input_for_candidates( + *, + work_id: int, + chapter: int, + candidate_ids: list[int], + run_id: str, + sample_id: str, + scenario: str, + authorization_snapshot_id: str, +) -> dict: + """组装合法盲评输入:候选匿名化、哈希全绑定、防泄漏校验。""" + + with connect(readonly=True) as conn: + rows = conn.execute( + "SELECT id, candidate_body FROM example_candidate WHERE id = ANY(%s)", + (list(candidate_ids),), + ).fetchall() + found = {row[0]: str(row[1]) for row in rows} + missing = [cid for cid in candidate_ids if cid not in found] + if missing: + raise SystemExit(f"候选缺失:{missing}") + + # 匿名化:盲 ID 与库内顺序脱钩(洗牌),身份只留在编排侧映射里。 + shuffled = list(candidate_ids) + random.shuffle(shuffled) + candidates = [] + blind_mapping: dict[str, int] = {} + for index, cid in enumerate(shuffled): + body = found[cid] + blind_id = f"candidate-{index + 1}" + blind_mapping[blind_id] = cid + candidates.append({ + "blindCandidateId": blind_id, + "candidateSha256": _sha256_text(body), + "candidateBody": body, + }) + candidate_order = [item["blindCandidateId"] for item in candidates] + + ctx = _latest_writer_context(work_id, chapter) + outline = ctx["fineOutline"] + fine_outline = { + "sourceRef": dict(outline["sourceRef"]), + "hardConstraints": [ + {"constraintId": f"hc-{i + 1}", "text": str(text)} + for i, text in enumerate(outline["hardConstraints"]) + ], + "adjustableBeats": [str(text) for text in outline.get("adjustableBeats") or []], + "declaredNewFacts": [ + {"factId": item["factId"], "text": item["text"], "sourceRef": dict(item["sourceRef"])} + for item in outline.get("declaredNewFacts") or [] + ], + } + + # oracle 真值包:目标章断言取细纲硬约束(本章必须发生的事);历史断言为空集。 + target_assertions = [ + { + "assertionId": f"target-hc-{i + 1}", + "text": item["text"], + "sourceVersion": str(outline["sourceRef"].get("sourceVersion") or "v1"), + "chapterStart": chapter, + "chapterEnd": chapter, + "contentSha256": _sha256_text(item["text"]), + } + for i, item in enumerate(fine_outline["hardConstraints"]) + ] + oracle_pack = { + "schemaVersion": ORACLE_VERSION, + "evaluationSetVersion": "production-preflight-v1", + "sampleId": sample_id, + "workId": work_id, + "asOf": chapter - 1, + "sourceSnapshotSha256": canonical_sha256(fine_outline), + "authorizationSnapshotId": authorization_snapshot_id, + "historicalAssertions": [], + "targetAssertions": target_assertions, + } + oracle_pack["packSha256"] = canonical_sha256(oracle_pack) + + blind_input = { + "schemaVersion": INPUT_VERSION, + "runId": run_id, + "sampleId": sample_id, + "reviewerInvocationId": f"{run_id}-judge-r1", + "candidateOrder": candidate_order, + "candidates": candidates, + "fineOutline": fine_outline, + "oracleTruthPack": oracle_pack, + "oracleTruthPackSha256": oracle_pack["packSha256"], + "authorizationSnapshotId": authorization_snapshot_id, + "scenario": scenario, + "rubricPolicyVersion": RUBRIC_POLICY_VERSION, + } + blind_input["blindInputSha256"] = canonical_sha256(blind_input) + validate_blind_judge_input(blind_input) # 组装即校验:防泄漏与哈希绑定不过就失败关闭 + return {"blindInput": blind_input, "blindMapping": blind_mapping} + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="派发评委智能体做生产预检盲评(评测链)") + parser.add_argument("work_id", type=int) + parser.add_argument("chapter", type=int) + parser.add_argument("--candidate-ids", required=True, help="逗号分隔,至少两个") + parser.add_argument("--provider", required=True) + parser.add_argument("--model", required=True) + parser.add_argument("--thinking", default=None) + parser.add_argument("--scenario", default="battle") + parser.add_argument("--mode", default=MODE_PRODUCTION_PREFLIGHT) + args = parser.parse_args(argv) + + candidate_ids = [int(x) for x in args.candidate_ids.split(",") if x.strip()] + if len(candidate_ids) < 2: + raise SystemExit("盲评是对比设计:至少两个候选") + + run_id = new_run_id("run-blind-judge", work_id=args.work_id, target_chapter=args.chapter) + start_run(run_id=run_id, work_id=args.work_id, target_chapter=args.chapter, + trigger_detail={"stage": "judge-evaluation", "mode": args.mode, + "candidateIds": candidate_ids}, + creator=CREATOR) + sample_id = f"preflight-work{args.work_id}-ch{args.chapter}-{datetime.now(timezone.utc).strftime('%Y%m%dT%H%M%SZ')}" + try: + built = build_blind_input_for_candidates( + work_id=args.work_id, chapter=args.chapter, candidate_ids=candidate_ids, + run_id=run_id, sample_id=sample_id, scenario=args.scenario, + authorization_snapshot_id="auth-work12-production-v1", + ) + blind_input = built["blindInput"] + ARTIFACTS.mkdir(exist_ok=True) + (ARTIFACTS / f"{run_id}-blind-input.json").write_text( + json.dumps(blind_input, ensure_ascii=False, indent=1), encoding="utf-8") + # 盲 ID → 候选映射只落编排侧工件,不进评委输入(解读报告用)。 + (ARTIFACTS / f"{run_id}-blind-mapping.json").write_text( + json.dumps(built["blindMapping"], ensure_ascii=False, indent=1), encoding="utf-8") + + runner = DispatchJudgeRunner( + run_id=run_id, work_id=args.work_id, repo_root=REPO_ROOT, + provider=args.provider, model=args.model, mode=args.mode, + thinking=args.thinking, spec_dir=ARTIFACTS, + ) + result = run_writer_blind_judge(blind_input, model_runner=runner) + if not result.get("ok"): + finish_run(run_id, "failed", creator=CREATOR, + trigger_detail={"stage": "judge-evaluation", "primaryCode": result.get("primaryCode")}) + print(json.dumps(result, ensure_ascii=False, default=str)[:600]) + print(f"RUN_ID={run_id}") + return 1 + + report = result["report"] + (ARTIFACTS / f"{run_id}-blind-judge-report.json").write_text( + json.dumps(report, ensure_ascii=False, indent=1), encoding="utf-8") + with connect() as conn: + conn.execute( + "INSERT INTO example_quality_result(run_id, judge_kind, conclusion, detail) " + "VALUES (%s, %s, %s, %s::jsonb)", + (run_id, "scoring", report["status"], json.dumps(report, ensure_ascii=False)), + ) + conn.commit() + finish_run(run_id, "completed", creator=CREATOR, + trigger_detail={"stage": "judge-evaluation", "dispatchRuns": runner.dispatch_run_ids}) + + # 盲 ID → 候选映射只在编排侧输出,不进评委输入。 + mapping = built["blindMapping"] + scores = { + f"{item['blindCandidateId']}(候选 {mapping[item['blindCandidateId']]})": item["scores"] + for item in report["candidateScores"] + } + print(json.dumps({"runId": run_id, "sampleId": sample_id, + "candidateOrder": report["candidateOrder"], + "scores": scores}, ensure_ascii=False, default=str)[:900]) + print(f"RUN_ID={run_id}") + print("盲评报告已落 example_quality_result(judge_kind=blind_judge)。") + return 0 + except BaseException as exc: + try: + finish_run(run_id, "failed", creator=CREATOR, + trigger_detail={"stage": "judge-evaluation", "error_type": type(exc).__name__}) + except Exception: + pass + raise + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/plans/2026-08-22-agent-example整体收敛总plan.md b/docs/plans/2026-08-22-agent-example整体收敛总plan.md index da577e1..e46708e 100644 --- a/docs/plans/2026-08-22-agent-example整体收敛总plan.md +++ b/docs/plans/2026-08-22-agent-example整体收敛总plan.md @@ -174,6 +174,7 @@ A、B 可并行;C 依赖 A;E 依赖 D;F 依赖 E;G 依赖 D(链路透 - 意图:评委与抽取智能体接入框架派发(评委保持圈定授权隔离);用对照实验数据裁决直调链去留;处理写手探索冗余——生产链改为只给最小冻结输入(任务提示词 + 授权范围),由写作智能体真正自主探索取材,预组装完整上下文降为对照模式专用(2026-08-22 烟测实证:预组装下写手 0 次工具调用,探索无发生空间)。 - 第一部分(已完成,离线绿):两阶段写手落地——探索阶段(只读工具,产出探索清单)与生成阶段(无工具,按回放资料单次成稿)分离;`--two-phase` 旗标;详见 `docs/plans/2026-08-22-阶段F-第一部分-两阶段写手-探索与生成分离.md`。 - 第二部分(已完成,真实派发通过):抽取智能体接入框架派发(全量正文注入任务输入,工具白名单留给查重探索;机械校验失败一轮修复重派,仍不合法走保守收口);候选 164 采纳后对第 3 章首次真实抽取产出 19 条草稿待人审;详见 `docs/plans/2026-08-22-阶段F-第二部分-知识链派发-抽取智能体.md`。 +- 第三部分(已完成,真实盲评通过):评委智能体经 `ModelRunner` 协议接入框架派发(圈定授权:实验无工具/生产预检只读检索;盲评合同与防泄漏校验全复用既有适配层);候选 123 vs 164 真实盲评通过,164 四维领先与人采纳方向一致;详见 `docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md`。 - 边界:对照只在显式对照模式运行;评测候选四层强制不可接受不变;知识草稿须经人确认转正。 - 验证:盲评隔离测试(禁看清单越权拒绝);对照产出可比正文与成本数据;退役项零引用后删除。 diff --git a/docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md b/docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md new file mode 100644 index 0000000..9c54efe --- /dev/null +++ b/docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md @@ -0,0 +1,51 @@ +# 阶段 F 第三部分:评测链派发——评委智能体 + +日期:2026-08-23 +状态:已完成(离线绿 + 真实盲评通过) +上游事实:第 3 章候选 164 已采纳入正典;候选 123(v7,被机械门拒)与 164(v9,采纳)构成真实对比盲评样本。 + +## 1. 意图 + +评委智能体接入框架派发,保持圈定授权隔离;盲评编排(输入防泄漏校验、草稿逐字引文校验、有界纠错、报告哈希绑定)全部复用既有可信适配层,派发桥只实现 `ModelRunner` 协议。 + +## 2. 设计 + +```text +编排侧:取两个候选 → 匿名化(盲 ID 洗牌,映射只留编排侧工件) + → 组装盲评输入(blind-judge v4:哈希全绑定,组装即过官方校验器) + ↓ +派发评委(每位评委独立新会话;任务输入 = 盲评模型输入本体,桥不附加身份) + 圈定授权:实验场景无工具;生产预检开放 read_chapter_text/search_entities + ↓ +编排侧:草稿逐字引文校验(防编造)→ 有界纠错一轮 → 报告绑定回执哈希 + ↓ +落库:example_quality_result(judge_kind=scoring) +``` + +关键取舍: + +- 接入点是 `ModelRunner` 协议(`run_writer_blind_judge(blind_input, model_runner=...)`):派发不触碰盲评合同本身。 +- 隔离三层机械强制:输入侧防泄漏走查(官方校验器)、任务包与模型输入全等(桥不附加身份)、评委会话互相独立。 +- 引文纪律是提示词层问题(模型可被说服):任务提示词给出 evidenceRefs 的四种 sourceType 与精确 ID 空间(逐字选用,禁拼接描述词)。 +- 报告落库枚举以库级 CHECK 约束为准(`scoring`),不自造未登记枚举。 + +## 3. 改动台账 + +| 文件 | 动作 | 原因 | +|---|---|---| +| `.agent/skills/score-content-quality/scripts/dispatch_judge_bridge.py` | 新增 | 圈定授权白名单 + `DispatchJudgeRunner`(ModelRunner 协议) | +| `.agent/skills/score-content-quality/scripts/judge_via_dispatch.py` | 新增 | 生产预检盲评入口:候选匿名化、盲评输入组装、报告落库 | +| `tests/skills/score-content-quality/test_dispatch_judge_bridge.py` | 新增 | 离线测试(10 项):圈定白名单、任务包零身份、会话独立、失败关闭、盲评输入官方校验器回环 | + +## 4. 验证结果 + +- 离线测试 10/10 通过。 +- 真实盲评(候选 123 vs 164,battle 场景,opus-5): + - 前两轮草稿因引文格式被机械校验拒绝(evidenceRefs 的 sourceId 拼接描述词)——失败关闭与有界纠错按合同工作;任务提示词补全引用合同(四种 sourceType 与精确 ID 空间)后第三轮通过。 + - 第三轮报告(`run-blind-judge-w12-c3-20260823T155855`,judge_kind=scoring):候选 164 在 5 维中 4 维领先(设定保真/细纲保真/文风一致/叙事张力),候选 123 仅正文可读性领先 0.5 分——与人采纳 164 的决定方向一致(独立证据)。 + - 落库枚举错误(误用 `blind_judge`,库级 CHECK 只许 detection/scoring/review/experiment)当场被约束拦下——按 `scoring` 补落库并如实收口运行。 + +## 5. 未决 + +- 评委圈定授权的"禁看清单越权拒绝"专项测试(总 plan 验证项)留待对照实验阶段一起做。 +- 直调链终态裁决(对照实验数据)与退役项删除按总 plan 阶段 F 收尾。 diff --git a/tests/skills/score-content-quality/test_dispatch_judge_bridge.py b/tests/skills/score-content-quality/test_dispatch_judge_bridge.py new file mode 100644 index 0000000..a3059fa --- /dev/null +++ b/tests/skills/score-content-quality/test_dispatch_judge_bridge.py @@ -0,0 +1,180 @@ +#!/usr/bin/env python3 +"""评委智能体派发桥离线测试(阶段 F 第三部分)。 + +固定合同:圈定授权工具白名单(实验无工具/生产预检只读)、任务包不带身份 +(防泄漏走查通过、输入与盲评模型输入全等)、每位评委独立会话、派发失败 +失败关闭、盲评输入组装哈希绑定经官方校验器回环。 +""" +from __future__ import annotations + +import json +import pathlib +import sys +import tempfile +import unittest +from unittest import mock + +PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3] +SCRIPT_DIR = PROJECT_ROOT / ".agent" / "skills" / "score-content-quality" / "scripts" +for path in (SCRIPT_DIR,): + if str(path) not in sys.path: + sys.path.insert(0, str(path)) + +import dispatch_judge_bridge as bridge # noqa: E402 +import judge_via_dispatch as cli # noqa: E402 +from dispatch_judge_bridge import ( # noqa: E402 + DispatchJudgeError, + DispatchJudgeRunner, + judge_tool_allowlist, +) +from run_writer_blind_judge import ( # noqa: E402 + FORBIDDEN_KEYS, + BlindJudgeContractError, + _walk_for_leakage, + validate_blind_judge_input, +) + +MODEL_INPUT = { + "candidates": [{"blindCandidateId": "candidate-1", "candidateBody": "正文甲"}], + "rubric": {"dimensions": []}, +} + + +class AllowlistTest(unittest.TestCase): + + def test_experiment_has_no_tools(self): + self.assertEqual(judge_tool_allowlist("experiment"), ()) + + def test_production_preflight_read_tools(self): + self.assertEqual(judge_tool_allowlist("production_preflight"), + ("read_chapter_text", "search_entities")) + + def test_unknown_mode_fails_closed(self): + with self.assertRaises(DispatchJudgeError): + judge_tool_allowlist("anything-else") + + +class RunnerTest(unittest.TestCase): + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.tmp = pathlib.Path(self._tmp.name) + self.captured = [] + + def tearDown(self): + self._tmp.cleanup() + + def _runner(self, mode="production_preflight"): + return DispatchJudgeRunner( + run_id="run-judge-test", work_id=12, repo_root=self.tmp, + provider="catproxy-anthropic", model="claude-opus-5", mode=mode, + spec_dir=self.tmp, + ) + + def _fake_dispatch(self, output, status="completed", code=0): + def fake(spec_file, **kwargs): + self.captured.append({"spec": json.loads(pathlib.Path(spec_file).read_text(encoding="utf-8")), + "kwargs": kwargs}) + run_dir = self.tmp / kwargs["run_id"] + run_dir.mkdir(parents=True) + if status == "completed": + (run_dir / "output.json").write_text(json.dumps(output, ensure_ascii=False), encoding="utf-8") + return {"status": status, "runDir": str(run_dir), "errorCode": None if code == 0 else "TIMEOUT"}, code + return fake + + def test_run_returns_structured_output_and_receipt_hash(self): + runner = self._runner() + draft = {"schemaVersion": "blind-judge-draft-v3", "candidateScores": []} + with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch(draft)): + out = runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={"type": "object"}) + self.assertEqual(out["structuredOutput"], draft) + self.assertTrue(out["modelReceiptSha256"].startswith("sha256:")) + self.assertEqual(runner.dispatch_run_ids, ["run-judge-test-judge-v1"]) + + def test_task_spec_carries_no_identity(self): + runner = self._runner() + with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})): + runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={"type": "object"}) + spec = self.captured[0]["spec"] + self.assertEqual(spec["role"], "judge") + # 输入与盲评模型输入全等:桥不附加身份字段 + self.assertEqual(spec["input"], MODEL_INPUT) + # 防泄漏走查:任务包整体不含禁看键 + _walk_for_leakage(spec) + for key in FORBIDDEN_KEYS: + self.assertNotIn(key, spec) + + def test_each_judge_gets_fresh_session(self): + runner = self._runner() + with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})): + runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={}) + runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={}) + sessions = [c["kwargs"]["session_id"] for c in self.captured] + self.assertEqual(sessions, ["judge-run-judge-test-judge-v1", "judge-run-judge-test-judge-v2"]) + self.assertEqual(len(set(sessions)), 2) + + def test_mode_controls_tool_allowlist_in_spec(self): + runner = self._runner(mode="experiment") + with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})): + runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={}) + self.assertEqual(self.captured[0]["spec"]["toolAllowlist"], []) + self.assertFalse(self.captured[0]["kwargs"]["enable_read_tools"]) + + def test_dispatch_failure_fails_closed(self): + runner = self._runner() + with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({}, status="failed", code=1)): + with self.assertRaises(BlindJudgeContractError) as ctx: + runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={}) + self.assertEqual(ctx.exception.code, "BLIND_JUDGE_RUNTIME_FAILED") + + def test_wrong_adapter_role_rejected(self): + runner = self._runner() + with self.assertRaises(BlindJudgeContractError): + runner.run(adapter_role="writer", model_input=MODEL_INPUT, output_schema={}) + + +class BlindInputBuilderTest(unittest.TestCase): + + def test_assembled_input_passes_official_validator(self): + bodies = {123: "甲候选正文:茧撕开舱门。", 164: "乙候选正文:林深盯着深渊。"} + outline = { + "sourceRef": {"sourceId": "outline:volume_1", "sourceVersion": "v1"}, + "hardConstraints": ["硬约束甲", "硬约束乙"], + "adjustableBeats": ["拍一"], + "declaredNewFacts": [ + {"factId": "fact-ch3-1", "text": "新事实", "sourceRef": {"sourceId": "outline:volume_1", "sourceVersion": "v1"}}, + ], + } + + class _Conn: + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def execute(self, sql, params=None): + class _Rows: + def fetchall(self): + return [(cid, bodies[cid]) for cid in params[0]] + return _Rows() + + with mock.patch.object(cli, "connect", lambda readonly=True: _Conn()), \ + mock.patch.object(cli, "_latest_writer_context", lambda work_id, chapter: {"fineOutline": outline}): + built = cli.build_blind_input_for_candidates( + work_id=12, chapter=3, candidate_ids=[123, 164], + run_id="run-judge-test", sample_id="sample-1", scenario="battle", + authorization_snapshot_id="auth-work12-production-v1", + ) + blind_input = built["blindInput"] + # 官方校验器回环:防泄漏、哈希绑定、候选哈希全部通过 + validate_blind_judge_input(blind_input) + # 盲 ID 与库内候选脱钩但映射完整 + self.assertEqual(sorted(built["blindMapping"].values()), [123, 164]) + self.assertEqual(sorted(blind_input["candidateOrder"]), ["candidate-1", "candidate-2"]) + # 目标章断言来自细纲硬约束 + self.assertEqual(len(blind_input["oracleTruthPack"]["targetAssertions"]), 2) + + +if __name__ == "__main__": + unittest.main()