阶段F第三部分: 评测链派发——评委智能体经ModelRunner协议接入框架派发(离线10项绿+候选123vs164真实盲评通过,164四维领先与人采纳方向一致)
This commit is contained in:
parent
640e17499d
commit
befb4714c0
@ -0,0 +1,179 @@
|
||||
#!/usr/bin/env python3
|
||||
"""评委智能体的框架派发 runner(阶段 F 第三部分)。
|
||||
|
||||
实现 run_writer_blind_judge 的 ModelRunner 协议:盲评编排(输入防泄漏校验、
|
||||
草稿校验、有界纠错、报告绑定)全部复用既有可信适配层,本桥只负责把模型
|
||||
调用切到智能体框架派发。
|
||||
|
||||
圈定授权隔离(边界合同):
|
||||
- 任务包 = 盲评模型输入本体(已过 _walk_for_leakage 防泄漏校验),桥不附加
|
||||
任何身份字段;
|
||||
- 实验场景(回放盲评):工具白名单为空,只用冻结快照;
|
||||
- 生产预检场景:开放只读工具检索正典事实,用于对照设定一致性;
|
||||
- 每位评委独立新会话,不共享其他评委的上下文。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Mapping
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DISPATCH_SCRIPTS = SCRIPT_DIR.parents[1] / "dispatch-agent-task" / "scripts"
|
||||
if str(DISPATCH_SCRIPTS) not in sys.path:
|
||||
sys.path.insert(0, str(DISPATCH_SCRIPTS))
|
||||
|
||||
from run_writer_blind_judge import BlindJudgeContractError, canonical_sha256 # noqa: E402
|
||||
from dispatch_agent_task import run_dispatch # noqa: E402
|
||||
from pi_runner import ExecutionPolicy # noqa: E402
|
||||
from read_tools import TOOL_REGISTRY # noqa: E402
|
||||
|
||||
MODE_EXPERIMENT = "experiment"
|
||||
MODE_PRODUCTION_PREFLIGHT = "production_preflight"
|
||||
|
||||
# 圈定授权:实验场景只用冻结快照(无工具);生产预检开放正典对照检索。
|
||||
MODE_TOOL_ALLOWLIST: dict[str, tuple[str, ...]] = {
|
||||
MODE_EXPERIMENT: (),
|
||||
MODE_PRODUCTION_PREFLIGHT: ("read_chapter_text", "search_entities"),
|
||||
}
|
||||
|
||||
JUDGE_MAX_DURATION_SECONDS = 1800
|
||||
SESSION_ROOT = Path("/tmp/muse-agent-runs/judge-sessions")
|
||||
|
||||
|
||||
class DispatchJudgeError(RuntimeError):
|
||||
"""评委派发失败:携带稳定错误码,编排方按失败关闭处理。"""
|
||||
|
||||
def __init__(self, code: str, message: str, *, details: Mapping[str, Any] | None = None):
|
||||
super().__init__(message)
|
||||
self.code = code
|
||||
self.details = dict(details or {})
|
||||
|
||||
|
||||
def judge_tool_allowlist(mode: str) -> tuple[str, ...]:
|
||||
"""按场景取圈定工具白名单;未登记场景失败关闭。"""
|
||||
|
||||
allowlist = MODE_TOOL_ALLOWLIST.get(mode)
|
||||
if allowlist is None:
|
||||
raise DispatchJudgeError("JUDGE_MODE_UNKNOWN", f"评委场景未登记:{mode}")
|
||||
for name in allowlist:
|
||||
if name not in TOOL_REGISTRY:
|
||||
raise DispatchJudgeError("JUDGE_TOOL_UNREGISTERED", f"评委白名单工具未登记:{name}")
|
||||
return allowlist
|
||||
|
||||
|
||||
class DispatchJudgeRunner:
|
||||
"""ModelRunner 协议的框架派发实现:每次 run 派发一位独立评委。"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
run_id: str,
|
||||
work_id: int,
|
||||
repo_root: str | Path,
|
||||
provider: str,
|
||||
model: str,
|
||||
mode: str = MODE_PRODUCTION_PREFLIGHT,
|
||||
thinking: str | None = None,
|
||||
spec_dir: str | Path | None = None,
|
||||
launcher: Callable[..., Any] | None = None,
|
||||
connect_factory: Callable[..., Any] | None = None,
|
||||
) -> None:
|
||||
self.run_id = run_id
|
||||
self.work_id = work_id
|
||||
self.repo_root = repo_root
|
||||
self.provider = provider
|
||||
self.model = model
|
||||
self.mode = mode
|
||||
self.thinking = thinking
|
||||
self.spec_dir = Path(spec_dir) if spec_dir is not None else SCRIPT_DIR
|
||||
self.launcher = launcher
|
||||
self.connect_factory = connect_factory
|
||||
self.allowlist = judge_tool_allowlist(mode)
|
||||
self._attempt = 0
|
||||
self.dispatch_run_ids: list[str] = []
|
||||
|
||||
def run(self, *, adapter_role: str, model_input: Mapping[str, Any], output_schema: Mapping[str, Any]) -> Mapping[str, Any]:
|
||||
if adapter_role != "blind_judge":
|
||||
raise BlindJudgeContractError("BLIND_JUDGE_RUNNER_INVALID", f"派发桥只承接 blind_judge,收到 {adapter_role}")
|
||||
self._attempt += 1
|
||||
dispatch_run_id = f"{self.run_id}-judge-v{self._attempt}"
|
||||
self.dispatch_run_ids.append(dispatch_run_id)
|
||||
|
||||
spec = {
|
||||
"specVersion": "agent-task-v1",
|
||||
"role": "judge",
|
||||
"taskPrompt": (
|
||||
"你是质量评委,执行一次独立盲评:只按盲评输入中的量表与证据逐维打分,"
|
||||
"每维分数必须给可定位引文;输入不包含任何候选身份,不要推测候选归属。"
|
||||
"evidenceRefs 每项严格为 {sourceType, sourceId},sourceId 必须从下列既有 ID 中逐字选用,"
|
||||
"不得拼接、改写或附加描述词:"
|
||||
"sourceType=candidate 时 sourceId=被打分候选的 blindCandidateId 本身;"
|
||||
"sourceType=fine_outline 时 sourceId=fineOutline.hardConstraints 中的某个 constraintId;"
|
||||
"sourceType=oracle_assertion 时 sourceId=oracleTruthPack 中的某个 assertionId;"
|
||||
"sourceType=judge_inference 时 sourceId=当前打分维度名。"
|
||||
"引文必须是候选正文中的逐字连续片段,不得改写、省略或加省略号。"
|
||||
"按 outputContract 的完整集合一次返回全部对象,只输出一个草稿 JSON。"
|
||||
),
|
||||
# 盲评模型输入本体即任务输入:已通过防泄漏校验,桥不附加身份字段。
|
||||
"input": dict(model_input),
|
||||
"outputSchema": dict(output_schema),
|
||||
"outputSchemaId": "blind-judge-draft-v3",
|
||||
"toolAllowlist": list(self.allowlist),
|
||||
"maxDurationSeconds": JUDGE_MAX_DURATION_SECONDS,
|
||||
}
|
||||
self.spec_dir.mkdir(parents=True, exist_ok=True)
|
||||
spec_file = self.spec_dir / f"{dispatch_run_id}-task.json"
|
||||
spec_file.write_text(json.dumps(spec, ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
|
||||
# 每位评委独立新会话:盲评独立性要求不共享其他评委上下文。
|
||||
session_id = f"judge-{dispatch_run_id}"
|
||||
session_dir = SESSION_ROOT / session_id
|
||||
session_dir.mkdir(parents=True, mode=0o700, exist_ok=True)
|
||||
|
||||
policy = ExecutionPolicy(provider=self.provider, model=self.model, thinking=self.thinking)
|
||||
receipt, code = run_dispatch(
|
||||
spec_file,
|
||||
repo_root=self.repo_root,
|
||||
policy=policy,
|
||||
run_id=dispatch_run_id,
|
||||
trigger_source="user",
|
||||
trigger_detail={"stage": "judge-dispatch", "evaluationRunId": self.run_id,
|
||||
"mode": self.mode, "attempt": self._attempt},
|
||||
session_id=session_id,
|
||||
session_dir=session_dir,
|
||||
enable_read_tools=bool(self.allowlist),
|
||||
launcher=self.launcher,
|
||||
connect_factory=self.connect_factory,
|
||||
)
|
||||
if code != 0 or receipt.get("status") != "completed":
|
||||
raise BlindJudgeContractError(
|
||||
"BLIND_JUDGE_RUNTIME_FAILED",
|
||||
f"评委派发未成功:{receipt.get('error') or receipt.get('errorCode')}",
|
||||
)
|
||||
|
||||
output_file = Path(str(receipt.get("runDir") or "")) / "output.json"
|
||||
try:
|
||||
structured = json.loads(output_file.read_text(encoding="utf-8"))
|
||||
except (OSError, ValueError) as exc:
|
||||
raise BlindJudgeContractError(
|
||||
"BLIND_JUDGE_RUNTIME_FAILED", f"评委派发未产出可用草稿:{exc}"
|
||||
) from exc
|
||||
if not isinstance(structured, Mapping):
|
||||
raise BlindJudgeContractError("BLIND_JUDGE_RUNTIME_FAILED", "评委草稿必须是 JSON 对象")
|
||||
|
||||
# 回执哈希绑定派发回执本体(规范化 JSON 哈希),报告经它回指本次模型调用。
|
||||
receipt_hash = canonical_sha256(json.loads(json.dumps(receipt, ensure_ascii=False, default=str)))
|
||||
return {"structuredOutput": dict(structured), "modelReceiptSha256": receipt_hash}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"JUDGE_MAX_DURATION_SECONDS",
|
||||
"MODE_EXPERIMENT",
|
||||
"MODE_PRODUCTION_PREFLIGHT",
|
||||
"MODE_TOOL_ALLOWLIST",
|
||||
"DispatchJudgeError",
|
||||
"DispatchJudgeRunner",
|
||||
"judge_tool_allowlist",
|
||||
]
|
||||
@ -0,0 +1,246 @@
|
||||
#!/usr/bin/env python3
|
||||
"""评委智能体派发入口(阶段 F 第三部分):生产预检盲评。
|
||||
|
||||
对同一章的两个(及以上)候选做匿名对比盲评:组装合法盲评输入
|
||||
(blind-judge v4,哈希全绑定、防泄漏校验),经框架派发评委,报告落
|
||||
example_quality_result(judge_kind=scoring,枚举以库级 CHECK 约束为准)。
|
||||
|
||||
真实模型调用必须显式授权并显式给出 provider/model:
|
||||
.venv/bin/python .agent/skills/score-content-quality/scripts/judge_via_dispatch.py \
|
||||
12 3 --candidate-ids 123,164 --provider catproxy-anthropic --model claude-opus-5 --thinking medium
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import random
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
REPO_ROOT = SCRIPT_DIR.parents[3]
|
||||
for _path in (SCRIPT_DIR,
|
||||
REPO_ROOT / ".agent" / "skills" / "record-run-evidence" / "scripts",
|
||||
REPO_ROOT / ".agent" / "skills" / "access-database" / "scripts"):
|
||||
if str(_path) not in sys.path:
|
||||
sys.path.insert(0, str(_path))
|
||||
|
||||
from dispatch_judge_bridge import ( # noqa: E402
|
||||
MODE_PRODUCTION_PREFLIGHT,
|
||||
DispatchJudgeRunner,
|
||||
)
|
||||
from run_writer_blind_judge import ( # noqa: E402
|
||||
INPUT_VERSION,
|
||||
ORACLE_VERSION,
|
||||
canonical_sha256,
|
||||
run_writer_blind_judge,
|
||||
validate_blind_judge_input,
|
||||
)
|
||||
from writer_rubric import RUBRIC_POLICY_VERSION # noqa: E402
|
||||
from muse_db import connect # noqa: E402
|
||||
from run_registry import finish_run, new_run_id, start_run # noqa: E402
|
||||
|
||||
ARTIFACTS = REPO_ROOT / "docs" / "write-chapter" / "artifacts"
|
||||
CREATOR = "judge-dispatch"
|
||||
|
||||
|
||||
def _sha256_text(text: str) -> str:
|
||||
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _latest_writer_context(work_id: int, chapter: int) -> dict:
|
||||
"""取目标章最近一次生产冻结上下文:细纲形状已按写手合同校验过。"""
|
||||
|
||||
candidates = sorted(
|
||||
ARTIFACTS.glob(f"run-prod-work{work_id}-ch{chapter}-*-writer-context.json"),
|
||||
key=lambda p: p.stat().st_mtime,
|
||||
)
|
||||
if not candidates:
|
||||
raise SystemExit(f"未找到作品{work_id}第{chapter}章的冻结上下文工件,先跑生产链")
|
||||
return json.loads(candidates[-1].read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def build_blind_input_for_candidates(
|
||||
*,
|
||||
work_id: int,
|
||||
chapter: int,
|
||||
candidate_ids: list[int],
|
||||
run_id: str,
|
||||
sample_id: str,
|
||||
scenario: str,
|
||||
authorization_snapshot_id: str,
|
||||
) -> dict:
|
||||
"""组装合法盲评输入:候选匿名化、哈希全绑定、防泄漏校验。"""
|
||||
|
||||
with connect(readonly=True) as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT id, candidate_body FROM example_candidate WHERE id = ANY(%s)",
|
||||
(list(candidate_ids),),
|
||||
).fetchall()
|
||||
found = {row[0]: str(row[1]) for row in rows}
|
||||
missing = [cid for cid in candidate_ids if cid not in found]
|
||||
if missing:
|
||||
raise SystemExit(f"候选缺失:{missing}")
|
||||
|
||||
# 匿名化:盲 ID 与库内顺序脱钩(洗牌),身份只留在编排侧映射里。
|
||||
shuffled = list(candidate_ids)
|
||||
random.shuffle(shuffled)
|
||||
candidates = []
|
||||
blind_mapping: dict[str, int] = {}
|
||||
for index, cid in enumerate(shuffled):
|
||||
body = found[cid]
|
||||
blind_id = f"candidate-{index + 1}"
|
||||
blind_mapping[blind_id] = cid
|
||||
candidates.append({
|
||||
"blindCandidateId": blind_id,
|
||||
"candidateSha256": _sha256_text(body),
|
||||
"candidateBody": body,
|
||||
})
|
||||
candidate_order = [item["blindCandidateId"] for item in candidates]
|
||||
|
||||
ctx = _latest_writer_context(work_id, chapter)
|
||||
outline = ctx["fineOutline"]
|
||||
fine_outline = {
|
||||
"sourceRef": dict(outline["sourceRef"]),
|
||||
"hardConstraints": [
|
||||
{"constraintId": f"hc-{i + 1}", "text": str(text)}
|
||||
for i, text in enumerate(outline["hardConstraints"])
|
||||
],
|
||||
"adjustableBeats": [str(text) for text in outline.get("adjustableBeats") or []],
|
||||
"declaredNewFacts": [
|
||||
{"factId": item["factId"], "text": item["text"], "sourceRef": dict(item["sourceRef"])}
|
||||
for item in outline.get("declaredNewFacts") or []
|
||||
],
|
||||
}
|
||||
|
||||
# oracle 真值包:目标章断言取细纲硬约束(本章必须发生的事);历史断言为空集。
|
||||
target_assertions = [
|
||||
{
|
||||
"assertionId": f"target-hc-{i + 1}",
|
||||
"text": item["text"],
|
||||
"sourceVersion": str(outline["sourceRef"].get("sourceVersion") or "v1"),
|
||||
"chapterStart": chapter,
|
||||
"chapterEnd": chapter,
|
||||
"contentSha256": _sha256_text(item["text"]),
|
||||
}
|
||||
for i, item in enumerate(fine_outline["hardConstraints"])
|
||||
]
|
||||
oracle_pack = {
|
||||
"schemaVersion": ORACLE_VERSION,
|
||||
"evaluationSetVersion": "production-preflight-v1",
|
||||
"sampleId": sample_id,
|
||||
"workId": work_id,
|
||||
"asOf": chapter - 1,
|
||||
"sourceSnapshotSha256": canonical_sha256(fine_outline),
|
||||
"authorizationSnapshotId": authorization_snapshot_id,
|
||||
"historicalAssertions": [],
|
||||
"targetAssertions": target_assertions,
|
||||
}
|
||||
oracle_pack["packSha256"] = canonical_sha256(oracle_pack)
|
||||
|
||||
blind_input = {
|
||||
"schemaVersion": INPUT_VERSION,
|
||||
"runId": run_id,
|
||||
"sampleId": sample_id,
|
||||
"reviewerInvocationId": f"{run_id}-judge-r1",
|
||||
"candidateOrder": candidate_order,
|
||||
"candidates": candidates,
|
||||
"fineOutline": fine_outline,
|
||||
"oracleTruthPack": oracle_pack,
|
||||
"oracleTruthPackSha256": oracle_pack["packSha256"],
|
||||
"authorizationSnapshotId": authorization_snapshot_id,
|
||||
"scenario": scenario,
|
||||
"rubricPolicyVersion": RUBRIC_POLICY_VERSION,
|
||||
}
|
||||
blind_input["blindInputSha256"] = canonical_sha256(blind_input)
|
||||
validate_blind_judge_input(blind_input) # 组装即校验:防泄漏与哈希绑定不过就失败关闭
|
||||
return {"blindInput": blind_input, "blindMapping": blind_mapping}
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description="派发评委智能体做生产预检盲评(评测链)")
|
||||
parser.add_argument("work_id", type=int)
|
||||
parser.add_argument("chapter", type=int)
|
||||
parser.add_argument("--candidate-ids", required=True, help="逗号分隔,至少两个")
|
||||
parser.add_argument("--provider", required=True)
|
||||
parser.add_argument("--model", required=True)
|
||||
parser.add_argument("--thinking", default=None)
|
||||
parser.add_argument("--scenario", default="battle")
|
||||
parser.add_argument("--mode", default=MODE_PRODUCTION_PREFLIGHT)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
candidate_ids = [int(x) for x in args.candidate_ids.split(",") if x.strip()]
|
||||
if len(candidate_ids) < 2:
|
||||
raise SystemExit("盲评是对比设计:至少两个候选")
|
||||
|
||||
run_id = new_run_id("run-blind-judge", work_id=args.work_id, target_chapter=args.chapter)
|
||||
start_run(run_id=run_id, work_id=args.work_id, target_chapter=args.chapter,
|
||||
trigger_detail={"stage": "judge-evaluation", "mode": args.mode,
|
||||
"candidateIds": candidate_ids},
|
||||
creator=CREATOR)
|
||||
sample_id = f"preflight-work{args.work_id}-ch{args.chapter}-{datetime.now(timezone.utc).strftime('%Y%m%dT%H%M%SZ')}"
|
||||
try:
|
||||
built = build_blind_input_for_candidates(
|
||||
work_id=args.work_id, chapter=args.chapter, candidate_ids=candidate_ids,
|
||||
run_id=run_id, sample_id=sample_id, scenario=args.scenario,
|
||||
authorization_snapshot_id="auth-work12-production-v1",
|
||||
)
|
||||
blind_input = built["blindInput"]
|
||||
ARTIFACTS.mkdir(exist_ok=True)
|
||||
(ARTIFACTS / f"{run_id}-blind-input.json").write_text(
|
||||
json.dumps(blind_input, ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
# 盲 ID → 候选映射只落编排侧工件,不进评委输入(解读报告用)。
|
||||
(ARTIFACTS / f"{run_id}-blind-mapping.json").write_text(
|
||||
json.dumps(built["blindMapping"], ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
|
||||
runner = DispatchJudgeRunner(
|
||||
run_id=run_id, work_id=args.work_id, repo_root=REPO_ROOT,
|
||||
provider=args.provider, model=args.model, mode=args.mode,
|
||||
thinking=args.thinking, spec_dir=ARTIFACTS,
|
||||
)
|
||||
result = run_writer_blind_judge(blind_input, model_runner=runner)
|
||||
if not result.get("ok"):
|
||||
finish_run(run_id, "failed", creator=CREATOR,
|
||||
trigger_detail={"stage": "judge-evaluation", "primaryCode": result.get("primaryCode")})
|
||||
print(json.dumps(result, ensure_ascii=False, default=str)[:600])
|
||||
print(f"RUN_ID={run_id}")
|
||||
return 1
|
||||
|
||||
report = result["report"]
|
||||
(ARTIFACTS / f"{run_id}-blind-judge-report.json").write_text(
|
||||
json.dumps(report, ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
with connect() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO example_quality_result(run_id, judge_kind, conclusion, detail) "
|
||||
"VALUES (%s, %s, %s, %s::jsonb)",
|
||||
(run_id, "scoring", report["status"], json.dumps(report, ensure_ascii=False)),
|
||||
)
|
||||
conn.commit()
|
||||
finish_run(run_id, "completed", creator=CREATOR,
|
||||
trigger_detail={"stage": "judge-evaluation", "dispatchRuns": runner.dispatch_run_ids})
|
||||
|
||||
# 盲 ID → 候选映射只在编排侧输出,不进评委输入。
|
||||
mapping = built["blindMapping"]
|
||||
scores = {
|
||||
f"{item['blindCandidateId']}(候选 {mapping[item['blindCandidateId']]})": item["scores"]
|
||||
for item in report["candidateScores"]
|
||||
}
|
||||
print(json.dumps({"runId": run_id, "sampleId": sample_id,
|
||||
"candidateOrder": report["candidateOrder"],
|
||||
"scores": scores}, ensure_ascii=False, default=str)[:900])
|
||||
print(f"RUN_ID={run_id}")
|
||||
print("盲评报告已落 example_quality_result(judge_kind=blind_judge)。")
|
||||
return 0
|
||||
except BaseException as exc:
|
||||
try:
|
||||
finish_run(run_id, "failed", creator=CREATOR,
|
||||
trigger_detail={"stage": "judge-evaluation", "error_type": type(exc).__name__})
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@ -174,6 +174,7 @@ A、B 可并行;C 依赖 A;E 依赖 D;F 依赖 E;G 依赖 D(链路透
|
||||
- 意图:评委与抽取智能体接入框架派发(评委保持圈定授权隔离);用对照实验数据裁决直调链去留;处理写手探索冗余——生产链改为只给最小冻结输入(任务提示词 + 授权范围),由写作智能体真正自主探索取材,预组装完整上下文降为对照模式专用(2026-08-22 烟测实证:预组装下写手 0 次工具调用,探索无发生空间)。
|
||||
- 第一部分(已完成,离线绿):两阶段写手落地——探索阶段(只读工具,产出探索清单)与生成阶段(无工具,按回放资料单次成稿)分离;`--two-phase` 旗标;详见 `docs/plans/2026-08-22-阶段F-第一部分-两阶段写手-探索与生成分离.md`。
|
||||
- 第二部分(已完成,真实派发通过):抽取智能体接入框架派发(全量正文注入任务输入,工具白名单留给查重探索;机械校验失败一轮修复重派,仍不合法走保守收口);候选 164 采纳后对第 3 章首次真实抽取产出 19 条草稿待人审;详见 `docs/plans/2026-08-22-阶段F-第二部分-知识链派发-抽取智能体.md`。
|
||||
- 第三部分(已完成,真实盲评通过):评委智能体经 `ModelRunner` 协议接入框架派发(圈定授权:实验无工具/生产预检只读检索;盲评合同与防泄漏校验全复用既有适配层);候选 123 vs 164 真实盲评通过,164 四维领先与人采纳方向一致;详见 `docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md`。
|
||||
- 边界:对照只在显式对照模式运行;评测候选四层强制不可接受不变;知识草稿须经人确认转正。
|
||||
- 验证:盲评隔离测试(禁看清单越权拒绝);对照产出可比正文与成本数据;退役项零引用后删除。
|
||||
|
||||
|
||||
51
docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md
Normal file
51
docs/plans/2026-08-22-阶段F-第三部分-评测链派发-评委智能体.md
Normal file
@ -0,0 +1,51 @@
|
||||
# 阶段 F 第三部分:评测链派发——评委智能体
|
||||
|
||||
日期:2026-08-23
|
||||
状态:已完成(离线绿 + 真实盲评通过)
|
||||
上游事实:第 3 章候选 164 已采纳入正典;候选 123(v7,被机械门拒)与 164(v9,采纳)构成真实对比盲评样本。
|
||||
|
||||
## 1. 意图
|
||||
|
||||
评委智能体接入框架派发,保持圈定授权隔离;盲评编排(输入防泄漏校验、草稿逐字引文校验、有界纠错、报告哈希绑定)全部复用既有可信适配层,派发桥只实现 `ModelRunner` 协议。
|
||||
|
||||
## 2. 设计
|
||||
|
||||
```text
|
||||
编排侧:取两个候选 → 匿名化(盲 ID 洗牌,映射只留编排侧工件)
|
||||
→ 组装盲评输入(blind-judge v4:哈希全绑定,组装即过官方校验器)
|
||||
↓
|
||||
派发评委(每位评委独立新会话;任务输入 = 盲评模型输入本体,桥不附加身份)
|
||||
圈定授权:实验场景无工具;生产预检开放 read_chapter_text/search_entities
|
||||
↓
|
||||
编排侧:草稿逐字引文校验(防编造)→ 有界纠错一轮 → 报告绑定回执哈希
|
||||
↓
|
||||
落库:example_quality_result(judge_kind=scoring)
|
||||
```
|
||||
|
||||
关键取舍:
|
||||
|
||||
- 接入点是 `ModelRunner` 协议(`run_writer_blind_judge(blind_input, model_runner=...)`):派发不触碰盲评合同本身。
|
||||
- 隔离三层机械强制:输入侧防泄漏走查(官方校验器)、任务包与模型输入全等(桥不附加身份)、评委会话互相独立。
|
||||
- 引文纪律是提示词层问题(模型可被说服):任务提示词给出 evidenceRefs 的四种 sourceType 与精确 ID 空间(逐字选用,禁拼接描述词)。
|
||||
- 报告落库枚举以库级 CHECK 约束为准(`scoring`),不自造未登记枚举。
|
||||
|
||||
## 3. 改动台账
|
||||
|
||||
| 文件 | 动作 | 原因 |
|
||||
|---|---|---|
|
||||
| `.agent/skills/score-content-quality/scripts/dispatch_judge_bridge.py` | 新增 | 圈定授权白名单 + `DispatchJudgeRunner`(ModelRunner 协议) |
|
||||
| `.agent/skills/score-content-quality/scripts/judge_via_dispatch.py` | 新增 | 生产预检盲评入口:候选匿名化、盲评输入组装、报告落库 |
|
||||
| `tests/skills/score-content-quality/test_dispatch_judge_bridge.py` | 新增 | 离线测试(10 项):圈定白名单、任务包零身份、会话独立、失败关闭、盲评输入官方校验器回环 |
|
||||
|
||||
## 4. 验证结果
|
||||
|
||||
- 离线测试 10/10 通过。
|
||||
- 真实盲评(候选 123 vs 164,battle 场景,opus-5):
|
||||
- 前两轮草稿因引文格式被机械校验拒绝(evidenceRefs 的 sourceId 拼接描述词)——失败关闭与有界纠错按合同工作;任务提示词补全引用合同(四种 sourceType 与精确 ID 空间)后第三轮通过。
|
||||
- 第三轮报告(`run-blind-judge-w12-c3-20260823T155855`,judge_kind=scoring):候选 164 在 5 维中 4 维领先(设定保真/细纲保真/文风一致/叙事张力),候选 123 仅正文可读性领先 0.5 分——与人采纳 164 的决定方向一致(独立证据)。
|
||||
- 落库枚举错误(误用 `blind_judge`,库级 CHECK 只许 detection/scoring/review/experiment)当场被约束拦下——按 `scoring` 补落库并如实收口运行。
|
||||
|
||||
## 5. 未决
|
||||
|
||||
- 评委圈定授权的"禁看清单越权拒绝"专项测试(总 plan 验证项)留待对照实验阶段一起做。
|
||||
- 直调链终态裁决(对照实验数据)与退役项删除按总 plan 阶段 F 收尾。
|
||||
180
tests/skills/score-content-quality/test_dispatch_judge_bridge.py
Normal file
180
tests/skills/score-content-quality/test_dispatch_judge_bridge.py
Normal file
@ -0,0 +1,180 @@
|
||||
#!/usr/bin/env python3
|
||||
"""评委智能体派发桥离线测试(阶段 F 第三部分)。
|
||||
|
||||
固定合同:圈定授权工具白名单(实验无工具/生产预检只读)、任务包不带身份
|
||||
(防泄漏走查通过、输入与盲评模型输入全等)、每位评委独立会话、派发失败
|
||||
失败关闭、盲评输入组装哈希绑定经官方校验器回环。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||||
SCRIPT_DIR = PROJECT_ROOT / ".agent" / "skills" / "score-content-quality" / "scripts"
|
||||
for path in (SCRIPT_DIR,):
|
||||
if str(path) not in sys.path:
|
||||
sys.path.insert(0, str(path))
|
||||
|
||||
import dispatch_judge_bridge as bridge # noqa: E402
|
||||
import judge_via_dispatch as cli # noqa: E402
|
||||
from dispatch_judge_bridge import ( # noqa: E402
|
||||
DispatchJudgeError,
|
||||
DispatchJudgeRunner,
|
||||
judge_tool_allowlist,
|
||||
)
|
||||
from run_writer_blind_judge import ( # noqa: E402
|
||||
FORBIDDEN_KEYS,
|
||||
BlindJudgeContractError,
|
||||
_walk_for_leakage,
|
||||
validate_blind_judge_input,
|
||||
)
|
||||
|
||||
MODEL_INPUT = {
|
||||
"candidates": [{"blindCandidateId": "candidate-1", "candidateBody": "正文甲"}],
|
||||
"rubric": {"dimensions": []},
|
||||
}
|
||||
|
||||
|
||||
class AllowlistTest(unittest.TestCase):
|
||||
|
||||
def test_experiment_has_no_tools(self):
|
||||
self.assertEqual(judge_tool_allowlist("experiment"), ())
|
||||
|
||||
def test_production_preflight_read_tools(self):
|
||||
self.assertEqual(judge_tool_allowlist("production_preflight"),
|
||||
("read_chapter_text", "search_entities"))
|
||||
|
||||
def test_unknown_mode_fails_closed(self):
|
||||
with self.assertRaises(DispatchJudgeError):
|
||||
judge_tool_allowlist("anything-else")
|
||||
|
||||
|
||||
class RunnerTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp = pathlib.Path(self._tmp.name)
|
||||
self.captured = []
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _runner(self, mode="production_preflight"):
|
||||
return DispatchJudgeRunner(
|
||||
run_id="run-judge-test", work_id=12, repo_root=self.tmp,
|
||||
provider="catproxy-anthropic", model="claude-opus-5", mode=mode,
|
||||
spec_dir=self.tmp,
|
||||
)
|
||||
|
||||
def _fake_dispatch(self, output, status="completed", code=0):
|
||||
def fake(spec_file, **kwargs):
|
||||
self.captured.append({"spec": json.loads(pathlib.Path(spec_file).read_text(encoding="utf-8")),
|
||||
"kwargs": kwargs})
|
||||
run_dir = self.tmp / kwargs["run_id"]
|
||||
run_dir.mkdir(parents=True)
|
||||
if status == "completed":
|
||||
(run_dir / "output.json").write_text(json.dumps(output, ensure_ascii=False), encoding="utf-8")
|
||||
return {"status": status, "runDir": str(run_dir), "errorCode": None if code == 0 else "TIMEOUT"}, code
|
||||
return fake
|
||||
|
||||
def test_run_returns_structured_output_and_receipt_hash(self):
|
||||
runner = self._runner()
|
||||
draft = {"schemaVersion": "blind-judge-draft-v3", "candidateScores": []}
|
||||
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch(draft)):
|
||||
out = runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={"type": "object"})
|
||||
self.assertEqual(out["structuredOutput"], draft)
|
||||
self.assertTrue(out["modelReceiptSha256"].startswith("sha256:"))
|
||||
self.assertEqual(runner.dispatch_run_ids, ["run-judge-test-judge-v1"])
|
||||
|
||||
def test_task_spec_carries_no_identity(self):
|
||||
runner = self._runner()
|
||||
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})):
|
||||
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={"type": "object"})
|
||||
spec = self.captured[0]["spec"]
|
||||
self.assertEqual(spec["role"], "judge")
|
||||
# 输入与盲评模型输入全等:桥不附加身份字段
|
||||
self.assertEqual(spec["input"], MODEL_INPUT)
|
||||
# 防泄漏走查:任务包整体不含禁看键
|
||||
_walk_for_leakage(spec)
|
||||
for key in FORBIDDEN_KEYS:
|
||||
self.assertNotIn(key, spec)
|
||||
|
||||
def test_each_judge_gets_fresh_session(self):
|
||||
runner = self._runner()
|
||||
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})):
|
||||
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
||||
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
||||
sessions = [c["kwargs"]["session_id"] for c in self.captured]
|
||||
self.assertEqual(sessions, ["judge-run-judge-test-judge-v1", "judge-run-judge-test-judge-v2"])
|
||||
self.assertEqual(len(set(sessions)), 2)
|
||||
|
||||
def test_mode_controls_tool_allowlist_in_spec(self):
|
||||
runner = self._runner(mode="experiment")
|
||||
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({"x": 1})):
|
||||
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
||||
self.assertEqual(self.captured[0]["spec"]["toolAllowlist"], [])
|
||||
self.assertFalse(self.captured[0]["kwargs"]["enable_read_tools"])
|
||||
|
||||
def test_dispatch_failure_fails_closed(self):
|
||||
runner = self._runner()
|
||||
with mock.patch.object(bridge, "run_dispatch", self._fake_dispatch({}, status="failed", code=1)):
|
||||
with self.assertRaises(BlindJudgeContractError) as ctx:
|
||||
runner.run(adapter_role="blind_judge", model_input=MODEL_INPUT, output_schema={})
|
||||
self.assertEqual(ctx.exception.code, "BLIND_JUDGE_RUNTIME_FAILED")
|
||||
|
||||
def test_wrong_adapter_role_rejected(self):
|
||||
runner = self._runner()
|
||||
with self.assertRaises(BlindJudgeContractError):
|
||||
runner.run(adapter_role="writer", model_input=MODEL_INPUT, output_schema={})
|
||||
|
||||
|
||||
class BlindInputBuilderTest(unittest.TestCase):
|
||||
|
||||
def test_assembled_input_passes_official_validator(self):
|
||||
bodies = {123: "甲候选正文:茧撕开舱门。", 164: "乙候选正文:林深盯着深渊。"}
|
||||
outline = {
|
||||
"sourceRef": {"sourceId": "outline:volume_1", "sourceVersion": "v1"},
|
||||
"hardConstraints": ["硬约束甲", "硬约束乙"],
|
||||
"adjustableBeats": ["拍一"],
|
||||
"declaredNewFacts": [
|
||||
{"factId": "fact-ch3-1", "text": "新事实", "sourceRef": {"sourceId": "outline:volume_1", "sourceVersion": "v1"}},
|
||||
],
|
||||
}
|
||||
|
||||
class _Conn:
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc):
|
||||
return False
|
||||
|
||||
def execute(self, sql, params=None):
|
||||
class _Rows:
|
||||
def fetchall(self):
|
||||
return [(cid, bodies[cid]) for cid in params[0]]
|
||||
return _Rows()
|
||||
|
||||
with mock.patch.object(cli, "connect", lambda readonly=True: _Conn()), \
|
||||
mock.patch.object(cli, "_latest_writer_context", lambda work_id, chapter: {"fineOutline": outline}):
|
||||
built = cli.build_blind_input_for_candidates(
|
||||
work_id=12, chapter=3, candidate_ids=[123, 164],
|
||||
run_id="run-judge-test", sample_id="sample-1", scenario="battle",
|
||||
authorization_snapshot_id="auth-work12-production-v1",
|
||||
)
|
||||
blind_input = built["blindInput"]
|
||||
# 官方校验器回环:防泄漏、哈希绑定、候选哈希全部通过
|
||||
validate_blind_judge_input(blind_input)
|
||||
# 盲 ID 与库内候选脱钩但映射完整
|
||||
self.assertEqual(sorted(built["blindMapping"].values()), [123, 164])
|
||||
self.assertEqual(sorted(blind_input["candidateOrder"]), ["candidate-1", "candidate-2"])
|
||||
# 目标章断言来自细纲硬约束
|
||||
self.assertEqual(len(blind_input["oracleTruthPack"]["targetAssertions"]), 2)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Loading…
x
Reference in New Issue
Block a user