外部 runner 各阶段失败写入明确终态,模型阶段保持 ok=false;detector 类别改为闭集并把目标章新角色缺卡归 judge/eval。同步双评编排状态、真实授权边界和离线测试口径。
141 lines
5.2 KiB
Python
141 lines
5.2 KiB
Python
#!/usr/bin/env python3
|
|
"""细纲回放 rubric 的确定性校验器。
|
|
|
|
它不替代 judge 打分,只检查评分报告是否使用正确维度、分数范围和证据字段。
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any, Mapping
|
|
|
|
|
|
RUBRIC_PROFILE = "fine_outline_replay"
|
|
DIMENSIONS = (
|
|
"structure_completeness",
|
|
"direction_causality",
|
|
"order_pacing",
|
|
"entity_state",
|
|
"foreshadowing_action",
|
|
"handoff_hook",
|
|
)
|
|
PROSE_DIMENSIONS = frozenset(
|
|
{"style_fit", "readability", "文风一致性", "文笔", "pacing_tension", "information_density"}
|
|
)
|
|
|
|
|
|
def validate_scores(scores: Mapping[str, Any]) -> list[str]:
|
|
"""返回报告问题;每个维度必须有 1-5 分和非空证据。"""
|
|
|
|
errors: list[str] = []
|
|
if not isinstance(scores, Mapping):
|
|
return ["scores 必须是对象"]
|
|
missing = [dimension for dimension in DIMENSIONS if dimension not in scores]
|
|
if missing:
|
|
errors.append(f"缺少 rubric 维度: {','.join(missing)}")
|
|
unexpected = sorted(set(scores) - set(DIMENSIONS))
|
|
if unexpected:
|
|
errors.append(f"存在未登记 rubric 维度: {','.join(unexpected)}")
|
|
prose = sorted(set(unexpected) & PROSE_DIMENSIONS)
|
|
if prose:
|
|
errors.append(f"细纲 rubric 禁止正文质量维度: {','.join(prose)}")
|
|
for dimension in DIMENSIONS:
|
|
value = scores.get(dimension)
|
|
if not isinstance(value, Mapping):
|
|
errors.append(f"维度必须包含 score/evidence 对象: {dimension}")
|
|
continue
|
|
score = value.get("score")
|
|
if isinstance(score, bool) or not isinstance(score, (int, float)) or not 1 <= score <= 5:
|
|
errors.append(f"分数必须在 1-5: {dimension}")
|
|
evidence = value.get("evidence")
|
|
if not isinstance(evidence, str) or not evidence.strip():
|
|
errors.append(f"分数缺少证据: {dimension}")
|
|
return errors
|
|
|
|
|
|
def stability_warning(
|
|
first: Mapping[str, float],
|
|
second: Mapping[str, float],
|
|
threshold: float = 0.5,
|
|
) -> dict[str, Any]:
|
|
"""比较两次评审,返回差异和是否需要人工复核。"""
|
|
|
|
missing = [
|
|
dimension
|
|
for dimension in DIMENSIONS
|
|
if dimension not in first or dimension not in second
|
|
]
|
|
gaps = {
|
|
dimension: abs(float(first[dimension]) - float(second[dimension]))
|
|
for dimension in DIMENSIONS
|
|
if dimension in first and dimension in second
|
|
}
|
|
return {
|
|
"stable": not missing and all(gap <= threshold for gap in gaps.values()),
|
|
"gaps": gaps,
|
|
"missingDimensions": missing,
|
|
}
|
|
|
|
|
|
def validate_report(
|
|
report: Mapping[str, Any],
|
|
*,
|
|
expected_judge_id: str | None = None,
|
|
expected_candidate_ids: tuple[str, ...] | None = None,
|
|
) -> list[str]:
|
|
"""校验单候选或批量评委报告,批量模式必须覆盖精确候选集合。"""
|
|
|
|
errors: list[str] = []
|
|
if not isinstance(report, Mapping):
|
|
return ["judge 报告必须是对象"]
|
|
if report.get("profile") != RUBRIC_PROFILE:
|
|
errors.append(f"profile 必须是 {RUBRIC_PROFILE}")
|
|
if expected_judge_id is None and expected_candidate_ids is None:
|
|
unexpected = sorted(set(report) - {"profile", "scores"})
|
|
if unexpected:
|
|
errors.append(f"judge 报告包含未登记字段: {','.join(unexpected)}")
|
|
errors.extend(validate_scores(report.get("scores", {})))
|
|
return errors
|
|
|
|
unexpected = sorted(set(report) - {"profile", "judgeId", "evaluations"})
|
|
if unexpected:
|
|
errors.append(f"judge 批报告包含未登记字段: {','.join(unexpected)}")
|
|
|
|
judge_id = report.get("judgeId")
|
|
if not isinstance(judge_id, str) or not judge_id.strip():
|
|
errors.append("judgeId 必须是非空字符串")
|
|
elif expected_judge_id is not None and judge_id != expected_judge_id:
|
|
errors.append(f"judgeId 不一致: expected={expected_judge_id}")
|
|
|
|
evaluations = report.get("evaluations")
|
|
if not isinstance(evaluations, list):
|
|
errors.append("evaluations 必须是数组")
|
|
return errors
|
|
candidate_ids = [
|
|
item.get("candidateId")
|
|
for item in evaluations
|
|
if isinstance(item, Mapping)
|
|
]
|
|
valid_candidate_ids = len(candidate_ids) == len(evaluations) and all(
|
|
isinstance(candidate_id, str) and bool(candidate_id)
|
|
for candidate_id in candidate_ids
|
|
)
|
|
if not valid_candidate_ids:
|
|
errors.append("每个 evaluation 必须包含非空 candidateId")
|
|
else:
|
|
if len(candidate_ids) != len(set(candidate_ids)):
|
|
errors.append("evaluations 包含重复 candidateId")
|
|
if expected_candidate_ids is not None and set(candidate_ids) != set(expected_candidate_ids):
|
|
errors.append("evaluations 未精确覆盖盲化候选集合")
|
|
for index, evaluation in enumerate(evaluations):
|
|
if not isinstance(evaluation, Mapping):
|
|
errors.append(f"evaluations[{index}] 必须是对象")
|
|
continue
|
|
errors.extend(
|
|
f"evaluations[{index}]: {error}"
|
|
for error in validate_scores(evaluation.get("scores", {}))
|
|
)
|
|
summary = evaluation.get("summary")
|
|
if not isinstance(summary, str) or not summary.strip():
|
|
errors.append(f"evaluations[{index}].summary 必须是非空摘要")
|
|
return errors
|