将角色与 Skill 从 .claude 迁入 .agent,移除 Claude CLI 运行时并接入固定 Opus 角色 profile、完整 schema、预算 deadline、raw 与回执证据链。 同步拆分 Skill 职责、复利 lesson、Gate 回放、Dashboard 人审入口、数据库登记和机械门禁;候选设计正文不包含在本提交中。
936 lines
40 KiB
Python
936 lines
40 KiB
Python
#!/usr/bin/env python3
|
||
"""正文回放五维量表与确定性稳定性仲裁。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from statistics import median
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
|
||
RUBRIC_PROFILE = "writer_replay"
|
||
RUBRIC_POLICY_VERSION = "writer-replay-rubric-v2"
|
||
COMMON_DIMENSIONS = (
|
||
"setting_entity_fidelity",
|
||
"fine_outline_fidelity",
|
||
"style_consistency",
|
||
"prose_readability",
|
||
)
|
||
# WHY: 保留既有 ID,避免破坏 blind-judge v3 的输出 schema 与历史 Gate 汇总;从 v2
|
||
# 策略起,它不再表示所有章节共用的抽象“紧张度”,而是按当前场景策略解释的执行分。
|
||
SCENARIO_DIMENSION = "narrative_tension"
|
||
DIMENSIONS = (*COMMON_DIMENSIONS[:3], SCENARIO_DIMENSION, COMMON_DIMENSIONS[3])
|
||
SCENARIO_TYPES = (
|
||
"battle",
|
||
"character_dialogue",
|
||
"turning_point",
|
||
"information_reveal",
|
||
"returning_character",
|
||
)
|
||
|
||
COMMON_DIMENSION_POLICIES: dict[str, dict[str, Any]] = {
|
||
"setting_entity_fidelity": {
|
||
"question": "设定、实体、关系、能力边界和冻结时点状态是否保真",
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "核心与局部事实均保真,无实质冲突或错置"},
|
||
{"scoreBand": "7-8.5", "meaning": "核心事实保真,仅有不影响理解的局部含混或轻微偏差"},
|
||
{"scoreBand": "5-6.5", "meaning": "主关系可辨,但存在一处实质歧义、属性错置或边界不清"},
|
||
{"scoreBand": "3-4.5", "meaning": "存在多处实质冲突,人物、实体或能力状态难以自洽"},
|
||
{"scoreBand": "0-2.5", "meaning": "核心身份、阵营、状态或世界规则被反转或破坏"},
|
||
],
|
||
},
|
||
"fine_outline_fidelity": {
|
||
"question": "细纲硬事件、结果方向、伏笔动作、出场实体和章末钩子是否忠实",
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "全部硬骨架语义成立,顺序、伏笔状态和章末钩子均准确"},
|
||
{"scoreBand": "7-8.5", "meaning": "硬约束全部成立,仅有不改变结果的节拍或展开偏差"},
|
||
{"scoreBand": "5-6.5", "meaning": "主事件成立,但遗漏、提前兑现或弱化一个关键节拍或钩子"},
|
||
{"scoreBand": "3-4.5", "meaning": "多个硬事件、结果方向或伏笔状态发生实质漂移"},
|
||
{"scoreBand": "0-2.5", "meaning": "章节核心目标缺失、反转或与细纲方向相反"},
|
||
],
|
||
},
|
||
"style_consistency": {
|
||
"question": "是否延续本作品的人物声音、动作习惯、句式和叙事质感",
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "人物声音和叙事节奏稳定,整体可自然接入相邻章节"},
|
||
{"scoreBand": "7-8.5", "meaning": "整体同风格,仅有少量局部措辞、比喻或节奏漂移"},
|
||
{"scoreBand": "5-6.5", "meaning": "作品仍可辨认,但文体、人物声音或叙事密度明显变化"},
|
||
{"scoreBand": "3-4.5", "meaning": "大部分段落与既有作品声音不一致或人物趋同"},
|
||
{"scoreBand": "0-2.5", "meaning": "文体、人物声音或类型感与作品基线完全不相容"},
|
||
],
|
||
},
|
||
"prose_readability": {
|
||
"question": "语言、指代、空间、动作和转场是否准确、自然、清晰",
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "表达清楚流畅,指代、动作、空间和转场均无阅读阻力"},
|
||
{"scoreBand": "7-8.5", "meaning": "整体顺畅,仅有少量局部重复、跳接或措辞生硬"},
|
||
{"scoreBand": "5-6.5", "meaning": "可以读懂,但反复出现指代、空间、节奏或信息组织问题"},
|
||
{"scoreBand": "3-4.5", "meaning": "频繁需要回读,动作主体、场景关系或因果难以确认"},
|
||
{"scoreBand": "0-2.5", "meaning": "主要段落无法连贯理解或存在大面积语言结构故障"},
|
||
],
|
||
},
|
||
}
|
||
|
||
SCENARIO_POLICIES: dict[str, dict[str, Any]] = {
|
||
"battle": {
|
||
"displayName": "战斗执行",
|
||
"question": "战斗的空间、攻防因果、策略升级、力量边界和结果代价是否成立",
|
||
"criteria": ["空间与动作可追踪", "攻防与策略有因果", "力量边界不漂移", "升级与结果产生后果"],
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "动作与空间清楚,攻防有策略因果,升级和代价共同推动局势"},
|
||
{"scoreBand": "7-8.5", "meaning": "主战斗链成立,仅有局部空间、节拍或策略交代不足"},
|
||
{"scoreBand": "5-6.5", "meaning": "胜负可读,但动作拼接、力量边界或升级逻辑有一处明显薄弱"},
|
||
{"scoreBand": "3-4.5", "meaning": "多处攻防、空间或能力因果断裂,胜负主要靠宣告"},
|
||
{"scoreBand": "0-2.5", "meaning": "战斗目标未形成,或核心规则与结果无法成立"},
|
||
],
|
||
},
|
||
"character_dialogue": {
|
||
"displayName": "人物对话执行",
|
||
"question": "人物声音、对话目标、潜台词、关系推进和信息交换是否成立",
|
||
"criteria": ["声音可区分", "双方各有目标", "潜台词与反应自然", "关系或局面发生变化"],
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "遮住名字仍能辨人,对话有目标与潜台词,并推动关系或局面"},
|
||
{"scoreBand": "7-8.5", "meaning": "声音和目标基本清楚,仅有少量直白解释或反应不足"},
|
||
{"scoreBand": "5-6.5", "meaning": "信息传达完成,但人物声音趋同、潜台词弱或关系推进有限"},
|
||
{"scoreBand": "3-4.5", "meaning": "对白主要承担说明,人物目标和互动因果多处缺失"},
|
||
{"scoreBand": "0-2.5", "meaning": "关键对话没有成立,人物声音或行为与既有角色严重冲突"},
|
||
],
|
||
},
|
||
"turning_point": {
|
||
"displayName": "转折执行",
|
||
"question": "转折是否有前因、触发、不可逆变化、人物反应和后续驱动力",
|
||
"criteria": ["前置条件充分", "触发因果明确", "状态变化不可逆", "反应与章末驱动力成立"],
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "转折由充分前因触发,变化不可逆,人物反应和后续钩子完整"},
|
||
{"scoreBand": "7-8.5", "meaning": "核心转折可信,仅有铺垫、反应或后果的一处局部不足"},
|
||
{"scoreBand": "5-6.5", "meaning": "转折发生,但前因不足、后果偏弱或章末过早兑现关键悬念"},
|
||
{"scoreBand": "3-4.5", "meaning": "转折依赖突发宣告,人物反应或状态变化难以自洽"},
|
||
{"scoreBand": "0-2.5", "meaning": "核心转折缺失、被反转或没有产生应有状态变化"},
|
||
],
|
||
},
|
||
"information_reveal": {
|
||
"displayName": "信息揭示执行",
|
||
"question": "信息来源、知情边界、揭示时机、清晰度和揭示后的影响是否成立",
|
||
"criteria": ["来源与证据可追踪", "知情范围正确", "时机不过早不过晚", "揭示改变理解或行动"],
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "来源、知情边界和时机准确,揭示清楚且有效改变理解或行动"},
|
||
{"scoreBand": "7-8.5", "meaning": "揭示目标成立,仅有局部解释偏多、节奏偏直或影响不足"},
|
||
{"scoreBand": "5-6.5", "meaning": "信息可以理解,但来源、时机、知情边界或戏剧影响有一处明显问题"},
|
||
{"scoreBand": "3-4.5", "meaning": "信息缺少证据链、提前泄露、顺序混乱或大量依赖说明"},
|
||
{"scoreBand": "0-2.5", "meaning": "核心信息未揭示、揭示错误或破坏作品的事实与知情边界"},
|
||
],
|
||
},
|
||
"returning_character": {
|
||
"displayName": "老角色回归执行",
|
||
"question": "角色状态与声音是否连续,回归动机、他人反应和剧情作用是否成立",
|
||
"criteria": ["状态与声音连续", "回归动机可信", "识别与关系反应自然", "回归推动当前剧情"],
|
||
"anchors": [
|
||
{"scoreBand": "9-10", "meaning": "角色状态、声音和关系连续,回归自然并实质推动当前剧情"},
|
||
{"scoreBand": "7-8.5", "meaning": "回归可信有效,仅有局部反应、铺垫或剧情作用不足"},
|
||
{"scoreBand": "5-6.5", "meaning": "角色能接回故事,但状态解释生硬、关系反应弱或作用偏工具化"},
|
||
{"scoreBand": "3-4.5", "meaning": "角色声音或状态多处断裂,回归主要依赖说明或巧合"},
|
||
{"scoreBand": "0-2.5", "meaning": "角色连续性被破坏,回归动机或身份关系无法成立"},
|
||
],
|
||
},
|
||
}
|
||
|
||
|
||
def rubric_for_scenario(scenario: str) -> dict[str, Any]:
|
||
"""返回完整且可序列化的通用指标 + 当前场景评分策略。"""
|
||
|
||
if scenario not in SCENARIO_POLICIES:
|
||
raise ValueError(f"未登记场景类型: {scenario}")
|
||
return {
|
||
"policyVersion": RUBRIC_POLICY_VERSION,
|
||
"scenarioType": scenario,
|
||
"dimensions": list(DIMENSIONS),
|
||
"scoreRange": [0, 10],
|
||
"scoreStep": 0.5,
|
||
"commonDimensions": [
|
||
{"dimensionId": dimension, **COMMON_DIMENSION_POLICIES[dimension]}
|
||
for dimension in COMMON_DIMENSIONS
|
||
],
|
||
"scenarioDimension": {
|
||
"dimensionId": SCENARIO_DIMENSION,
|
||
"compatibilityNote": "此 ID 为兼容保留,必须按当前 scenarioType 的场景策略评分,不按抽象通用紧张度评分",
|
||
**SCENARIO_POLICIES[scenario],
|
||
},
|
||
}
|
||
EVIDENCE_SOURCE_TYPES = frozenset(
|
||
{"fine_outline", "historical_prose", "judge_inference"}
|
||
)
|
||
STRUCTURED_EVIDENCE_SOURCE_TYPES = frozenset(
|
||
{
|
||
"candidate",
|
||
"fine_outline",
|
||
"historical_prose",
|
||
"judge_inference",
|
||
"oracle_assertion",
|
||
"preregistered_constraint",
|
||
}
|
||
)
|
||
REPORT_FIELDS = frozenset(
|
||
{"profile", "reviewerId", "sampleId", "blindCandidateId", "candidateOrder", "scores"}
|
||
)
|
||
STRUCTURED_REPORT_FIELDS = frozenset(
|
||
{
|
||
"schemaVersion",
|
||
"runId",
|
||
"sampleId",
|
||
"scenario",
|
||
"rubricPolicyVersion",
|
||
"reviewerInvocationId",
|
||
"blindInputSha256",
|
||
"oracleTruthPackSha256",
|
||
"candidateOrder",
|
||
"candidateScores",
|
||
"dimensionPreferences",
|
||
"oracleAssertionVerdicts",
|
||
"hardConstraintVerdicts",
|
||
"modelReceiptSha256",
|
||
"status",
|
||
"reportSha256",
|
||
}
|
||
)
|
||
STRUCTURED_REPORT_VERSION = "blind-judge-report-v4"
|
||
STRUCTURED_VERDICTS = frozenset({"pass", "fail", "unknown"})
|
||
|
||
|
||
def _is_half_step(value: Any) -> bool:
|
||
"""只接受 0 到 10 的数字和 0.5 步长,布尔值不算数字。"""
|
||
|
||
return (
|
||
not isinstance(value, bool)
|
||
and isinstance(value, (int, float))
|
||
and 0 <= float(value) <= 10
|
||
and float(value) * 2 == int(float(value) * 2)
|
||
)
|
||
|
||
|
||
def validate_scores(scores: Any) -> list[str]:
|
||
"""校验五维分数、步长和逐项证据归因。"""
|
||
|
||
if not isinstance(scores, Mapping):
|
||
return ["scores 必须是对象"]
|
||
errors: list[str] = []
|
||
missing = [dimension for dimension in DIMENSIONS if dimension not in scores]
|
||
unexpected = sorted(set(scores) - set(DIMENSIONS))
|
||
if missing:
|
||
errors.append(f"缺少正文 rubric 维度: {','.join(missing)}")
|
||
if unexpected:
|
||
errors.append(f"存在未登记正文 rubric 维度: {','.join(unexpected)}")
|
||
for dimension in DIMENSIONS:
|
||
item = scores.get(dimension)
|
||
if not isinstance(item, Mapping):
|
||
errors.append(f"维度必须包含 score/evidence 对象: {dimension}")
|
||
continue
|
||
if set(item) != {"score", "evidence"}:
|
||
errors.append(f"维度字段必须精确为 score/evidence: {dimension}")
|
||
if not _is_half_step(item.get("score")):
|
||
errors.append(f"分数必须在 0-10 且使用 0.5 步长: {dimension}")
|
||
evidence = item.get("evidence")
|
||
if not isinstance(evidence, list) or not evidence:
|
||
errors.append(f"分数缺少证据: {dimension}")
|
||
continue
|
||
for index, raw in enumerate(evidence):
|
||
if not isinstance(raw, Mapping):
|
||
errors.append(f"证据必须是对象: {dimension}[{index}]")
|
||
continue
|
||
if set(raw) != {"sourceType", "sourceRef", "excerpt"}:
|
||
errors.append(f"证据字段必须精确为 sourceType/sourceRef/excerpt: {dimension}[{index}]")
|
||
continue
|
||
if raw.get("sourceType") not in EVIDENCE_SOURCE_TYPES:
|
||
errors.append(f"证据来源类型未登记: {dimension}[{index}]")
|
||
for field in ("sourceRef", "excerpt"):
|
||
if not isinstance(raw.get(field), str) or not raw[field].strip():
|
||
errors.append(f"证据 {field} 不能为空: {dimension}[{index}]")
|
||
return errors
|
||
|
||
|
||
def _validate_structured_evidence(value: Any, path: str) -> list[str]:
|
||
"""校验 v3 报告证据只引用候选、共同 oracle 或预注册约束。"""
|
||
|
||
if not isinstance(value, list) or not value:
|
||
return [f"{path} 必须是非空证据数组"]
|
||
errors: list[str] = []
|
||
for index, raw in enumerate(value):
|
||
if not isinstance(raw, Mapping):
|
||
errors.append(f"{path}[{index}] 必须是对象")
|
||
continue
|
||
if set(raw) != {"sourceType", "sourceRef"}:
|
||
errors.append(f"{path}[{index}] 存在额外字段或缺字段")
|
||
continue
|
||
if raw.get("sourceType") not in STRUCTURED_EVIDENCE_SOURCE_TYPES:
|
||
errors.append(f"{path}[{index}].sourceType 未登记")
|
||
if not isinstance(raw.get("sourceRef"), str) or not raw["sourceRef"].strip():
|
||
errors.append(f"{path}[{index}].sourceRef 不能为空")
|
||
return errors
|
||
|
||
|
||
def _validate_structured_scores(value: Any, path: str) -> list[str]:
|
||
"""校验 v3 候选五维评分,沿用 0 到 10 的 0.5 步长。"""
|
||
|
||
if not isinstance(value, Mapping):
|
||
return [f"{path} 必须是对象"]
|
||
errors: list[str] = []
|
||
if set(value) != set(DIMENSIONS):
|
||
errors.append(f"{path} 必须精确包含五个正文维度")
|
||
for dimension in DIMENSIONS:
|
||
item = value.get(dimension)
|
||
if not isinstance(item, Mapping):
|
||
errors.append(f"{path}.{dimension} 必须是对象")
|
||
continue
|
||
if set(item) != {"score", "evidence"}:
|
||
errors.append(f"{path}.{dimension} 存在额外字段或缺字段")
|
||
if not _is_half_step(item.get("score")):
|
||
errors.append(f"{path}.{dimension}.score 必须使用 0.5 步长")
|
||
evidence = item.get("evidence")
|
||
if not isinstance(evidence, list) or not evidence:
|
||
errors.append(f"{path}.{dimension}.evidence 必须是非空数组")
|
||
continue
|
||
for evidence_index, raw in enumerate(evidence):
|
||
evidence_path = f"{path}.{dimension}.evidence[{evidence_index}]"
|
||
if not isinstance(raw, Mapping):
|
||
errors.append(f"{evidence_path} 必须是对象")
|
||
continue
|
||
if set(raw) != {"sourceType", "sourceRef", "excerpt"}:
|
||
errors.append(f"{evidence_path} 存在额外字段或缺字段")
|
||
continue
|
||
if raw.get("sourceType") not in STRUCTURED_EVIDENCE_SOURCE_TYPES:
|
||
errors.append(f"{evidence_path}.sourceType 未登记")
|
||
for field in ("sourceRef", "excerpt"):
|
||
if not isinstance(raw.get(field), str) or not raw[field].strip():
|
||
errors.append(f"{evidence_path}.{field} 不能为空")
|
||
return errors
|
||
|
||
|
||
def _validate_structured_verdicts(value: Any, path: str, id_field: str) -> list[str]:
|
||
"""校验 v3 verdict 的候选绑定形状、枚举、证据和复合键唯一性。"""
|
||
|
||
if not isinstance(value, list):
|
||
return [f"{path} 必须是数组"]
|
||
errors: list[str] = []
|
||
keys: list[tuple[Any, Any]] = []
|
||
expected_fields = {id_field, "candidateSha256", "verdict", "reason", "evidenceRefs"}
|
||
for index, raw in enumerate(value):
|
||
item_path = f"{path}[{index}]"
|
||
if not isinstance(raw, Mapping):
|
||
errors.append(f"{item_path} 必须是对象")
|
||
continue
|
||
if set(raw) != expected_fields:
|
||
errors.append(f"{item_path} 存在额外字段或缺字段")
|
||
stable_id = raw.get(id_field)
|
||
candidate_hash = raw.get("candidateSha256")
|
||
if not isinstance(stable_id, str) or not stable_id.strip():
|
||
errors.append(f"{item_path}.{id_field} 不能为空")
|
||
if not isinstance(candidate_hash, str) or not candidate_hash.startswith("sha256:"):
|
||
errors.append(f"{item_path}.candidateSha256 非法")
|
||
keys.append((stable_id, candidate_hash))
|
||
verdict = raw.get("verdict")
|
||
if verdict not in STRUCTURED_VERDICTS:
|
||
errors.append(f"{item_path}.verdict 枚举非法")
|
||
elif verdict == "unknown":
|
||
errors.append(f"{item_path}.verdict=unknown 失败关闭")
|
||
if not isinstance(raw.get("reason"), str) or not raw["reason"].strip():
|
||
errors.append(f"{item_path}.reason 不能为空")
|
||
errors.extend(
|
||
_validate_structured_evidence(raw.get("evidenceRefs"), f"{item_path}.evidenceRefs")
|
||
)
|
||
if len(keys) != len(set(keys)):
|
||
errors.append(f"{path} 的稳定 ID 与候选哈希复合键不得重复")
|
||
return errors
|
||
|
||
|
||
def validate_structured_report(report: Any) -> list[str]:
|
||
"""校验 blind-judge-report-v4 的 closed-fields 与逐项结构。"""
|
||
|
||
if not isinstance(report, Mapping):
|
||
return ["v4 评委报告必须是对象"]
|
||
errors: list[str] = []
|
||
if set(report) != STRUCTURED_REPORT_FIELDS:
|
||
errors.append("v4 评委报告存在额外字段或缺字段")
|
||
if report.get("schemaVersion") != STRUCTURED_REPORT_VERSION:
|
||
errors.append(f"schemaVersion 必须是 {STRUCTURED_REPORT_VERSION}")
|
||
for field in ("runId", "sampleId", "reviewerInvocationId"):
|
||
if not isinstance(report.get(field), str) or not report[field].strip():
|
||
errors.append(f"{field} 必须是非空字符串")
|
||
if report.get("scenario") not in SCENARIO_TYPES:
|
||
errors.append("scenario 必须是已登记场景类型")
|
||
if report.get("rubricPolicyVersion") != RUBRIC_POLICY_VERSION:
|
||
errors.append("rubricPolicyVersion 必须是当前正文评分策略")
|
||
for field in (
|
||
"blindInputSha256",
|
||
"oracleTruthPackSha256",
|
||
"modelReceiptSha256",
|
||
"reportSha256",
|
||
):
|
||
value = report.get(field)
|
||
if (
|
||
not isinstance(value, str)
|
||
or len(value) != 71
|
||
or not value.startswith("sha256:")
|
||
):
|
||
errors.append(f"{field} 必须是带前缀的 SHA-256")
|
||
order = report.get("candidateOrder")
|
||
if (
|
||
not isinstance(order, list)
|
||
or len(order) < 2
|
||
or any(not isinstance(item, str) or not item for item in order)
|
||
or len(set(order)) != len(order)
|
||
):
|
||
errors.append("candidateOrder 必须是至少两个无重复盲候选 ID")
|
||
order = []
|
||
candidate_scores = report.get("candidateScores")
|
||
score_ids: list[str] = []
|
||
score_hashes: list[str] = []
|
||
if not isinstance(candidate_scores, list):
|
||
errors.append("candidateScores 必须是数组")
|
||
else:
|
||
for index, raw in enumerate(candidate_scores):
|
||
path = f"candidateScores[{index}]"
|
||
if not isinstance(raw, Mapping):
|
||
errors.append(f"{path} 必须是对象")
|
||
continue
|
||
if set(raw) != {"blindCandidateId", "candidateSha256", "scores"}:
|
||
errors.append(f"{path} 存在额外字段或缺字段")
|
||
blind_id = raw.get("blindCandidateId")
|
||
candidate_hash = raw.get("candidateSha256")
|
||
if not isinstance(blind_id, str) or not blind_id:
|
||
errors.append(f"{path}.blindCandidateId 不能为空")
|
||
else:
|
||
score_ids.append(blind_id)
|
||
if not isinstance(candidate_hash, str) or not candidate_hash.startswith("sha256:"):
|
||
errors.append(f"{path}.candidateSha256 非法")
|
||
else:
|
||
score_hashes.append(candidate_hash)
|
||
errors.extend(_validate_structured_scores(raw.get("scores"), f"{path}.scores"))
|
||
if order and score_ids != order:
|
||
errors.append("candidateScores 顺序必须与 candidateOrder 完全一致")
|
||
if len(score_ids) != len(set(score_ids)) or len(score_hashes) != len(set(score_hashes)):
|
||
errors.append("candidateScores 的候选 ID 和哈希必须一一对应且无重复")
|
||
preferences = report.get("dimensionPreferences")
|
||
if not isinstance(preferences, list) or [item.get("dimension") for item in preferences if isinstance(item, Mapping)] != list(DIMENSIONS):
|
||
errors.append("dimensionPreferences 必须按五维完整返回")
|
||
else:
|
||
for index, item in enumerate(preferences):
|
||
if not isinstance(item, Mapping) or set(item) != {"dimension", "orderedCandidateIds", "reason"}:
|
||
errors.append(f"dimensionPreferences[{index}] 字段非法")
|
||
continue
|
||
ranked = item.get("orderedCandidateIds")
|
||
if not isinstance(ranked, list) or len(ranked) != len(order) or set(ranked) != set(order):
|
||
errors.append(f"dimensionPreferences[{index}] 候选排序非法")
|
||
if not isinstance(item.get("reason"), str) or not item["reason"].strip():
|
||
errors.append(f"dimensionPreferences[{index}].reason 不能为空")
|
||
errors.extend(
|
||
_validate_structured_verdicts(
|
||
report.get("oracleAssertionVerdicts"),
|
||
"oracleAssertionVerdicts",
|
||
"assertionId",
|
||
)
|
||
)
|
||
errors.extend(
|
||
_validate_structured_verdicts(
|
||
report.get("hardConstraintVerdicts"),
|
||
"hardConstraintVerdicts",
|
||
"constraintId",
|
||
)
|
||
)
|
||
if report.get("status") != "completed":
|
||
errors.append("status 必须是 completed")
|
||
return errors
|
||
|
||
|
||
def _structured_candidate_map(report: Mapping[str, Any]) -> dict[str, str]:
|
||
"""从已校验 v3 报告提取盲候选 ID 到候选哈希映射。"""
|
||
|
||
return {
|
||
item["blindCandidateId"]: item["candidateSha256"]
|
||
for item in report["candidateScores"]
|
||
}
|
||
|
||
|
||
def _structured_score_map(report: Mapping[str, Any]) -> dict[tuple[str, str], float]:
|
||
"""把 v3 五维评分展开成稳定复合键映射。"""
|
||
|
||
return {
|
||
(item["candidateSha256"], dimension): float(item["scores"][dimension]["score"])
|
||
for item in report["candidateScores"]
|
||
for dimension in DIMENSIONS
|
||
}
|
||
|
||
|
||
def _structured_verdict_map(
|
||
report: Mapping[str, Any], field: str, id_field: str
|
||
) -> dict[tuple[str, str], str]:
|
||
"""把逐断言或逐约束 verdict 展开成稳定复合键映射。"""
|
||
|
||
return {
|
||
(item["candidateSha256"], item[id_field]): item["verdict"]
|
||
for item in report[field]
|
||
}
|
||
|
||
|
||
def validate_structured_pair(first: Any, second: Any) -> list[str]:
|
||
"""校验双评独立、共同 truth pack、候选集合和严格反序。"""
|
||
|
||
errors = [*validate_structured_report(first), *validate_structured_report(second)]
|
||
if errors or not isinstance(first, Mapping) or not isinstance(second, Mapping):
|
||
return errors
|
||
if first["reviewerInvocationId"] == second["reviewerInvocationId"]:
|
||
errors.append("双评必须使用独立 reviewerInvocationId")
|
||
for field in (
|
||
"runId",
|
||
"sampleId",
|
||
"scenario",
|
||
"rubricPolicyVersion",
|
||
"oracleTruthPackSha256",
|
||
):
|
||
if first[field] != second[field]:
|
||
errors.append(f"双评的 {field} 必须一致")
|
||
if _structured_candidate_map(first) != _structured_candidate_map(second):
|
||
errors.append("双评必须使用相同候选 ID/hash 集")
|
||
if second["candidateOrder"] != list(reversed(first["candidateOrder"])):
|
||
errors.append("第二评必须对候选顺序严格反序")
|
||
first_assertions = set(
|
||
_structured_verdict_map(first, "oracleAssertionVerdicts", "assertionId")
|
||
)
|
||
second_assertions = set(
|
||
_structured_verdict_map(second, "oracleAssertionVerdicts", "assertionId")
|
||
)
|
||
if first_assertions != second_assertions:
|
||
errors.append("双评 assertion verdict 复合键集合必须一致")
|
||
first_constraints = set(
|
||
_structured_verdict_map(first, "hardConstraintVerdicts", "constraintId")
|
||
)
|
||
second_constraints = set(
|
||
_structured_verdict_map(second, "hardConstraintVerdicts", "constraintId")
|
||
)
|
||
if first_constraints != second_constraints:
|
||
errors.append("双评 constraint verdict 复合键集合必须一致")
|
||
return errors
|
||
|
||
|
||
def _stable_verdict(values: Sequence[str]) -> str | None:
|
||
"""仅当至少两个 verdict 完全一致时返回多数稳定值。"""
|
||
|
||
for value in ("pass", "fail"):
|
||
if sum(item == value for item in values) >= 2:
|
||
return value
|
||
return None
|
||
|
||
|
||
def _final_structured_scores(
|
||
report: Mapping[str, Any], values: Mapping[tuple[str, str], float]
|
||
) -> list[dict[str, Any]]:
|
||
"""按第一评候选顺序组装不带自由文本的稳定最终分数。"""
|
||
|
||
return [
|
||
{
|
||
"blindCandidateId": blind_id,
|
||
"candidateSha256": candidate_hash,
|
||
"scores": {
|
||
dimension: values[(candidate_hash, dimension)] for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
for blind_id in report["candidateOrder"]
|
||
for candidate_hash in [_structured_candidate_map(report)[blind_id]]
|
||
]
|
||
|
||
|
||
def _final_structured_verdicts(
|
||
keys: Sequence[tuple[str, str]],
|
||
values: Mapping[tuple[str, str], str],
|
||
id_field: str,
|
||
) -> list[dict[str, Any]]:
|
||
"""按候选哈希和稳定 ID 排序组装稳定 verdict。"""
|
||
|
||
return [
|
||
{"candidateSha256": candidate_hash, id_field: stable_id, "verdict": values[key]}
|
||
for key in sorted(keys)
|
||
for candidate_hash, stable_id in [key]
|
||
]
|
||
|
||
|
||
def adjudicate_structured_reviews(
|
||
first: Mapping[str, Any],
|
||
second: Mapping[str, Any],
|
||
third: Mapping[str, Any] | None = None,
|
||
*,
|
||
threshold: float = 0.5,
|
||
) -> dict[str, Any]:
|
||
"""对 v3 全报告执行双评和最多一次第三评的确定性稳定仲裁。"""
|
||
|
||
pair_errors = validate_structured_pair(first, second)
|
||
if pair_errors:
|
||
return {"status": "failed_judge_invalid", "errors": pair_errors, "reviewCount": 0}
|
||
first_scores = _structured_score_map(first)
|
||
second_scores = _structured_score_map(second)
|
||
first_assertions = _structured_verdict_map(
|
||
first, "oracleAssertionVerdicts", "assertionId"
|
||
)
|
||
second_assertions = _structured_verdict_map(
|
||
second, "oracleAssertionVerdicts", "assertionId"
|
||
)
|
||
first_constraints = _structured_verdict_map(
|
||
first, "hardConstraintVerdicts", "constraintId"
|
||
)
|
||
second_constraints = _structured_verdict_map(
|
||
second, "hardConstraintVerdicts", "constraintId"
|
||
)
|
||
unstable_scores = [
|
||
{"candidateSha256": key[0], "dimension": key[1]}
|
||
for key in sorted(first_scores)
|
||
if abs(first_scores[key] - second_scores[key]) > threshold
|
||
]
|
||
unstable_assertions = [
|
||
{"candidateSha256": key[0], "assertionId": key[1]}
|
||
for key in sorted(first_assertions)
|
||
if first_assertions[key] != second_assertions[key]
|
||
]
|
||
unstable_constraints = [
|
||
{"candidateSha256": key[0], "constraintId": key[1]}
|
||
for key in sorted(first_constraints)
|
||
if first_constraints[key] != second_constraints[key]
|
||
]
|
||
if not unstable_scores and not unstable_assertions and not unstable_constraints:
|
||
averaged_scores = {
|
||
key: (first_scores[key] + second_scores[key]) / 2 for key in first_scores
|
||
}
|
||
return {
|
||
"status": "stable_report",
|
||
"reviewCount": 2,
|
||
"candidateScores": _final_structured_scores(first, averaged_scores),
|
||
"oracleAssertionVerdicts": _final_structured_verdicts(
|
||
list(first_assertions), first_assertions, "assertionId"
|
||
),
|
||
"hardConstraintVerdicts": _final_structured_verdicts(
|
||
list(first_constraints), first_constraints, "constraintId"
|
||
),
|
||
"unstableScores": [],
|
||
"unstableOracleVerdicts": [],
|
||
"unstableConstraintVerdicts": [],
|
||
}
|
||
if third is None:
|
||
return {
|
||
"status": "needs_third_reviewer",
|
||
"reviewCount": 2,
|
||
"unstableScores": unstable_scores,
|
||
"unstableOracleVerdicts": unstable_assertions,
|
||
"unstableConstraintVerdicts": unstable_constraints,
|
||
}
|
||
third_errors = validate_structured_report(third)
|
||
if third_errors:
|
||
return {"status": "failed_judge_invalid", "errors": third_errors, "reviewCount": 3}
|
||
if third["reviewerInvocationId"] in {
|
||
first["reviewerInvocationId"],
|
||
second["reviewerInvocationId"],
|
||
}:
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": ["第三评必须使用独立 reviewerInvocationId"],
|
||
"reviewCount": 3,
|
||
}
|
||
for field in (
|
||
"runId",
|
||
"sampleId",
|
||
"scenario",
|
||
"rubricPolicyVersion",
|
||
"oracleTruthPackSha256",
|
||
):
|
||
if third[field] != first[field]:
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": [f"第三评的 {field} 必须与双评一致"],
|
||
"reviewCount": 3,
|
||
}
|
||
if _structured_candidate_map(third) != _structured_candidate_map(first):
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": ["第三评必须使用相同候选 ID/hash 集"],
|
||
"reviewCount": 3,
|
||
}
|
||
if third["candidateOrder"] not in (
|
||
first["candidateOrder"],
|
||
second["candidateOrder"],
|
||
):
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": ["第三评顺序必须等于第一评或严格反序"],
|
||
"reviewCount": 3,
|
||
}
|
||
third_scores = _structured_score_map(third)
|
||
third_assertions = _structured_verdict_map(
|
||
third, "oracleAssertionVerdicts", "assertionId"
|
||
)
|
||
third_constraints = _structured_verdict_map(
|
||
third, "hardConstraintVerdicts", "constraintId"
|
||
)
|
||
if set(third_scores) != set(first_scores):
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": ["第三评 score 复合键集合不完整"],
|
||
"reviewCount": 3,
|
||
}
|
||
if set(third_assertions) != set(first_assertions):
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": ["第三评 assertion verdict 复合键集合不完整"],
|
||
"reviewCount": 3,
|
||
}
|
||
if set(third_constraints) != set(first_constraints):
|
||
return {
|
||
"status": "failed_judge_invalid",
|
||
"errors": ["第三评 constraint verdict 复合键集合不完整"],
|
||
"reviewCount": 3,
|
||
}
|
||
unresolved_scores = [
|
||
{"candidateSha256": key[0], "dimension": key[1]}
|
||
for key in sorted(first_scores)
|
||
if not _stable_pair_exists(
|
||
[first_scores[key], second_scores[key], third_scores[key]], threshold
|
||
)
|
||
]
|
||
final_assertions = {
|
||
key: _stable_verdict(
|
||
[first_assertions[key], second_assertions[key], third_assertions[key]]
|
||
)
|
||
for key in first_assertions
|
||
}
|
||
final_constraints = {
|
||
key: _stable_verdict(
|
||
[first_constraints[key], second_constraints[key], third_constraints[key]]
|
||
)
|
||
for key in first_constraints
|
||
}
|
||
unresolved_assertions = [
|
||
{"candidateSha256": key[0], "assertionId": key[1]}
|
||
for key, verdict in sorted(final_assertions.items())
|
||
if verdict is None
|
||
]
|
||
unresolved_constraints = [
|
||
{"candidateSha256": key[0], "constraintId": key[1]}
|
||
for key, verdict in sorted(final_constraints.items())
|
||
if verdict is None
|
||
]
|
||
if unresolved_scores or unresolved_assertions or unresolved_constraints:
|
||
return {
|
||
"status": "failed_judge_unstable",
|
||
"reviewCount": 3,
|
||
"unstableScores": unresolved_scores,
|
||
"unstableOracleVerdicts": unresolved_assertions,
|
||
"unstableConstraintVerdicts": unresolved_constraints,
|
||
}
|
||
median_scores = {
|
||
key: float(median([first_scores[key], second_scores[key], third_scores[key]]))
|
||
for key in first_scores
|
||
}
|
||
return {
|
||
"status": "adjudicated_report",
|
||
"reviewCount": 3,
|
||
"candidateScores": _final_structured_scores(first, median_scores),
|
||
"oracleAssertionVerdicts": _final_structured_verdicts(
|
||
list(first_assertions),
|
||
{key: str(value) for key, value in final_assertions.items()},
|
||
"assertionId",
|
||
),
|
||
"hardConstraintVerdicts": _final_structured_verdicts(
|
||
list(first_constraints),
|
||
{key: str(value) for key, value in final_constraints.items()},
|
||
"constraintId",
|
||
),
|
||
"unstableScores": unstable_scores,
|
||
"unstableOracleVerdicts": unstable_assertions,
|
||
"unstableConstraintVerdicts": unstable_constraints,
|
||
}
|
||
|
||
|
||
def validate_report(report: Any) -> list[str]:
|
||
"""校验单个盲评报告,真实臂名或额外字段一律失败关闭。"""
|
||
|
||
if not isinstance(report, Mapping):
|
||
return ["评委报告必须是对象"]
|
||
errors: list[str] = []
|
||
if set(report) != REPORT_FIELDS:
|
||
errors.append("评委报告字段非法,去盲前不得包含真实臂名或额外信息")
|
||
if report.get("profile") != RUBRIC_PROFILE:
|
||
errors.append(f"profile 必须是 {RUBRIC_PROFILE}")
|
||
for field in ("reviewerId", "sampleId", "blindCandidateId"):
|
||
if not isinstance(report.get(field), str) or not report[field].strip():
|
||
errors.append(f"{field} 必须是非空字符串")
|
||
order = report.get("candidateOrder")
|
||
if (
|
||
not isinstance(order, list)
|
||
or len(order) < 2
|
||
or any(not isinstance(item, str) or not item for item in order)
|
||
or len(set(order)) != len(order)
|
||
):
|
||
errors.append("candidateOrder 必须是无重复盲化候选 ID 数组")
|
||
elif report.get("blindCandidateId") not in order:
|
||
errors.append("blindCandidateId 不在本轮 candidateOrder 中")
|
||
errors.extend(validate_scores(report.get("scores")))
|
||
return errors
|
||
|
||
|
||
def validate_blind_pair(first: Any, second: Any) -> list[str]:
|
||
"""确认双评独立、同样本、同候选集合且第二评委严格反序。"""
|
||
|
||
errors = [*validate_report(first), *validate_report(second)]
|
||
if errors or not isinstance(first, Mapping) or not isinstance(second, Mapping):
|
||
return errors
|
||
if first["reviewerId"] == second["reviewerId"]:
|
||
errors.append("双评委必须使用独立 reviewerId")
|
||
if first["sampleId"] != second["sampleId"]:
|
||
errors.append("双评委必须评审同一样本")
|
||
if first["blindCandidateId"] != second["blindCandidateId"]:
|
||
errors.append("双评委必须评审同一盲化候选")
|
||
if set(first["candidateOrder"]) != set(second["candidateOrder"]):
|
||
errors.append("双评委的 candidateOrder 必须包含同一盲化候选集合")
|
||
if second["candidateOrder"] != list(reversed(first["candidateOrder"])):
|
||
errors.append("第二评委必须对相同盲化候选严格反序")
|
||
return errors
|
||
|
||
|
||
def deblind_reports(
|
||
reports: Sequence[Mapping[str, Any]], mapping: Mapping[str, str]
|
||
) -> list[dict[str, Any]]:
|
||
"""评分结束后按预注册映射去盲,绝不依赖展示位置推断真实臂。"""
|
||
|
||
if set(mapping.values()) != {"A", "B", "C"} or len(mapping) != 3:
|
||
raise ValueError("去盲映射必须将三个盲 ID 一一映射到 A/B/C")
|
||
result: list[dict[str, Any]] = []
|
||
for index, report in enumerate(reports):
|
||
errors = validate_report(report)
|
||
if errors:
|
||
raise ValueError(f"reports[{index}] 非法: {'; '.join(errors)}")
|
||
blind_id = str(report["blindCandidateId"])
|
||
if blind_id not in mapping:
|
||
raise ValueError(f"reports[{index}] 的 blindCandidateId 未预注册")
|
||
result.append({**dict(report), "arm": mapping[blind_id]})
|
||
return result
|
||
|
||
|
||
def _numeric_scores(report: Mapping[str, Any]) -> dict[str, float]:
|
||
"""从已校验报告提取确定性浮点分数。"""
|
||
|
||
return {
|
||
dimension: float(report["scores"][dimension]["score"])
|
||
for dimension in DIMENSIONS
|
||
}
|
||
|
||
|
||
def _stable_pair_exists(values: Sequence[float], threshold: float) -> bool:
|
||
"""判断三个评分中是否至少存在一对落在稳定阈值内。"""
|
||
|
||
return any(
|
||
abs(values[left] - values[right]) <= threshold
|
||
for left in range(len(values))
|
||
for right in range(left + 1, len(values))
|
||
)
|
||
|
||
|
||
def adjudicate_reviews(
|
||
first: Mapping[str, Any],
|
||
second: Mapping[str, Any],
|
||
third: Mapping[str, Any] | None = None,
|
||
*,
|
||
threshold: float = 0.5,
|
||
) -> dict[str, Any]:
|
||
"""按双评差异触发最多一次第三评委,并输出稳定终态。"""
|
||
|
||
pair_errors = validate_blind_pair(first, second)
|
||
if pair_errors:
|
||
return {"status": "invalid_report", "errors": pair_errors}
|
||
first_scores = _numeric_scores(first)
|
||
second_scores = _numeric_scores(second)
|
||
unstable = [
|
||
dimension
|
||
for dimension in DIMENSIONS
|
||
if abs(first_scores[dimension] - second_scores[dimension]) > threshold
|
||
]
|
||
if not unstable:
|
||
return {
|
||
"status": "stable_report",
|
||
"scores": {
|
||
dimension: (first_scores[dimension] + second_scores[dimension]) / 2
|
||
for dimension in DIMENSIONS
|
||
},
|
||
"unstableDimensions": [],
|
||
"reviewCount": 2,
|
||
}
|
||
if third is None:
|
||
return {
|
||
"status": "needs_third_reviewer",
|
||
"unstableDimensions": unstable,
|
||
"reviewCount": 2,
|
||
}
|
||
third_errors = validate_report(third)
|
||
if third_errors:
|
||
return {"status": "invalid_report", "errors": third_errors}
|
||
if third["sampleId"] != first["sampleId"]:
|
||
return {"status": "invalid_report", "errors": ["第三评委必须评审同一样本"]}
|
||
if third["blindCandidateId"] != first["blindCandidateId"]:
|
||
return {"status": "invalid_report", "errors": ["第三评委必须评审同一盲化候选"]}
|
||
allowed_third_orders = (
|
||
first["candidateOrder"],
|
||
list(reversed(first["candidateOrder"])),
|
||
)
|
||
if third["candidateOrder"] not in allowed_third_orders:
|
||
return {
|
||
"status": "invalid_report",
|
||
"errors": ["第三评委必须使用与双评相同的盲化候选集合及第一评或反序顺序"],
|
||
}
|
||
if third["reviewerId"] in {first["reviewerId"], second["reviewerId"]}:
|
||
return {"status": "invalid_report", "errors": ["第三评委必须使用独立 reviewerId"]}
|
||
|
||
third_scores = _numeric_scores(third)
|
||
unresolved = [
|
||
dimension
|
||
for dimension in unstable
|
||
if not _stable_pair_exists(
|
||
[first_scores[dimension], second_scores[dimension], third_scores[dimension]],
|
||
threshold,
|
||
)
|
||
]
|
||
if unresolved:
|
||
return {
|
||
"status": "invalid_unstable",
|
||
"unstableDimensions": unresolved,
|
||
"reviewCount": 3,
|
||
}
|
||
return {
|
||
"status": "adjudicated_report",
|
||
"scores": {
|
||
dimension: float(
|
||
median(
|
||
[first_scores[dimension], second_scores[dimension], third_scores[dimension]]
|
||
)
|
||
)
|
||
for dimension in DIMENSIONS
|
||
},
|
||
"unstableDimensions": unstable,
|
||
"reviewCount": 3,
|
||
}
|
||
|
||
|
||
__all__ = [
|
||
"RUBRIC_PROFILE",
|
||
"RUBRIC_POLICY_VERSION",
|
||
"COMMON_DIMENSIONS",
|
||
"DIMENSIONS",
|
||
"SCENARIO_DIMENSION",
|
||
"SCENARIO_TYPES",
|
||
"COMMON_DIMENSION_POLICIES",
|
||
"SCENARIO_POLICIES",
|
||
"rubric_for_scenario",
|
||
"EVIDENCE_SOURCE_TYPES",
|
||
"STRUCTURED_EVIDENCE_SOURCE_TYPES",
|
||
"STRUCTURED_REPORT_VERSION",
|
||
"validate_scores",
|
||
"validate_report",
|
||
"validate_blind_pair",
|
||
"validate_structured_report",
|
||
"validate_structured_pair",
|
||
"deblind_reports",
|
||
"adjudicate_reviews",
|
||
"adjudicate_structured_reviews",
|
||
]
|