zizi 091b66a9bb 重构: 收敛 Agent/Skill 运行时与创作质量闭环
将角色与 Skill 从 .claude 迁入 .agent,移除 Claude CLI 运行时并接入固定 Opus 角色 profile、完整 schema、预算 deadline、raw 与回执证据链。

同步拆分 Skill 职责、复利 lesson、Gate 回放、Dashboard 人审入口、数据库登记和机械门禁;候选设计正文不包含在本提交中。
2026-08-22 02:12:32 +08:00

936 lines
40 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""正文回放五维量表与确定性稳定性仲裁。"""
from __future__ import annotations
from statistics import median
from typing import Any, Mapping, Sequence
RUBRIC_PROFILE = "writer_replay"
RUBRIC_POLICY_VERSION = "writer-replay-rubric-v2"
COMMON_DIMENSIONS = (
"setting_entity_fidelity",
"fine_outline_fidelity",
"style_consistency",
"prose_readability",
)
# WHY: 保留既有 ID,避免破坏 blind-judge v3 的输出 schema 与历史 Gate 汇总;从 v2
# 策略起,它不再表示所有章节共用的抽象“紧张度”,而是按当前场景策略解释的执行分。
SCENARIO_DIMENSION = "narrative_tension"
DIMENSIONS = (*COMMON_DIMENSIONS[:3], SCENARIO_DIMENSION, COMMON_DIMENSIONS[3])
SCENARIO_TYPES = (
"battle",
"character_dialogue",
"turning_point",
"information_reveal",
"returning_character",
)
COMMON_DIMENSION_POLICIES: dict[str, dict[str, Any]] = {
"setting_entity_fidelity": {
"question": "设定、实体、关系、能力边界和冻结时点状态是否保真",
"anchors": [
{"scoreBand": "9-10", "meaning": "核心与局部事实均保真,无实质冲突或错置"},
{"scoreBand": "7-8.5", "meaning": "核心事实保真,仅有不影响理解的局部含混或轻微偏差"},
{"scoreBand": "5-6.5", "meaning": "主关系可辨,但存在一处实质歧义、属性错置或边界不清"},
{"scoreBand": "3-4.5", "meaning": "存在多处实质冲突,人物、实体或能力状态难以自洽"},
{"scoreBand": "0-2.5", "meaning": "核心身份、阵营、状态或世界规则被反转或破坏"},
],
},
"fine_outline_fidelity": {
"question": "细纲硬事件、结果方向、伏笔动作、出场实体和章末钩子是否忠实",
"anchors": [
{"scoreBand": "9-10", "meaning": "全部硬骨架语义成立,顺序、伏笔状态和章末钩子均准确"},
{"scoreBand": "7-8.5", "meaning": "硬约束全部成立,仅有不改变结果的节拍或展开偏差"},
{"scoreBand": "5-6.5", "meaning": "主事件成立,但遗漏、提前兑现或弱化一个关键节拍或钩子"},
{"scoreBand": "3-4.5", "meaning": "多个硬事件、结果方向或伏笔状态发生实质漂移"},
{"scoreBand": "0-2.5", "meaning": "章节核心目标缺失、反转或与细纲方向相反"},
],
},
"style_consistency": {
"question": "是否延续本作品的人物声音、动作习惯、句式和叙事质感",
"anchors": [
{"scoreBand": "9-10", "meaning": "人物声音和叙事节奏稳定,整体可自然接入相邻章节"},
{"scoreBand": "7-8.5", "meaning": "整体同风格,仅有少量局部措辞、比喻或节奏漂移"},
{"scoreBand": "5-6.5", "meaning": "作品仍可辨认,但文体、人物声音或叙事密度明显变化"},
{"scoreBand": "3-4.5", "meaning": "大部分段落与既有作品声音不一致或人物趋同"},
{"scoreBand": "0-2.5", "meaning": "文体、人物声音或类型感与作品基线完全不相容"},
],
},
"prose_readability": {
"question": "语言、指代、空间、动作和转场是否准确、自然、清晰",
"anchors": [
{"scoreBand": "9-10", "meaning": "表达清楚流畅,指代、动作、空间和转场均无阅读阻力"},
{"scoreBand": "7-8.5", "meaning": "整体顺畅,仅有少量局部重复、跳接或措辞生硬"},
{"scoreBand": "5-6.5", "meaning": "可以读懂,但反复出现指代、空间、节奏或信息组织问题"},
{"scoreBand": "3-4.5", "meaning": "频繁需要回读,动作主体、场景关系或因果难以确认"},
{"scoreBand": "0-2.5", "meaning": "主要段落无法连贯理解或存在大面积语言结构故障"},
],
},
}
SCENARIO_POLICIES: dict[str, dict[str, Any]] = {
"battle": {
"displayName": "战斗执行",
"question": "战斗的空间、攻防因果、策略升级、力量边界和结果代价是否成立",
"criteria": ["空间与动作可追踪", "攻防与策略有因果", "力量边界不漂移", "升级与结果产生后果"],
"anchors": [
{"scoreBand": "9-10", "meaning": "动作与空间清楚,攻防有策略因果,升级和代价共同推动局势"},
{"scoreBand": "7-8.5", "meaning": "主战斗链成立,仅有局部空间、节拍或策略交代不足"},
{"scoreBand": "5-6.5", "meaning": "胜负可读,但动作拼接、力量边界或升级逻辑有一处明显薄弱"},
{"scoreBand": "3-4.5", "meaning": "多处攻防、空间或能力因果断裂,胜负主要靠宣告"},
{"scoreBand": "0-2.5", "meaning": "战斗目标未形成,或核心规则与结果无法成立"},
],
},
"character_dialogue": {
"displayName": "人物对话执行",
"question": "人物声音、对话目标、潜台词、关系推进和信息交换是否成立",
"criteria": ["声音可区分", "双方各有目标", "潜台词与反应自然", "关系或局面发生变化"],
"anchors": [
{"scoreBand": "9-10", "meaning": "遮住名字仍能辨人,对话有目标与潜台词,并推动关系或局面"},
{"scoreBand": "7-8.5", "meaning": "声音和目标基本清楚,仅有少量直白解释或反应不足"},
{"scoreBand": "5-6.5", "meaning": "信息传达完成,但人物声音趋同、潜台词弱或关系推进有限"},
{"scoreBand": "3-4.5", "meaning": "对白主要承担说明,人物目标和互动因果多处缺失"},
{"scoreBand": "0-2.5", "meaning": "关键对话没有成立,人物声音或行为与既有角色严重冲突"},
],
},
"turning_point": {
"displayName": "转折执行",
"question": "转折是否有前因、触发、不可逆变化、人物反应和后续驱动力",
"criteria": ["前置条件充分", "触发因果明确", "状态变化不可逆", "反应与章末驱动力成立"],
"anchors": [
{"scoreBand": "9-10", "meaning": "转折由充分前因触发,变化不可逆,人物反应和后续钩子完整"},
{"scoreBand": "7-8.5", "meaning": "核心转折可信,仅有铺垫、反应或后果的一处局部不足"},
{"scoreBand": "5-6.5", "meaning": "转折发生,但前因不足、后果偏弱或章末过早兑现关键悬念"},
{"scoreBand": "3-4.5", "meaning": "转折依赖突发宣告,人物反应或状态变化难以自洽"},
{"scoreBand": "0-2.5", "meaning": "核心转折缺失、被反转或没有产生应有状态变化"},
],
},
"information_reveal": {
"displayName": "信息揭示执行",
"question": "信息来源、知情边界、揭示时机、清晰度和揭示后的影响是否成立",
"criteria": ["来源与证据可追踪", "知情范围正确", "时机不过早不过晚", "揭示改变理解或行动"],
"anchors": [
{"scoreBand": "9-10", "meaning": "来源、知情边界和时机准确,揭示清楚且有效改变理解或行动"},
{"scoreBand": "7-8.5", "meaning": "揭示目标成立,仅有局部解释偏多、节奏偏直或影响不足"},
{"scoreBand": "5-6.5", "meaning": "信息可以理解,但来源、时机、知情边界或戏剧影响有一处明显问题"},
{"scoreBand": "3-4.5", "meaning": "信息缺少证据链、提前泄露、顺序混乱或大量依赖说明"},
{"scoreBand": "0-2.5", "meaning": "核心信息未揭示、揭示错误或破坏作品的事实与知情边界"},
],
},
"returning_character": {
"displayName": "老角色回归执行",
"question": "角色状态与声音是否连续,回归动机、他人反应和剧情作用是否成立",
"criteria": ["状态与声音连续", "回归动机可信", "识别与关系反应自然", "回归推动当前剧情"],
"anchors": [
{"scoreBand": "9-10", "meaning": "角色状态、声音和关系连续,回归自然并实质推动当前剧情"},
{"scoreBand": "7-8.5", "meaning": "回归可信有效,仅有局部反应、铺垫或剧情作用不足"},
{"scoreBand": "5-6.5", "meaning": "角色能接回故事,但状态解释生硬、关系反应弱或作用偏工具化"},
{"scoreBand": "3-4.5", "meaning": "角色声音或状态多处断裂,回归主要依赖说明或巧合"},
{"scoreBand": "0-2.5", "meaning": "角色连续性被破坏,回归动机或身份关系无法成立"},
],
},
}
def rubric_for_scenario(scenario: str) -> dict[str, Any]:
"""返回完整且可序列化的通用指标 + 当前场景评分策略。"""
if scenario not in SCENARIO_POLICIES:
raise ValueError(f"未登记场景类型: {scenario}")
return {
"policyVersion": RUBRIC_POLICY_VERSION,
"scenarioType": scenario,
"dimensions": list(DIMENSIONS),
"scoreRange": [0, 10],
"scoreStep": 0.5,
"commonDimensions": [
{"dimensionId": dimension, **COMMON_DIMENSION_POLICIES[dimension]}
for dimension in COMMON_DIMENSIONS
],
"scenarioDimension": {
"dimensionId": SCENARIO_DIMENSION,
"compatibilityNote": "此 ID 为兼容保留,必须按当前 scenarioType 的场景策略评分,不按抽象通用紧张度评分",
**SCENARIO_POLICIES[scenario],
},
}
EVIDENCE_SOURCE_TYPES = frozenset(
{"fine_outline", "historical_prose", "judge_inference"}
)
STRUCTURED_EVIDENCE_SOURCE_TYPES = frozenset(
{
"candidate",
"fine_outline",
"historical_prose",
"judge_inference",
"oracle_assertion",
"preregistered_constraint",
}
)
REPORT_FIELDS = frozenset(
{"profile", "reviewerId", "sampleId", "blindCandidateId", "candidateOrder", "scores"}
)
STRUCTURED_REPORT_FIELDS = frozenset(
{
"schemaVersion",
"runId",
"sampleId",
"scenario",
"rubricPolicyVersion",
"reviewerInvocationId",
"blindInputSha256",
"oracleTruthPackSha256",
"candidateOrder",
"candidateScores",
"dimensionPreferences",
"oracleAssertionVerdicts",
"hardConstraintVerdicts",
"modelReceiptSha256",
"status",
"reportSha256",
}
)
STRUCTURED_REPORT_VERSION = "blind-judge-report-v4"
STRUCTURED_VERDICTS = frozenset({"pass", "fail", "unknown"})
def _is_half_step(value: Any) -> bool:
"""只接受 0 到 10 的数字和 0.5 步长,布尔值不算数字。"""
return (
not isinstance(value, bool)
and isinstance(value, (int, float))
and 0 <= float(value) <= 10
and float(value) * 2 == int(float(value) * 2)
)
def validate_scores(scores: Any) -> list[str]:
"""校验五维分数、步长和逐项证据归因。"""
if not isinstance(scores, Mapping):
return ["scores 必须是对象"]
errors: list[str] = []
missing = [dimension for dimension in DIMENSIONS if dimension not in scores]
unexpected = sorted(set(scores) - set(DIMENSIONS))
if missing:
errors.append(f"缺少正文 rubric 维度: {','.join(missing)}")
if unexpected:
errors.append(f"存在未登记正文 rubric 维度: {','.join(unexpected)}")
for dimension in DIMENSIONS:
item = scores.get(dimension)
if not isinstance(item, Mapping):
errors.append(f"维度必须包含 score/evidence 对象: {dimension}")
continue
if set(item) != {"score", "evidence"}:
errors.append(f"维度字段必须精确为 score/evidence: {dimension}")
if not _is_half_step(item.get("score")):
errors.append(f"分数必须在 0-10 且使用 0.5 步长: {dimension}")
evidence = item.get("evidence")
if not isinstance(evidence, list) or not evidence:
errors.append(f"分数缺少证据: {dimension}")
continue
for index, raw in enumerate(evidence):
if not isinstance(raw, Mapping):
errors.append(f"证据必须是对象: {dimension}[{index}]")
continue
if set(raw) != {"sourceType", "sourceRef", "excerpt"}:
errors.append(f"证据字段必须精确为 sourceType/sourceRef/excerpt: {dimension}[{index}]")
continue
if raw.get("sourceType") not in EVIDENCE_SOURCE_TYPES:
errors.append(f"证据来源类型未登记: {dimension}[{index}]")
for field in ("sourceRef", "excerpt"):
if not isinstance(raw.get(field), str) or not raw[field].strip():
errors.append(f"证据 {field} 不能为空: {dimension}[{index}]")
return errors
def _validate_structured_evidence(value: Any, path: str) -> list[str]:
"""校验 v3 报告证据只引用候选、共同 oracle 或预注册约束。"""
if not isinstance(value, list) or not value:
return [f"{path} 必须是非空证据数组"]
errors: list[str] = []
for index, raw in enumerate(value):
if not isinstance(raw, Mapping):
errors.append(f"{path}[{index}] 必须是对象")
continue
if set(raw) != {"sourceType", "sourceRef"}:
errors.append(f"{path}[{index}] 存在额外字段或缺字段")
continue
if raw.get("sourceType") not in STRUCTURED_EVIDENCE_SOURCE_TYPES:
errors.append(f"{path}[{index}].sourceType 未登记")
if not isinstance(raw.get("sourceRef"), str) or not raw["sourceRef"].strip():
errors.append(f"{path}[{index}].sourceRef 不能为空")
return errors
def _validate_structured_scores(value: Any, path: str) -> list[str]:
"""校验 v3 候选五维评分,沿用 0 到 10 的 0.5 步长。"""
if not isinstance(value, Mapping):
return [f"{path} 必须是对象"]
errors: list[str] = []
if set(value) != set(DIMENSIONS):
errors.append(f"{path} 必须精确包含五个正文维度")
for dimension in DIMENSIONS:
item = value.get(dimension)
if not isinstance(item, Mapping):
errors.append(f"{path}.{dimension} 必须是对象")
continue
if set(item) != {"score", "evidence"}:
errors.append(f"{path}.{dimension} 存在额外字段或缺字段")
if not _is_half_step(item.get("score")):
errors.append(f"{path}.{dimension}.score 必须使用 0.5 步长")
evidence = item.get("evidence")
if not isinstance(evidence, list) or not evidence:
errors.append(f"{path}.{dimension}.evidence 必须是非空数组")
continue
for evidence_index, raw in enumerate(evidence):
evidence_path = f"{path}.{dimension}.evidence[{evidence_index}]"
if not isinstance(raw, Mapping):
errors.append(f"{evidence_path} 必须是对象")
continue
if set(raw) != {"sourceType", "sourceRef", "excerpt"}:
errors.append(f"{evidence_path} 存在额外字段或缺字段")
continue
if raw.get("sourceType") not in STRUCTURED_EVIDENCE_SOURCE_TYPES:
errors.append(f"{evidence_path}.sourceType 未登记")
for field in ("sourceRef", "excerpt"):
if not isinstance(raw.get(field), str) or not raw[field].strip():
errors.append(f"{evidence_path}.{field} 不能为空")
return errors
def _validate_structured_verdicts(value: Any, path: str, id_field: str) -> list[str]:
"""校验 v3 verdict 的候选绑定形状、枚举、证据和复合键唯一性。"""
if not isinstance(value, list):
return [f"{path} 必须是数组"]
errors: list[str] = []
keys: list[tuple[Any, Any]] = []
expected_fields = {id_field, "candidateSha256", "verdict", "reason", "evidenceRefs"}
for index, raw in enumerate(value):
item_path = f"{path}[{index}]"
if not isinstance(raw, Mapping):
errors.append(f"{item_path} 必须是对象")
continue
if set(raw) != expected_fields:
errors.append(f"{item_path} 存在额外字段或缺字段")
stable_id = raw.get(id_field)
candidate_hash = raw.get("candidateSha256")
if not isinstance(stable_id, str) or not stable_id.strip():
errors.append(f"{item_path}.{id_field} 不能为空")
if not isinstance(candidate_hash, str) or not candidate_hash.startswith("sha256:"):
errors.append(f"{item_path}.candidateSha256 非法")
keys.append((stable_id, candidate_hash))
verdict = raw.get("verdict")
if verdict not in STRUCTURED_VERDICTS:
errors.append(f"{item_path}.verdict 枚举非法")
elif verdict == "unknown":
errors.append(f"{item_path}.verdict=unknown 失败关闭")
if not isinstance(raw.get("reason"), str) or not raw["reason"].strip():
errors.append(f"{item_path}.reason 不能为空")
errors.extend(
_validate_structured_evidence(raw.get("evidenceRefs"), f"{item_path}.evidenceRefs")
)
if len(keys) != len(set(keys)):
errors.append(f"{path} 的稳定 ID 与候选哈希复合键不得重复")
return errors
def validate_structured_report(report: Any) -> list[str]:
"""校验 blind-judge-report-v4 的 closed-fields 与逐项结构。"""
if not isinstance(report, Mapping):
return ["v4 评委报告必须是对象"]
errors: list[str] = []
if set(report) != STRUCTURED_REPORT_FIELDS:
errors.append("v4 评委报告存在额外字段或缺字段")
if report.get("schemaVersion") != STRUCTURED_REPORT_VERSION:
errors.append(f"schemaVersion 必须是 {STRUCTURED_REPORT_VERSION}")
for field in ("runId", "sampleId", "reviewerInvocationId"):
if not isinstance(report.get(field), str) or not report[field].strip():
errors.append(f"{field} 必须是非空字符串")
if report.get("scenario") not in SCENARIO_TYPES:
errors.append("scenario 必须是已登记场景类型")
if report.get("rubricPolicyVersion") != RUBRIC_POLICY_VERSION:
errors.append("rubricPolicyVersion 必须是当前正文评分策略")
for field in (
"blindInputSha256",
"oracleTruthPackSha256",
"modelReceiptSha256",
"reportSha256",
):
value = report.get(field)
if (
not isinstance(value, str)
or len(value) != 71
or not value.startswith("sha256:")
):
errors.append(f"{field} 必须是带前缀的 SHA-256")
order = report.get("candidateOrder")
if (
not isinstance(order, list)
or len(order) < 2
or any(not isinstance(item, str) or not item for item in order)
or len(set(order)) != len(order)
):
errors.append("candidateOrder 必须是至少两个无重复盲候选 ID")
order = []
candidate_scores = report.get("candidateScores")
score_ids: list[str] = []
score_hashes: list[str] = []
if not isinstance(candidate_scores, list):
errors.append("candidateScores 必须是数组")
else:
for index, raw in enumerate(candidate_scores):
path = f"candidateScores[{index}]"
if not isinstance(raw, Mapping):
errors.append(f"{path} 必须是对象")
continue
if set(raw) != {"blindCandidateId", "candidateSha256", "scores"}:
errors.append(f"{path} 存在额外字段或缺字段")
blind_id = raw.get("blindCandidateId")
candidate_hash = raw.get("candidateSha256")
if not isinstance(blind_id, str) or not blind_id:
errors.append(f"{path}.blindCandidateId 不能为空")
else:
score_ids.append(blind_id)
if not isinstance(candidate_hash, str) or not candidate_hash.startswith("sha256:"):
errors.append(f"{path}.candidateSha256 非法")
else:
score_hashes.append(candidate_hash)
errors.extend(_validate_structured_scores(raw.get("scores"), f"{path}.scores"))
if order and score_ids != order:
errors.append("candidateScores 顺序必须与 candidateOrder 完全一致")
if len(score_ids) != len(set(score_ids)) or len(score_hashes) != len(set(score_hashes)):
errors.append("candidateScores 的候选 ID 和哈希必须一一对应且无重复")
preferences = report.get("dimensionPreferences")
if not isinstance(preferences, list) or [item.get("dimension") for item in preferences if isinstance(item, Mapping)] != list(DIMENSIONS):
errors.append("dimensionPreferences 必须按五维完整返回")
else:
for index, item in enumerate(preferences):
if not isinstance(item, Mapping) or set(item) != {"dimension", "orderedCandidateIds", "reason"}:
errors.append(f"dimensionPreferences[{index}] 字段非法")
continue
ranked = item.get("orderedCandidateIds")
if not isinstance(ranked, list) or len(ranked) != len(order) or set(ranked) != set(order):
errors.append(f"dimensionPreferences[{index}] 候选排序非法")
if not isinstance(item.get("reason"), str) or not item["reason"].strip():
errors.append(f"dimensionPreferences[{index}].reason 不能为空")
errors.extend(
_validate_structured_verdicts(
report.get("oracleAssertionVerdicts"),
"oracleAssertionVerdicts",
"assertionId",
)
)
errors.extend(
_validate_structured_verdicts(
report.get("hardConstraintVerdicts"),
"hardConstraintVerdicts",
"constraintId",
)
)
if report.get("status") != "completed":
errors.append("status 必须是 completed")
return errors
def _structured_candidate_map(report: Mapping[str, Any]) -> dict[str, str]:
"""从已校验 v3 报告提取盲候选 ID 到候选哈希映射。"""
return {
item["blindCandidateId"]: item["candidateSha256"]
for item in report["candidateScores"]
}
def _structured_score_map(report: Mapping[str, Any]) -> dict[tuple[str, str], float]:
"""把 v3 五维评分展开成稳定复合键映射。"""
return {
(item["candidateSha256"], dimension): float(item["scores"][dimension]["score"])
for item in report["candidateScores"]
for dimension in DIMENSIONS
}
def _structured_verdict_map(
report: Mapping[str, Any], field: str, id_field: str
) -> dict[tuple[str, str], str]:
"""把逐断言或逐约束 verdict 展开成稳定复合键映射。"""
return {
(item["candidateSha256"], item[id_field]): item["verdict"]
for item in report[field]
}
def validate_structured_pair(first: Any, second: Any) -> list[str]:
"""校验双评独立、共同 truth pack、候选集合和严格反序。"""
errors = [*validate_structured_report(first), *validate_structured_report(second)]
if errors or not isinstance(first, Mapping) or not isinstance(second, Mapping):
return errors
if first["reviewerInvocationId"] == second["reviewerInvocationId"]:
errors.append("双评必须使用独立 reviewerInvocationId")
for field in (
"runId",
"sampleId",
"scenario",
"rubricPolicyVersion",
"oracleTruthPackSha256",
):
if first[field] != second[field]:
errors.append(f"双评的 {field} 必须一致")
if _structured_candidate_map(first) != _structured_candidate_map(second):
errors.append("双评必须使用相同候选 ID/hash 集")
if second["candidateOrder"] != list(reversed(first["candidateOrder"])):
errors.append("第二评必须对候选顺序严格反序")
first_assertions = set(
_structured_verdict_map(first, "oracleAssertionVerdicts", "assertionId")
)
second_assertions = set(
_structured_verdict_map(second, "oracleAssertionVerdicts", "assertionId")
)
if first_assertions != second_assertions:
errors.append("双评 assertion verdict 复合键集合必须一致")
first_constraints = set(
_structured_verdict_map(first, "hardConstraintVerdicts", "constraintId")
)
second_constraints = set(
_structured_verdict_map(second, "hardConstraintVerdicts", "constraintId")
)
if first_constraints != second_constraints:
errors.append("双评 constraint verdict 复合键集合必须一致")
return errors
def _stable_verdict(values: Sequence[str]) -> str | None:
"""仅当至少两个 verdict 完全一致时返回多数稳定值。"""
for value in ("pass", "fail"):
if sum(item == value for item in values) >= 2:
return value
return None
def _final_structured_scores(
report: Mapping[str, Any], values: Mapping[tuple[str, str], float]
) -> list[dict[str, Any]]:
"""按第一评候选顺序组装不带自由文本的稳定最终分数。"""
return [
{
"blindCandidateId": blind_id,
"candidateSha256": candidate_hash,
"scores": {
dimension: values[(candidate_hash, dimension)] for dimension in DIMENSIONS
},
}
for blind_id in report["candidateOrder"]
for candidate_hash in [_structured_candidate_map(report)[blind_id]]
]
def _final_structured_verdicts(
keys: Sequence[tuple[str, str]],
values: Mapping[tuple[str, str], str],
id_field: str,
) -> list[dict[str, Any]]:
"""按候选哈希和稳定 ID 排序组装稳定 verdict。"""
return [
{"candidateSha256": candidate_hash, id_field: stable_id, "verdict": values[key]}
for key in sorted(keys)
for candidate_hash, stable_id in [key]
]
def adjudicate_structured_reviews(
first: Mapping[str, Any],
second: Mapping[str, Any],
third: Mapping[str, Any] | None = None,
*,
threshold: float = 0.5,
) -> dict[str, Any]:
"""对 v3 全报告执行双评和最多一次第三评的确定性稳定仲裁。"""
pair_errors = validate_structured_pair(first, second)
if pair_errors:
return {"status": "failed_judge_invalid", "errors": pair_errors, "reviewCount": 0}
first_scores = _structured_score_map(first)
second_scores = _structured_score_map(second)
first_assertions = _structured_verdict_map(
first, "oracleAssertionVerdicts", "assertionId"
)
second_assertions = _structured_verdict_map(
second, "oracleAssertionVerdicts", "assertionId"
)
first_constraints = _structured_verdict_map(
first, "hardConstraintVerdicts", "constraintId"
)
second_constraints = _structured_verdict_map(
second, "hardConstraintVerdicts", "constraintId"
)
unstable_scores = [
{"candidateSha256": key[0], "dimension": key[1]}
for key in sorted(first_scores)
if abs(first_scores[key] - second_scores[key]) > threshold
]
unstable_assertions = [
{"candidateSha256": key[0], "assertionId": key[1]}
for key in sorted(first_assertions)
if first_assertions[key] != second_assertions[key]
]
unstable_constraints = [
{"candidateSha256": key[0], "constraintId": key[1]}
for key in sorted(first_constraints)
if first_constraints[key] != second_constraints[key]
]
if not unstable_scores and not unstable_assertions and not unstable_constraints:
averaged_scores = {
key: (first_scores[key] + second_scores[key]) / 2 for key in first_scores
}
return {
"status": "stable_report",
"reviewCount": 2,
"candidateScores": _final_structured_scores(first, averaged_scores),
"oracleAssertionVerdicts": _final_structured_verdicts(
list(first_assertions), first_assertions, "assertionId"
),
"hardConstraintVerdicts": _final_structured_verdicts(
list(first_constraints), first_constraints, "constraintId"
),
"unstableScores": [],
"unstableOracleVerdicts": [],
"unstableConstraintVerdicts": [],
}
if third is None:
return {
"status": "needs_third_reviewer",
"reviewCount": 2,
"unstableScores": unstable_scores,
"unstableOracleVerdicts": unstable_assertions,
"unstableConstraintVerdicts": unstable_constraints,
}
third_errors = validate_structured_report(third)
if third_errors:
return {"status": "failed_judge_invalid", "errors": third_errors, "reviewCount": 3}
if third["reviewerInvocationId"] in {
first["reviewerInvocationId"],
second["reviewerInvocationId"],
}:
return {
"status": "failed_judge_invalid",
"errors": ["第三评必须使用独立 reviewerInvocationId"],
"reviewCount": 3,
}
for field in (
"runId",
"sampleId",
"scenario",
"rubricPolicyVersion",
"oracleTruthPackSha256",
):
if third[field] != first[field]:
return {
"status": "failed_judge_invalid",
"errors": [f"第三评的 {field} 必须与双评一致"],
"reviewCount": 3,
}
if _structured_candidate_map(third) != _structured_candidate_map(first):
return {
"status": "failed_judge_invalid",
"errors": ["第三评必须使用相同候选 ID/hash 集"],
"reviewCount": 3,
}
if third["candidateOrder"] not in (
first["candidateOrder"],
second["candidateOrder"],
):
return {
"status": "failed_judge_invalid",
"errors": ["第三评顺序必须等于第一评或严格反序"],
"reviewCount": 3,
}
third_scores = _structured_score_map(third)
third_assertions = _structured_verdict_map(
third, "oracleAssertionVerdicts", "assertionId"
)
third_constraints = _structured_verdict_map(
third, "hardConstraintVerdicts", "constraintId"
)
if set(third_scores) != set(first_scores):
return {
"status": "failed_judge_invalid",
"errors": ["第三评 score 复合键集合不完整"],
"reviewCount": 3,
}
if set(third_assertions) != set(first_assertions):
return {
"status": "failed_judge_invalid",
"errors": ["第三评 assertion verdict 复合键集合不完整"],
"reviewCount": 3,
}
if set(third_constraints) != set(first_constraints):
return {
"status": "failed_judge_invalid",
"errors": ["第三评 constraint verdict 复合键集合不完整"],
"reviewCount": 3,
}
unresolved_scores = [
{"candidateSha256": key[0], "dimension": key[1]}
for key in sorted(first_scores)
if not _stable_pair_exists(
[first_scores[key], second_scores[key], third_scores[key]], threshold
)
]
final_assertions = {
key: _stable_verdict(
[first_assertions[key], second_assertions[key], third_assertions[key]]
)
for key in first_assertions
}
final_constraints = {
key: _stable_verdict(
[first_constraints[key], second_constraints[key], third_constraints[key]]
)
for key in first_constraints
}
unresolved_assertions = [
{"candidateSha256": key[0], "assertionId": key[1]}
for key, verdict in sorted(final_assertions.items())
if verdict is None
]
unresolved_constraints = [
{"candidateSha256": key[0], "constraintId": key[1]}
for key, verdict in sorted(final_constraints.items())
if verdict is None
]
if unresolved_scores or unresolved_assertions or unresolved_constraints:
return {
"status": "failed_judge_unstable",
"reviewCount": 3,
"unstableScores": unresolved_scores,
"unstableOracleVerdicts": unresolved_assertions,
"unstableConstraintVerdicts": unresolved_constraints,
}
median_scores = {
key: float(median([first_scores[key], second_scores[key], third_scores[key]]))
for key in first_scores
}
return {
"status": "adjudicated_report",
"reviewCount": 3,
"candidateScores": _final_structured_scores(first, median_scores),
"oracleAssertionVerdicts": _final_structured_verdicts(
list(first_assertions),
{key: str(value) for key, value in final_assertions.items()},
"assertionId",
),
"hardConstraintVerdicts": _final_structured_verdicts(
list(first_constraints),
{key: str(value) for key, value in final_constraints.items()},
"constraintId",
),
"unstableScores": unstable_scores,
"unstableOracleVerdicts": unstable_assertions,
"unstableConstraintVerdicts": unstable_constraints,
}
def validate_report(report: Any) -> list[str]:
"""校验单个盲评报告,真实臂名或额外字段一律失败关闭。"""
if not isinstance(report, Mapping):
return ["评委报告必须是对象"]
errors: list[str] = []
if set(report) != REPORT_FIELDS:
errors.append("评委报告字段非法,去盲前不得包含真实臂名或额外信息")
if report.get("profile") != RUBRIC_PROFILE:
errors.append(f"profile 必须是 {RUBRIC_PROFILE}")
for field in ("reviewerId", "sampleId", "blindCandidateId"):
if not isinstance(report.get(field), str) or not report[field].strip():
errors.append(f"{field} 必须是非空字符串")
order = report.get("candidateOrder")
if (
not isinstance(order, list)
or len(order) < 2
or any(not isinstance(item, str) or not item for item in order)
or len(set(order)) != len(order)
):
errors.append("candidateOrder 必须是无重复盲化候选 ID 数组")
elif report.get("blindCandidateId") not in order:
errors.append("blindCandidateId 不在本轮 candidateOrder 中")
errors.extend(validate_scores(report.get("scores")))
return errors
def validate_blind_pair(first: Any, second: Any) -> list[str]:
"""确认双评独立、同样本、同候选集合且第二评委严格反序。"""
errors = [*validate_report(first), *validate_report(second)]
if errors or not isinstance(first, Mapping) or not isinstance(second, Mapping):
return errors
if first["reviewerId"] == second["reviewerId"]:
errors.append("双评委必须使用独立 reviewerId")
if first["sampleId"] != second["sampleId"]:
errors.append("双评委必须评审同一样本")
if first["blindCandidateId"] != second["blindCandidateId"]:
errors.append("双评委必须评审同一盲化候选")
if set(first["candidateOrder"]) != set(second["candidateOrder"]):
errors.append("双评委的 candidateOrder 必须包含同一盲化候选集合")
if second["candidateOrder"] != list(reversed(first["candidateOrder"])):
errors.append("第二评委必须对相同盲化候选严格反序")
return errors
def deblind_reports(
reports: Sequence[Mapping[str, Any]], mapping: Mapping[str, str]
) -> list[dict[str, Any]]:
"""评分结束后按预注册映射去盲,绝不依赖展示位置推断真实臂。"""
if set(mapping.values()) != {"A", "B", "C"} or len(mapping) != 3:
raise ValueError("去盲映射必须将三个盲 ID 一一映射到 A/B/C")
result: list[dict[str, Any]] = []
for index, report in enumerate(reports):
errors = validate_report(report)
if errors:
raise ValueError(f"reports[{index}] 非法: {'; '.join(errors)}")
blind_id = str(report["blindCandidateId"])
if blind_id not in mapping:
raise ValueError(f"reports[{index}] 的 blindCandidateId 未预注册")
result.append({**dict(report), "arm": mapping[blind_id]})
return result
def _numeric_scores(report: Mapping[str, Any]) -> dict[str, float]:
"""从已校验报告提取确定性浮点分数。"""
return {
dimension: float(report["scores"][dimension]["score"])
for dimension in DIMENSIONS
}
def _stable_pair_exists(values: Sequence[float], threshold: float) -> bool:
"""判断三个评分中是否至少存在一对落在稳定阈值内。"""
return any(
abs(values[left] - values[right]) <= threshold
for left in range(len(values))
for right in range(left + 1, len(values))
)
def adjudicate_reviews(
first: Mapping[str, Any],
second: Mapping[str, Any],
third: Mapping[str, Any] | None = None,
*,
threshold: float = 0.5,
) -> dict[str, Any]:
"""按双评差异触发最多一次第三评委,并输出稳定终态。"""
pair_errors = validate_blind_pair(first, second)
if pair_errors:
return {"status": "invalid_report", "errors": pair_errors}
first_scores = _numeric_scores(first)
second_scores = _numeric_scores(second)
unstable = [
dimension
for dimension in DIMENSIONS
if abs(first_scores[dimension] - second_scores[dimension]) > threshold
]
if not unstable:
return {
"status": "stable_report",
"scores": {
dimension: (first_scores[dimension] + second_scores[dimension]) / 2
for dimension in DIMENSIONS
},
"unstableDimensions": [],
"reviewCount": 2,
}
if third is None:
return {
"status": "needs_third_reviewer",
"unstableDimensions": unstable,
"reviewCount": 2,
}
third_errors = validate_report(third)
if third_errors:
return {"status": "invalid_report", "errors": third_errors}
if third["sampleId"] != first["sampleId"]:
return {"status": "invalid_report", "errors": ["第三评委必须评审同一样本"]}
if third["blindCandidateId"] != first["blindCandidateId"]:
return {"status": "invalid_report", "errors": ["第三评委必须评审同一盲化候选"]}
allowed_third_orders = (
first["candidateOrder"],
list(reversed(first["candidateOrder"])),
)
if third["candidateOrder"] not in allowed_third_orders:
return {
"status": "invalid_report",
"errors": ["第三评委必须使用与双评相同的盲化候选集合及第一评或反序顺序"],
}
if third["reviewerId"] in {first["reviewerId"], second["reviewerId"]}:
return {"status": "invalid_report", "errors": ["第三评委必须使用独立 reviewerId"]}
third_scores = _numeric_scores(third)
unresolved = [
dimension
for dimension in unstable
if not _stable_pair_exists(
[first_scores[dimension], second_scores[dimension], third_scores[dimension]],
threshold,
)
]
if unresolved:
return {
"status": "invalid_unstable",
"unstableDimensions": unresolved,
"reviewCount": 3,
}
return {
"status": "adjudicated_report",
"scores": {
dimension: float(
median(
[first_scores[dimension], second_scores[dimension], third_scores[dimension]]
)
)
for dimension in DIMENSIONS
},
"unstableDimensions": unstable,
"reviewCount": 3,
}
__all__ = [
"RUBRIC_PROFILE",
"RUBRIC_POLICY_VERSION",
"COMMON_DIMENSIONS",
"DIMENSIONS",
"SCENARIO_DIMENSION",
"SCENARIO_TYPES",
"COMMON_DIMENSION_POLICIES",
"SCENARIO_POLICIES",
"rubric_for_scenario",
"EVIDENCE_SOURCE_TYPES",
"STRUCTURED_EVIDENCE_SOURCE_TYPES",
"STRUCTURED_REPORT_VERSION",
"validate_scores",
"validate_report",
"validate_blind_pair",
"validate_structured_report",
"validate_structured_pair",
"deblind_reports",
"adjudicate_reviews",
"adjudicate_structured_reviews",
]