范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
718 lines
29 KiB
Python
718 lines
29 KiB
Python
#!/usr/bin/env python3
|
||
"""GateInputBuilder、Gate 裁决与 FileCas receipt 的联合测试。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import json
|
||
import pathlib
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
|
||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||
SKILLS_DIR = PROJECT_ROOT / ".claude" / "skills"
|
||
SCRIPT_DIR = SKILLS_DIR / "evaluate-frozen-replay" / "scripts"
|
||
TEST_DIR = PROJECT_ROOT / "tests" / "skills" / "evaluate-frozen-replay"
|
||
QUALITY_GATE_DIR = SKILLS_DIR / "score-content-quality" / "scripts"
|
||
EVIDENCE_DIR = SKILLS_DIR / "record-run-evidence" / "scripts"
|
||
for _import_dir in (SCRIPT_DIR, TEST_DIR, QUALITY_GATE_DIR, EVIDENCE_DIR):
|
||
if str(_import_dir) not in sys.path:
|
||
sys.path.insert(0, str(_import_dir))
|
||
|
||
from gate_input_builder import GateInputBuilder, canonical_sha256
|
||
from writer_eval_preregister import build_balanced_preregistration
|
||
from writer_gate import (
|
||
decide_gate,
|
||
issue_gate_report_and_receipt,
|
||
verify_writer_gate_receipt,
|
||
)
|
||
from writer_rubric import (
|
||
DIMENSIONS,
|
||
RUBRIC_POLICY_VERSION,
|
||
SCENARIO_DIMENSION,
|
||
adjudicate_structured_reviews,
|
||
)
|
||
|
||
|
||
SCENARIOS = (
|
||
"battle",
|
||
"character_dialogue",
|
||
"turning_point",
|
||
"information_reveal",
|
||
"returning_character",
|
||
)
|
||
|
||
|
||
def _with_hash(payload: dict, field: str = "receiptSha256") -> dict:
|
||
"""给测试来源补入与生产一致的规范自哈希。"""
|
||
|
||
return {**payload, field: canonical_sha256(payload)}
|
||
|
||
|
||
def _runtime_receipt(
|
||
role: str,
|
||
sample_id: str,
|
||
arm: str,
|
||
invocation: int,
|
||
*,
|
||
structured_output: dict | None = None,
|
||
) -> dict:
|
||
"""构造字段完整且由原始哈希绑定的成功 ExecutionReceipt。"""
|
||
|
||
return {
|
||
"adapterRole": role,
|
||
"invocationId": f"{sample_id}-{arm}-{role}-{invocation}",
|
||
"executionProfileSha256": canonical_sha256({"role": role, "profile": 1}),
|
||
"requestedModelId": "claude-opus-4-1-20250805",
|
||
"actualModelId": "claude-opus-4-1-20250805",
|
||
"modelMatch": True,
|
||
"effort": "high",
|
||
"maxBudgetUsdPerCall": "1.000000",
|
||
"totalCostUsd": "0.010000",
|
||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||
"modelUsage": {"claude-opus-4-1-20250805": {"costUSD": "0.010000"}},
|
||
"stopReason": "end_turn",
|
||
"terminalReason": "success",
|
||
"isError": False,
|
||
"apiErrorStatus": None,
|
||
"exitCode": 0,
|
||
"durationMs": 1,
|
||
"inputSha256": canonical_sha256({"sampleId": sample_id, "arm": arm, "role": role}),
|
||
"structuredOutputSha256": canonical_sha256(
|
||
structured_output
|
||
if structured_output is not None
|
||
else {"sampleId": sample_id, "arm": arm, "role": role, "invocation": invocation}
|
||
),
|
||
"jsonSchemaSha256": canonical_sha256({"role": role, "schema": 1}),
|
||
}
|
||
|
||
|
||
def _reviewer_raw_report(
|
||
*,
|
||
run_id: str,
|
||
sample_id: str,
|
||
scenario: str,
|
||
reviewer: int,
|
||
order: list[str],
|
||
candidates: dict[str, str],
|
||
) -> dict:
|
||
"""构造 adapter 已绑定场景策略、身份与证据位置的 blind judge v4 报告。"""
|
||
|
||
blind_to_arm = {"blind-1": "A", "blind-2": "B", "blind-3": "C"}
|
||
base_scores = {"A": 7.0, "B": 2.0, "C": 7.5}
|
||
return {
|
||
"schemaVersion": "blind-judge-report-v4",
|
||
"runId": run_id,
|
||
"sampleId": sample_id,
|
||
"scenario": scenario,
|
||
"rubricPolicyVersion": RUBRIC_POLICY_VERSION,
|
||
"reviewerInvocationId": f"{sample_id}-reviewer-{reviewer}",
|
||
"blindInputSha256": canonical_sha256(
|
||
{"sampleId": sample_id, "reviewer": reviewer, "order": order}
|
||
),
|
||
"oracleTruthPackSha256": canonical_sha256({"sampleId": sample_id, "oracle": 1}),
|
||
"candidateOrder": list(order),
|
||
"candidateScores": [
|
||
{
|
||
"blindCandidateId": blind_id,
|
||
"candidateSha256": candidates[blind_to_arm[blind_id]],
|
||
"scores": {
|
||
dimension: {
|
||
"score": base_scores[blind_to_arm[blind_id]],
|
||
"evidence": [
|
||
{
|
||
"sourceType": "judge_inference",
|
||
"sourceRef": f"judge-inference:{dimension}",
|
||
"excerpt": "测试判断依据",
|
||
}
|
||
],
|
||
}
|
||
for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
for blind_id in order
|
||
],
|
||
"dimensionPreferences": [
|
||
{
|
||
"dimension": dimension,
|
||
"orderedCandidateIds": list(order),
|
||
"reason": "共同 rubric 下的匿名偏好顺序",
|
||
}
|
||
for dimension in DIMENSIONS
|
||
],
|
||
"oracleAssertionVerdicts": [
|
||
{
|
||
"assertionId": "assertion-1",
|
||
"candidateSha256": candidates[arm],
|
||
"verdict": "pass",
|
||
"reason": "共同 oracle 断言核对通过",
|
||
"evidenceRefs": [
|
||
{"sourceType": "oracle_assertion", "sourceRef": "assertion-1"}
|
||
],
|
||
}
|
||
for arm in ("A", "B", "C")
|
||
],
|
||
"hardConstraintVerdicts": [
|
||
{
|
||
"constraintId": "constraint-1",
|
||
"candidateSha256": candidates[arm],
|
||
"verdict": "pass",
|
||
"reason": "共同细纲约束核对通过",
|
||
"evidenceRefs": [
|
||
{
|
||
"sourceType": "preregistered_constraint",
|
||
"sourceRef": "constraint-1",
|
||
}
|
||
],
|
||
}
|
||
for arm in ("A", "B", "C")
|
||
],
|
||
"modelReceiptSha256": "sha256:" + "0" * 64,
|
||
"status": "completed",
|
||
"reportSha256": "sha256:" + "0" * 64,
|
||
}
|
||
|
||
|
||
def _reviewer_raw_draft(
|
||
*,
|
||
sample_id: str,
|
||
reviewer: int,
|
||
order: list[str],
|
||
) -> dict:
|
||
"""构造与 reviewer 报告对应的模型原始 structured output(归一化前的 draft)。
|
||
|
||
WHY: structuredOutputSha256 的语义是「哈希模型原始输出」,而不是哈希带绑定占位字段
|
||
(modelReceiptSha256/reportSha256)的报告。draft 与同名 reviewer 报告同分数、同候选顺序,
|
||
表达「报告是原始输出经归一化的产物」;builder 用它做 canonical(draft)==回执哈希 的只读复核。
|
||
"""
|
||
|
||
blind_to_arm = {"blind-1": "A", "blind-2": "B", "blind-3": "C"}
|
||
base_scores = {"A": 7.0, "B": 2.0, "C": 7.5}
|
||
return {
|
||
"schemaVersion": "blind-judge-draft-v3",
|
||
"reviewerInvocationId": f"{sample_id}-reviewer-{reviewer}",
|
||
"candidateOrder": list(order),
|
||
"candidateScores": [
|
||
{
|
||
"blindCandidateId": blind_id,
|
||
"scores": {
|
||
dimension: {
|
||
"score": base_scores[blind_to_arm[blind_id]],
|
||
"reason": "模型原始输出的判断理由",
|
||
"candidateQuote": "候选正文原始引用",
|
||
"evidenceRefs": [
|
||
{"sourceType": "candidate", "sourceId": blind_id}
|
||
],
|
||
}
|
||
for dimension in DIMENSIONS
|
||
},
|
||
}
|
||
for blind_id in order
|
||
],
|
||
}
|
||
|
||
|
||
def source_bundle(sample_count: int = 10, *, gate: str = "B") -> dict:
|
||
"""构造不含正文且所有身份、候选和 verdict 可闭合的来源包。"""
|
||
|
||
sample_ids = [f"sample-{index + 1}" for index in range(sample_count)]
|
||
run_id = "gate-test-run"
|
||
evaluation_set_version = "gate-test-set-v1"
|
||
manifest_samples = []
|
||
diff_receipts = {}
|
||
retrieval_manifests = {}
|
||
execution_receipts = {}
|
||
detector_reports = {}
|
||
judge_reports = {}
|
||
oracle_receipts = {}
|
||
leakage_receipts = {}
|
||
for index, sample_id in enumerate(sample_ids):
|
||
candidates = {
|
||
arm: canonical_sha256({"sampleId": sample_id, "arm": arm})
|
||
for arm in ("A", "B", "C")
|
||
}
|
||
scenario = SCENARIOS[index % len(SCENARIOS)]
|
||
manifest_samples.append(
|
||
{
|
||
"sampleId": sample_id,
|
||
"workId": 1 if index < 5 else 2,
|
||
"scenario": scenario,
|
||
"newCharacterRatio": (0.0, 0.25, 0.75, None)[index % 4],
|
||
"newCharacterBasis": {"absentBeforeAsOf": []},
|
||
"comparableAssertionIds": ["assertion-1"],
|
||
"hardConstraintIds": ["constraint-1"],
|
||
"arms": {
|
||
arm: {"candidateSha256": candidate_hash}
|
||
for arm, candidate_hash in candidates.items()
|
||
},
|
||
}
|
||
)
|
||
diff_receipts[sample_id] = _with_hash(
|
||
{
|
||
"schemaVersion": "writer-creative-input-allowlist-diff-v2",
|
||
"sampleId": sample_id,
|
||
"arms": ["A", "C"],
|
||
"ok": True,
|
||
}
|
||
)
|
||
retrieval_manifests[sample_id] = {
|
||
arm: {"manifestId": canonical_sha256({"sampleId": sample_id, "arm": arm}), "cardEntityNames": []}
|
||
for arm in ("A", "B", "C")
|
||
}
|
||
execution_receipts[sample_id] = {}
|
||
detector_reports[sample_id] = {}
|
||
leakage_receipts[sample_id] = {}
|
||
for arm in ("A", "B", "C"):
|
||
execution_receipts[sample_id][arm] = {}
|
||
for role in ("writer", "semantic_detector"):
|
||
inner_receipts = [
|
||
_runtime_receipt(
|
||
role,
|
||
sample_id,
|
||
arm,
|
||
invocation,
|
||
)
|
||
for invocation in range(1, 2)
|
||
]
|
||
execution_receipts[sample_id][arm][role] = _with_hash(
|
||
{
|
||
"sampleId": sample_id,
|
||
"arm": arm,
|
||
"role": role,
|
||
"candidateSha256": candidates[arm],
|
||
"status": "completed",
|
||
"executionReceipts": inner_receipts,
|
||
"executionReceiptSha256": [
|
||
canonical_sha256(receipt) for receipt in inner_receipts
|
||
],
|
||
}
|
||
)
|
||
detector_payload = {
|
||
"schemaVersion": "semantic-detection-v3",
|
||
"runId": run_id,
|
||
"sampleId": sample_id,
|
||
"opaqueArmId": f"blind-{arm.lower()}",
|
||
"inputSha256": canonical_sha256({"sampleId": sample_id, "arm": arm, "input": 1}),
|
||
"candidateVersion": 1,
|
||
"candidateSha256": candidates[arm],
|
||
"contextSnapshotSha256": canonical_sha256({"sampleId": sample_id, "arm": arm, "context": 1}),
|
||
"modelReceiptSha256": execution_receipts[sample_id][arm][
|
||
"semantic_detector"
|
||
]["executionReceiptSha256"][0],
|
||
"claims": [],
|
||
"findings": [],
|
||
"assertionVerdicts": [
|
||
{
|
||
"assertionId": "assertion-1",
|
||
"candidateSha256": candidates[arm],
|
||
"verdict": "pass",
|
||
}
|
||
],
|
||
"hardConstraintVerdicts": [
|
||
{
|
||
"constraintId": "constraint-1",
|
||
"candidateSha256": candidates[arm],
|
||
"verdict": "pass",
|
||
}
|
||
],
|
||
"newSettingCandidates": [],
|
||
"evidenceGaps": [],
|
||
"status": "passed",
|
||
}
|
||
detector_reports[sample_id][arm] = {
|
||
**detector_payload,
|
||
"reportSha256": canonical_sha256(detector_payload),
|
||
}
|
||
leakage_receipts[sample_id][arm] = _with_hash(
|
||
{"sampleId": sample_id, "arm": arm, "status": "passed", "findingCount": 0}
|
||
)
|
||
reviewer_outputs = [
|
||
_reviewer_raw_report(
|
||
run_id=run_id,
|
||
sample_id=sample_id,
|
||
scenario=scenario,
|
||
reviewer=1,
|
||
order=["blind-1", "blind-2", "blind-3"],
|
||
candidates=candidates,
|
||
),
|
||
_reviewer_raw_report(
|
||
run_id=run_id,
|
||
sample_id=sample_id,
|
||
scenario=scenario,
|
||
reviewer=2,
|
||
order=["blind-3", "blind-2", "blind-1"],
|
||
candidates=candidates,
|
||
),
|
||
]
|
||
# WHY: reviewer 原始 draft 与 reviewerReports 按 reviewer 顺序一一对应(reviewer 1/2,
|
||
# 候选顺序与同名报告一致),作为 reviewerStructuredOutputs 随报告进 builder。
|
||
reviewer_drafts = [
|
||
_reviewer_raw_draft(
|
||
sample_id=sample_id,
|
||
reviewer=1,
|
||
order=["blind-1", "blind-2", "blind-3"],
|
||
),
|
||
_reviewer_raw_draft(
|
||
sample_id=sample_id,
|
||
reviewer=2,
|
||
order=["blind-3", "blind-2", "blind-1"],
|
||
),
|
||
]
|
||
judge_receipts = [
|
||
_runtime_receipt(
|
||
"blind_judge",
|
||
sample_id,
|
||
"panel",
|
||
invocation,
|
||
# WHY: 回执的 structuredOutputSha256 哈希模型原始 draft,而不是带占位绑定字段的报告,
|
||
# 与生产语义「哈希模型原始输出」统一。
|
||
structured_output=reviewer_draft,
|
||
)
|
||
for invocation, reviewer_draft in enumerate(reviewer_drafts, 1)
|
||
]
|
||
for arm in ("A", "B", "C"):
|
||
execution_receipts[sample_id][arm]["blind_judge"] = _with_hash(
|
||
{
|
||
"sampleId": sample_id,
|
||
"arm": arm,
|
||
"role": "blind_judge",
|
||
"candidateSha256": candidates[arm],
|
||
"status": "completed",
|
||
"executionReceipts": copy.deepcopy(judge_receipts),
|
||
"executionReceiptSha256": [
|
||
canonical_sha256(receipt) for receipt in judge_receipts
|
||
],
|
||
}
|
||
)
|
||
bound_reviewer_reports = []
|
||
for reviewer_output, receipt in zip(reviewer_outputs, judge_receipts, strict=True):
|
||
bound = copy.deepcopy(reviewer_output)
|
||
bound["modelReceiptSha256"] = canonical_sha256(receipt)
|
||
bound["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in bound.items() if key != "reportSha256"}
|
||
)
|
||
bound_reviewer_reports.append(bound)
|
||
panel = adjudicate_structured_reviews(
|
||
bound_reviewer_reports[0], bound_reviewer_reports[1]
|
||
)
|
||
judge_payload = {
|
||
**panel,
|
||
"reviewerReports": bound_reviewer_reports,
|
||
# WHY: 模型原始 draft 按 reviewer 顺序与 reviewerReports 一一对应,随报告进 builder,
|
||
# 供 builder 只读复核 canonical(draft)==回执 structuredOutputSha256,恢复评分内容↔原始输出绑定。
|
||
"reviewerStructuredOutputs": copy.deepcopy(reviewer_drafts),
|
||
"reviewerFinalReceiptIndexes": [0, 1],
|
||
"modelReceiptSha256": canonical_sha256(
|
||
execution_receipts[sample_id]["A"]["blind_judge"][
|
||
"executionReceiptSha256"
|
||
]
|
||
),
|
||
}
|
||
judge_reports[sample_id] = {
|
||
**judge_payload,
|
||
"reportSha256": canonical_sha256(judge_payload),
|
||
}
|
||
oracle_receipts[sample_id] = _with_hash(
|
||
{
|
||
"sampleId": sample_id,
|
||
"status": "authorized",
|
||
"assertionIds": ["assertion-1"],
|
||
"packSha256": canonical_sha256({"sampleId": sample_id, "pack": 1}),
|
||
}
|
||
)
|
||
gate_a_payload = {
|
||
"schemaVersion": "writer-gate-report-v2",
|
||
"gate": "A",
|
||
"status": "passed",
|
||
"runId": "gate-a-prior-run",
|
||
"evaluationSetVersion": evaluation_set_version,
|
||
"gateInputSha256": canonical_sha256({"gate": "A", "set": evaluation_set_version}),
|
||
"primaryReason": "all_gate_a_conditions_met",
|
||
"reasons": ["all_gate_a_conditions_met"],
|
||
"metrics": {},
|
||
"confounds": {},
|
||
}
|
||
return {
|
||
"gate": gate,
|
||
"run_id": run_id,
|
||
"evaluation_set_version": evaluation_set_version,
|
||
"manifest": {
|
||
"schemaVersion": "writer-replay-manifest-v3",
|
||
"runId": run_id,
|
||
"evaluationSetVersion": evaluation_set_version,
|
||
"samples": manifest_samples,
|
||
},
|
||
"order_table": build_balanced_preregistration(
|
||
evaluation_set_version=evaluation_set_version,
|
||
sample_ids=sample_ids,
|
||
),
|
||
"diff_receipts": diff_receipts,
|
||
"retrieval_manifests": retrieval_manifests,
|
||
"execution_receipts": execution_receipts,
|
||
"detector_reports": detector_reports,
|
||
"judge_reports": judge_reports,
|
||
"oracle_receipts": oracle_receipts,
|
||
"leakage_receipts": leakage_receipts,
|
||
"authorization_receipt": _with_hash(
|
||
{"runId": run_id, "status": "authorized", "authorizationSnapshotId": "auth-1"}
|
||
),
|
||
"gate_a_report": (
|
||
{**gate_a_payload, "gateReportSha256": canonical_sha256(gate_a_payload)}
|
||
if gate == "B"
|
||
else None
|
||
),
|
||
}
|
||
|
||
|
||
def build_input(bundle: dict) -> dict:
|
||
"""调用唯一 builder,测试不得直接拼 Gate 输入。"""
|
||
|
||
return GateInputBuilder().build(**bundle)
|
||
|
||
|
||
def _rebuild_judge_source(bundle: dict, sample_id: str, mutate) -> None:
|
||
"""修改模型原始输出后重建 runtime 回执与 panel,供非篡改质量场景测试使用。"""
|
||
|
||
report = bundle["judge_reports"][sample_id]
|
||
reviewer_reports = copy.deepcopy(report["reviewerReports"])
|
||
reviewer_drafts = copy.deepcopy(report["reviewerStructuredOutputs"])
|
||
for reviewer_report in reviewer_reports:
|
||
mutate(reviewer_report)
|
||
# WHY: 报告是原始输出归一化的产物。mutate 改了报告评分后,把同序候选的同维分数同步回 draft,
|
||
# 保持「原始输出 ↔ 报告评分」一致(合法改分场景),回执 structuredOutputSha256 仍哈希新原始输出。
|
||
for draft, reviewer_report in zip(reviewer_drafts, reviewer_reports, strict=True):
|
||
for draft_row, report_row in zip(
|
||
draft["candidateScores"], reviewer_report["candidateScores"], strict=True
|
||
):
|
||
for dimension in DIMENSIONS:
|
||
draft_row["scores"][dimension]["score"] = report_row["scores"][dimension]["score"]
|
||
old_receipts = bundle["execution_receipts"][sample_id]["A"]["blind_judge"][
|
||
"executionReceipts"
|
||
]
|
||
receipts = []
|
||
for index, (old_receipt, draft) in enumerate(
|
||
zip(old_receipts, reviewer_drafts, strict=True), 1
|
||
):
|
||
receipt = copy.deepcopy(old_receipt)
|
||
receipt["structuredOutputSha256"] = canonical_sha256(draft)
|
||
receipt["invocationId"] = f"{sample_id}-panel-rebuilt-{index}"
|
||
receipts.append(receipt)
|
||
receipt_hashes = [canonical_sha256(receipt) for receipt in receipts]
|
||
for arm in ("A", "B", "C"):
|
||
wrapper = bundle["execution_receipts"][sample_id][arm]["blind_judge"]
|
||
wrapper["executionReceipts"] = copy.deepcopy(receipts)
|
||
wrapper["executionReceiptSha256"] = list(receipt_hashes)
|
||
wrapper["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in wrapper.items() if key != "receiptSha256"}
|
||
)
|
||
bound_reports = []
|
||
for reviewer_report, receipt in zip(reviewer_reports, receipts, strict=True):
|
||
bound = copy.deepcopy(reviewer_report)
|
||
bound["modelReceiptSha256"] = canonical_sha256(receipt)
|
||
bound["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in bound.items() if key != "reportSha256"}
|
||
)
|
||
bound_reports.append(bound)
|
||
panel = adjudicate_structured_reviews(bound_reports[0], bound_reports[1])
|
||
payload = {
|
||
**panel,
|
||
"reviewerReports": bound_reports,
|
||
"reviewerStructuredOutputs": reviewer_drafts,
|
||
"reviewerFinalReceiptIndexes": [0, 1],
|
||
"modelReceiptSha256": canonical_sha256(receipt_hashes),
|
||
}
|
||
bundle["judge_reports"][sample_id] = {
|
||
**payload,
|
||
"reportSha256": canonical_sha256(payload),
|
||
}
|
||
|
||
|
||
class WriterGateTest(unittest.TestCase):
|
||
def test_manual_gate_input_is_rejected(self):
|
||
with self.assertRaisesRegex(ValueError, "writer-gate-input-v3"):
|
||
decide_gate({"gate": "A", "samples": []})
|
||
|
||
def test_system_failure_precedes_insufficient_evidence(self):
|
||
bundle = source_bundle(1, gate="A")
|
||
receipt = bundle["execution_receipts"]["sample-1"]["A"]["writer"]
|
||
receipt["status"] = "failed"
|
||
receipt["receiptSha256"] = canonical_sha256(
|
||
{key: value for key, value in receipt.items() if key != "receiptSha256"}
|
||
)
|
||
report = decide_gate(build_input(bundle))
|
||
self.assertEqual(report["status"], "failed")
|
||
self.assertEqual(report["primaryReason"], "system_failure")
|
||
|
||
def test_b_is_diagnostic_and_c_minus_a_drives_gate(self):
|
||
report = decide_gate(build_input(source_bundle()))
|
||
self.assertEqual(report["status"], "passed")
|
||
self.assertLess(report["metrics"]["bMinusAAverageDeltas"]["setting_entity_fidelity"], 0)
|
||
self.assertGreaterEqual(report["metrics"]["cMinusAAverageDeltas"]["setting_entity_fidelity"], 0.25)
|
||
|
||
def test_real_detector_report_prevents_constant_false_green(self):
|
||
bundle = source_bundle(5, gate="A")
|
||
report = bundle["detector_reports"]["sample-1"]["C"]
|
||
report["findings"] = [{"findingId": "high-1", "severity": "high"}]
|
||
report["status"] = "failed"
|
||
report["reportSha256"] = canonical_sha256(
|
||
{key: value for key, value in report.items() if key != "reportSha256"}
|
||
)
|
||
decision = decide_gate(build_input(bundle))
|
||
self.assertEqual(decision["status"], "failed")
|
||
self.assertIn("c_arm_high_severity_residual", decision["reasons"])
|
||
|
||
def test_nonempty_stratum_regression_cannot_be_hidden_by_average(self):
|
||
bundle = source_bundle()
|
||
for sample in bundle["manifest"]["samples"]:
|
||
if sample["newCharacterRatio"] != 0.0:
|
||
continue
|
||
candidate_c = sample["arms"]["C"]["candidateSha256"]
|
||
|
||
def lower_c_style(reviewer):
|
||
"""只降低当前样本 C 候选的风格分。"""
|
||
|
||
for row in reviewer["candidateScores"]:
|
||
if row["candidateSha256"] == candidate_c:
|
||
row["scores"]["style_consistency"]["score"] = 6.0
|
||
|
||
_rebuild_judge_source(
|
||
bundle,
|
||
sample["sampleId"],
|
||
lower_c_style,
|
||
)
|
||
decision = decide_gate(build_input(bundle))
|
||
self.assertEqual(decision["status"], "failed")
|
||
self.assertIn("zero_style_consistency_regression", decision["reasons"])
|
||
|
||
def test_scenario_regression_cannot_be_hidden_by_other_scenario_types(self):
|
||
bundle = source_bundle()
|
||
for sample in bundle["manifest"]["samples"]:
|
||
if sample["scenario"] != "battle":
|
||
continue
|
||
candidate_c = sample["arms"]["C"]["candidateSha256"]
|
||
|
||
def lower_battle_execution(reviewer):
|
||
for row in reviewer["candidateScores"]:
|
||
if row["candidateSha256"] == candidate_c:
|
||
row["scores"][SCENARIO_DIMENSION]["score"] = 6.0
|
||
|
||
_rebuild_judge_source(
|
||
bundle,
|
||
sample["sampleId"],
|
||
lower_battle_execution,
|
||
)
|
||
|
||
decision = decide_gate(build_input(bundle))
|
||
|
||
self.assertEqual(decision["status"], "failed")
|
||
self.assertIn("battle_scenario_execution_regression", decision["reasons"])
|
||
battle = decision["metrics"]["scenarioExecutionByType"]["battle"]
|
||
self.assertEqual(battle["sampleCount"], 2)
|
||
self.assertEqual(battle["stableSampleCount"], 2)
|
||
self.assertEqual(battle["cMinusAAverageDelta"], -1.0)
|
||
self.assertEqual(
|
||
decision["metrics"]["scenarioExecutionByType"]["character_dialogue"][
|
||
"cMinusAAverageDelta"
|
||
],
|
||
0.5,
|
||
)
|
||
|
||
def test_non_passed_gate_b_does_not_issue_receipt(self):
|
||
bundle = source_bundle()
|
||
for sample_id in list(bundle["judge_reports"]):
|
||
def set_all_scores(reviewer):
|
||
"""把每位 reviewer 的五维分数统一为无增益基线。"""
|
||
|
||
for row in reviewer["candidateScores"]:
|
||
for dimension in DIMENSIONS:
|
||
row["scores"][dimension]["score"] = 7.0
|
||
|
||
_rebuild_judge_source(bundle, sample_id, set_all_scores)
|
||
gate_input = build_input(bundle)
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
root = pathlib.Path(directory)
|
||
report, receipt = issue_gate_report_and_receipt(
|
||
gate_input=gate_input,
|
||
expected_input_sha256=gate_input["gateInputSha256"],
|
||
output_dir=root,
|
||
cas_dir=root / "cas",
|
||
)
|
||
self.assertEqual(report["status"], "no_gain")
|
||
self.assertIsNone(receipt)
|
||
self.assertFalse((root / "writer-gate-b-passed-receipt.json").exists())
|
||
|
||
def test_passed_gate_b_signs_and_tampering_is_rejected(self):
|
||
gate_input = build_input(source_bundle())
|
||
with tempfile.TemporaryDirectory(dir="/private/tmp") as directory:
|
||
root = pathlib.Path(directory)
|
||
input_path = root / "gate-input.json"
|
||
input_path.write_text(json.dumps(gate_input, ensure_ascii=False), encoding="utf-8")
|
||
report, receipt = issue_gate_report_and_receipt(
|
||
gate_input=gate_input,
|
||
expected_input_sha256=gate_input["gateInputSha256"],
|
||
output_dir=root,
|
||
cas_dir=root / "cas",
|
||
)
|
||
self.assertEqual(report["status"], "passed")
|
||
self.assertIsNotNone(receipt)
|
||
receipt_path = root / "writer-gate-b-passed-receipt.json"
|
||
verified = verify_writer_gate_receipt(
|
||
receipt_path=receipt_path,
|
||
gate_report_path=root / "gate-report.json",
|
||
gate_input_path=input_path,
|
||
cas_dir=root / "cas",
|
||
)
|
||
self.assertEqual(verified["receiptSha256"], receipt["receiptSha256"])
|
||
tampered = json.loads((root / "gate-report.json").read_text(encoding="utf-8"))
|
||
tampered["status"] = "failed"
|
||
(root / "gate-report.json").write_text(json.dumps(tampered), encoding="utf-8")
|
||
with self.assertRaisesRegex(ValueError, "gate-report"):
|
||
verify_writer_gate_receipt(
|
||
receipt_path=receipt_path,
|
||
gate_report_path=root / "gate-report.json",
|
||
gate_input_path=input_path,
|
||
cas_dir=root / "cas",
|
||
)
|
||
|
||
def test_scene_selection_bias_flags_single_work_on_passed_gate_a(self):
|
||
"""五样本全来自单一作品时,即使 Gate A 通过也必须标注场景选择偏差。"""
|
||
|
||
bundle = source_bundle(5, gate="A")
|
||
report = decide_gate(build_input(bundle))
|
||
# 五样本覆盖全部场景且 C 臂硬门通过,Gate A 终态为 passed;但 workId 只有一个,
|
||
# 第六类混淆项仍须如实标注,证明它独立于终态门、只作代表性报告而不改写终态。
|
||
self.assertEqual(report["status"], "passed")
|
||
bias = report["confounds"]["sceneSelectionBias"]
|
||
self.assertIn({"code": "single_work", "workCount": 1}, bias)
|
||
# 五类场景全部覆盖,不应误报场景覆盖不全。
|
||
self.assertNotIn(
|
||
"scenario_coverage_incomplete", [item["code"] for item in bias]
|
||
)
|
||
|
||
def test_scene_selection_bias_empty_for_balanced_multi_work_set(self):
|
||
"""跨两作品且覆盖全部场景的均衡评测集不产出场景选择偏差。"""
|
||
|
||
report = decide_gate(build_input(source_bundle()))
|
||
self.assertEqual(report["confounds"]["sceneSelectionBias"], [])
|
||
|
||
def test_scene_selection_bias_flags_incomplete_scenario_coverage(self):
|
||
"""只选部分预注册场景类型时产出场景覆盖不全混淆项并列出缺失场景。"""
|
||
|
||
bundle = source_bundle(5, gate="A")
|
||
# 把第五个样本改成与前四个重复的合法场景,机械构造「只选部分场景」的代表性不足。
|
||
bundle["manifest"]["samples"][4]["scenario"] = "battle"
|
||
report = decide_gate(build_input(bundle))
|
||
bias = report["confounds"]["sceneSelectionBias"]
|
||
self.assertIn("scenario_coverage_incomplete", [item["code"] for item in bias])
|
||
missing = next(
|
||
item["missingScenarios"]
|
||
for item in bias
|
||
if item["code"] == "scenario_coverage_incomplete"
|
||
)
|
||
self.assertEqual(missing, ["returning_character"])
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|