范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
638 lines
31 KiB
Python
638 lines
31 KiB
Python
#!/usr/bin/env python3
|
||
"""SemanticDetection v3 模型边界与可信绑定的离线测试。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import hashlib
|
||
import pathlib
|
||
import sys
|
||
import unittest
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "check-content-consistency" / "scripts"
|
||
if str(SCRIPT_DIR) not in sys.path:
|
||
sys.path.insert(0, str(SCRIPT_DIR))
|
||
|
||
from run_writer_semantic_detector import ( # noqa: E402
|
||
SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA,
|
||
SemanticDetectorContractError,
|
||
_quote_location,
|
||
build_safe_semantic_diagnostic,
|
||
canonical_sha256,
|
||
calculate_semantic_metrics,
|
||
run_writer_semantic_detector,
|
||
validate_semantic_detector_input,
|
||
validate_semantic_detector_report,
|
||
)
|
||
|
||
|
||
def _hash_text(value: str) -> str:
|
||
return "sha256:" + hashlib.sha256(value.encode("utf-8")).hexdigest()
|
||
|
||
|
||
def semantic_input() -> dict[str, Any]:
|
||
body = "林澈守住城门。旧徽章在他掌心发热。"
|
||
payload: dict[str, Any] = {
|
||
"schemaVersion": "semantic-detector-input-v3",
|
||
"runId": "run-semantic-1",
|
||
"sampleId": "sample-489",
|
||
"opaqueArmId": "blind-semantic-7",
|
||
"candidateVersion": 3,
|
||
"candidateSha256": _hash_text(body),
|
||
"candidateBody": body,
|
||
"contextSnapshotSha256": "sha256:" + "1" * 64,
|
||
"fineOutline": {
|
||
"sourceRef": {"sourceId": "fine-outline:489", "sourceVersion": "v1", "chapter": 489},
|
||
"hardConstraints": [{"constraintId": "constraint-1", "text": "林澈必须守住城门"}],
|
||
"adjustableBeats": [],
|
||
"declaredNewFacts": [],
|
||
},
|
||
"hardConstraints": [{"constraintId": "constraint-1", "text": "林澈必须守住城门"}],
|
||
"factEvidence": [{
|
||
"evidenceId": "evidence-1",
|
||
"fact": "林澈在城门",
|
||
"sourceType": "canonical_state",
|
||
"sourceRef": {"sourceId": "state:488", "sourceVersion": "v1", "chapter": 488},
|
||
"contentSha256": "sha256:" + "2" * 64,
|
||
"riskLevel": "low",
|
||
}],
|
||
"proseEvidence": [],
|
||
"asOf": 488,
|
||
"authorizationSnapshotId": "authorization-offline-1",
|
||
}
|
||
payload["inputSha256"] = canonical_sha256(payload)
|
||
return payload
|
||
|
||
|
||
def semantic_draft(*, gap: bool = False, failed: bool = False) -> dict[str, Any]:
|
||
draft: dict[str, Any] = {
|
||
"schemaVersion": "semantic-detection-draft-v3",
|
||
"claims": [{
|
||
"claimId": "claim-1",
|
||
"factType": "character_state",
|
||
"text": "林澈守住城门",
|
||
"candidateQuote": "林澈守住城门",
|
||
"coverageState": "supported",
|
||
"evidenceIds": ["evidence-1"],
|
||
}],
|
||
"findings": ([{
|
||
"findingId": "finding-1",
|
||
"severity": "high",
|
||
"category": "fact_conflict",
|
||
"candidateQuote": "旧徽章",
|
||
"evidenceIds": ["evidence-1"],
|
||
"message": "候选出现与既有事实冲突的旧徽章",
|
||
}] if failed else []),
|
||
"assertionVerdicts": [{
|
||
"assertionId": "evidence-1",
|
||
"verdict": "unknown" if gap else "pass",
|
||
"candidateQuote": "林澈守住城门",
|
||
"evidenceIds": ["evidence-1"],
|
||
**({"gapReason": "缺少旧徽章来源"} if gap else {}),
|
||
}],
|
||
"hardConstraintVerdicts": [{
|
||
"constraintId": "constraint-1",
|
||
"verdict": "pass",
|
||
"candidateQuote": "林澈守住城门",
|
||
"evidenceIds": ["constraint-1"],
|
||
}],
|
||
"newSettingCandidates": [],
|
||
"evidenceGaps": ([{
|
||
"gapId": "gap-1",
|
||
"query": "旧徽章来源",
|
||
"reason": "候选出现未覆盖物品",
|
||
"priority": "high",
|
||
"candidateQuote": "旧徽章",
|
||
}] if gap else []),
|
||
}
|
||
return draft
|
||
|
||
|
||
class QuoteLocationTest(unittest.TestCase):
|
||
"""`_quote_location` 三情形:恰好 1 次、≥2 次(绑定首次)、0 次(失败关闭)。"""
|
||
|
||
BODY = "林澈守住城门。旧徽章在他掌心发热。"
|
||
|
||
def test_quote_appears_once_returns_location(self) -> None:
|
||
text, start, end = _quote_location(self.BODY, "旧徽章", "$.candidateQuote")
|
||
self.assertEqual(text, "旧徽章")
|
||
self.assertEqual(start, self.BODY.index("旧徽章"))
|
||
self.assertEqual(end, start + len("旧徽章"))
|
||
self.assertEqual(self.BODY[start:end], "旧徽章")
|
||
|
||
def test_quote_appears_multiple_times_binds_first_occurrence(self) -> None:
|
||
# "。" 在正文出现两次:不再抛错,绑定首次出现。
|
||
text, start, end = _quote_location(self.BODY, "。", "$.candidateQuote")
|
||
self.assertEqual(text, "。")
|
||
self.assertEqual(start, self.BODY.index("。")) # 首次出现位置
|
||
self.assertEqual(end, start + len("。"))
|
||
self.assertEqual(self.BODY[start:end], "。")
|
||
|
||
def test_quote_absent_fails_closed(self) -> None:
|
||
with self.assertRaises(SemanticDetectorContractError) as caught:
|
||
_quote_location(self.BODY, "不存在的引文", "$.candidateQuote")
|
||
self.assertEqual(caught.exception.code, "SEMANTIC_DETECTOR_QUOTE_NOT_FOUND")
|
||
|
||
|
||
class FakeRunner:
|
||
def __init__(self, output: Mapping[str, Any], receipt_hash: str = "sha256:" + "3" * 64) -> None:
|
||
self.output = copy.deepcopy(dict(output))
|
||
self.receipt_hash = receipt_hash
|
||
self.calls: list[dict[str, Any]] = []
|
||
|
||
def run(self, *, adapter_role: str, model_input: Mapping[str, Any], output_schema: Mapping[str, Any]) -> Mapping[str, Any]:
|
||
self.calls.append({"role": adapter_role, "input": copy.deepcopy(dict(model_input)), "schema": output_schema})
|
||
return {"structuredOutput": copy.deepcopy(self.output), "modelReceiptSha256": self.receipt_hash}
|
||
|
||
|
||
class SequenceFakeRunner:
|
||
"""按调用次序依次返回预设产出,用于驱动自我纠错环。"""
|
||
|
||
def __init__(self, outputs: Sequence[Any], receipt_hash: str = "sha256:" + "3" * 64) -> None:
|
||
self.outputs = [
|
||
output if isinstance(output, BaseException) else copy.deepcopy(dict(output))
|
||
for output in outputs
|
||
]
|
||
self.receipt_hash = receipt_hash
|
||
self.calls: list[dict[str, Any]] = []
|
||
|
||
def run(self, *, adapter_role: str, model_input: Mapping[str, Any], output_schema: Mapping[str, Any]) -> Mapping[str, Any]:
|
||
self.calls.append({"role": adapter_role, "input": copy.deepcopy(dict(model_input)), "schema": output_schema})
|
||
# 超出预设数量后一直复用最后一份产出,便于断言纠错轮次上界。
|
||
output = self.outputs[min(len(self.calls) - 1, len(self.outputs) - 1)]
|
||
if isinstance(output, BaseException):
|
||
raise output
|
||
return {"structuredOutput": copy.deepcopy(output), "modelReceiptSha256": self.receipt_hash}
|
||
|
||
|
||
class SemanticDetectorCorrectionTest(unittest.TestCase):
|
||
"""检测自我纠错环:首轮引文不合格 → 回喂上一轮产出+原因 → 纠错后合格。"""
|
||
|
||
def test_quote_not_found_then_corrected_returns_ok_after_two_calls(self) -> None:
|
||
# 首轮引了一句正文里没有的话(QUOTE_NOT_FOUND),纠错后换成正文里真实存在的原话。
|
||
bad = semantic_draft()
|
||
bad["assertionVerdicts"][0]["candidateQuote"] = "正文里根本没有的引文"
|
||
good = semantic_draft()
|
||
runner = SequenceFakeRunner([bad, good])
|
||
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "passed")
|
||
# 模型被调了 2 次:首轮不合格 + 一轮纠错。
|
||
self.assertEqual(len(runner.calls), 2)
|
||
# 首轮输入不带 correction。
|
||
self.assertNotIn("correction", runner.calls[0]["input"])
|
||
# 第二轮输入携带 correction:previousDraft 是首轮原始产出,error 说明引文不在正文里。
|
||
correction = runner.calls[1]["input"]["correction"]
|
||
self.assertEqual(correction["previousDraft"], bad)
|
||
self.assertIn("引文未在候选正文中出现", correction["error"])
|
||
# 纠错不放宽校验:第二轮合格产出仍被完整绑定(offset 由 adapter 计算)。
|
||
verdict = result["report"]["assertionVerdicts"][0]
|
||
self.assertEqual(verdict["candidateQuote"], "林澈守住城门")
|
||
self.assertEqual(verdict["startCodePoint"], 0)
|
||
|
||
def test_id_set_mismatch_correction_receives_expected_verdict_ids(self) -> None:
|
||
"""ID 集错误时把冻结顺序显式回喂,但仍由适配器执行完整覆盖校验。"""
|
||
|
||
bad = semantic_draft()
|
||
bad["assertionVerdicts"][0]["assertionId"] = "wrong-id"
|
||
runner = SequenceFakeRunner([bad, semantic_draft()])
|
||
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(
|
||
runner.calls[1]["input"]["correction"]["expectedVerdictIds"],
|
||
{"assertionVerdicts": ["evidence-1"], "hardConstraintVerdicts": ["constraint-1"]},
|
||
)
|
||
|
||
def test_transient_api_error_retries_same_input_without_fake_correction(self) -> None:
|
||
"""可信 API 瞬时错误占用现有槽位,重发原输入后可恢复。"""
|
||
|
||
api_error = SemanticDetectorContractError(
|
||
"SEMANTIC_DETECTOR_API_ERROR", "瞬时 API 错误"
|
||
)
|
||
runner = SequenceFakeRunner([api_error, semantic_draft()])
|
||
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(result["attemptCount"], 2)
|
||
self.assertEqual(result["correctionCount"], 0)
|
||
self.assertEqual(len(runner.calls), 2)
|
||
self.assertNotIn("correction", runner.calls[0]["input"])
|
||
self.assertNotIn("correction", runner.calls[1]["input"])
|
||
|
||
def test_second_transient_api_error_fails_without_third_call(self) -> None:
|
||
"""API 瞬时错误最多重试一次,不能吃掉所有格式纠错槽位后继续盲重试。"""
|
||
|
||
first = SemanticDetectorContractError(
|
||
"SEMANTIC_DETECTOR_API_ERROR", "第一次瞬时 API 错误"
|
||
)
|
||
second = SemanticDetectorContractError(
|
||
"SEMANTIC_DETECTOR_API_ERROR", "第二次瞬时 API 错误"
|
||
)
|
||
runner = SequenceFakeRunner([first, second, semantic_draft()])
|
||
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_API_ERROR")
|
||
self.assertEqual(result["safeDiagnostic"]["attemptCount"], 2)
|
||
self.assertEqual(result["safeDiagnostic"]["correctionCount"], 0)
|
||
self.assertEqual(len(runner.calls), 2)
|
||
|
||
def test_all_rounds_invalid_exhausts_corrections_and_fails_closed(self) -> None:
|
||
# 三轮都引错(max_corrections=2 → 总共最多 3 次调用),最终返回最后一轮的失败。
|
||
bad = semantic_draft()
|
||
bad["assertionVerdicts"][0]["candidateQuote"] = "始终不在正文里的引文"
|
||
runner = SequenceFakeRunner([bad])
|
||
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_QUOTE_NOT_FOUND")
|
||
self.assertEqual(len(runner.calls), 3)
|
||
# 第二、三轮都带纠错反馈,首轮不带。
|
||
self.assertNotIn("correction", runner.calls[0]["input"])
|
||
self.assertIn("correction", runner.calls[1]["input"])
|
||
self.assertIn("correction", runner.calls[2]["input"])
|
||
self.assertEqual(
|
||
result["safeDiagnostic"],
|
||
{
|
||
"schemaVersion": "semantic-diagnostic-v1",
|
||
"outcome": "invalid",
|
||
"primaryCode": "SEMANTIC_DETECTOR_QUOTE_NOT_FOUND",
|
||
"reasonCode": "QUOTE_NOT_FOUND",
|
||
"section": "assertion_verdicts",
|
||
"attemptCount": 3,
|
||
"correctionCount": 2,
|
||
"blockingCounts": {
|
||
"highFindings": 0,
|
||
"failedAssertions": 0,
|
||
"failedHardConstraints": 0,
|
||
"conflictingClaims": 0,
|
||
"evidenceGaps": 0,
|
||
"unknownAssertions": 0,
|
||
"unknownHardConstraints": 0,
|
||
"unknownClaims": 0,
|
||
},
|
||
},
|
||
)
|
||
|
||
def test_hostile_extra_key_never_enters_safe_diagnostic(self) -> None:
|
||
hostile = semantic_draft()
|
||
hostile["raw-/private/tmp-候选正文"] = "不应出现在安全摘要"
|
||
|
||
result = run_writer_semantic_detector(
|
||
semantic_input(), model_runner=FakeRunner(hostile), max_corrections=0
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
diagnostic = result["safeDiagnostic"]
|
||
self.assertEqual(diagnostic["reasonCode"], "FIELD_SET_INVALID")
|
||
self.assertEqual(diagnostic["section"], "model_output")
|
||
serialized = str(diagnostic)
|
||
for forbidden in ("raw-/private/tmp-候选正文", "不应出现在安全摘要", "missing", "extra"):
|
||
self.assertNotIn(forbidden, serialized)
|
||
|
||
def test_safe_diagnostic_rejects_untrusted_code_and_unbounded_counts(self) -> None:
|
||
diagnostic = build_safe_semantic_diagnostic(
|
||
{
|
||
"ok": False,
|
||
"primaryCode": "SEMANTIC_BAD\n/private/tmp/raw",
|
||
"safeDiagnostic": {
|
||
"primaryCode": "SEMANTIC_BAD\n/private/tmp/raw",
|
||
"reasonCode": "不可信原因",
|
||
"section": "$.candidateBody",
|
||
"attemptCount": 100_000_000,
|
||
"correctionCount": 100_000_000,
|
||
},
|
||
}
|
||
)
|
||
|
||
self.assertEqual(diagnostic["primaryCode"], "SEMANTIC_DETECTOR_INVALID")
|
||
self.assertEqual(diagnostic["reasonCode"], "CONTRACT_INVALID")
|
||
self.assertEqual(diagnostic["section"], "model_output")
|
||
self.assertEqual(diagnostic["attemptCount"], 10_000)
|
||
self.assertEqual(diagnostic["correctionCount"], 9_999)
|
||
self.assertNotIn("/private/tmp", str(diagnostic))
|
||
|
||
def test_runner_failure_without_draft_is_not_corrected(self) -> None:
|
||
# runner 底座失败没有模型原始产出,纠错帮不上忙:只调一次即失败关闭。
|
||
class BrokenRunner:
|
||
def __init__(self) -> None:
|
||
self.calls = 0
|
||
|
||
def run(self, *, adapter_role: str, model_input: Mapping[str, Any], output_schema: Mapping[str, Any]) -> Mapping[str, Any]:
|
||
self.calls += 1
|
||
return {"structuredOutput": None} # 缺 modelReceiptSha256 → _invoke_model_runner 抛错
|
||
|
||
runner = BrokenRunner()
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_RECEIPT_BINDING_MISMATCH")
|
||
self.assertEqual(runner.calls, 1)
|
||
|
||
|
||
class SemanticDetectorV3Test(unittest.TestCase):
|
||
def test_model_schema_rejects_hash_offset_and_runtime_identity(self) -> None:
|
||
schema = SEMANTIC_DETECTOR_REPORT_JSON_SCHEMA
|
||
self.assertFalse(schema["additionalProperties"])
|
||
forbidden = {"runId", "candidateSha256", "contextSnapshotSha256", "modelReceiptSha256", "reportSha256", "startCodePoint", "endCodePoint"}
|
||
self.assertTrue(forbidden.isdisjoint(schema["properties"]))
|
||
finding_properties = schema["properties"]["findings"]["items"]["properties"]
|
||
self.assertTrue({"candidateSha256", "startCodePoint", "endCodePoint"}.isdisjoint(finding_properties))
|
||
claim_properties = schema["properties"]["claims"]["items"]["properties"]
|
||
self.assertTrue({"candidateSha256", "startCodePoint", "endCodePoint"}.isdisjoint(claim_properties))
|
||
|
||
forged = semantic_draft()
|
||
forged["candidateSha256"] = "sha256:" + "f" * 64
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(forged))
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_MODEL_OUTPUT_INVALID")
|
||
|
||
def test_adapter_projects_only_semantic_content_and_binds_deterministically(self) -> None:
|
||
runner = FakeRunner(semantic_draft())
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=runner)
|
||
self.assertTrue(result["ok"])
|
||
report = result["report"]
|
||
self.assertEqual(report["schemaVersion"], "semantic-detection-v3")
|
||
self.assertEqual(report["candidateVersion"], 3)
|
||
self.assertEqual(report["candidateSha256"], semantic_input()["candidateSha256"])
|
||
self.assertEqual(report["contextSnapshotSha256"], semantic_input()["contextSnapshotSha256"])
|
||
verdict = report["assertionVerdicts"][0]
|
||
self.assertEqual(verdict["startCodePoint"], 0)
|
||
self.assertEqual(verdict["endCodePoint"], len("林澈守住城门"))
|
||
self.assertEqual(report["reportSha256"], canonical_sha256({k: v for k, v in report.items() if k != "reportSha256"}))
|
||
self.assertEqual(report["claims"][0]["candidateSha256"], report["candidateSha256"])
|
||
|
||
model_input = runner.calls[0]["input"]
|
||
serialized = str(model_input)
|
||
for forbidden in ("runId", "sampleId", "opaqueArmId", "candidateSha256", "contextSnapshotSha256", "authorizationSnapshotId", "inputSha256"):
|
||
self.assertNotIn(forbidden, serialized)
|
||
|
||
def test_verdict_id_set_is_normalized_to_input_order(self) -> None:
|
||
"""同一完整 ID 集乱序时确定性重排,缺失/重复仍由合同拒绝。"""
|
||
|
||
payload = semantic_input()
|
||
payload["factEvidence"].append({
|
||
"evidenceId": "evidence-2",
|
||
"fact": "旧徽章在城门",
|
||
"sourceType": "historical_prose",
|
||
"sourceRef": {"sourceId": "prose:487", "sourceVersion": "v1", "chapter": 487},
|
||
"contentSha256": "sha256:" + "3" * 64,
|
||
"riskLevel": "low",
|
||
})
|
||
payload["inputSha256"] = canonical_sha256({
|
||
key: value for key, value in payload.items() if key != "inputSha256"
|
||
})
|
||
draft = semantic_draft()
|
||
draft["assertionVerdicts"].append({
|
||
"assertionId": "evidence-2",
|
||
"verdict": "pass",
|
||
"candidateQuote": "林澈守住城门",
|
||
"evidenceIds": ["evidence-2"],
|
||
})
|
||
draft["assertionVerdicts"].reverse()
|
||
|
||
result = run_writer_semantic_detector(payload, model_runner=FakeRunner(draft))
|
||
|
||
self.assertTrue(result["ok"], result)
|
||
self.assertEqual(
|
||
[item["assertionId"] for item in result["report"]["assertionVerdicts"]],
|
||
["evidence-1", "evidence-2"],
|
||
)
|
||
|
||
def test_duplicate_quote_now_binds_first_occurrence(self) -> None:
|
||
# 引文在候选中出现多次不再失败关闭:绑定到首次出现("。" 在正文里出现两次)。
|
||
duplicate = semantic_draft()
|
||
duplicate["assertionVerdicts"][0]["candidateQuote"] = "。"
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(duplicate))
|
||
self.assertTrue(result["ok"])
|
||
body = semantic_input()["candidateBody"]
|
||
verdict = result["report"]["assertionVerdicts"][0]
|
||
self.assertEqual(verdict["startCodePoint"], body.index("。"))
|
||
self.assertEqual(verdict["endCodePoint"], body.index("。") + len("。"))
|
||
|
||
def test_absent_quote_fails_closed_with_not_found(self) -> None:
|
||
# 引文 0 次出现 = 编造证据,仍失败关闭,错误码为 SEMANTIC_DETECTOR_QUOTE_NOT_FOUND。
|
||
absent = semantic_draft()
|
||
absent["assertionVerdicts"][0]["candidateQuote"] = "正文里根本没有的引文"
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(absent))
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_QUOTE_NOT_FOUND")
|
||
|
||
def test_unknown_is_legal_and_evidence_gap_drives_needs_evidence(self) -> None:
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(semantic_draft(gap=True)))
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "needs_evidence")
|
||
self.assertEqual(result["report"]["evidenceGaps"][0]["startCodePoint"], len("林澈守住城门。"))
|
||
|
||
def test_high_severity_finding_drives_failed_status(self) -> None:
|
||
# failed 主分支:无证据缺口、无 unknown,但存在 high 严重度 finding → status=failed。
|
||
# status 优先级为 needs_evidence > failed > passed(见 build_semantic_detection)。
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(semantic_draft(failed=True)))
|
||
self.assertTrue(result["ok"])
|
||
self.assertEqual(result["status"], "failed")
|
||
self.assertEqual(result["report"]["status"], "failed")
|
||
self.assertEqual(result["report"]["findings"][0]["severity"], "high")
|
||
self.assertEqual(result["metrics"]["highSeverityCount"], 1)
|
||
diagnostic = build_safe_semantic_diagnostic(result)
|
||
self.assertEqual(diagnostic["outcome"], "failed")
|
||
self.assertEqual(diagnostic["reasonCode"], "SEMANTIC_BLOCKED")
|
||
self.assertEqual(diagnostic["blockingCounts"]["highFindings"], 1)
|
||
self.assertEqual(diagnostic["blockingCounts"]["failedHardConstraints"], 0)
|
||
serialized = str(diagnostic)
|
||
for forbidden in ("candidateQuote", "message", "旧徽章", "constraint-1"):
|
||
self.assertNotIn(forbidden, serialized)
|
||
|
||
def test_corrected_blocking_report_preserves_attempt_count(self) -> None:
|
||
invalid = semantic_draft()
|
||
invalid["assertionVerdicts"][0]["candidateQuote"] = "正文里不存在的引文"
|
||
result = run_writer_semantic_detector(
|
||
semantic_input(),
|
||
model_runner=SequenceFakeRunner([invalid, semantic_draft(failed=True)]),
|
||
)
|
||
|
||
diagnostic = build_safe_semantic_diagnostic(result)
|
||
|
||
self.assertEqual(result["attemptCount"], 2)
|
||
self.assertEqual(diagnostic["outcome"], "failed")
|
||
self.assertEqual(diagnostic["attemptCount"], 2)
|
||
self.assertEqual(diagnostic["correctionCount"], 1)
|
||
|
||
def test_needs_evidence_safe_diagnostic_contains_counts_only(self) -> None:
|
||
result = run_writer_semantic_detector(
|
||
semantic_input(), model_runner=FakeRunner(semantic_draft(gap=True))
|
||
)
|
||
|
||
diagnostic = build_safe_semantic_diagnostic(result)
|
||
|
||
self.assertEqual(diagnostic["outcome"], "needs_evidence")
|
||
self.assertEqual(diagnostic["primaryCode"], "SEMANTIC_EVIDENCE_REQUIRED")
|
||
self.assertEqual(diagnostic["blockingCounts"]["evidenceGaps"], 1)
|
||
self.assertEqual(diagnostic["blockingCounts"]["unknownAssertions"], 1)
|
||
serialized = str(diagnostic)
|
||
for forbidden in (
|
||
"candidateQuote", "旧徽章来源", "候选出现未覆盖物品", "gap-1"
|
||
):
|
||
self.assertNotIn(forbidden, serialized)
|
||
|
||
def test_ellipsis_reference_not_flagged_but_real_traversal_blocked(self) -> None:
|
||
# 省略号 `...` 含子串 `..`,旧的 `".." in text` 会误判为路径穿越;精确判定必须放行。
|
||
ellipsis = semantic_input()
|
||
ellipsis["factEvidence"][0]["sourceRef"]["sourceId"] = "林澈说:等等..."
|
||
ellipsis["inputSha256"] = canonical_sha256({k: v for k, v in ellipsis.items() if k != "inputSha256"})
|
||
validated = validate_semantic_detector_input(ellipsis) # 不抛 SEMANTIC_DETECTOR_LEAKAGE_DETECTED
|
||
self.assertEqual(validated["factEvidence"][0]["sourceRef"]["sourceId"], "林澈说:等等...")
|
||
|
||
# 反例:真实路径穿越 a/../b 仍被拦下。
|
||
traversal = semantic_input()
|
||
traversal["factEvidence"][0]["sourceRef"]["sourceId"] = "vault/a/../b"
|
||
traversal["inputSha256"] = canonical_sha256({k: v for k, v in traversal.items() if k != "inputSha256"})
|
||
with self.assertRaises(SemanticDetectorContractError) as caught:
|
||
validate_semantic_detector_input(traversal)
|
||
self.assertEqual(caught.exception.code, "SEMANTIC_DETECTOR_LEAKAGE_DETECTED")
|
||
|
||
def test_unknown_without_gap_reason_fails_closed(self) -> None:
|
||
draft = semantic_draft(gap=True)
|
||
draft["assertionVerdicts"][0].pop("gapReason")
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(draft))
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_MODEL_OUTPUT_INVALID")
|
||
|
||
def test_non_unknown_claim_gap_reason_is_deterministically_discarded(self) -> None:
|
||
# WHY: supported/conflict/declared_new 已有闭集结论,模型残留的解释不应阻断整份报告,
|
||
# 也不得进入可信报告参与状态或哈希计算。
|
||
for coverage_state in ("supported", "conflict", "declared_new"):
|
||
with self.subTest(coverage_state=coverage_state):
|
||
draft = semantic_draft()
|
||
draft["claims"][0]["coverageState"] = coverage_state
|
||
draft["claims"][0]["gapReason"] = "模型残留的冗余解释"
|
||
|
||
result = run_writer_semantic_detector(
|
||
semantic_input(), model_runner=FakeRunner(draft), max_corrections=0
|
||
)
|
||
|
||
self.assertTrue(result["ok"])
|
||
bound_claim = result["report"]["claims"][0]
|
||
self.assertEqual(bound_claim["coverageState"], coverage_state)
|
||
self.assertNotIn("gapReason", bound_claim)
|
||
|
||
def test_non_unknown_verdict_gap_reason_is_deterministically_discarded(self) -> None:
|
||
# assertion 与 hard constraint 共用同一绑定器;分别覆盖 pass/fail,确保两类列表都收敛。
|
||
cases = (
|
||
("assertionVerdicts", "pass"),
|
||
("assertionVerdicts", "fail"),
|
||
("hardConstraintVerdicts", "pass"),
|
||
("hardConstraintVerdicts", "fail"),
|
||
)
|
||
for field, verdict in cases:
|
||
with self.subTest(field=field, verdict=verdict):
|
||
draft = semantic_draft()
|
||
draft[field][0]["verdict"] = verdict
|
||
draft[field][0]["gapReason"] = "模型残留的冗余解释"
|
||
|
||
result = run_writer_semantic_detector(
|
||
semantic_input(), model_runner=FakeRunner(draft), max_corrections=0
|
||
)
|
||
|
||
self.assertTrue(result["ok"])
|
||
bound_verdict = result["report"][field][0]
|
||
self.assertEqual(bound_verdict["verdict"], verdict)
|
||
self.assertNotIn("gapReason", bound_verdict)
|
||
|
||
def test_unknown_claim_and_hard_constraint_without_gap_reason_fail_closed(self) -> None:
|
||
# unknown 的解释不是冗余字段:缺失时仍须失败关闭,防止“未知”成为无理由逃生口。
|
||
cases = (
|
||
("claims", "coverageState"),
|
||
("hardConstraintVerdicts", "verdict"),
|
||
)
|
||
for field, state_field in cases:
|
||
with self.subTest(field=field):
|
||
draft = semantic_draft()
|
||
draft[field][0][state_field] = "unknown"
|
||
|
||
result = run_writer_semantic_detector(
|
||
semantic_input(), model_runner=FakeRunner(draft), max_corrections=0
|
||
)
|
||
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_MODEL_OUTPUT_INVALID")
|
||
self.assertEqual(result["safeDiagnostic"]["reasonCode"], "GAP_REASON_REQUIRED")
|
||
|
||
def test_old_v2_and_v1_reports_fail_closed(self) -> None:
|
||
payload = semantic_input()
|
||
for version in ("semantic-detector-report-v2", "semantic-detector-report-v1"):
|
||
with self.subTest(version=version), self.assertRaises(SemanticDetectorContractError):
|
||
validate_semantic_detector_report({"schemaVersion": version}, payload)
|
||
|
||
def test_input_hash_and_candidate_hash_fail_closed(self) -> None:
|
||
stale = semantic_input()
|
||
stale["candidateBody"] += "旧"
|
||
with self.assertRaises(SemanticDetectorContractError) as caught:
|
||
validate_semantic_detector_input(stale)
|
||
self.assertEqual(caught.exception.code, "SEMANTIC_DETECTOR_CANDIDATE_HASH_MISMATCH")
|
||
|
||
def test_nested_sources_freeze_and_content_hash_fail_closed(self) -> None:
|
||
raw_source = semantic_input()
|
||
raw_source["factEvidence"][0]["sourceRef"]["sourceId"] = "/private/tmp/raw/fact.json"
|
||
raw_source["inputSha256"] = canonical_sha256({k: v for k, v in raw_source.items() if k != "inputSha256"})
|
||
with self.assertRaises(SemanticDetectorContractError) as caught:
|
||
validate_semantic_detector_input(raw_source)
|
||
self.assertEqual(caught.exception.code, "SEMANTIC_DETECTOR_LEAKAGE_DETECTED")
|
||
|
||
future = semantic_input()
|
||
text = "未来正文"
|
||
future["proseEvidence"] = [{
|
||
"evidenceId": "prose-1", "chapter": 489,
|
||
"sourceRef": {"sourceId": "chapter:489", "sourceVersion": "v1", "chapter": 489},
|
||
"contentSha256": _hash_text(text), "purpose": "continuity", "text": text,
|
||
"isRecentBaseline": False,
|
||
}]
|
||
future["inputSha256"] = canonical_sha256({k: v for k, v in future.items() if k != "inputSha256"})
|
||
with self.assertRaises(SemanticDetectorContractError):
|
||
validate_semantic_detector_input(future)
|
||
|
||
nested_extra = semantic_input()
|
||
nested_extra["factEvidence"][0]["debug"] = True
|
||
nested_extra["inputSha256"] = canonical_sha256({k: v for k, v in nested_extra.items() if k != "inputSha256"})
|
||
with self.assertRaises(SemanticDetectorContractError):
|
||
validate_semantic_detector_input(nested_extra)
|
||
|
||
def test_missing_verdict_id_and_receipt_rebinding_fail_closed(self) -> None:
|
||
missing = semantic_draft()
|
||
missing["assertionVerdicts"] = []
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=FakeRunner(missing))
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_MODEL_OUTPUT_INVALID")
|
||
|
||
payload = semantic_input()
|
||
good = run_writer_semantic_detector(payload, model_runner=FakeRunner(semantic_draft()))["report"]
|
||
with self.assertRaises(SemanticDetectorContractError) as caught:
|
||
validate_semantic_detector_report(good, payload, model_receipt_sha256="sha256:" + "9" * 64)
|
||
self.assertEqual(caught.exception.code, "SEMANTIC_DETECTOR_RECEIPT_BINDING_MISMATCH")
|
||
|
||
def test_metrics_are_computed_from_bound_report(self) -> None:
|
||
payload = semantic_input()
|
||
result = run_writer_semantic_detector(payload, model_runner=FakeRunner(semantic_draft()))
|
||
self.assertEqual(calculate_semantic_metrics(result["report"], payload), {"highSeverityCount": 0, "hardConstraintCoverage": 1.0, "evidenceGapCount": 0})
|
||
|
||
def test_runner_must_supply_receipt_hash(self) -> None:
|
||
class DirectRunner:
|
||
def run(self, **_kwargs: Any) -> Mapping[str, Any]:
|
||
return semantic_draft()
|
||
|
||
result = run_writer_semantic_detector(semantic_input(), model_runner=DirectRunner())
|
||
self.assertFalse(result["ok"])
|
||
self.assertEqual(result["primaryCode"], "SEMANTIC_DETECTOR_RECEIPT_BINDING_MISMATCH")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|