340 lines
14 KiB
Python
340 lines
14 KiB
Python
#!/usr/bin/env python3
|
|
"""正文候选机械硬门;不调用模型、数据库或外部服务。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import pathlib
|
|
import sys
|
|
from typing import Any, Mapping, Sequence
|
|
|
|
SCRIPT_DIR = pathlib.Path(__file__).resolve().parent
|
|
READ_CONTEXT_DIR = SCRIPT_DIR.parents[1] / "read-context" / "scripts"
|
|
if str(READ_CONTEXT_DIR) not in sys.path:
|
|
sys.path.insert(0, str(READ_CONTEXT_DIR))
|
|
|
|
from writer_contract import ( # noqa: E402
|
|
ContractError,
|
|
han_count,
|
|
normalize_text,
|
|
validate_writer_context,
|
|
validate_writer_output,
|
|
)
|
|
|
|
|
|
SEMANTIC_CHECKS = (
|
|
"hard_event_semantics",
|
|
"claim_truth_alignment",
|
|
"knowledge_scope",
|
|
"ability_cost_consistency",
|
|
)
|
|
|
|
|
|
def _failure(code: str, message: str, **details: Any) -> dict[str, Any]:
|
|
"""生成稳定且可追踪的阻塞项。"""
|
|
|
|
return {"code": code, "message": message, **details}
|
|
|
|
|
|
def _anchors_present(body: str, anchors: Any) -> bool:
|
|
"""只执行确定性的文本锚点匹配,语义等价判断留给模型 detector。"""
|
|
|
|
return isinstance(anchors, list) and bool(anchors) and any(
|
|
isinstance(anchor, str) and anchor and normalize_text(anchor) in body for anchor in anchors
|
|
)
|
|
|
|
|
|
def _safe_sequence(value: Any) -> Sequence[Any]:
|
|
"""把非法数组降为空序列,随后由输入合同阻塞项统一报告。"""
|
|
|
|
return value if isinstance(value, list) else ()
|
|
|
|
|
|
def _precheck_candidate_integrity(candidate: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
"""在严格 schema 前检查候选与 claim 身份,保留精确失败码。"""
|
|
|
|
failures: list[dict[str, Any]] = []
|
|
body = candidate.get("candidateBody")
|
|
candidate_hash = candidate.get("candidateSha256")
|
|
if not isinstance(body, str):
|
|
return [_failure("OUTPUT_CONTRACT_INVALID", "candidateBody 必须是字符串")]
|
|
try:
|
|
normalized_body = normalize_text(body)
|
|
except ContractError as exc:
|
|
return [_failure("OUTPUT_CONTRACT_INVALID", str(exc))]
|
|
expected_hash = "sha256:" + hashlib.sha256(normalized_body.encode("utf-8")).hexdigest()
|
|
if candidate_hash != expected_hash:
|
|
failures.append(
|
|
_failure(
|
|
"CANDIDATE_HASH_MISMATCH",
|
|
"candidateSha256 与规范化候选正文不一致",
|
|
expectedSha256=expected_hash,
|
|
)
|
|
)
|
|
claims = candidate.get("claimLedger")
|
|
if not isinstance(claims, list):
|
|
failures.append(_failure("OUTPUT_CONTRACT_INVALID", "claimLedger 必须是数组"))
|
|
return failures
|
|
for index, claim in enumerate(claims):
|
|
if not isinstance(claim, Mapping):
|
|
failures.append(_failure("OUTPUT_CONTRACT_INVALID", "claim 必须是对象", claimIndex=index))
|
|
continue
|
|
if claim.get("candidateSha256") != expected_hash:
|
|
failures.append(
|
|
_failure(
|
|
"CLAIM_HASH_MISMATCH",
|
|
"claim 未绑定当前候选正文哈希",
|
|
claimIndex=index,
|
|
claimId=claim.get("claimId"),
|
|
)
|
|
)
|
|
start = claim.get("startCodePoint")
|
|
end = claim.get("endCodePoint")
|
|
if (
|
|
isinstance(start, bool)
|
|
or isinstance(end, bool)
|
|
or not isinstance(start, int)
|
|
or not isinstance(end, int)
|
|
or start < 0
|
|
or end <= start
|
|
or end > len(normalized_body)
|
|
):
|
|
failures.append(
|
|
_failure(
|
|
"CLAIM_UNICODE_OFFSET_INVALID",
|
|
"claim 偏移必须是规范正文的 Unicode code point 左闭右开区间",
|
|
claimIndex=index,
|
|
claimId=claim.get("claimId"),
|
|
)
|
|
)
|
|
return failures
|
|
|
|
|
|
def _check_context_binding(
|
|
context: Mapping[str, Any], candidate: Mapping[str, Any]
|
|
) -> list[dict[str, Any]]:
|
|
"""检查候选元数据和所有证据 ID 均来自本次冻结上下文。"""
|
|
|
|
failures: list[dict[str, Any]] = []
|
|
expected = {
|
|
"runId": context["runId"],
|
|
"attempt": context["attempt"],
|
|
"mode": context["mode"],
|
|
"qualityPolicyVersion": context["qualityPolicyVersion"],
|
|
"contextSnapshotId": context["contextSnapshot"]["manifestId"],
|
|
"contextSnapshotSha256": context["contextSnapshot"]["contextSha256"],
|
|
"acceptanceEligible": context["acceptanceEligible"],
|
|
}
|
|
mismatches = [field for field, value in expected.items() if candidate.get(field) != value]
|
|
if mismatches:
|
|
failures.append(
|
|
_failure("CONTEXT_BINDING_MISMATCH", "候选未绑定当前冻结上下文", fields=mismatches)
|
|
)
|
|
fact_ids = {item["evidenceId"] for item in context["factEvidence"]}
|
|
prose_ids = {item["evidenceId"] for item in context["proseEvidence"]}
|
|
for index, claim in enumerate(_safe_sequence(candidate.get("claimLedger"))):
|
|
if not isinstance(claim, Mapping):
|
|
continue
|
|
fact_id = claim.get("factEvidenceId")
|
|
prose_id = claim.get("proseEvidenceId")
|
|
if fact_id not in fact_ids or (prose_id is not None and prose_id not in prose_ids):
|
|
failures.append(
|
|
_failure(
|
|
"CLAIM_EVIDENCE_MISSING",
|
|
"claim 引用了当前上下文之外的事实或文风证据",
|
|
claimIndex=index,
|
|
claimId=claim.get("claimId"),
|
|
)
|
|
)
|
|
if claim.get("coverageState") in {"unsupported", "conflict"}:
|
|
failures.append(
|
|
_failure(
|
|
"FROZEN_FACT_CONFLICT",
|
|
"claim 使用了不受支持或互相冲突的冻结事实",
|
|
claimIndex=index,
|
|
claimId=claim.get("claimId"),
|
|
)
|
|
)
|
|
return failures
|
|
|
|
|
|
def _check_outline_anchors(body: str, requirements: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
"""机械检查硬事件、必须出场角色、伏笔动作和章末钩子。"""
|
|
|
|
failures: list[dict[str, Any]] = []
|
|
for item in _safe_sequence(requirements.get("requiredEvents")):
|
|
if not isinstance(item, Mapping) or not _anchors_present(body, item.get("anchors")):
|
|
failures.append(
|
|
_failure(
|
|
"HARD_EVENT_MISSING",
|
|
"候选未命中细纲硬事件锚点",
|
|
requirementId=item.get("requirementId") if isinstance(item, Mapping) else None,
|
|
)
|
|
)
|
|
for character in _safe_sequence(requirements.get("requiredCharacters")):
|
|
if not isinstance(character, str) or not character or normalize_text(character) not in body:
|
|
failures.append(
|
|
_failure("REQUIRED_CHARACTER_MISSING", "细纲要求的角色未出场", character=character)
|
|
)
|
|
for item in _safe_sequence(requirements.get("foreshadowingActions")):
|
|
if not isinstance(item, Mapping) or not _anchors_present(body, item.get("anchors")):
|
|
failures.append(
|
|
_failure(
|
|
"FORESHADOWING_ACTION_MISSING",
|
|
"候选未命中细纲伏笔动作锚点",
|
|
requirementId=item.get("requirementId") if isinstance(item, Mapping) else None,
|
|
)
|
|
)
|
|
hook = requirements.get("chapterEndHook")
|
|
if isinstance(hook, Mapping):
|
|
max_distance = hook.get("maxDistanceFromEnd")
|
|
if isinstance(max_distance, bool) or not isinstance(max_distance, int) or max_distance <= 0:
|
|
failures.append(_failure("DETECTOR_INPUT_INVALID", "章末钩子距离必须是正整数"))
|
|
else:
|
|
tail = body[-max_distance:]
|
|
if not _anchors_present(tail, hook.get("anchors")):
|
|
failures.append(
|
|
_failure(
|
|
"CHAPTER_END_HOOK_MISSING",
|
|
"候选结尾未命中细纲章末钩子",
|
|
requirementId=hook.get("requirementId"),
|
|
)
|
|
)
|
|
else:
|
|
failures.append(_failure("DETECTOR_INPUT_INVALID", "chapterEndHook 必须是对象"))
|
|
return failures
|
|
|
|
|
|
def _spans_overlap(left_start: int, left_end: int, right_start: int, right_end: int) -> bool:
|
|
"""判断两个 Unicode code point 左闭右开区间是否相交。"""
|
|
|
|
return left_start < right_end and right_start < left_end
|
|
|
|
|
|
def _check_new_settings_and_conflicts(
|
|
candidate: Mapping[str, Any], requirements: Mapping[str, Any]
|
|
) -> list[dict[str, Any]]:
|
|
"""检查可信实体识别结果是否有对应申报,并回显冻结冲突来源。"""
|
|
|
|
failures: list[dict[str, Any]] = []
|
|
declarations = [
|
|
item
|
|
for item in _safe_sequence(candidate.get("newSettingDeclarations"))
|
|
if isinstance(item, Mapping)
|
|
]
|
|
for setting in _safe_sequence(requirements.get("detectedNewSettings")):
|
|
if not isinstance(setting, Mapping):
|
|
failures.append(_failure("DETECTOR_INPUT_INVALID", "detectedNewSettings 项必须是对象"))
|
|
continue
|
|
start = setting.get("startCodePoint")
|
|
end = setting.get("endCodePoint")
|
|
matched = False
|
|
if isinstance(start, int) and not isinstance(start, bool) and isinstance(end, int) and not isinstance(end, bool):
|
|
matched = any(
|
|
declaration.get("factType") == setting.get("factType")
|
|
and isinstance(declaration.get("startCodePoint"), int)
|
|
and isinstance(declaration.get("endCodePoint"), int)
|
|
and _spans_overlap(
|
|
start,
|
|
end,
|
|
declaration["startCodePoint"],
|
|
declaration["endCodePoint"],
|
|
)
|
|
for declaration in declarations
|
|
)
|
|
if not matched:
|
|
failures.append(
|
|
_failure(
|
|
"NEW_SETTING_UNDECLARED",
|
|
"候选中的新设定没有对应申报",
|
|
settingId=setting.get("settingId"),
|
|
)
|
|
)
|
|
for conflict in _safe_sequence(requirements.get("frozenConflicts")):
|
|
if not isinstance(conflict, Mapping):
|
|
failures.append(_failure("DETECTOR_INPUT_INVALID", "frozenConflicts 项必须是对象"))
|
|
continue
|
|
failures.append(
|
|
_failure(
|
|
"FROZEN_FACT_CONFLICT",
|
|
str(conflict.get("message") or "候选与冻结事实冲突"),
|
|
conflictId=conflict.get("conflictId"),
|
|
sourceRef=conflict.get("sourceRef"),
|
|
)
|
|
)
|
|
return failures
|
|
|
|
|
|
def check_writer_candidate(
|
|
context: Mapping[str, Any], candidate: Mapping[str, Any], requirements: Mapping[str, Any]
|
|
) -> dict[str, Any]:
|
|
"""运行不含模型调用的机械硬门,并返回结构化 detector 报告。"""
|
|
|
|
failures: list[dict[str, Any]] = []
|
|
if not isinstance(context, Mapping) or not isinstance(candidate, Mapping) or not isinstance(requirements, Mapping):
|
|
failures.append(_failure("DETECTOR_INPUT_INVALID", "context、candidate、requirements 必须是对象"))
|
|
return {
|
|
"schemaVersion": "writer-detector-report-v1",
|
|
"passed": False,
|
|
"blockingFailures": failures,
|
|
"semanticReview": {"status": "pending", "checks": list(SEMANTIC_CHECKS)},
|
|
}
|
|
try:
|
|
normalized_context = validate_writer_context(context)
|
|
except ContractError as exc:
|
|
failures.append(_failure("CONTEXT_CONTRACT_INVALID", str(exc)))
|
|
normalized_context = None
|
|
failures.extend(_precheck_candidate_integrity(candidate))
|
|
try:
|
|
normalized_candidate = validate_writer_output(candidate)
|
|
except ContractError as exc:
|
|
failures.append(_failure("OUTPUT_CONTRACT_INVALID", str(exc)))
|
|
normalized_candidate = None
|
|
if normalized_context is not None:
|
|
failures.extend(_check_context_binding(normalized_context, candidate))
|
|
body = candidate.get("candidateBody")
|
|
if isinstance(body, str):
|
|
normalized_body = normalize_text(body)
|
|
if normalized_context is not None:
|
|
contract = normalized_context["outputContract"]
|
|
actual_han_chars = han_count(normalized_body)
|
|
if not contract["minChars"] <= actual_han_chars <= contract["maxChars"]:
|
|
failures.append(
|
|
_failure(
|
|
"CANDIDATE_LENGTH_OUT_OF_RANGE",
|
|
"候选正文汉字数超出动态篇幅合同",
|
|
actualHanChars=actual_han_chars,
|
|
minChars=contract["minChars"],
|
|
maxChars=contract["maxChars"],
|
|
targetChars=contract["targetChars"],
|
|
)
|
|
)
|
|
failures.extend(_check_outline_anchors(normalized_body, requirements))
|
|
failures.extend(_check_new_settings_and_conflicts(candidate, requirements))
|
|
# 按 code 与定位去重,避免严格合同与预检重复报告同一处问题。
|
|
unique: list[dict[str, Any]] = []
|
|
seen: set[tuple[Any, ...]] = set()
|
|
for item in failures:
|
|
key = (item.get("code"), item.get("claimIndex"), item.get("requirementId"), item.get("settingId"), item.get("conflictId"))
|
|
if key not in seen:
|
|
seen.add(key)
|
|
unique.append(item)
|
|
source = normalized_candidate or candidate
|
|
return {
|
|
"schemaVersion": "writer-detector-report-v1",
|
|
"runId": source.get("runId"),
|
|
"attempt": source.get("attempt"),
|
|
"candidateVersion": source.get("candidateVersion"),
|
|
"candidateSha256": source.get("candidateSha256"),
|
|
"passed": not unique,
|
|
"blockingFailures": unique,
|
|
"semanticReview": {
|
|
"status": "pending",
|
|
"checks": list(SEMANTIC_CHECKS),
|
|
"inputContract": "context + candidate + mechanicalReport -> semanticDetectorReport",
|
|
},
|
|
}
|
|
|
|
|
|
__all__ = ["SEMANTIC_CHECKS", "check_writer_candidate"]
|