619 lines
32 KiB
Python
619 lines
32 KiB
Python
#!/usr/bin/env python3
|
||
"""正文写手上下文与输出的严格合同。
|
||
|
||
本模块只处理纯数据,不读取文件、数据库或网络。所有进入写手的文本先做
|
||
Unicode NFC 与换行归一化,所有身份哈希都来自同一份规范 JSON。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import json
|
||
import math
|
||
import re
|
||
import unicodedata
|
||
from decimal import Decimal, ROUND_HALF_UP
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
|
||
CONTEXT_VERSION = "writer-context-v1"
|
||
OUTPUT_VERSION = "writer-output-v1"
|
||
PLAN_VERSION = "writer-retrieval-plan-v1"
|
||
MANIFEST_VERSION = "writer-retrieval-manifest-v1"
|
||
TIE_BREAK = "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC"
|
||
|
||
_HASH_RE = re.compile(r"^sha256:[0-9a-f]{64}$")
|
||
_RUN_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$")
|
||
_VOLATILE_IDENTITY_FIELDS = frozenset(
|
||
{"runId", "generatedAt", "timestamp", "executionNode", "contextSha256"}
|
||
)
|
||
|
||
# Unicode Script=Han 覆盖的标准区段。兼容表意文字也按“汉字”计数,
|
||
# 但标点、Markdown、拉丁字母和数字不会落入这些区段。
|
||
_HAN_RANGES = (
|
||
(0x3400, 0x4DBF),
|
||
(0x4E00, 0x9FFF),
|
||
(0xF900, 0xFAFF),
|
||
(0x20000, 0x2EBEF),
|
||
(0x2F800, 0x2FA1F),
|
||
(0x30000, 0x323AF),
|
||
)
|
||
|
||
|
||
class ContractError(ValueError):
|
||
"""输入不符合严格合同时抛出,调用方必须失败关闭。"""
|
||
|
||
|
||
def normalize_text(value: str) -> str:
|
||
"""把文本统一为 NFC 与 LF,供哈希和 Unicode 偏移共同使用。"""
|
||
|
||
if not isinstance(value, str):
|
||
raise ContractError("待归一化文本必须是字符串")
|
||
return unicodedata.normalize("NFC", value.replace("\r\n", "\n").replace("\r", "\n"))
|
||
|
||
|
||
def _normalize_json(value: Any) -> Any:
|
||
"""递归归一化 JSON 值,并拒绝 JSON 之外或不可复现的数值。"""
|
||
|
||
if isinstance(value, str):
|
||
return normalize_text(value)
|
||
if value is None or isinstance(value, (bool, int)):
|
||
return value
|
||
if isinstance(value, float):
|
||
if not math.isfinite(value):
|
||
raise ContractError("规范 JSON 不允许 NaN 或 Infinity")
|
||
return value
|
||
if isinstance(value, list):
|
||
return [_normalize_json(item) for item in value]
|
||
if isinstance(value, tuple):
|
||
return [_normalize_json(item) for item in value]
|
||
if isinstance(value, Mapping):
|
||
result: dict[str, Any] = {}
|
||
for raw_key, item in value.items():
|
||
if not isinstance(raw_key, str):
|
||
raise ContractError("规范 JSON 的对象键必须是字符串")
|
||
key = normalize_text(raw_key)
|
||
if key in result:
|
||
raise ContractError(f"对象键在 NFC 归一化后冲突: {key}")
|
||
result[key] = _normalize_json(item)
|
||
return result
|
||
raise ContractError(f"值不是 JSON 类型: {type(value).__name__}")
|
||
|
||
|
||
def canonical_json(value: Any) -> str:
|
||
"""输出 UTF-8 语义、键排序、无额外空白的规范 JSON 文本。"""
|
||
|
||
return json.dumps(
|
||
_normalize_json(value),
|
||
ensure_ascii=False,
|
||
sort_keys=True,
|
||
separators=(",", ":"),
|
||
allow_nan=False,
|
||
)
|
||
|
||
|
||
def _without_volatile_fields(value: Any) -> Any:
|
||
"""递归移除运行元数据,防止同一检索输入得到不同身份。"""
|
||
|
||
if isinstance(value, list):
|
||
return [_without_volatile_fields(item) for item in value]
|
||
if isinstance(value, Mapping):
|
||
return {
|
||
key: _without_volatile_fields(item)
|
||
for key, item in value.items()
|
||
if key not in _VOLATILE_IDENTITY_FIELDS
|
||
}
|
||
return value
|
||
|
||
|
||
def retrieval_identity(value: Any) -> str:
|
||
"""计算计划、manifest 或上下文的稳定 SHA-256 身份。"""
|
||
|
||
encoded = canonical_json(_without_volatile_fields(value)).encode("utf-8")
|
||
return "sha256:" + hashlib.sha256(encoded).hexdigest()
|
||
|
||
|
||
def han_count(value: str) -> int:
|
||
"""统计 Unicode Han code point,不把标点或 Markdown 算入正文长度。"""
|
||
|
||
text = normalize_text(value)
|
||
return sum(any(start <= ord(char) <= end for start, end in _HAN_RANGES) for char in text)
|
||
|
||
|
||
def _round_half_up(value: Decimal) -> int:
|
||
"""以十进制 ROUND_HALF_UP 规则取整,避免 Python 银行家舍入。"""
|
||
|
||
return int(value.quantize(Decimal("1"), rounding=ROUND_HALF_UP))
|
||
|
||
|
||
def _round_to_100(value: Decimal) -> int:
|
||
"""以百字为单位执行十进制半入取整。"""
|
||
|
||
return int((value / Decimal(100)).quantize(Decimal("1"), rounding=ROUND_HALF_UP) * 100)
|
||
|
||
|
||
def calculate_target_chars(
|
||
*,
|
||
explicit_target_chars: int | None = None,
|
||
recent_chapter_han_counts: Sequence[int] = (),
|
||
default_target_chars: int = 4000,
|
||
hard_event_count: int = 3,
|
||
foreshadowing_action_count: int = 0,
|
||
required_scene_count: int = 0,
|
||
min_chars: int = 2000,
|
||
max_chars: int = 10000,
|
||
) -> int:
|
||
"""按冻结历史中位数与细纲密度计算确定性目标汉字数。"""
|
||
|
||
counts = {
|
||
"default_target_chars": default_target_chars,
|
||
"hard_event_count": hard_event_count,
|
||
"foreshadowing_action_count": foreshadowing_action_count,
|
||
"required_scene_count": required_scene_count,
|
||
"min_chars": min_chars,
|
||
"max_chars": max_chars,
|
||
}
|
||
if any(isinstance(value, bool) or not isinstance(value, int) for value in counts.values()):
|
||
raise ContractError("篇幅参数必须是整数")
|
||
if default_target_chars <= 0 or min_chars <= 0 or max_chars < min_chars or any(value < 0 for key, value in counts.items() if "count" in key):
|
||
raise ContractError("篇幅边界或细纲计数非法")
|
||
if explicit_target_chars is not None:
|
||
if isinstance(explicit_target_chars, bool) or not isinstance(explicit_target_chars, int):
|
||
raise ContractError("显式 targetChars 必须是整数")
|
||
return min(max(explicit_target_chars, min_chars), max_chars)
|
||
|
||
valid_counts = list(recent_chapter_han_counts)[-20:]
|
||
if any(isinstance(value, bool) or not isinstance(value, int) or value < 500 for value in valid_counts):
|
||
raise ContractError("历史章汉字数必须是大于等于 500 的整数")
|
||
if len(valid_counts) >= 3:
|
||
ordered_counts = sorted(valid_counts)
|
||
midpoint = len(ordered_counts) // 2
|
||
if len(ordered_counts) % 2:
|
||
baseline = Decimal(ordered_counts[midpoint])
|
||
else:
|
||
baseline = (Decimal(ordered_counts[midpoint - 1]) + Decimal(ordered_counts[midpoint])) / 2
|
||
else:
|
||
baseline = Decimal(default_target_chars)
|
||
density = (
|
||
Decimal(hard_event_count)
|
||
+ Decimal("0.5") * foreshadowing_action_count
|
||
+ Decimal("0.5") * required_scene_count
|
||
)
|
||
factor = min(Decimal("1.15"), max(Decimal("0.85"), Decimal("0.85") + Decimal("0.05") * (density - 3)))
|
||
target = _round_to_100(Decimal(_round_half_up(baseline)) * factor)
|
||
return min(max(target, min_chars), max_chars)
|
||
|
||
|
||
def _object(
|
||
value: Any,
|
||
path: str,
|
||
required: set[str] | frozenset[str],
|
||
optional: set[str] | frozenset[str] = frozenset(),
|
||
) -> Mapping[str, Any]:
|
||
"""校验严格对象,任何缺字段或未知字段都立即失败。"""
|
||
|
||
if not isinstance(value, Mapping):
|
||
raise ContractError(f"{path} 必须是对象")
|
||
missing = sorted(required - set(value))
|
||
unknown = sorted(set(value) - required - optional)
|
||
if missing:
|
||
raise ContractError(f"{path} 缺少字段: {','.join(missing)}")
|
||
if unknown:
|
||
raise ContractError(f"{path} 包含未知字段: {','.join(unknown)}")
|
||
return value
|
||
|
||
|
||
def _string(value: Any, path: str, *, nonempty: bool = True) -> str:
|
||
"""校验字符串,并在需要时拒绝空值。"""
|
||
|
||
if not isinstance(value, str) or (nonempty and not value.strip()):
|
||
raise ContractError(f"{path} 必须是非空字符串")
|
||
if value != normalize_text(value):
|
||
raise ContractError(f"{path} 必须预先归一化为 Unicode NFC/LF")
|
||
return value
|
||
|
||
|
||
def _integer(value: Any, path: str, *, minimum: int = 0) -> int:
|
||
"""校验整数,显式排除 bool 这一 Python int 子类。"""
|
||
|
||
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
|
||
raise ContractError(f"{path} 必须是大于等于 {minimum} 的整数")
|
||
return value
|
||
|
||
|
||
def _boolean(value: Any, path: str) -> bool:
|
||
"""校验严格布尔值。"""
|
||
|
||
if not isinstance(value, bool):
|
||
raise ContractError(f"{path} 必须是布尔值")
|
||
return value
|
||
|
||
|
||
def _array(value: Any, path: str) -> list[Any]:
|
||
"""校验数组并返回原值,拒绝元组等隐式转换。"""
|
||
|
||
if not isinstance(value, list):
|
||
raise ContractError(f"{path} 必须是数组")
|
||
return value
|
||
|
||
|
||
def _hash(value: Any, path: str) -> str:
|
||
"""校验带算法前缀的 SHA-256。"""
|
||
|
||
text = _string(value, path)
|
||
if not _HASH_RE.fullmatch(text):
|
||
raise ContractError(f"{path} 必须是 sha256: 加 64 位小写十六进制")
|
||
return text
|
||
|
||
|
||
def _source_ref(value: Any, path: str) -> None:
|
||
"""校验不可变来源引用;历史原文可额外携带块和字符区间。"""
|
||
|
||
ref = _object(
|
||
value,
|
||
path,
|
||
frozenset({"sourceId", "sourceVersion"}),
|
||
frozenset({"chapter", "blockId", "startCodePoint", "endCodePoint", "contentSha256", "sourceType"}),
|
||
)
|
||
_string(ref["sourceId"], f"{path}.sourceId")
|
||
_string(ref["sourceVersion"], f"{path}.sourceVersion")
|
||
for field in ("chapter", "blockId", "startCodePoint", "endCodePoint"):
|
||
if field in ref:
|
||
_integer(ref[field], f"{path}.{field}", minimum=0 if "CodePoint" in field else 1)
|
||
if "startCodePoint" in ref and "endCodePoint" in ref and ref["endCodePoint"] <= ref["startCodePoint"]:
|
||
raise ContractError(f"{path} 字符区间必须是非空左闭右开区间")
|
||
if "contentSha256" in ref:
|
||
_hash(ref["contentSha256"], f"{path}.contentSha256")
|
||
if "sourceType" in ref:
|
||
_string(ref["sourceType"], f"{path}.sourceType")
|
||
|
||
|
||
def _validate_plan(value: Any, path: str) -> None:
|
||
"""校验固定检索计划,不允许写手临场扩张查询。"""
|
||
|
||
plan = _object(
|
||
value,
|
||
path,
|
||
frozenset(
|
||
{"planVersion", "planId", "runId", "asOf", "queries", "cardIndexVersion", "proseIndexVersion", "filters", "tieBreak", "tokenBudget"}
|
||
),
|
||
)
|
||
if plan["planVersion"] != PLAN_VERSION:
|
||
raise ContractError(f"{path}.planVersion 版本不支持")
|
||
_hash(plan["planId"], f"{path}.planId")
|
||
if not _RUN_ID_RE.fullmatch(_string(plan["runId"], f"{path}.runId")):
|
||
raise ContractError(f"{path}.runId 格式非法")
|
||
_integer(plan["asOf"], f"{path}.asOf", minimum=1)
|
||
for index, query in enumerate(_array(plan["queries"], f"{path}.queries")):
|
||
item = _object(query, f"{path}.queries[{index}]", frozenset({"queryId", "text", "entityTypes", "purpose", "topK"}))
|
||
_string(item["queryId"], f"{path}.queries[{index}].queryId")
|
||
_string(item["text"], f"{path}.queries[{index}].text")
|
||
if any(not isinstance(kind, str) or not kind for kind in _array(item["entityTypes"], f"{path}.queries[{index}].entityTypes")):
|
||
raise ContractError(f"{path}.queries[{index}].entityTypes 必须是非空字符串数组")
|
||
_string(item["purpose"], f"{path}.queries[{index}].purpose")
|
||
_integer(item["topK"], f"{path}.queries[{index}].topK", minimum=1)
|
||
_string(plan["cardIndexVersion"], f"{path}.cardIndexVersion")
|
||
_string(plan["proseIndexVersion"], f"{path}.proseIndexVersion")
|
||
filters = _object(plan["filters"], f"{path}.filters", frozenset({"workId", "asOfChapter", "sourceStatus", "authorizationRequired"}))
|
||
_integer(filters["workId"], f"{path}.filters.workId", minimum=1)
|
||
_integer(filters["asOfChapter"], f"{path}.filters.asOfChapter", minimum=1)
|
||
_string(filters["sourceStatus"], f"{path}.filters.sourceStatus")
|
||
_boolean(filters["authorizationRequired"], f"{path}.filters.authorizationRequired")
|
||
if plan["tieBreak"] != TIE_BREAK:
|
||
raise ContractError(f"{path}.tieBreak 不符合稳定排序合同")
|
||
budget = _object(plan["tokenBudget"], f"{path}.tokenBudget", frozenset({"maxContextChars"}), frozenset({"cardChars", "recentProseChars", "historicalProseChars", "patternChars"}))
|
||
for key, item in budget.items():
|
||
_integer(item, f"{path}.tokenBudget.{key}", minimum=0)
|
||
identity_payload = {key: item for key, item in plan.items() if key != "planId"}
|
||
if plan["planId"] != retrieval_identity(identity_payload):
|
||
raise ContractError(f"{path}.planId 与计划内容不匹配")
|
||
|
||
|
||
def _validate_manifest(value: Any, path: str) -> None:
|
||
"""校验检索清单的来源与裁剪回显。"""
|
||
|
||
manifest = _object(value, path, frozenset({"manifestVersion", "manifestId", "planId", "sources", "omittedSources"}))
|
||
if manifest["manifestVersion"] != MANIFEST_VERSION:
|
||
raise ContractError(f"{path}.manifestVersion 版本不支持")
|
||
_hash(manifest["manifestId"], f"{path}.manifestId")
|
||
_hash(manifest["planId"], f"{path}.planId")
|
||
for index, source in enumerate(_array(manifest["sources"], f"{path}.sources")):
|
||
_source_ref(source, f"{path}.sources[{index}]")
|
||
for index, omitted in enumerate(_array(manifest["omittedSources"], f"{path}.omittedSources")):
|
||
item = _object(omitted, f"{path}.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
|
||
_string(item["sourceId"], f"{path}.omittedSources[{index}].sourceId")
|
||
_string(item["reason"], f"{path}.omittedSources[{index}].reason")
|
||
identity_payload = {key: item for key, item in manifest.items() if key != "manifestId"}
|
||
if manifest["manifestId"] != retrieval_identity(identity_payload):
|
||
raise ContractError(f"{path}.manifestId 与来源清单不匹配")
|
||
|
||
|
||
def validate_writer_context(value: Any) -> dict[str, Any]:
|
||
"""校验 WriterContext v1;成功时返回可安全复制的规范 JSON 对象。"""
|
||
|
||
required = frozenset(
|
||
{
|
||
"schemaVersion", "runId", "attempt", "mode", "purpose", "qualityPolicyVersion",
|
||
"workId", "targetChapter", "asOf", "contextSnapshot", "sourceVersion",
|
||
"authorizationSnapshot", "sourceStatus", "retrievalPlan", "retrievalManifest",
|
||
"fineOutline", "narrativeState", "factEvidence", "proseEvidence",
|
||
"patternReferences", "evidenceCoverage", "outputContract", "tokenBudget",
|
||
"omittedSources", "acceptanceEligible",
|
||
}
|
||
)
|
||
context = _object(
|
||
value,
|
||
"$",
|
||
required,
|
||
frozenset({"evidenceStrategy", "indexHints"}),
|
||
)
|
||
if context["schemaVersion"] != CONTEXT_VERSION:
|
||
raise ContractError("$.schemaVersion 版本不支持")
|
||
run_id = _string(context["runId"], "$.runId")
|
||
if not _RUN_ID_RE.fullmatch(run_id):
|
||
raise ContractError("$.runId 格式非法")
|
||
_integer(context["attempt"], "$.attempt", minimum=1)
|
||
if context["mode"] not in {"production", "diagnostic_only"}:
|
||
raise ContractError("$.mode 枚举非法")
|
||
if context["purpose"] not in {"production", "evaluation", "diagnostic"}:
|
||
raise ContractError("$.purpose 枚举非法")
|
||
evidence_strategy = context.get("evidenceStrategy", "production_dual_evidence")
|
||
if evidence_strategy not in {
|
||
"production_dual_evidence",
|
||
"historical_prose_only",
|
||
"card_index_only",
|
||
"card_index_plus_prose",
|
||
}:
|
||
raise ContractError("$.evidenceStrategy 枚举非法")
|
||
if context["mode"] == "production" and evidence_strategy != "production_dual_evidence":
|
||
raise ContractError("生产上下文必须使用 production_dual_evidence")
|
||
_string(context["qualityPolicyVersion"], "$.qualityPolicyVersion")
|
||
_integer(context["workId"], "$.workId", minimum=1)
|
||
target = _integer(context["targetChapter"], "$.targetChapter", minimum=1)
|
||
as_of = _integer(context["asOf"], "$.asOf", minimum=1)
|
||
if as_of >= target:
|
||
raise ContractError("$.asOf 必须早于 targetChapter")
|
||
|
||
snapshot = _object(context["contextSnapshot"], "$.contextSnapshot", frozenset({"manifestId", "contextSha256", "generatedAt"}))
|
||
_hash(snapshot["manifestId"], "$.contextSnapshot.manifestId")
|
||
_hash(snapshot["contextSha256"], "$.contextSnapshot.contextSha256")
|
||
_string(snapshot["generatedAt"], "$.contextSnapshot.generatedAt")
|
||
_string(context["sourceVersion"], "$.sourceVersion")
|
||
authorization = _object(
|
||
context["authorizationSnapshot"],
|
||
"$.authorizationSnapshot",
|
||
frozenset({"snapshotId", "allowedPurpose", "verifiedAt"}),
|
||
frozenset({"expiresAt", "sourceVersion"}),
|
||
)
|
||
for key, item in authorization.items():
|
||
_string(item, f"$.authorizationSnapshot.{key}")
|
||
if context["sourceStatus"] not in {"active", "authorized", "frozen_authorized"}:
|
||
raise ContractError("$.sourceStatus 不允许生成")
|
||
|
||
_validate_plan(context["retrievalPlan"], "$.retrievalPlan")
|
||
_validate_manifest(context["retrievalManifest"], "$.retrievalManifest")
|
||
if context["retrievalPlan"]["runId"] != run_id or context["retrievalPlan"]["asOf"] != as_of:
|
||
raise ContractError("检索计划与上下文的 runId/asOf 不一致")
|
||
if context["retrievalManifest"]["planId"] != context["retrievalPlan"]["planId"]:
|
||
raise ContractError("检索 manifest 未绑定当前计划")
|
||
if context["contextSnapshot"]["manifestId"] != context["retrievalManifest"]["manifestId"]:
|
||
raise ContractError("上下文快照未绑定当前 manifest")
|
||
|
||
outline = _object(context["fineOutline"], "$.fineOutline", frozenset({"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"}))
|
||
_source_ref(outline["sourceRef"], "$.fineOutline.sourceRef")
|
||
for field in ("hardConstraints", "adjustableBeats"):
|
||
if any(not isinstance(item, str) or not item for item in _array(outline[field], f"$.fineOutline.{field}")):
|
||
raise ContractError(f"$.fineOutline.{field} 必须是非空字符串数组")
|
||
for index, fact in enumerate(_array(outline["declaredNewFacts"], "$.fineOutline.declaredNewFacts")):
|
||
item = _object(fact, f"$.fineOutline.declaredNewFacts[{index}]", frozenset({"factId", "text", "sourceRef"}))
|
||
_string(item["factId"], f"$.fineOutline.declaredNewFacts[{index}].factId")
|
||
_string(item["text"], f"$.fineOutline.declaredNewFacts[{index}].text")
|
||
_source_ref(item["sourceRef"], f"$.fineOutline.declaredNewFacts[{index}].sourceRef")
|
||
|
||
state = _object(context["narrativeState"], "$.narrativeState", frozenset({"time", "location", "characterPositions", "immediateSituation"}))
|
||
for field in ("time", "location", "immediateSituation"):
|
||
_string(state[field], f"$.narrativeState.{field}", nonempty=False)
|
||
positions = _object(state["characterPositions"], "$.narrativeState.characterPositions", frozenset(state["characterPositions"].keys()) if isinstance(state["characterPositions"], Mapping) else frozenset())
|
||
for key, item in positions.items():
|
||
_string(key, "$.narrativeState.characterPositions.<key>")
|
||
_string(item, f"$.narrativeState.characterPositions.{key}")
|
||
|
||
for index, evidence in enumerate(_array(context["factEvidence"], "$.factEvidence")):
|
||
item = _object(evidence, f"$.factEvidence[{index}]", frozenset({"evidenceId", "fact", "sourceType", "sourceRef", "contentSha256", "riskLevel"}))
|
||
for field in ("evidenceId", "fact", "sourceType"):
|
||
_string(item[field], f"$.factEvidence[{index}].{field}")
|
||
if item["sourceType"] not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}:
|
||
raise ContractError(f"$.factEvidence[{index}].sourceType 枚举非法")
|
||
_source_ref(item["sourceRef"], f"$.factEvidence[{index}].sourceRef")
|
||
if item["sourceType"] == "historical_prose":
|
||
required_location = {"chapter", "blockId", "startCodePoint", "endCodePoint"}
|
||
if not required_location.issubset(item["sourceRef"]):
|
||
raise ContractError(f"$.factEvidence[{index}] 历史事实必须回到章、块和字符区间")
|
||
_hash(item["contentSha256"], f"$.factEvidence[{index}].contentSha256")
|
||
if item["riskLevel"] not in {"low", "medium", "high"}:
|
||
raise ContractError(f"$.factEvidence[{index}].riskLevel 枚举非法")
|
||
|
||
for index, evidence in enumerate(_array(context["proseEvidence"], "$.proseEvidence")):
|
||
item = _object(evidence, f"$.proseEvidence[{index}]", frozenset({"evidenceId", "chapter", "sourceRef", "contentSha256", "purpose", "text", "isRecentBaseline"}))
|
||
_string(item["evidenceId"], f"$.proseEvidence[{index}].evidenceId")
|
||
chapter = _integer(item["chapter"], f"$.proseEvidence[{index}].chapter", minimum=1)
|
||
if chapter > as_of:
|
||
raise ContractError(f"$.proseEvidence[{index}] 超出冻结线")
|
||
_source_ref(item["sourceRef"], f"$.proseEvidence[{index}].sourceRef")
|
||
if not {"chapter", "blockId", "startCodePoint", "endCodePoint"}.issubset(item["sourceRef"]):
|
||
raise ContractError(f"$.proseEvidence[{index}] 必须带章、块和字符区间")
|
||
_hash(item["contentSha256"], f"$.proseEvidence[{index}].contentSha256")
|
||
_string(item["purpose"], f"$.proseEvidence[{index}].purpose")
|
||
text = _string(item["text"], f"$.proseEvidence[{index}].text")
|
||
_boolean(item["isRecentBaseline"], f"$.proseEvidence[{index}].isRecentBaseline")
|
||
expected = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||
if item["contentSha256"] != expected:
|
||
raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配")
|
||
|
||
index_hints = _array(context.get("indexHints", []), "$.indexHints")
|
||
for index, hint in enumerate(index_hints):
|
||
item = _object(
|
||
hint,
|
||
f"$.indexHints[{index}]",
|
||
frozenset(
|
||
{
|
||
"cardId",
|
||
"name",
|
||
"type",
|
||
"content",
|
||
"sourceId",
|
||
"sourceVersion",
|
||
"asOf",
|
||
}
|
||
),
|
||
)
|
||
for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion"):
|
||
_string(item[field], f"$.indexHints[{index}].{field}")
|
||
hint_as_of = _integer(item["asOf"], f"$.indexHints[{index}].asOf", minimum=1)
|
||
if hint_as_of > as_of:
|
||
raise ContractError(f"$.indexHints[{index}] 超出冻结线")
|
||
if index_hints and not (
|
||
context["mode"] == "diagnostic_only"
|
||
and context["purpose"] in {"evaluation", "diagnostic"}
|
||
and context["acceptanceEligible"] is False
|
||
):
|
||
raise ContractError("indexHints 只允许用于不可接受的评测或诊断上下文")
|
||
if context["mode"] == "production" and index_hints:
|
||
raise ContractError("生产上下文禁止 indexHints")
|
||
if evidence_strategy == "card_index_only":
|
||
if context["mode"] != "diagnostic_only" or context["purpose"] not in {"evaluation", "diagnostic"}:
|
||
raise ContractError("card_index_only 只允许用于诊断上下文")
|
||
if context["factEvidence"] or context["proseEvidence"] or not index_hints:
|
||
raise ContractError("card_index_only 必须仅包含 indexHints")
|
||
elif evidence_strategy == "historical_prose_only":
|
||
if index_hints or context["factEvidence"] or not context["proseEvidence"]:
|
||
raise ContractError("historical_prose_only 必须仅包含历史原文")
|
||
elif evidence_strategy == "card_index_plus_prose":
|
||
if context["factEvidence"] or not index_hints or not context["proseEvidence"]:
|
||
raise ContractError("card_index_plus_prose 必须包含 indexHints 与历史原文")
|
||
|
||
# v1 生产正文必须携带截至冻结点的连续前四章。诊断 A/C 保持同一
|
||
# 原文基线,只有显式 card_index_only 策略获准跳过该生产要求。
|
||
if evidence_strategy != "card_index_only":
|
||
baseline_chapters = sorted(
|
||
item["chapter"]
|
||
for item in context["proseEvidence"]
|
||
if item["isRecentBaseline"] is True
|
||
)
|
||
expected_baseline = list(range(max(1, as_of - 3), as_of + 1))
|
||
if baseline_chapters != expected_baseline:
|
||
raise ContractError("原文证据必须包含截至冻结点的连续前四章基线")
|
||
|
||
for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")):
|
||
_source_ref(reference, f"$.patternReferences[{index}]")
|
||
for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")):
|
||
item = _object(coverage, f"$.evidenceCoverage[{index}]", frozenset({"elementId", "elementType", "name", "status", "factEvidenceIds", "proseEvidenceIds", "gapReason"}))
|
||
for field in ("elementId", "elementType", "name", "gapReason"):
|
||
_string(item[field], f"$.evidenceCoverage[{index}].{field}", nonempty=field != "gapReason")
|
||
if item["status"] not in {"supported", "declared_new", "card_gap", "style_gap", "unsupported", "conflict"}:
|
||
raise ContractError(f"$.evidenceCoverage[{index}].status 枚举非法")
|
||
for field in ("factEvidenceIds", "proseEvidenceIds"):
|
||
if any(not isinstance(item_id, str) or not item_id for item_id in _array(item[field], f"$.evidenceCoverage[{index}].{field}")):
|
||
raise ContractError(f"$.evidenceCoverage[{index}].{field} 必须是字符串数组")
|
||
|
||
output = _object(context["outputContract"], "$.outputContract", frozenset({"targetChars", "minChars", "maxChars", "frontmatterRequired", "newSettingDeclarationRequired"}))
|
||
for field in ("targetChars", "minChars", "maxChars"):
|
||
_integer(output[field], f"$.outputContract.{field}", minimum=1)
|
||
if not output["minChars"] <= output["targetChars"] <= output["maxChars"]:
|
||
raise ContractError("$.outputContract 篇幅范围不包含目标值")
|
||
_boolean(output["frontmatterRequired"], "$.outputContract.frontmatterRequired")
|
||
_boolean(output["newSettingDeclarationRequired"], "$.outputContract.newSettingDeclarationRequired")
|
||
budget = _object(context["tokenBudget"], "$.tokenBudget", frozenset({"maxContextChars", "usedContextChars"}))
|
||
for field in budget:
|
||
_integer(budget[field], f"$.tokenBudget.{field}", minimum=0)
|
||
if budget["usedContextChars"] > budget["maxContextChars"]:
|
||
raise ContractError("$.tokenBudget 已超预算")
|
||
for index, omitted in enumerate(_array(context["omittedSources"], "$.omittedSources")):
|
||
item = _object(omitted, f"$.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
|
||
_string(item["sourceId"], f"$.omittedSources[{index}].sourceId")
|
||
_string(item["reason"], f"$.omittedSources[{index}].reason")
|
||
|
||
eligible = _boolean(context["acceptanceEligible"], "$.acceptanceEligible")
|
||
if context["purpose"] in {"evaluation", "diagnostic"} or context["mode"] == "diagnostic_only":
|
||
if eligible:
|
||
raise ContractError("评测或诊断上下文必须 acceptanceEligible=false")
|
||
elif context["qualityPolicyVersion"] != "writer-production-v1":
|
||
raise ContractError("生产上下文必须绑定 writer-production-v1")
|
||
if context["contextSnapshot"]["contextSha256"] != retrieval_identity(context):
|
||
raise ContractError("$.contextSnapshot.contextSha256 与上下文内容不匹配")
|
||
return json.loads(canonical_json(context))
|
||
|
||
|
||
def validate_writer_output(value: Any) -> dict[str, Any]:
|
||
"""校验 WriterOutput v1、正文哈希和全部来源引用。"""
|
||
|
||
output = _object(
|
||
value,
|
||
"$",
|
||
frozenset(
|
||
{"schemaVersion", "runId", "attempt", "mode", "qualityPolicyVersion", "contextSnapshotId", "contextSnapshotSha256", "candidateVersion", "candidateSha256", "acceptanceEligible", "candidateBody", "claimLedger", "evidenceRequests", "newSettingDeclarations", "selfCheck"}
|
||
),
|
||
)
|
||
if output["schemaVersion"] != OUTPUT_VERSION:
|
||
raise ContractError("$.schemaVersion 版本不支持")
|
||
if not _RUN_ID_RE.fullmatch(_string(output["runId"], "$.runId")):
|
||
raise ContractError("$.runId 格式非法")
|
||
_integer(output["attempt"], "$.attempt", minimum=1)
|
||
if output["mode"] not in {"production", "diagnostic_only"}:
|
||
raise ContractError("$.mode 枚举非法")
|
||
_string(output["qualityPolicyVersion"], "$.qualityPolicyVersion")
|
||
_hash(output["contextSnapshotId"], "$.contextSnapshotId")
|
||
_hash(output["contextSnapshotSha256"], "$.contextSnapshotSha256")
|
||
_integer(output["candidateVersion"], "$.candidateVersion", minimum=1)
|
||
body = _string(output["candidateBody"], "$.candidateBody")
|
||
candidate_hash = _hash(output["candidateSha256"], "$.candidateSha256")
|
||
expected_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
|
||
if candidate_hash != expected_hash:
|
||
raise ContractError("$.candidateSha256 与规范正文不匹配")
|
||
eligible = _boolean(output["acceptanceEligible"], "$.acceptanceEligible")
|
||
if output["mode"] == "diagnostic_only" and eligible:
|
||
raise ContractError("诊断输出必须 acceptanceEligible=false")
|
||
|
||
for index, claim in enumerate(_array(output["claimLedger"], "$.claimLedger")):
|
||
item = _object(
|
||
claim,
|
||
f"$.claimLedger[{index}]",
|
||
frozenset({"claimId", "candidateSha256", "startCodePoint", "endCodePoint", "factType", "factEvidenceId", "coverageState"}),
|
||
frozenset({"proseEvidenceId"}),
|
||
)
|
||
for field in ("claimId", "factType", "factEvidenceId", "coverageState"):
|
||
_string(item[field], f"$.claimLedger[{index}].{field}")
|
||
_hash(item["candidateSha256"], f"$.claimLedger[{index}].candidateSha256")
|
||
if item["candidateSha256"] != candidate_hash:
|
||
raise ContractError(f"$.claimLedger[{index}] 未绑定当前候选")
|
||
start = _integer(item["startCodePoint"], f"$.claimLedger[{index}].startCodePoint")
|
||
end = _integer(item["endCodePoint"], f"$.claimLedger[{index}].endCodePoint", minimum=1)
|
||
if end <= start or end > len(body):
|
||
raise ContractError(f"$.claimLedger[{index}] Unicode 偏移越界")
|
||
if "proseEvidenceId" in item and item["proseEvidenceId"] is not None:
|
||
_string(item["proseEvidenceId"], f"$.claimLedger[{index}].proseEvidenceId")
|
||
for index, request in enumerate(_array(output["evidenceRequests"], "$.evidenceRequests")):
|
||
item = _object(request, f"$.evidenceRequests[{index}]", frozenset({"requestId", "query", "reason", "priority"}))
|
||
for field in ("requestId", "query", "reason", "priority"):
|
||
_string(item[field], f"$.evidenceRequests[{index}].{field}")
|
||
for index, declaration in enumerate(_array(output["newSettingDeclarations"], "$.newSettingDeclarations")):
|
||
item = _object(declaration, f"$.newSettingDeclarations[{index}]", frozenset({"declarationId", "factType", "text", "startCodePoint", "endCodePoint"}))
|
||
for field in ("declarationId", "factType", "text"):
|
||
_string(item[field], f"$.newSettingDeclarations[{index}].{field}")
|
||
start = _integer(item["startCodePoint"], f"$.newSettingDeclarations[{index}].startCodePoint")
|
||
end = _integer(item["endCodePoint"], f"$.newSettingDeclarations[{index}].endCodePoint", minimum=1)
|
||
if end <= start or end > len(body):
|
||
raise ContractError(f"$.newSettingDeclarations[{index}] Unicode 偏移越界")
|
||
self_check = _object(output["selfCheck"], "$.selfCheck", frozenset({"hardConstraintsCovered", "notes"}))
|
||
_boolean(self_check["hardConstraintsCovered"], "$.selfCheck.hardConstraintsCovered")
|
||
if any(not isinstance(note, str) for note in _array(self_check["notes"], "$.selfCheck.notes")):
|
||
raise ContractError("$.selfCheck.notes 必须是字符串数组")
|
||
return json.loads(canonical_json(output))
|
||
|
||
|
||
__all__ = [
|
||
"CONTEXT_VERSION", "OUTPUT_VERSION", "PLAN_VERSION", "MANIFEST_VERSION", "TIE_BREAK",
|
||
"ContractError", "normalize_text", "canonical_json", "retrieval_identity", "han_count",
|
||
"calculate_target_chars", "validate_writer_context", "validate_writer_output",
|
||
]
|