619 lines
32 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""正文写手上下文与输出的严格合同。
本模块只处理纯数据,不读取文件、数据库或网络。所有进入写手的文本先做
Unicode NFC 与换行归一化,所有身份哈希都来自同一份规范 JSON。
"""
from __future__ import annotations
import hashlib
import json
import math
import re
import unicodedata
from decimal import Decimal, ROUND_HALF_UP
from typing import Any, Mapping, Sequence
CONTEXT_VERSION = "writer-context-v1"
OUTPUT_VERSION = "writer-output-v1"
PLAN_VERSION = "writer-retrieval-plan-v1"
MANIFEST_VERSION = "writer-retrieval-manifest-v1"
TIE_BREAK = "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC"
_HASH_RE = re.compile(r"^sha256:[0-9a-f]{64}$")
_RUN_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$")
_VOLATILE_IDENTITY_FIELDS = frozenset(
{"runId", "generatedAt", "timestamp", "executionNode", "contextSha256"}
)
# Unicode Script=Han 覆盖的标准区段。兼容表意文字也按“汉字”计数,
# 但标点、Markdown、拉丁字母和数字不会落入这些区段。
_HAN_RANGES = (
(0x3400, 0x4DBF),
(0x4E00, 0x9FFF),
(0xF900, 0xFAFF),
(0x20000, 0x2EBEF),
(0x2F800, 0x2FA1F),
(0x30000, 0x323AF),
)
class ContractError(ValueError):
"""输入不符合严格合同时抛出,调用方必须失败关闭。"""
def normalize_text(value: str) -> str:
"""把文本统一为 NFC 与 LF,供哈希和 Unicode 偏移共同使用。"""
if not isinstance(value, str):
raise ContractError("待归一化文本必须是字符串")
return unicodedata.normalize("NFC", value.replace("\r\n", "\n").replace("\r", "\n"))
def _normalize_json(value: Any) -> Any:
"""递归归一化 JSON 值,并拒绝 JSON 之外或不可复现的数值。"""
if isinstance(value, str):
return normalize_text(value)
if value is None or isinstance(value, (bool, int)):
return value
if isinstance(value, float):
if not math.isfinite(value):
raise ContractError("规范 JSON 不允许 NaN 或 Infinity")
return value
if isinstance(value, list):
return [_normalize_json(item) for item in value]
if isinstance(value, tuple):
return [_normalize_json(item) for item in value]
if isinstance(value, Mapping):
result: dict[str, Any] = {}
for raw_key, item in value.items():
if not isinstance(raw_key, str):
raise ContractError("规范 JSON 的对象键必须是字符串")
key = normalize_text(raw_key)
if key in result:
raise ContractError(f"对象键在 NFC 归一化后冲突: {key}")
result[key] = _normalize_json(item)
return result
raise ContractError(f"值不是 JSON 类型: {type(value).__name__}")
def canonical_json(value: Any) -> str:
"""输出 UTF-8 语义、键排序、无额外空白的规范 JSON 文本。"""
return json.dumps(
_normalize_json(value),
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
)
def _without_volatile_fields(value: Any) -> Any:
"""递归移除运行元数据,防止同一检索输入得到不同身份。"""
if isinstance(value, list):
return [_without_volatile_fields(item) for item in value]
if isinstance(value, Mapping):
return {
key: _without_volatile_fields(item)
for key, item in value.items()
if key not in _VOLATILE_IDENTITY_FIELDS
}
return value
def retrieval_identity(value: Any) -> str:
"""计算计划、manifest 或上下文的稳定 SHA-256 身份。"""
encoded = canonical_json(_without_volatile_fields(value)).encode("utf-8")
return "sha256:" + hashlib.sha256(encoded).hexdigest()
def han_count(value: str) -> int:
"""统计 Unicode Han code point,不把标点或 Markdown 算入正文长度。"""
text = normalize_text(value)
return sum(any(start <= ord(char) <= end for start, end in _HAN_RANGES) for char in text)
def _round_half_up(value: Decimal) -> int:
"""以十进制 ROUND_HALF_UP 规则取整,避免 Python 银行家舍入。"""
return int(value.quantize(Decimal("1"), rounding=ROUND_HALF_UP))
def _round_to_100(value: Decimal) -> int:
"""以百字为单位执行十进制半入取整。"""
return int((value / Decimal(100)).quantize(Decimal("1"), rounding=ROUND_HALF_UP) * 100)
def calculate_target_chars(
*,
explicit_target_chars: int | None = None,
recent_chapter_han_counts: Sequence[int] = (),
default_target_chars: int = 4000,
hard_event_count: int = 3,
foreshadowing_action_count: int = 0,
required_scene_count: int = 0,
min_chars: int = 2000,
max_chars: int = 10000,
) -> int:
"""按冻结历史中位数与细纲密度计算确定性目标汉字数。"""
counts = {
"default_target_chars": default_target_chars,
"hard_event_count": hard_event_count,
"foreshadowing_action_count": foreshadowing_action_count,
"required_scene_count": required_scene_count,
"min_chars": min_chars,
"max_chars": max_chars,
}
if any(isinstance(value, bool) or not isinstance(value, int) for value in counts.values()):
raise ContractError("篇幅参数必须是整数")
if default_target_chars <= 0 or min_chars <= 0 or max_chars < min_chars or any(value < 0 for key, value in counts.items() if "count" in key):
raise ContractError("篇幅边界或细纲计数非法")
if explicit_target_chars is not None:
if isinstance(explicit_target_chars, bool) or not isinstance(explicit_target_chars, int):
raise ContractError("显式 targetChars 必须是整数")
return min(max(explicit_target_chars, min_chars), max_chars)
valid_counts = list(recent_chapter_han_counts)[-20:]
if any(isinstance(value, bool) or not isinstance(value, int) or value < 500 for value in valid_counts):
raise ContractError("历史章汉字数必须是大于等于 500 的整数")
if len(valid_counts) >= 3:
ordered_counts = sorted(valid_counts)
midpoint = len(ordered_counts) // 2
if len(ordered_counts) % 2:
baseline = Decimal(ordered_counts[midpoint])
else:
baseline = (Decimal(ordered_counts[midpoint - 1]) + Decimal(ordered_counts[midpoint])) / 2
else:
baseline = Decimal(default_target_chars)
density = (
Decimal(hard_event_count)
+ Decimal("0.5") * foreshadowing_action_count
+ Decimal("0.5") * required_scene_count
)
factor = min(Decimal("1.15"), max(Decimal("0.85"), Decimal("0.85") + Decimal("0.05") * (density - 3)))
target = _round_to_100(Decimal(_round_half_up(baseline)) * factor)
return min(max(target, min_chars), max_chars)
def _object(
value: Any,
path: str,
required: set[str] | frozenset[str],
optional: set[str] | frozenset[str] = frozenset(),
) -> Mapping[str, Any]:
"""校验严格对象,任何缺字段或未知字段都立即失败。"""
if not isinstance(value, Mapping):
raise ContractError(f"{path} 必须是对象")
missing = sorted(required - set(value))
unknown = sorted(set(value) - required - optional)
if missing:
raise ContractError(f"{path} 缺少字段: {','.join(missing)}")
if unknown:
raise ContractError(f"{path} 包含未知字段: {','.join(unknown)}")
return value
def _string(value: Any, path: str, *, nonempty: bool = True) -> str:
"""校验字符串,并在需要时拒绝空值。"""
if not isinstance(value, str) or (nonempty and not value.strip()):
raise ContractError(f"{path} 必须是非空字符串")
if value != normalize_text(value):
raise ContractError(f"{path} 必须预先归一化为 Unicode NFC/LF")
return value
def _integer(value: Any, path: str, *, minimum: int = 0) -> int:
"""校验整数,显式排除 bool 这一 Python int 子类。"""
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
raise ContractError(f"{path} 必须是大于等于 {minimum} 的整数")
return value
def _boolean(value: Any, path: str) -> bool:
"""校验严格布尔值。"""
if not isinstance(value, bool):
raise ContractError(f"{path} 必须是布尔值")
return value
def _array(value: Any, path: str) -> list[Any]:
"""校验数组并返回原值,拒绝元组等隐式转换。"""
if not isinstance(value, list):
raise ContractError(f"{path} 必须是数组")
return value
def _hash(value: Any, path: str) -> str:
"""校验带算法前缀的 SHA-256。"""
text = _string(value, path)
if not _HASH_RE.fullmatch(text):
raise ContractError(f"{path} 必须是 sha256: 加 64 位小写十六进制")
return text
def _source_ref(value: Any, path: str) -> None:
"""校验不可变来源引用;历史原文可额外携带块和字符区间。"""
ref = _object(
value,
path,
frozenset({"sourceId", "sourceVersion"}),
frozenset({"chapter", "blockId", "startCodePoint", "endCodePoint", "contentSha256", "sourceType"}),
)
_string(ref["sourceId"], f"{path}.sourceId")
_string(ref["sourceVersion"], f"{path}.sourceVersion")
for field in ("chapter", "blockId", "startCodePoint", "endCodePoint"):
if field in ref:
_integer(ref[field], f"{path}.{field}", minimum=0 if "CodePoint" in field else 1)
if "startCodePoint" in ref and "endCodePoint" in ref and ref["endCodePoint"] <= ref["startCodePoint"]:
raise ContractError(f"{path} 字符区间必须是非空左闭右开区间")
if "contentSha256" in ref:
_hash(ref["contentSha256"], f"{path}.contentSha256")
if "sourceType" in ref:
_string(ref["sourceType"], f"{path}.sourceType")
def _validate_plan(value: Any, path: str) -> None:
"""校验固定检索计划,不允许写手临场扩张查询。"""
plan = _object(
value,
path,
frozenset(
{"planVersion", "planId", "runId", "asOf", "queries", "cardIndexVersion", "proseIndexVersion", "filters", "tieBreak", "tokenBudget"}
),
)
if plan["planVersion"] != PLAN_VERSION:
raise ContractError(f"{path}.planVersion 版本不支持")
_hash(plan["planId"], f"{path}.planId")
if not _RUN_ID_RE.fullmatch(_string(plan["runId"], f"{path}.runId")):
raise ContractError(f"{path}.runId 格式非法")
_integer(plan["asOf"], f"{path}.asOf", minimum=1)
for index, query in enumerate(_array(plan["queries"], f"{path}.queries")):
item = _object(query, f"{path}.queries[{index}]", frozenset({"queryId", "text", "entityTypes", "purpose", "topK"}))
_string(item["queryId"], f"{path}.queries[{index}].queryId")
_string(item["text"], f"{path}.queries[{index}].text")
if any(not isinstance(kind, str) or not kind for kind in _array(item["entityTypes"], f"{path}.queries[{index}].entityTypes")):
raise ContractError(f"{path}.queries[{index}].entityTypes 必须是非空字符串数组")
_string(item["purpose"], f"{path}.queries[{index}].purpose")
_integer(item["topK"], f"{path}.queries[{index}].topK", minimum=1)
_string(plan["cardIndexVersion"], f"{path}.cardIndexVersion")
_string(plan["proseIndexVersion"], f"{path}.proseIndexVersion")
filters = _object(plan["filters"], f"{path}.filters", frozenset({"workId", "asOfChapter", "sourceStatus", "authorizationRequired"}))
_integer(filters["workId"], f"{path}.filters.workId", minimum=1)
_integer(filters["asOfChapter"], f"{path}.filters.asOfChapter", minimum=1)
_string(filters["sourceStatus"], f"{path}.filters.sourceStatus")
_boolean(filters["authorizationRequired"], f"{path}.filters.authorizationRequired")
if plan["tieBreak"] != TIE_BREAK:
raise ContractError(f"{path}.tieBreak 不符合稳定排序合同")
budget = _object(plan["tokenBudget"], f"{path}.tokenBudget", frozenset({"maxContextChars"}), frozenset({"cardChars", "recentProseChars", "historicalProseChars", "patternChars"}))
for key, item in budget.items():
_integer(item, f"{path}.tokenBudget.{key}", minimum=0)
identity_payload = {key: item for key, item in plan.items() if key != "planId"}
if plan["planId"] != retrieval_identity(identity_payload):
raise ContractError(f"{path}.planId 与计划内容不匹配")
def _validate_manifest(value: Any, path: str) -> None:
"""校验检索清单的来源与裁剪回显。"""
manifest = _object(value, path, frozenset({"manifestVersion", "manifestId", "planId", "sources", "omittedSources"}))
if manifest["manifestVersion"] != MANIFEST_VERSION:
raise ContractError(f"{path}.manifestVersion 版本不支持")
_hash(manifest["manifestId"], f"{path}.manifestId")
_hash(manifest["planId"], f"{path}.planId")
for index, source in enumerate(_array(manifest["sources"], f"{path}.sources")):
_source_ref(source, f"{path}.sources[{index}]")
for index, omitted in enumerate(_array(manifest["omittedSources"], f"{path}.omittedSources")):
item = _object(omitted, f"{path}.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
_string(item["sourceId"], f"{path}.omittedSources[{index}].sourceId")
_string(item["reason"], f"{path}.omittedSources[{index}].reason")
identity_payload = {key: item for key, item in manifest.items() if key != "manifestId"}
if manifest["manifestId"] != retrieval_identity(identity_payload):
raise ContractError(f"{path}.manifestId 与来源清单不匹配")
def validate_writer_context(value: Any) -> dict[str, Any]:
"""校验 WriterContext v1;成功时返回可安全复制的规范 JSON 对象。"""
required = frozenset(
{
"schemaVersion", "runId", "attempt", "mode", "purpose", "qualityPolicyVersion",
"workId", "targetChapter", "asOf", "contextSnapshot", "sourceVersion",
"authorizationSnapshot", "sourceStatus", "retrievalPlan", "retrievalManifest",
"fineOutline", "narrativeState", "factEvidence", "proseEvidence",
"patternReferences", "evidenceCoverage", "outputContract", "tokenBudget",
"omittedSources", "acceptanceEligible",
}
)
context = _object(
value,
"$",
required,
frozenset({"evidenceStrategy", "indexHints"}),
)
if context["schemaVersion"] != CONTEXT_VERSION:
raise ContractError("$.schemaVersion 版本不支持")
run_id = _string(context["runId"], "$.runId")
if not _RUN_ID_RE.fullmatch(run_id):
raise ContractError("$.runId 格式非法")
_integer(context["attempt"], "$.attempt", minimum=1)
if context["mode"] not in {"production", "diagnostic_only"}:
raise ContractError("$.mode 枚举非法")
if context["purpose"] not in {"production", "evaluation", "diagnostic"}:
raise ContractError("$.purpose 枚举非法")
evidence_strategy = context.get("evidenceStrategy", "production_dual_evidence")
if evidence_strategy not in {
"production_dual_evidence",
"historical_prose_only",
"card_index_only",
"card_index_plus_prose",
}:
raise ContractError("$.evidenceStrategy 枚举非法")
if context["mode"] == "production" and evidence_strategy != "production_dual_evidence":
raise ContractError("生产上下文必须使用 production_dual_evidence")
_string(context["qualityPolicyVersion"], "$.qualityPolicyVersion")
_integer(context["workId"], "$.workId", minimum=1)
target = _integer(context["targetChapter"], "$.targetChapter", minimum=1)
as_of = _integer(context["asOf"], "$.asOf", minimum=1)
if as_of >= target:
raise ContractError("$.asOf 必须早于 targetChapter")
snapshot = _object(context["contextSnapshot"], "$.contextSnapshot", frozenset({"manifestId", "contextSha256", "generatedAt"}))
_hash(snapshot["manifestId"], "$.contextSnapshot.manifestId")
_hash(snapshot["contextSha256"], "$.contextSnapshot.contextSha256")
_string(snapshot["generatedAt"], "$.contextSnapshot.generatedAt")
_string(context["sourceVersion"], "$.sourceVersion")
authorization = _object(
context["authorizationSnapshot"],
"$.authorizationSnapshot",
frozenset({"snapshotId", "allowedPurpose", "verifiedAt"}),
frozenset({"expiresAt", "sourceVersion"}),
)
for key, item in authorization.items():
_string(item, f"$.authorizationSnapshot.{key}")
if context["sourceStatus"] not in {"active", "authorized", "frozen_authorized"}:
raise ContractError("$.sourceStatus 不允许生成")
_validate_plan(context["retrievalPlan"], "$.retrievalPlan")
_validate_manifest(context["retrievalManifest"], "$.retrievalManifest")
if context["retrievalPlan"]["runId"] != run_id or context["retrievalPlan"]["asOf"] != as_of:
raise ContractError("检索计划与上下文的 runId/asOf 不一致")
if context["retrievalManifest"]["planId"] != context["retrievalPlan"]["planId"]:
raise ContractError("检索 manifest 未绑定当前计划")
if context["contextSnapshot"]["manifestId"] != context["retrievalManifest"]["manifestId"]:
raise ContractError("上下文快照未绑定当前 manifest")
outline = _object(context["fineOutline"], "$.fineOutline", frozenset({"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"}))
_source_ref(outline["sourceRef"], "$.fineOutline.sourceRef")
for field in ("hardConstraints", "adjustableBeats"):
if any(not isinstance(item, str) or not item for item in _array(outline[field], f"$.fineOutline.{field}")):
raise ContractError(f"$.fineOutline.{field} 必须是非空字符串数组")
for index, fact in enumerate(_array(outline["declaredNewFacts"], "$.fineOutline.declaredNewFacts")):
item = _object(fact, f"$.fineOutline.declaredNewFacts[{index}]", frozenset({"factId", "text", "sourceRef"}))
_string(item["factId"], f"$.fineOutline.declaredNewFacts[{index}].factId")
_string(item["text"], f"$.fineOutline.declaredNewFacts[{index}].text")
_source_ref(item["sourceRef"], f"$.fineOutline.declaredNewFacts[{index}].sourceRef")
state = _object(context["narrativeState"], "$.narrativeState", frozenset({"time", "location", "characterPositions", "immediateSituation"}))
for field in ("time", "location", "immediateSituation"):
_string(state[field], f"$.narrativeState.{field}", nonempty=False)
positions = _object(state["characterPositions"], "$.narrativeState.characterPositions", frozenset(state["characterPositions"].keys()) if isinstance(state["characterPositions"], Mapping) else frozenset())
for key, item in positions.items():
_string(key, "$.narrativeState.characterPositions.<key>")
_string(item, f"$.narrativeState.characterPositions.{key}")
for index, evidence in enumerate(_array(context["factEvidence"], "$.factEvidence")):
item = _object(evidence, f"$.factEvidence[{index}]", frozenset({"evidenceId", "fact", "sourceType", "sourceRef", "contentSha256", "riskLevel"}))
for field in ("evidenceId", "fact", "sourceType"):
_string(item[field], f"$.factEvidence[{index}].{field}")
if item["sourceType"] not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}:
raise ContractError(f"$.factEvidence[{index}].sourceType 枚举非法")
_source_ref(item["sourceRef"], f"$.factEvidence[{index}].sourceRef")
if item["sourceType"] == "historical_prose":
required_location = {"chapter", "blockId", "startCodePoint", "endCodePoint"}
if not required_location.issubset(item["sourceRef"]):
raise ContractError(f"$.factEvidence[{index}] 历史事实必须回到章、块和字符区间")
_hash(item["contentSha256"], f"$.factEvidence[{index}].contentSha256")
if item["riskLevel"] not in {"low", "medium", "high"}:
raise ContractError(f"$.factEvidence[{index}].riskLevel 枚举非法")
for index, evidence in enumerate(_array(context["proseEvidence"], "$.proseEvidence")):
item = _object(evidence, f"$.proseEvidence[{index}]", frozenset({"evidenceId", "chapter", "sourceRef", "contentSha256", "purpose", "text", "isRecentBaseline"}))
_string(item["evidenceId"], f"$.proseEvidence[{index}].evidenceId")
chapter = _integer(item["chapter"], f"$.proseEvidence[{index}].chapter", minimum=1)
if chapter > as_of:
raise ContractError(f"$.proseEvidence[{index}] 超出冻结线")
_source_ref(item["sourceRef"], f"$.proseEvidence[{index}].sourceRef")
if not {"chapter", "blockId", "startCodePoint", "endCodePoint"}.issubset(item["sourceRef"]):
raise ContractError(f"$.proseEvidence[{index}] 必须带章、块和字符区间")
_hash(item["contentSha256"], f"$.proseEvidence[{index}].contentSha256")
_string(item["purpose"], f"$.proseEvidence[{index}].purpose")
text = _string(item["text"], f"$.proseEvidence[{index}].text")
_boolean(item["isRecentBaseline"], f"$.proseEvidence[{index}].isRecentBaseline")
expected = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
if item["contentSha256"] != expected:
raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配")
index_hints = _array(context.get("indexHints", []), "$.indexHints")
for index, hint in enumerate(index_hints):
item = _object(
hint,
f"$.indexHints[{index}]",
frozenset(
{
"cardId",
"name",
"type",
"content",
"sourceId",
"sourceVersion",
"asOf",
}
),
)
for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion"):
_string(item[field], f"$.indexHints[{index}].{field}")
hint_as_of = _integer(item["asOf"], f"$.indexHints[{index}].asOf", minimum=1)
if hint_as_of > as_of:
raise ContractError(f"$.indexHints[{index}] 超出冻结线")
if index_hints and not (
context["mode"] == "diagnostic_only"
and context["purpose"] in {"evaluation", "diagnostic"}
and context["acceptanceEligible"] is False
):
raise ContractError("indexHints 只允许用于不可接受的评测或诊断上下文")
if context["mode"] == "production" and index_hints:
raise ContractError("生产上下文禁止 indexHints")
if evidence_strategy == "card_index_only":
if context["mode"] != "diagnostic_only" or context["purpose"] not in {"evaluation", "diagnostic"}:
raise ContractError("card_index_only 只允许用于诊断上下文")
if context["factEvidence"] or context["proseEvidence"] or not index_hints:
raise ContractError("card_index_only 必须仅包含 indexHints")
elif evidence_strategy == "historical_prose_only":
if index_hints or context["factEvidence"] or not context["proseEvidence"]:
raise ContractError("historical_prose_only 必须仅包含历史原文")
elif evidence_strategy == "card_index_plus_prose":
if context["factEvidence"] or not index_hints or not context["proseEvidence"]:
raise ContractError("card_index_plus_prose 必须包含 indexHints 与历史原文")
# v1 生产正文必须携带截至冻结点的连续前四章。诊断 A/C 保持同一
# 原文基线,只有显式 card_index_only 策略获准跳过该生产要求。
if evidence_strategy != "card_index_only":
baseline_chapters = sorted(
item["chapter"]
for item in context["proseEvidence"]
if item["isRecentBaseline"] is True
)
expected_baseline = list(range(max(1, as_of - 3), as_of + 1))
if baseline_chapters != expected_baseline:
raise ContractError("原文证据必须包含截至冻结点的连续前四章基线")
for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")):
_source_ref(reference, f"$.patternReferences[{index}]")
for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")):
item = _object(coverage, f"$.evidenceCoverage[{index}]", frozenset({"elementId", "elementType", "name", "status", "factEvidenceIds", "proseEvidenceIds", "gapReason"}))
for field in ("elementId", "elementType", "name", "gapReason"):
_string(item[field], f"$.evidenceCoverage[{index}].{field}", nonempty=field != "gapReason")
if item["status"] not in {"supported", "declared_new", "card_gap", "style_gap", "unsupported", "conflict"}:
raise ContractError(f"$.evidenceCoverage[{index}].status 枚举非法")
for field in ("factEvidenceIds", "proseEvidenceIds"):
if any(not isinstance(item_id, str) or not item_id for item_id in _array(item[field], f"$.evidenceCoverage[{index}].{field}")):
raise ContractError(f"$.evidenceCoverage[{index}].{field} 必须是字符串数组")
output = _object(context["outputContract"], "$.outputContract", frozenset({"targetChars", "minChars", "maxChars", "frontmatterRequired", "newSettingDeclarationRequired"}))
for field in ("targetChars", "minChars", "maxChars"):
_integer(output[field], f"$.outputContract.{field}", minimum=1)
if not output["minChars"] <= output["targetChars"] <= output["maxChars"]:
raise ContractError("$.outputContract 篇幅范围不包含目标值")
_boolean(output["frontmatterRequired"], "$.outputContract.frontmatterRequired")
_boolean(output["newSettingDeclarationRequired"], "$.outputContract.newSettingDeclarationRequired")
budget = _object(context["tokenBudget"], "$.tokenBudget", frozenset({"maxContextChars", "usedContextChars"}))
for field in budget:
_integer(budget[field], f"$.tokenBudget.{field}", minimum=0)
if budget["usedContextChars"] > budget["maxContextChars"]:
raise ContractError("$.tokenBudget 已超预算")
for index, omitted in enumerate(_array(context["omittedSources"], "$.omittedSources")):
item = _object(omitted, f"$.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
_string(item["sourceId"], f"$.omittedSources[{index}].sourceId")
_string(item["reason"], f"$.omittedSources[{index}].reason")
eligible = _boolean(context["acceptanceEligible"], "$.acceptanceEligible")
if context["purpose"] in {"evaluation", "diagnostic"} or context["mode"] == "diagnostic_only":
if eligible:
raise ContractError("评测或诊断上下文必须 acceptanceEligible=false")
elif context["qualityPolicyVersion"] != "writer-production-v1":
raise ContractError("生产上下文必须绑定 writer-production-v1")
if context["contextSnapshot"]["contextSha256"] != retrieval_identity(context):
raise ContractError("$.contextSnapshot.contextSha256 与上下文内容不匹配")
return json.loads(canonical_json(context))
def validate_writer_output(value: Any) -> dict[str, Any]:
"""校验 WriterOutput v1、正文哈希和全部来源引用。"""
output = _object(
value,
"$",
frozenset(
{"schemaVersion", "runId", "attempt", "mode", "qualityPolicyVersion", "contextSnapshotId", "contextSnapshotSha256", "candidateVersion", "candidateSha256", "acceptanceEligible", "candidateBody", "claimLedger", "evidenceRequests", "newSettingDeclarations", "selfCheck"}
),
)
if output["schemaVersion"] != OUTPUT_VERSION:
raise ContractError("$.schemaVersion 版本不支持")
if not _RUN_ID_RE.fullmatch(_string(output["runId"], "$.runId")):
raise ContractError("$.runId 格式非法")
_integer(output["attempt"], "$.attempt", minimum=1)
if output["mode"] not in {"production", "diagnostic_only"}:
raise ContractError("$.mode 枚举非法")
_string(output["qualityPolicyVersion"], "$.qualityPolicyVersion")
_hash(output["contextSnapshotId"], "$.contextSnapshotId")
_hash(output["contextSnapshotSha256"], "$.contextSnapshotSha256")
_integer(output["candidateVersion"], "$.candidateVersion", minimum=1)
body = _string(output["candidateBody"], "$.candidateBody")
candidate_hash = _hash(output["candidateSha256"], "$.candidateSha256")
expected_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
if candidate_hash != expected_hash:
raise ContractError("$.candidateSha256 与规范正文不匹配")
eligible = _boolean(output["acceptanceEligible"], "$.acceptanceEligible")
if output["mode"] == "diagnostic_only" and eligible:
raise ContractError("诊断输出必须 acceptanceEligible=false")
for index, claim in enumerate(_array(output["claimLedger"], "$.claimLedger")):
item = _object(
claim,
f"$.claimLedger[{index}]",
frozenset({"claimId", "candidateSha256", "startCodePoint", "endCodePoint", "factType", "factEvidenceId", "coverageState"}),
frozenset({"proseEvidenceId"}),
)
for field in ("claimId", "factType", "factEvidenceId", "coverageState"):
_string(item[field], f"$.claimLedger[{index}].{field}")
_hash(item["candidateSha256"], f"$.claimLedger[{index}].candidateSha256")
if item["candidateSha256"] != candidate_hash:
raise ContractError(f"$.claimLedger[{index}] 未绑定当前候选")
start = _integer(item["startCodePoint"], f"$.claimLedger[{index}].startCodePoint")
end = _integer(item["endCodePoint"], f"$.claimLedger[{index}].endCodePoint", minimum=1)
if end <= start or end > len(body):
raise ContractError(f"$.claimLedger[{index}] Unicode 偏移越界")
if "proseEvidenceId" in item and item["proseEvidenceId"] is not None:
_string(item["proseEvidenceId"], f"$.claimLedger[{index}].proseEvidenceId")
for index, request in enumerate(_array(output["evidenceRequests"], "$.evidenceRequests")):
item = _object(request, f"$.evidenceRequests[{index}]", frozenset({"requestId", "query", "reason", "priority"}))
for field in ("requestId", "query", "reason", "priority"):
_string(item[field], f"$.evidenceRequests[{index}].{field}")
for index, declaration in enumerate(_array(output["newSettingDeclarations"], "$.newSettingDeclarations")):
item = _object(declaration, f"$.newSettingDeclarations[{index}]", frozenset({"declarationId", "factType", "text", "startCodePoint", "endCodePoint"}))
for field in ("declarationId", "factType", "text"):
_string(item[field], f"$.newSettingDeclarations[{index}].{field}")
start = _integer(item["startCodePoint"], f"$.newSettingDeclarations[{index}].startCodePoint")
end = _integer(item["endCodePoint"], f"$.newSettingDeclarations[{index}].endCodePoint", minimum=1)
if end <= start or end > len(body):
raise ContractError(f"$.newSettingDeclarations[{index}] Unicode 偏移越界")
self_check = _object(output["selfCheck"], "$.selfCheck", frozenset({"hardConstraintsCovered", "notes"}))
_boolean(self_check["hardConstraintsCovered"], "$.selfCheck.hardConstraintsCovered")
if any(not isinstance(note, str) for note in _array(self_check["notes"], "$.selfCheck.notes")):
raise ContractError("$.selfCheck.notes 必须是字符串数组")
return json.loads(canonical_json(output))
__all__ = [
"CONTEXT_VERSION", "OUTPUT_VERSION", "PLAN_VERSION", "MANIFEST_VERSION", "TIE_BREAK",
"ContractError", "normalize_text", "canonical_json", "retrieval_identity", "han_count",
"calculate_target_chars", "validate_writer_context", "validate_writer_output",
]