#!/usr/bin/env python3 """正文写手上下文与输出的严格合同。 本模块只处理纯数据,不读取文件、数据库或网络。所有进入写手的文本先做 Unicode NFC 与换行归一化,所有身份哈希都来自同一份规范 JSON。 """ from __future__ import annotations import copy import hashlib import json import math import re import unicodedata from decimal import Decimal, ROUND_HALF_UP from typing import Any, Mapping, Sequence CONTEXT_VERSION = "writer-context-v1" DRAFT_VERSION = "writer-draft-v2" OUTPUT_VERSION = "candidate-envelope-v2" PLAN_VERSION = "writer-retrieval-plan-v1" MANIFEST_VERSION = "writer-retrieval-manifest-v1" TIE_BREAK = "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC" _HASH_RE = re.compile(r"^sha256:[0-9a-f]{64}$") _RUN_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$") _VOLATILE_IDENTITY_FIELDS = frozenset( {"runId", "generatedAt", "timestamp", "executionNode", "contextSha256"} ) # Unicode Script=Han 覆盖的标准区段。兼容表意文字也按“汉字”计数, # 但标点、Markdown、拉丁字母和数字不会落入这些区段。 _HAN_RANGES = ( (0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF), (0x20000, 0x2EBEF), (0x2F800, 0x2FA1F), (0x30000, 0x323AF), ) class ContractError(ValueError): """输入不符合严格合同时抛出,调用方必须失败关闭。""" def normalize_text(value: str) -> str: """把文本统一为 NFC 与 LF,供哈希和 Unicode 偏移共同使用。 模型在 JSON 输出里常把换行双重转义成字面 ``\\n``(反斜杠+n 两个字符),这里连同 真实的 CRLF/CR 一并还原为真正的换行符 LF,避免正文带着字面 ``\\n`` 显示异常、 以及检测/盲评的跨段引文因换行表示不同而匹配失败。 """ if not isinstance(value, str): raise ContractError("待归一化文本必须是字符串") value = value.replace("\r\n", "\n").replace("\r", "\n") # 真实 CRLF/CR → LF # 字面转义还原:先处理 \r\n(4 字符)再处理 \n / \r(2 字符),顺序避免半截替换 value = value.replace("\\r\\n", "\n").replace("\\n", "\n").replace("\\r", "\n") return unicodedata.normalize("NFC", value) def _normalize_json(value: Any) -> Any: """递归归一化 JSON 值,并拒绝 JSON 之外或不可复现的数值。""" if isinstance(value, str): return normalize_text(value) if value is None or isinstance(value, (bool, int)): return value if isinstance(value, float): if not math.isfinite(value): raise ContractError("规范 JSON 不允许 NaN 或 Infinity") return value if isinstance(value, list): return [_normalize_json(item) for item in value] if isinstance(value, tuple): return [_normalize_json(item) for item in value] if isinstance(value, Mapping): result: dict[str, Any] = {} for raw_key, item in value.items(): if not isinstance(raw_key, str): raise ContractError("规范 JSON 的对象键必须是字符串") key = normalize_text(raw_key) if key in result: raise ContractError(f"对象键在 NFC 归一化后冲突: {key}") result[key] = _normalize_json(item) return result raise ContractError(f"值不是 JSON 类型: {type(value).__name__}") def canonical_json(value: Any) -> str: """输出 UTF-8 语义、键排序、无额外空白的规范 JSON 文本。""" return json.dumps( _normalize_json(value), ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False, ) def _without_volatile_fields(value: Any) -> Any: """递归移除运行元数据,防止同一检索输入得到不同身份。""" if isinstance(value, list): return [_without_volatile_fields(item) for item in value] if isinstance(value, Mapping): return { key: _without_volatile_fields(item) for key, item in value.items() if key not in _VOLATILE_IDENTITY_FIELDS } return value def retrieval_identity(value: Any) -> str: """计算计划、manifest 或上下文的稳定 SHA-256 身份。""" encoded = canonical_json(_without_volatile_fields(value)).encode("utf-8") return "sha256:" + hashlib.sha256(encoded).hexdigest() def han_count(value: str) -> int: """统计 Unicode Han code point,不把标点或 Markdown 算入正文长度。""" text = normalize_text(value) return sum(any(start <= ord(char) <= end for start, end in _HAN_RANGES) for char in text) def _round_half_up(value: Decimal) -> int: """以十进制 ROUND_HALF_UP 规则取整,避免 Python 银行家舍入。""" return int(value.quantize(Decimal("1"), rounding=ROUND_HALF_UP)) def _round_to_100(value: Decimal) -> int: """以百字为单位执行十进制半入取整。""" return int((value / Decimal(100)).quantize(Decimal("1"), rounding=ROUND_HALF_UP) * 100) def calculate_target_chars( *, explicit_target_chars: int | None = None, recent_chapter_han_counts: Sequence[int] = (), default_target_chars: int = 4000, hard_event_count: int = 3, foreshadowing_action_count: int = 0, required_scene_count: int = 0, min_chars: int = 2000, max_chars: int = 10000, ) -> int: """按冻结历史中位数与细纲密度计算确定性目标汉字数。""" counts = { "default_target_chars": default_target_chars, "hard_event_count": hard_event_count, "foreshadowing_action_count": foreshadowing_action_count, "required_scene_count": required_scene_count, "min_chars": min_chars, "max_chars": max_chars, } if any(isinstance(value, bool) or not isinstance(value, int) for value in counts.values()): raise ContractError("篇幅参数必须是整数") if default_target_chars <= 0 or min_chars <= 0 or max_chars < min_chars or any(value < 0 for key, value in counts.items() if "count" in key): raise ContractError("篇幅边界或细纲计数非法") if explicit_target_chars is not None: if isinstance(explicit_target_chars, bool) or not isinstance(explicit_target_chars, int): raise ContractError("显式 targetChars 必须是整数") return min(max(explicit_target_chars, min_chars), max_chars) valid_counts = list(recent_chapter_han_counts)[-20:] if any(isinstance(value, bool) or not isinstance(value, int) or value < 500 for value in valid_counts): raise ContractError("历史章汉字数必须是大于等于 500 的整数") if len(valid_counts) >= 3: ordered_counts = sorted(valid_counts) midpoint = len(ordered_counts) // 2 if len(ordered_counts) % 2: baseline = Decimal(ordered_counts[midpoint]) else: baseline = (Decimal(ordered_counts[midpoint - 1]) + Decimal(ordered_counts[midpoint])) / 2 else: baseline = Decimal(default_target_chars) density = ( Decimal(hard_event_count) + Decimal("0.5") * foreshadowing_action_count + Decimal("0.5") * required_scene_count ) factor = min(Decimal("1.15"), max(Decimal("0.85"), Decimal("0.85") + Decimal("0.05") * (density - 3))) target = _round_to_100(Decimal(_round_half_up(baseline)) * factor) return min(max(target, min_chars), max_chars) def _object( value: Any, path: str, required: set[str] | frozenset[str], optional: set[str] | frozenset[str] = frozenset(), ) -> Mapping[str, Any]: """校验严格对象,任何缺字段或未知字段都立即失败。""" if not isinstance(value, Mapping): raise ContractError(f"{path} 必须是对象") missing = sorted(required - set(value)) unknown = sorted(set(value) - required - optional) if missing: raise ContractError(f"{path} 缺少字段: {','.join(missing)}") if unknown: raise ContractError(f"{path} 包含未知字段: {','.join(unknown)}") return value def _string(value: Any, path: str, *, nonempty: bool = True) -> str: """校验字符串,并在需要时拒绝空值。""" if not isinstance(value, str) or (nonempty and not value.strip()): raise ContractError(f"{path} 必须是非空字符串") if value != normalize_text(value): raise ContractError(f"{path} 必须预先归一化为 Unicode NFC/LF") return value def _integer(value: Any, path: str, *, minimum: int = 0) -> int: """校验整数,显式排除 bool 这一 Python int 子类。""" if isinstance(value, bool) or not isinstance(value, int) or value < minimum: raise ContractError(f"{path} 必须是大于等于 {minimum} 的整数") return value def _boolean(value: Any, path: str) -> bool: """校验严格布尔值。""" if not isinstance(value, bool): raise ContractError(f"{path} 必须是布尔值") return value def _array(value: Any, path: str) -> list[Any]: """校验数组并返回原值,拒绝元组等隐式转换。""" if not isinstance(value, list): raise ContractError(f"{path} 必须是数组") return value def _hash(value: Any, path: str) -> str: """校验带算法前缀的 SHA-256。""" text = _string(value, path) if not _HASH_RE.fullmatch(text): raise ContractError(f"{path} 必须是 sha256: 加 64 位小写十六进制") return text _SOURCE_REF_REQUIRED = frozenset({"sourceId", "sourceVersion"}) _SOURCE_REF_OPTIONAL = frozenset( {"chapter", "blockId", "startCodePoint", "endCodePoint", "contentSha256", "sourceType"} ) # 范式引用(patternReferences)在严格来源指针之外额外允许的内容字段。 # WHY(SoT 变更):此前 patternReferences 只能带来源指针,写手最终只看到一个空标签 # (referenceId+kind),读不到范式卡的名字/摘要/写法,「范式指导」这个实验单变量 # 形同虚设。放宽这三个内容字段只针对 patternReferences,其它来源指针不受影响。 PATTERN_CONTENT_FIELDS = frozenset({"name", "summary", "writingPoints"}) # 范式引用内容字段的体量硬上限(name/summary/写法要点值按 code point 计,字段数按个计)。 # WHY:范式卡原始字段可能长达数千字,直接灌给写手会撑爆上下文预算。合同侧按这些上限 # 失败关闭——无论检索端将来怎么换,超量内容都进不了写手输入;检索端投影时应先截断到 # 上限以内,合同复核是第二道闸。 PATTERN_NAME_MAX_CHARS = 40 PATTERN_SUMMARY_MAX_CHARS = 120 PATTERN_POINTS_MAX_FIELDS = 6 PATTERN_POINT_MAX_CHARS = 200 def _validate_source_ref_pointers(ref: Mapping[str, Any], path: str) -> None: """校验来源指针自身(sourceId/sourceVersion 必填及定位字段),供严格与放宽校验复用。""" _string(ref["sourceId"], f"{path}.sourceId") _string(ref["sourceVersion"], f"{path}.sourceVersion") for field in ("chapter", "blockId", "startCodePoint", "endCodePoint"): if field in ref: _integer(ref[field], f"{path}.{field}", minimum=0 if "CodePoint" in field else 1) if "startCodePoint" in ref and "endCodePoint" in ref and ref["endCodePoint"] <= ref["startCodePoint"]: raise ContractError(f"{path} 字符区间必须是非空左闭右开区间") if "contentSha256" in ref: _hash(ref["contentSha256"], f"{path}.contentSha256") if "sourceType" in ref: _string(ref["sourceType"], f"{path}.sourceType") def _source_ref(value: Any, path: str) -> None: """校验不可变来源引用;历史原文可额外携带块和字符区间。""" ref = _object(value, path, _SOURCE_REF_REQUIRED, _SOURCE_REF_OPTIONAL) _validate_source_ref_pointers(ref, path) def _pattern_source_ref(value: Any, path: str) -> None: """校验范式引用:严格来源指针 + 放宽且限量的内容字段。 WHY(SoT 变更):让写手真正读到范式卡——名字、一句话摘要、写法要点——而不是 只看到一个来源标签。来源指针仍必填,保证可回读、可审计;内容字段全部限量并失败 关闭,防止撑爆写手上下文。放宽只针对 patternReferences:proseEvidence/factEvidence/ manifest 等其它来源指针继续走严格的 _source_ref,任何名字/摘要字段仍按「未知字段」拒收。 """ ref = _object(value, path, _SOURCE_REF_REQUIRED, _SOURCE_REF_OPTIONAL | PATTERN_CONTENT_FIELDS) _validate_source_ref_pointers(ref, path) if "name" in ref and len(_string(ref["name"], f"{path}.name")) > PATTERN_NAME_MAX_CHARS: raise ContractError(f"{path}.name 超出体量上限 {PATTERN_NAME_MAX_CHARS} 字") if "summary" in ref and len(_string(ref["summary"], f"{path}.summary")) > PATTERN_SUMMARY_MAX_CHARS: raise ContractError(f"{path}.summary 超出体量上限 {PATTERN_SUMMARY_MAX_CHARS} 字") if "writingPoints" in ref: points = ref["writingPoints"] if not isinstance(points, Mapping): raise ContractError(f"{path}.writingPoints 必须是对象") if len(points) > PATTERN_POINTS_MAX_FIELDS: raise ContractError(f"{path}.writingPoints 超出 {PATTERN_POINTS_MAX_FIELDS} 个字段上限") for key, item in points.items(): key_text = _string(key, f"{path}.writingPoints.") if len(key_text) > PATTERN_NAME_MAX_CHARS: raise ContractError(f"{path}.writingPoints 字段名超出体量上限 {PATTERN_NAME_MAX_CHARS} 字") if len(_string(item, f"{path}.writingPoints.{key_text}")) > PATTERN_POINT_MAX_CHARS: raise ContractError( f"{path}.writingPoints.{key_text} 超出体量上限 {PATTERN_POINT_MAX_CHARS} 字" ) def pattern_references_for_arm( arm: str, c_references: Sequence[Mapping[str, Any]] ) -> list[dict[str, Any]]: """按臂分配范式引用的唯一事实源:A 臂恒空,其余臂(B/C)拿 C 臂候选范式卡。 WHY:Gate A 的唯一实验变量是「有无卡(含范式卡)」。A 臂是纯历史原文对照,必须 恒空,否则 A/C 单变量对照被破坏。范式卡的链路有两段独立 assemble:装配端 (load_writer_reference_work)检索出 C 臂候选并冻结进 config.json 的 ``writerContextInput.patternReferences``;回放端(run_writer_replay)真写时再从 config.json 读出候选、重新 assemble 各臂上下文。两段必须按完全相同的规则分臂, 因此把规则收敛到合同模块这一处由两端复用——任一段各写一套,就会出现「C 臂真写 读不到范式卡(实验失效)」或「A 臂混入范式卡(对照破坏)」。返回深拷贝,避免 各臂上下文与冻结候选互相串改。 """ if arm == "A": return [] return [copy.deepcopy(dict(item)) for item in c_references] def project_pattern_pointers(value: Mapping[str, Any]) -> dict[str, Any]: """从可能携带内容字段的引用中投影出纯来源指针(供 manifest 等审计账本使用)。 WHY:manifest 记录「哪些来源入包」,只承载可回读指针,不承载范式卡正文;范式卡 内容只由上下文内的 patternReferences 承载(并计入上下文身份哈希)。 """ return {key: value[key] for key in (_SOURCE_REF_REQUIRED | _SOURCE_REF_OPTIONAL) if key in value} def _validate_plan(value: Any, path: str) -> None: """校验固定检索计划,不允许写手临场扩张查询。""" plan = _object( value, path, frozenset( {"planVersion", "planId", "runId", "asOf", "queries", "cardIndexVersion", "proseIndexVersion", "filters", "tieBreak", "tokenBudget"} ), ) if plan["planVersion"] != PLAN_VERSION: raise ContractError(f"{path}.planVersion 版本不支持") _hash(plan["planId"], f"{path}.planId") if not _RUN_ID_RE.fullmatch(_string(plan["runId"], f"{path}.runId")): raise ContractError(f"{path}.runId 格式非法") # asOf=0 表示开篇前冻结线(第 1 章之前):此刻只有设定/大纲/细纲,无历史正文 _integer(plan["asOf"], f"{path}.asOf", minimum=0) for index, query in enumerate(_array(plan["queries"], f"{path}.queries")): item = _object(query, f"{path}.queries[{index}]", frozenset({"queryId", "text", "entityTypes", "purpose", "topK"})) _string(item["queryId"], f"{path}.queries[{index}].queryId") _string(item["text"], f"{path}.queries[{index}].text") if any(not isinstance(kind, str) or not kind for kind in _array(item["entityTypes"], f"{path}.queries[{index}].entityTypes")): raise ContractError(f"{path}.queries[{index}].entityTypes 必须是非空字符串数组") _string(item["purpose"], f"{path}.queries[{index}].purpose") _integer(item["topK"], f"{path}.queries[{index}].topK", minimum=1) _string(plan["cardIndexVersion"], f"{path}.cardIndexVersion") _string(plan["proseIndexVersion"], f"{path}.proseIndexVersion") filters = _object(plan["filters"], f"{path}.filters", frozenset({"workId", "asOfChapter", "sourceStatus", "authorizationRequired"})) _integer(filters["workId"], f"{path}.filters.workId", minimum=1) _integer(filters["asOfChapter"], f"{path}.filters.asOfChapter", minimum=0) _string(filters["sourceStatus"], f"{path}.filters.sourceStatus") _boolean(filters["authorizationRequired"], f"{path}.filters.authorizationRequired") if plan["tieBreak"] != TIE_BREAK: raise ContractError(f"{path}.tieBreak 不符合稳定排序合同") budget = _object(plan["tokenBudget"], f"{path}.tokenBudget", frozenset({"maxContextChars"}), frozenset({"cardChars", "recentProseChars", "historicalProseChars", "patternChars"})) for key, item in budget.items(): _integer(item, f"{path}.tokenBudget.{key}", minimum=0) identity_payload = {key: item for key, item in plan.items() if key != "planId"} if plan["planId"] != retrieval_identity(identity_payload): raise ContractError(f"{path}.planId 与计划内容不匹配") def _validate_manifest(value: Any, path: str) -> None: """校验检索清单的来源与裁剪回显。""" manifest = _object(value, path, frozenset({"manifestVersion", "manifestId", "planId", "sources", "omittedSources"})) if manifest["manifestVersion"] != MANIFEST_VERSION: raise ContractError(f"{path}.manifestVersion 版本不支持") _hash(manifest["manifestId"], f"{path}.manifestId") _hash(manifest["planId"], f"{path}.planId") for index, source in enumerate(_array(manifest["sources"], f"{path}.sources")): _source_ref(source, f"{path}.sources[{index}]") for index, omitted in enumerate(_array(manifest["omittedSources"], f"{path}.omittedSources")): item = _object(omitted, f"{path}.omittedSources[{index}]", frozenset({"sourceId", "reason"})) _string(item["sourceId"], f"{path}.omittedSources[{index}].sourceId") _string(item["reason"], f"{path}.omittedSources[{index}].reason") identity_payload = {key: item for key, item in manifest.items() if key != "manifestId"} if manifest["manifestId"] != retrieval_identity(identity_payload): raise ContractError(f"{path}.manifestId 与来源清单不匹配") def validate_writer_context(value: Any) -> dict[str, Any]: """校验 WriterContext v1;成功时返回可安全复制的规范 JSON 对象。""" required = frozenset( { "schemaVersion", "runId", "attempt", "mode", "purpose", "qualityPolicyVersion", "workId", "targetChapter", "asOf", "contextSnapshot", "sourceVersion", "authorizationSnapshot", "sourceStatus", "retrievalPlan", "retrievalManifest", "fineOutline", "narrativeState", "factEvidence", "proseEvidence", "patternReferences", "evidenceCoverage", "outputContract", "tokenBudget", "omittedSources", "acceptanceEligible", } ) context = _object( value, "$", required, frozenset({ "evidenceStrategy", "indexHints", "styleConstraints", "humanizationProvenance", "generationLengthContract", }), ) if context["schemaVersion"] != CONTEXT_VERSION: raise ContractError("$.schemaVersion 版本不支持") run_id = _string(context["runId"], "$.runId") if not _RUN_ID_RE.fullmatch(run_id): raise ContractError("$.runId 格式非法") _integer(context["attempt"], "$.attempt", minimum=1) if context["mode"] not in {"production", "diagnostic_only"}: raise ContractError("$.mode 枚举非法") if context["purpose"] not in {"production", "evaluation", "diagnostic"}: raise ContractError("$.purpose 枚举非法") evidence_strategy = context.get("evidenceStrategy", "production_dual_evidence") if evidence_strategy not in { "production_dual_evidence", "historical_prose_only", "card_index_only", "card_index_plus_prose", }: raise ContractError("$.evidenceStrategy 枚举非法") if context["mode"] == "production" and evidence_strategy != "production_dual_evidence": raise ContractError("生产上下文必须使用 production_dual_evidence") _string(context["qualityPolicyVersion"], "$.qualityPolicyVersion") _integer(context["workId"], "$.workId", minimum=1) target = _integer(context["targetChapter"], "$.targetChapter", minimum=1) # asOf=0 合法:开篇前冻结线,正文基线必为空(任何历史正文都会被后续"超出冻结线"挡下) as_of = _integer(context["asOf"], "$.asOf", minimum=0) if as_of >= target: raise ContractError("$.asOf 必须早于 targetChapter") snapshot = _object(context["contextSnapshot"], "$.contextSnapshot", frozenset({"manifestId", "contextSha256", "generatedAt"})) _hash(snapshot["manifestId"], "$.contextSnapshot.manifestId") _hash(snapshot["contextSha256"], "$.contextSnapshot.contextSha256") _string(snapshot["generatedAt"], "$.contextSnapshot.generatedAt") _string(context["sourceVersion"], "$.sourceVersion") authorization = _object( context["authorizationSnapshot"], "$.authorizationSnapshot", frozenset({"snapshotId", "allowedPurpose", "verifiedAt"}), frozenset({"expiresAt", "sourceVersion"}), ) for key, item in authorization.items(): _string(item, f"$.authorizationSnapshot.{key}") if context["sourceStatus"] not in {"active", "authorized", "frozen_authorized"}: raise ContractError("$.sourceStatus 不允许生成") _validate_plan(context["retrievalPlan"], "$.retrievalPlan") _validate_manifest(context["retrievalManifest"], "$.retrievalManifest") if context["retrievalPlan"]["runId"] != run_id or context["retrievalPlan"]["asOf"] != as_of: raise ContractError("检索计划与上下文的 runId/asOf 不一致") if context["retrievalManifest"]["planId"] != context["retrievalPlan"]["planId"]: raise ContractError("检索 manifest 未绑定当前计划") if context["contextSnapshot"]["manifestId"] != context["retrievalManifest"]["manifestId"]: raise ContractError("上下文快照未绑定当前 manifest") outline = _object(context["fineOutline"], "$.fineOutline", frozenset({"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"})) _source_ref(outline["sourceRef"], "$.fineOutline.sourceRef") for field in ("hardConstraints", "adjustableBeats"): if any(not isinstance(item, str) or not item for item in _array(outline[field], f"$.fineOutline.{field}")): raise ContractError(f"$.fineOutline.{field} 必须是非空字符串数组") for index, fact in enumerate(_array(outline["declaredNewFacts"], "$.fineOutline.declaredNewFacts")): item = _object(fact, f"$.fineOutline.declaredNewFacts[{index}]", frozenset({"factId", "text", "sourceRef"})) _string(item["factId"], f"$.fineOutline.declaredNewFacts[{index}].factId") _string(item["text"], f"$.fineOutline.declaredNewFacts[{index}].text") _source_ref(item["sourceRef"], f"$.fineOutline.declaredNewFacts[{index}].sourceRef") state = _object(context["narrativeState"], "$.narrativeState", frozenset({"time", "location", "characterPositions", "immediateSituation"})) for field in ("time", "location", "immediateSituation"): _string(state[field], f"$.narrativeState.{field}", nonempty=False) positions = _object(state["characterPositions"], "$.narrativeState.characterPositions", frozenset(state["characterPositions"].keys()) if isinstance(state["characterPositions"], Mapping) else frozenset()) for key, item in positions.items(): _string(key, "$.narrativeState.characterPositions.") _string(item, f"$.narrativeState.characterPositions.{key}") for index, evidence in enumerate(_array(context["factEvidence"], "$.factEvidence")): item = _object(evidence, f"$.factEvidence[{index}]", frozenset({"evidenceId", "fact", "sourceType", "sourceRef", "contentSha256", "riskLevel"})) for field in ("evidenceId", "fact", "sourceType"): _string(item[field], f"$.factEvidence[{index}].{field}") if item["sourceType"] not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}: raise ContractError(f"$.factEvidence[{index}].sourceType 枚举非法") _source_ref(item["sourceRef"], f"$.factEvidence[{index}].sourceRef") if item["sourceType"] == "historical_prose": required_location = {"chapter", "blockId", "startCodePoint", "endCodePoint"} if not required_location.issubset(item["sourceRef"]): raise ContractError(f"$.factEvidence[{index}] 历史事实必须回到章、块和字符区间") _hash(item["contentSha256"], f"$.factEvidence[{index}].contentSha256") if item["riskLevel"] not in {"low", "medium", "high"}: raise ContractError(f"$.factEvidence[{index}].riskLevel 枚举非法") for index, evidence in enumerate(_array(context["proseEvidence"], "$.proseEvidence")): item = _object(evidence, f"$.proseEvidence[{index}]", frozenset({"evidenceId", "chapter", "sourceRef", "contentSha256", "purpose", "text", "isRecentBaseline"})) _string(item["evidenceId"], f"$.proseEvidence[{index}].evidenceId") chapter = _integer(item["chapter"], f"$.proseEvidence[{index}].chapter", minimum=1) if chapter > as_of: raise ContractError(f"$.proseEvidence[{index}] 超出冻结线") _source_ref(item["sourceRef"], f"$.proseEvidence[{index}].sourceRef") if not {"chapter", "blockId", "startCodePoint", "endCodePoint"}.issubset(item["sourceRef"]): raise ContractError(f"$.proseEvidence[{index}] 必须带章、块和字符区间") _hash(item["contentSha256"], f"$.proseEvidence[{index}].contentSha256") _string(item["purpose"], f"$.proseEvidence[{index}].purpose") text = _string(item["text"], f"$.proseEvidence[{index}].text") _boolean(item["isRecentBaseline"], f"$.proseEvidence[{index}].isRecentBaseline") expected = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() if item["contentSha256"] != expected: raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配") index_hints = _array(context.get("indexHints", []), "$.indexHints") for index, hint in enumerate(index_hints): item = _object( hint, f"$.indexHints[{index}]", frozenset( { "cardId", "name", "type", "content", "sourceId", "sourceVersion", "asOf", } ), ) for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion"): _string(item[field], f"$.indexHints[{index}].{field}") hint_as_of = _integer(item["asOf"], f"$.indexHints[{index}].asOf", minimum=1) if hint_as_of > as_of: raise ContractError(f"$.indexHints[{index}] 超出冻结线") if index_hints and not ( context["mode"] == "diagnostic_only" and context["purpose"] in {"evaluation", "diagnostic"} and context["acceptanceEligible"] is False ): raise ContractError("indexHints 只允许用于不可接受的评测或诊断上下文") if context["mode"] == "production" and index_hints: raise ContractError("生产上下文禁止 indexHints") if evidence_strategy == "card_index_only": if context["mode"] != "diagnostic_only" or context["purpose"] not in {"evaluation", "diagnostic"}: raise ContractError("card_index_only 只允许用于诊断上下文") if context["factEvidence"] or context["proseEvidence"] or not index_hints: raise ContractError("card_index_only 必须仅包含 indexHints") elif evidence_strategy == "historical_prose_only": if index_hints or context["factEvidence"] or not context["proseEvidence"]: raise ContractError("historical_prose_only 必须仅包含历史原文") elif evidence_strategy == "card_index_plus_prose": if context["factEvidence"] or not index_hints or not context["proseEvidence"]: raise ContractError("card_index_plus_prose 必须包含 indexHints 与历史原文") # 生产正文必须携带截至冻结点的连续前四章。诊断 A/C 保持同一 # 原文基线,只有显式 card_index_only 策略获准跳过该生产要求。 if evidence_strategy != "card_index_only": baseline_chapters = sorted( item["chapter"] for item in context["proseEvidence"] if item["isRecentBaseline"] is True ) expected_baseline = list(range(max(1, as_of - 3), as_of + 1)) if baseline_chapters != expected_baseline: raise ContractError("原文证据必须包含截至冻结点的连续前四章基线") for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")): # 范式引用走放宽校验:来源指针仍严格,另允许 name/summary/writingPoints 内容字段。 _pattern_source_ref(reference, f"$.patternReferences[{index}]") if "humanizationProvenance" in context: provenance = _object( context["humanizationProvenance"], "$.humanizationProvenance", frozenset({"schemaVersion", "ruleLibraryVersion", "voiceLedgerSha256", "constraintCount"}), ) _string(provenance["schemaVersion"], "$.humanizationProvenance.schemaVersion") _string(provenance["ruleLibraryVersion"], "$.humanizationProvenance.ruleLibraryVersion") if provenance["voiceLedgerSha256"] is not None: _hash(provenance["voiceLedgerSha256"], "$.humanizationProvenance.voiceLedgerSha256") _integer(provenance["constraintCount"], "$.humanizationProvenance.constraintCount", minimum=0) if "styleConstraints" in context: # 文风约束(规划期选定的 style 画像投影):非空字符串数组;缺省时整个键省略, # 上下文逐字节不变(不破坏既有冻结哈希),build 端按空数组投影。 style_rules = _array(context["styleConstraints"], "$.styleConstraints") for index, rule in enumerate(style_rules): _string(rule, f"$.styleConstraints[{index}]", nonempty=True) context["styleConstraints"] = [str(rule) for rule in style_rules] for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")): item = _object(coverage, f"$.evidenceCoverage[{index}]", frozenset({"elementId", "elementType", "name", "status", "factEvidenceIds", "proseEvidenceIds", "gapReason"})) for field in ("elementId", "elementType", "name", "gapReason"): _string(item[field], f"$.evidenceCoverage[{index}].{field}", nonempty=field != "gapReason") if item["status"] not in {"supported", "declared_new", "card_gap", "style_gap", "unsupported", "conflict"}: raise ContractError(f"$.evidenceCoverage[{index}].status 枚举非法") for field in ("factEvidenceIds", "proseEvidenceIds"): if any(not isinstance(item_id, str) or not item_id for item_id in _array(item[field], f"$.evidenceCoverage[{index}].{field}")): raise ContractError(f"$.evidenceCoverage[{index}].{field} 必须是字符串数组") def validate_length_contract(raw: Any, path: str) -> Mapping[str, Any]: contract = _object( raw, path, frozenset({"targetChars", "minChars", "maxChars", "frontmatterRequired"}), ) for field in ("targetChars", "minChars", "maxChars"): _integer(contract[field], f"{path}.{field}", minimum=1) if not contract["minChars"] <= contract["targetChars"] <= contract["maxChars"]: raise ContractError(f"{path} 篇幅范围不包含目标值") _boolean(contract["frontmatterRequired"], f"{path}.frontmatterRequired") return contract output = validate_length_contract(context["outputContract"], "$.outputContract") if "generationLengthContract" in context: generation = validate_length_contract( context["generationLengthContract"], "$.generationLengthContract" ) if not ( output["minChars"] <= generation["minChars"] <= generation["targetChars"] <= generation["maxChars"] <= output["maxChars"] ): raise ContractError("$.generationLengthContract 必须完整落在 $.outputContract 接受区间内") if generation["frontmatterRequired"] != output["frontmatterRequired"]: raise ContractError("生成篇幅合同与接受篇幅合同的 frontmatterRequired 必须一致") budget = _object(context["tokenBudget"], "$.tokenBudget", frozenset({"maxContextChars", "usedContextChars"})) for field in budget: _integer(budget[field], f"$.tokenBudget.{field}", minimum=0) if budget["usedContextChars"] > budget["maxContextChars"]: raise ContractError("$.tokenBudget 已超预算") for index, omitted in enumerate(_array(context["omittedSources"], "$.omittedSources")): item = _object(omitted, f"$.omittedSources[{index}]", frozenset({"sourceId", "reason"})) _string(item["sourceId"], f"$.omittedSources[{index}].sourceId") _string(item["reason"], f"$.omittedSources[{index}].reason") eligible = _boolean(context["acceptanceEligible"], "$.acceptanceEligible") if context["purpose"] in {"evaluation", "diagnostic"} or context["mode"] == "diagnostic_only": if eligible: raise ContractError("评测或诊断上下文必须 acceptanceEligible=false") elif context["qualityPolicyVersion"] != "writer-production-v1": raise ContractError("生产上下文必须绑定 writer-production-v1") if context["contextSnapshot"]["contextSha256"] != retrieval_identity(context): raise ContractError("$.contextSnapshot.contextSha256 与上下文内容不匹配") return json.loads(canonical_json(context)) def build_writer_creative_input(value: Any) -> dict[str, Any]: """从完整冻结上下文投影 writer 唯一可见的创作输入。""" context = validate_writer_context(value) outline = context["fineOutline"] fact_constraints = [ { "constraintId": item["evidenceId"], "text": item["fact"], "kind": "established", "riskLevel": item["riskLevel"], } for item in context["factEvidence"] ] fact_constraints.extend( { "constraintId": item["factId"], "text": item["text"], "kind": "declared_new", "riskLevel": "low", } for item in outline["declaredNewFacts"] ) prose_excerpts = [ { "excerptId": item["evidenceId"], "chapter": item["chapter"], "purpose": item["purpose"], "text": item["text"], "isRecentBaseline": item["isRecentBaseline"], } for item in context["proseEvidence"] ] # SoT 变更:把范式卡内容投影给写手。WHY——此前只投影 referenceId+kind 两个标签, # 写手看不到范式卡写什么,「范式指导」单变量实际为空;这里把名字、一句话摘要和写法 # 要点 surface 出来(均为可选,存在且非空才给)。来源指针(sourceId/sourceVersion) # 一律不进写手输入,只留在冻结上下文供审计回读。 pattern_references = [] for index, item in enumerate(context["patternReferences"]): reference: dict[str, Any] = { "referenceId": f"pattern-{index + 1}", "kind": item.get("sourceType", "authorized_pattern"), } if item.get("name"): reference["name"] = item["name"] if item.get("summary"): reference["summary"] = item["summary"] if item.get("writingPoints"): reference["writingPoints"] = dict(item["writingPoints"]) pattern_references.append(reference) output_contract = context.get("generationLengthContract", context["outputContract"]) target_chars = output_contract.get("目标字数") or output_contract.get("targetChars", 7000) min_chars = output_contract.get("字数下限") or output_contract.get("minChars", 3000) max_chars = output_contract.get("字数上限") or output_contract.get("maxChars", 10000) fm_required = output_contract.get("需要前言") or output_contract.get("frontmatterRequired", False) creative_input = { "fineOutline": { "hardConstraints": list(outline.get("硬约束") or outline.get("hardConstraints", [])), "adjustableBeats": list(outline.get("可调整节拍") or outline.get("adjustableBeats", [])), "declaredNewFacts": [ {"factId": item.get("事实ID") or item.get("factId", ""), "text": item.get("文本") or item.get("text", "")} for item in (outline.get("新事实宣告") or outline.get("declaredNewFacts", [])) ], }, "narrativeState": context["narrativeState"], "factConstraints": fact_constraints, "proseExcerpts": prose_excerpts, "patternReferences": pattern_references, "lengthContract": { "targetChars": target_chars, "minChars": min_chars, "maxChars": max_chars, "frontmatterRequired": fm_required, }, "styleConstraints": [str(rule) for rule in context.get("styleConstraints", [])], } return json.loads(canonical_json(creative_input)) def validate_writer_draft(value: Any) -> dict[str, Any]: """校验 writer 模型的单字段创作输出。""" draft = _object(value, "$", frozenset({"candidateBody"})) body = draft["candidateBody"] if not isinstance(body, str) or not body.strip(): raise ContractError("$.candidateBody 必须是非空字符串") return {"candidateBody": body} def build_candidate_envelope( context_value: Any, draft_value: Any, *, candidate_version: int = 1, ) -> dict[str, Any]: """规范化模型正文并绑定可信候选身份。""" context = validate_writer_context(context_value) draft = validate_writer_draft(draft_value) version = _integer(candidate_version, "$.candidateVersion", minimum=1) body = normalize_text(draft["candidateBody"]) if not body.strip(): raise ContractError("$.candidateBody 规范化后不得为空") candidate_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest() envelope = { "schemaVersion": OUTPUT_VERSION, "runId": context["runId"], "attempt": context["attempt"], "mode": context["mode"], "qualityPolicyVersion": context["qualityPolicyVersion"], "contextSnapshotId": context["contextSnapshot"]["manifestId"], "contextSnapshotSha256": context["contextSnapshot"]["contextSha256"], "candidateVersion": version, "candidateSha256": candidate_hash, "acceptanceEligible": context["acceptanceEligible"], "candidateBody": body, } return validate_writer_output(envelope) def validate_writer_output(value: Any) -> dict[str, Any]: """校验 CandidateEnvelope v2 的信任绑定与正文哈希。""" output = _object( value, "$", frozenset( { "schemaVersion", "runId", "attempt", "mode", "qualityPolicyVersion", "contextSnapshotId", "contextSnapshotSha256", "candidateVersion", "candidateSha256", "acceptanceEligible", "candidateBody", } ), ) if output["schemaVersion"] != OUTPUT_VERSION: raise ContractError("$.schemaVersion 版本不支持") if not _RUN_ID_RE.fullmatch(_string(output["runId"], "$.runId")): raise ContractError("$.runId 格式非法") _integer(output["attempt"], "$.attempt", minimum=1) if output["mode"] not in {"production", "diagnostic_only"}: raise ContractError("$.mode 枚举非法") _string(output["qualityPolicyVersion"], "$.qualityPolicyVersion") _hash(output["contextSnapshotId"], "$.contextSnapshotId") _hash(output["contextSnapshotSha256"], "$.contextSnapshotSha256") _integer(output["candidateVersion"], "$.candidateVersion", minimum=1) body = _string(output["candidateBody"], "$.candidateBody") candidate_hash = _hash(output["candidateSha256"], "$.candidateSha256") expected_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest() if candidate_hash != expected_hash: raise ContractError("$.candidateSha256 与规范正文不匹配") eligible = _boolean(output["acceptanceEligible"], "$.acceptanceEligible") if output["mode"] == "diagnostic_only" and eligible: raise ContractError("诊断候选必须 acceptanceEligible=false") return json.loads(canonical_json(output)) __all__ = [ "CONTEXT_VERSION", "DRAFT_VERSION", "OUTPUT_VERSION", "PLAN_VERSION", "MANIFEST_VERSION", "TIE_BREAK", "PATTERN_CONTENT_FIELDS", "PATTERN_NAME_MAX_CHARS", "PATTERN_SUMMARY_MAX_CHARS", "PATTERN_POINTS_MAX_FIELDS", "PATTERN_POINT_MAX_CHARS", "ContractError", "normalize_text", "canonical_json", "retrieval_identity", "han_count", "calculate_target_chars", "validate_writer_context", "build_writer_creative_input", "validate_writer_draft", "build_candidate_envelope", "validate_writer_output", "project_pattern_pointers", ]