831 lines
41 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""正文写手上下文与输出的严格合同。
本模块只处理纯数据,不读取文件、数据库或网络。所有进入写手的文本先做
Unicode NFC 与换行归一化,所有身份哈希都来自同一份规范 JSON。
"""
from __future__ import annotations
import copy
import hashlib
import json
import math
import re
import unicodedata
from decimal import Decimal, ROUND_HALF_UP
from typing import Any, Mapping, Sequence
CONTEXT_VERSION = "writer-context-v1"
DRAFT_VERSION = "writer-draft-v2"
OUTPUT_VERSION = "candidate-envelope-v2"
PLAN_VERSION = "writer-retrieval-plan-v1"
MANIFEST_VERSION = "writer-retrieval-manifest-v1"
TIE_BREAK = "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC"
_HASH_RE = re.compile(r"^sha256:[0-9a-f]{64}$")
_RUN_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$")
_VOLATILE_IDENTITY_FIELDS = frozenset(
{"runId", "generatedAt", "timestamp", "executionNode", "contextSha256"}
)
# Unicode Script=Han 覆盖的标准区段。兼容表意文字也按“汉字”计数,
# 但标点、Markdown、拉丁字母和数字不会落入这些区段。
_HAN_RANGES = (
(0x3400, 0x4DBF),
(0x4E00, 0x9FFF),
(0xF900, 0xFAFF),
(0x20000, 0x2EBEF),
(0x2F800, 0x2FA1F),
(0x30000, 0x323AF),
)
class ContractError(ValueError):
"""输入不符合严格合同时抛出,调用方必须失败关闭。"""
def normalize_text(value: str) -> str:
"""把文本统一为 NFC 与 LF,供哈希和 Unicode 偏移共同使用。
模型在 JSON 输出里常把换行双重转义成字面 ``\\n``(反斜杠+n 两个字符),这里连同
真实的 CRLF/CR 一并还原为真正的换行符 LF,避免正文带着字面 ``\\n`` 显示异常、
以及检测/盲评的跨段引文因换行表示不同而匹配失败。
"""
if not isinstance(value, str):
raise ContractError("待归一化文本必须是字符串")
value = value.replace("\r\n", "\n").replace("\r", "\n") # 真实 CRLF/CR → LF
# 字面转义还原:先处理 \r\n(4 字符)再处理 \n / \r(2 字符),顺序避免半截替换
value = value.replace("\\r\\n", "\n").replace("\\n", "\n").replace("\\r", "\n")
return unicodedata.normalize("NFC", value)
def _normalize_json(value: Any) -> Any:
"""递归归一化 JSON 值,并拒绝 JSON 之外或不可复现的数值。"""
if isinstance(value, str):
return normalize_text(value)
if value is None or isinstance(value, (bool, int)):
return value
if isinstance(value, float):
if not math.isfinite(value):
raise ContractError("规范 JSON 不允许 NaN 或 Infinity")
return value
if isinstance(value, list):
return [_normalize_json(item) for item in value]
if isinstance(value, tuple):
return [_normalize_json(item) for item in value]
if isinstance(value, Mapping):
result: dict[str, Any] = {}
for raw_key, item in value.items():
if not isinstance(raw_key, str):
raise ContractError("规范 JSON 的对象键必须是字符串")
key = normalize_text(raw_key)
if key in result:
raise ContractError(f"对象键在 NFC 归一化后冲突: {key}")
result[key] = _normalize_json(item)
return result
raise ContractError(f"值不是 JSON 类型: {type(value).__name__}")
def canonical_json(value: Any) -> str:
"""输出 UTF-8 语义、键排序、无额外空白的规范 JSON 文本。"""
return json.dumps(
_normalize_json(value),
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
)
def _without_volatile_fields(value: Any) -> Any:
"""递归移除运行元数据,防止同一检索输入得到不同身份。"""
if isinstance(value, list):
return [_without_volatile_fields(item) for item in value]
if isinstance(value, Mapping):
return {
key: _without_volatile_fields(item)
for key, item in value.items()
if key not in _VOLATILE_IDENTITY_FIELDS
}
return value
def retrieval_identity(value: Any) -> str:
"""计算计划、manifest 或上下文的稳定 SHA-256 身份。"""
encoded = canonical_json(_without_volatile_fields(value)).encode("utf-8")
return "sha256:" + hashlib.sha256(encoded).hexdigest()
def han_count(value: str) -> int:
"""统计 Unicode Han code point,不把标点或 Markdown 算入正文长度。"""
text = normalize_text(value)
return sum(any(start <= ord(char) <= end for start, end in _HAN_RANGES) for char in text)
def _round_half_up(value: Decimal) -> int:
"""以十进制 ROUND_HALF_UP 规则取整,避免 Python 银行家舍入。"""
return int(value.quantize(Decimal("1"), rounding=ROUND_HALF_UP))
def _round_to_100(value: Decimal) -> int:
"""以百字为单位执行十进制半入取整。"""
return int((value / Decimal(100)).quantize(Decimal("1"), rounding=ROUND_HALF_UP) * 100)
def calculate_target_chars(
*,
explicit_target_chars: int | None = None,
recent_chapter_han_counts: Sequence[int] = (),
default_target_chars: int = 4000,
hard_event_count: int = 3,
foreshadowing_action_count: int = 0,
required_scene_count: int = 0,
min_chars: int = 2000,
max_chars: int = 10000,
) -> int:
"""按冻结历史中位数与细纲密度计算确定性目标汉字数。"""
counts = {
"default_target_chars": default_target_chars,
"hard_event_count": hard_event_count,
"foreshadowing_action_count": foreshadowing_action_count,
"required_scene_count": required_scene_count,
"min_chars": min_chars,
"max_chars": max_chars,
}
if any(isinstance(value, bool) or not isinstance(value, int) for value in counts.values()):
raise ContractError("篇幅参数必须是整数")
if default_target_chars <= 0 or min_chars <= 0 or max_chars < min_chars or any(value < 0 for key, value in counts.items() if "count" in key):
raise ContractError("篇幅边界或细纲计数非法")
if explicit_target_chars is not None:
if isinstance(explicit_target_chars, bool) or not isinstance(explicit_target_chars, int):
raise ContractError("显式 targetChars 必须是整数")
return min(max(explicit_target_chars, min_chars), max_chars)
valid_counts = list(recent_chapter_han_counts)[-20:]
if any(isinstance(value, bool) or not isinstance(value, int) or value < 500 for value in valid_counts):
raise ContractError("历史章汉字数必须是大于等于 500 的整数")
if len(valid_counts) >= 3:
ordered_counts = sorted(valid_counts)
midpoint = len(ordered_counts) // 2
if len(ordered_counts) % 2:
baseline = Decimal(ordered_counts[midpoint])
else:
baseline = (Decimal(ordered_counts[midpoint - 1]) + Decimal(ordered_counts[midpoint])) / 2
else:
baseline = Decimal(default_target_chars)
density = (
Decimal(hard_event_count)
+ Decimal("0.5") * foreshadowing_action_count
+ Decimal("0.5") * required_scene_count
)
factor = min(Decimal("1.15"), max(Decimal("0.85"), Decimal("0.85") + Decimal("0.05") * (density - 3)))
target = _round_to_100(Decimal(_round_half_up(baseline)) * factor)
return min(max(target, min_chars), max_chars)
def _object(
value: Any,
path: str,
required: set[str] | frozenset[str],
optional: set[str] | frozenset[str] = frozenset(),
) -> Mapping[str, Any]:
"""校验严格对象,任何缺字段或未知字段都立即失败。"""
if not isinstance(value, Mapping):
raise ContractError(f"{path} 必须是对象")
missing = sorted(required - set(value))
unknown = sorted(set(value) - required - optional)
if missing:
raise ContractError(f"{path} 缺少字段: {','.join(missing)}")
if unknown:
raise ContractError(f"{path} 包含未知字段: {','.join(unknown)}")
return value
def _string(value: Any, path: str, *, nonempty: bool = True) -> str:
"""校验字符串,并在需要时拒绝空值。"""
if not isinstance(value, str) or (nonempty and not value.strip()):
raise ContractError(f"{path} 必须是非空字符串")
if value != normalize_text(value):
raise ContractError(f"{path} 必须预先归一化为 Unicode NFC/LF")
return value
def _integer(value: Any, path: str, *, minimum: int = 0) -> int:
"""校验整数,显式排除 bool 这一 Python int 子类。"""
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
raise ContractError(f"{path} 必须是大于等于 {minimum} 的整数")
return value
def _boolean(value: Any, path: str) -> bool:
"""校验严格布尔值。"""
if not isinstance(value, bool):
raise ContractError(f"{path} 必须是布尔值")
return value
def _array(value: Any, path: str) -> list[Any]:
"""校验数组并返回原值,拒绝元组等隐式转换。"""
if not isinstance(value, list):
raise ContractError(f"{path} 必须是数组")
return value
def _hash(value: Any, path: str) -> str:
"""校验带算法前缀的 SHA-256。"""
text = _string(value, path)
if not _HASH_RE.fullmatch(text):
raise ContractError(f"{path} 必须是 sha256: 加 64 位小写十六进制")
return text
_SOURCE_REF_REQUIRED = frozenset({"sourceId", "sourceVersion"})
_SOURCE_REF_OPTIONAL = frozenset(
{"chapter", "blockId", "startCodePoint", "endCodePoint", "contentSha256", "sourceType"}
)
# 范式引用(patternReferences)在严格来源指针之外额外允许的内容字段。
# WHY(SoT 变更):此前 patternReferences 只能带来源指针,写手最终只看到一个空标签
# (referenceId+kind),读不到范式卡的名字/摘要/写法,「范式指导」这个实验单变量
# 形同虚设。放宽这三个内容字段只针对 patternReferences,其它来源指针不受影响。
PATTERN_CONTENT_FIELDS = frozenset({"name", "summary", "writingPoints"})
# 范式引用内容字段的体量硬上限(name/summary/写法要点值按 code point 计,字段数按个计)。
# WHY:范式卡原始字段可能长达数千字,直接灌给写手会撑爆上下文预算。合同侧按这些上限
# 失败关闭——无论检索端将来怎么换,超量内容都进不了写手输入;检索端投影时应先截断到
# 上限以内,合同复核是第二道闸。
PATTERN_NAME_MAX_CHARS = 40
PATTERN_SUMMARY_MAX_CHARS = 120
PATTERN_POINTS_MAX_FIELDS = 6
PATTERN_POINT_MAX_CHARS = 200
def _validate_source_ref_pointers(ref: Mapping[str, Any], path: str) -> None:
"""校验来源指针自身(sourceId/sourceVersion 必填及定位字段),供严格与放宽校验复用。"""
_string(ref["sourceId"], f"{path}.sourceId")
_string(ref["sourceVersion"], f"{path}.sourceVersion")
for field in ("chapter", "blockId", "startCodePoint", "endCodePoint"):
if field in ref:
_integer(ref[field], f"{path}.{field}", minimum=0 if "CodePoint" in field else 1)
if "startCodePoint" in ref and "endCodePoint" in ref and ref["endCodePoint"] <= ref["startCodePoint"]:
raise ContractError(f"{path} 字符区间必须是非空左闭右开区间")
if "contentSha256" in ref:
_hash(ref["contentSha256"], f"{path}.contentSha256")
if "sourceType" in ref:
_string(ref["sourceType"], f"{path}.sourceType")
def _source_ref(value: Any, path: str) -> None:
"""校验不可变来源引用;历史原文可额外携带块和字符区间。"""
ref = _object(value, path, _SOURCE_REF_REQUIRED, _SOURCE_REF_OPTIONAL)
_validate_source_ref_pointers(ref, path)
def _pattern_source_ref(value: Any, path: str) -> None:
"""校验范式引用:严格来源指针 + 放宽且限量的内容字段。
WHY(SoT 变更):让写手真正读到范式卡——名字、一句话摘要、写法要点——而不是
只看到一个来源标签。来源指针仍必填,保证可回读、可审计;内容字段全部限量并失败
关闭,防止撑爆写手上下文。放宽只针对 patternReferences:proseEvidence/factEvidence/
manifest 等其它来源指针继续走严格的 _source_ref,任何名字/摘要字段仍按「未知字段」拒收。
"""
ref = _object(value, path, _SOURCE_REF_REQUIRED, _SOURCE_REF_OPTIONAL | PATTERN_CONTENT_FIELDS)
_validate_source_ref_pointers(ref, path)
if "name" in ref and len(_string(ref["name"], f"{path}.name")) > PATTERN_NAME_MAX_CHARS:
raise ContractError(f"{path}.name 超出体量上限 {PATTERN_NAME_MAX_CHARS} 字")
if "summary" in ref and len(_string(ref["summary"], f"{path}.summary")) > PATTERN_SUMMARY_MAX_CHARS:
raise ContractError(f"{path}.summary 超出体量上限 {PATTERN_SUMMARY_MAX_CHARS} 字")
if "writingPoints" in ref:
points = ref["writingPoints"]
if not isinstance(points, Mapping):
raise ContractError(f"{path}.writingPoints 必须是对象")
if len(points) > PATTERN_POINTS_MAX_FIELDS:
raise ContractError(f"{path}.writingPoints 超出 {PATTERN_POINTS_MAX_FIELDS} 个字段上限")
for key, item in points.items():
key_text = _string(key, f"{path}.writingPoints.<key>")
if len(key_text) > PATTERN_NAME_MAX_CHARS:
raise ContractError(f"{path}.writingPoints 字段名超出体量上限 {PATTERN_NAME_MAX_CHARS} 字")
if len(_string(item, f"{path}.writingPoints.{key_text}")) > PATTERN_POINT_MAX_CHARS:
raise ContractError(
f"{path}.writingPoints.{key_text} 超出体量上限 {PATTERN_POINT_MAX_CHARS} 字"
)
def pattern_references_for_arm(
arm: str, c_references: Sequence[Mapping[str, Any]]
) -> list[dict[str, Any]]:
"""按臂分配范式引用的唯一事实源:A 臂恒空,其余臂(B/C)拿 C 臂候选范式卡。
WHY:Gate A 的唯一实验变量是「有无卡(含范式卡)」。A 臂是纯历史原文对照,必须
恒空,否则 A/C 单变量对照被破坏。范式卡的链路有两段独立 assemble:装配端
(load_writer_reference_work)检索出 C 臂候选并冻结进 config.json 的
``writerContextInput.patternReferences``;回放端(run_writer_replay)真写时再从
config.json 读出候选、重新 assemble 各臂上下文。两段必须按完全相同的规则分臂,
因此把规则收敛到合同模块这一处由两端复用——任一段各写一套,就会出现「C 臂真写
读不到范式卡(实验失效)」或「A 臂混入范式卡(对照破坏)」。返回深拷贝,避免
各臂上下文与冻结候选互相串改。
"""
if arm == "A":
return []
return [copy.deepcopy(dict(item)) for item in c_references]
def project_pattern_pointers(value: Mapping[str, Any]) -> dict[str, Any]:
"""从可能携带内容字段的引用中投影出纯来源指针(供 manifest 等审计账本使用)。
WHY:manifest 记录「哪些来源入包」,只承载可回读指针,不承载范式卡正文;范式卡
内容只由上下文内的 patternReferences 承载(并计入上下文身份哈希)。
"""
return {key: value[key] for key in (_SOURCE_REF_REQUIRED | _SOURCE_REF_OPTIONAL) if key in value}
def _validate_plan(value: Any, path: str) -> None:
"""校验固定检索计划,不允许写手临场扩张查询。"""
plan = _object(
value,
path,
frozenset(
{"planVersion", "planId", "runId", "asOf", "queries", "cardIndexVersion", "proseIndexVersion", "filters", "tieBreak", "tokenBudget"}
),
)
if plan["planVersion"] != PLAN_VERSION:
raise ContractError(f"{path}.planVersion 版本不支持")
_hash(plan["planId"], f"{path}.planId")
if not _RUN_ID_RE.fullmatch(_string(plan["runId"], f"{path}.runId")):
raise ContractError(f"{path}.runId 格式非法")
# asOf=0 表示开篇前冻结线(第 1 章之前):此刻只有设定/大纲/细纲,无历史正文
_integer(plan["asOf"], f"{path}.asOf", minimum=0)
for index, query in enumerate(_array(plan["queries"], f"{path}.queries")):
item = _object(query, f"{path}.queries[{index}]", frozenset({"queryId", "text", "entityTypes", "purpose", "topK"}))
_string(item["queryId"], f"{path}.queries[{index}].queryId")
_string(item["text"], f"{path}.queries[{index}].text")
if any(not isinstance(kind, str) or not kind for kind in _array(item["entityTypes"], f"{path}.queries[{index}].entityTypes")):
raise ContractError(f"{path}.queries[{index}].entityTypes 必须是非空字符串数组")
_string(item["purpose"], f"{path}.queries[{index}].purpose")
_integer(item["topK"], f"{path}.queries[{index}].topK", minimum=1)
_string(plan["cardIndexVersion"], f"{path}.cardIndexVersion")
_string(plan["proseIndexVersion"], f"{path}.proseIndexVersion")
filters = _object(plan["filters"], f"{path}.filters", frozenset({"workId", "asOfChapter", "sourceStatus", "authorizationRequired"}))
_integer(filters["workId"], f"{path}.filters.workId", minimum=1)
_integer(filters["asOfChapter"], f"{path}.filters.asOfChapter", minimum=0)
_string(filters["sourceStatus"], f"{path}.filters.sourceStatus")
_boolean(filters["authorizationRequired"], f"{path}.filters.authorizationRequired")
if plan["tieBreak"] != TIE_BREAK:
raise ContractError(f"{path}.tieBreak 不符合稳定排序合同")
budget = _object(plan["tokenBudget"], f"{path}.tokenBudget", frozenset({"maxContextChars"}), frozenset({"cardChars", "recentProseChars", "historicalProseChars", "patternChars"}))
for key, item in budget.items():
_integer(item, f"{path}.tokenBudget.{key}", minimum=0)
identity_payload = {key: item for key, item in plan.items() if key != "planId"}
if plan["planId"] != retrieval_identity(identity_payload):
raise ContractError(f"{path}.planId 与计划内容不匹配")
def _validate_manifest(value: Any, path: str) -> None:
"""校验检索清单的来源与裁剪回显。"""
manifest = _object(value, path, frozenset({"manifestVersion", "manifestId", "planId", "sources", "omittedSources"}))
if manifest["manifestVersion"] != MANIFEST_VERSION:
raise ContractError(f"{path}.manifestVersion 版本不支持")
_hash(manifest["manifestId"], f"{path}.manifestId")
_hash(manifest["planId"], f"{path}.planId")
for index, source in enumerate(_array(manifest["sources"], f"{path}.sources")):
_source_ref(source, f"{path}.sources[{index}]")
for index, omitted in enumerate(_array(manifest["omittedSources"], f"{path}.omittedSources")):
item = _object(omitted, f"{path}.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
_string(item["sourceId"], f"{path}.omittedSources[{index}].sourceId")
_string(item["reason"], f"{path}.omittedSources[{index}].reason")
identity_payload = {key: item for key, item in manifest.items() if key != "manifestId"}
if manifest["manifestId"] != retrieval_identity(identity_payload):
raise ContractError(f"{path}.manifestId 与来源清单不匹配")
def validate_writer_context(value: Any) -> dict[str, Any]:
"""校验 WriterContext v1;成功时返回可安全复制的规范 JSON 对象。"""
required = frozenset(
{
"schemaVersion", "runId", "attempt", "mode", "purpose", "qualityPolicyVersion",
"workId", "targetChapter", "asOf", "contextSnapshot", "sourceVersion",
"authorizationSnapshot", "sourceStatus", "retrievalPlan", "retrievalManifest",
"fineOutline", "narrativeState", "factEvidence", "proseEvidence",
"patternReferences", "evidenceCoverage", "outputContract", "tokenBudget",
"omittedSources", "acceptanceEligible",
}
)
context = _object(
value,
"$",
required,
frozenset({"evidenceStrategy", "indexHints", "styleConstraints", "humanizationProvenance"}),
)
if context["schemaVersion"] != CONTEXT_VERSION:
raise ContractError("$.schemaVersion 版本不支持")
run_id = _string(context["runId"], "$.runId")
if not _RUN_ID_RE.fullmatch(run_id):
raise ContractError("$.runId 格式非法")
_integer(context["attempt"], "$.attempt", minimum=1)
if context["mode"] not in {"production", "diagnostic_only"}:
raise ContractError("$.mode 枚举非法")
if context["purpose"] not in {"production", "evaluation", "diagnostic"}:
raise ContractError("$.purpose 枚举非法")
evidence_strategy = context.get("evidenceStrategy", "production_dual_evidence")
if evidence_strategy not in {
"production_dual_evidence",
"historical_prose_only",
"card_index_only",
"card_index_plus_prose",
}:
raise ContractError("$.evidenceStrategy 枚举非法")
if context["mode"] == "production" and evidence_strategy != "production_dual_evidence":
raise ContractError("生产上下文必须使用 production_dual_evidence")
_string(context["qualityPolicyVersion"], "$.qualityPolicyVersion")
_integer(context["workId"], "$.workId", minimum=1)
target = _integer(context["targetChapter"], "$.targetChapter", minimum=1)
# asOf=0 合法:开篇前冻结线,正文基线必为空(任何历史正文都会被后续"超出冻结线"挡下)
as_of = _integer(context["asOf"], "$.asOf", minimum=0)
if as_of >= target:
raise ContractError("$.asOf 必须早于 targetChapter")
snapshot = _object(context["contextSnapshot"], "$.contextSnapshot", frozenset({"manifestId", "contextSha256", "generatedAt"}))
_hash(snapshot["manifestId"], "$.contextSnapshot.manifestId")
_hash(snapshot["contextSha256"], "$.contextSnapshot.contextSha256")
_string(snapshot["generatedAt"], "$.contextSnapshot.generatedAt")
_string(context["sourceVersion"], "$.sourceVersion")
authorization = _object(
context["authorizationSnapshot"],
"$.authorizationSnapshot",
frozenset({"snapshotId", "allowedPurpose", "verifiedAt"}),
frozenset({"expiresAt", "sourceVersion"}),
)
for key, item in authorization.items():
_string(item, f"$.authorizationSnapshot.{key}")
if context["sourceStatus"] not in {"active", "authorized", "frozen_authorized"}:
raise ContractError("$.sourceStatus 不允许生成")
_validate_plan(context["retrievalPlan"], "$.retrievalPlan")
_validate_manifest(context["retrievalManifest"], "$.retrievalManifest")
if context["retrievalPlan"]["runId"] != run_id or context["retrievalPlan"]["asOf"] != as_of:
raise ContractError("检索计划与上下文的 runId/asOf 不一致")
if context["retrievalManifest"]["planId"] != context["retrievalPlan"]["planId"]:
raise ContractError("检索 manifest 未绑定当前计划")
if context["contextSnapshot"]["manifestId"] != context["retrievalManifest"]["manifestId"]:
raise ContractError("上下文快照未绑定当前 manifest")
outline = _object(context["fineOutline"], "$.fineOutline", frozenset({"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"}))
_source_ref(outline["sourceRef"], "$.fineOutline.sourceRef")
for field in ("hardConstraints", "adjustableBeats"):
if any(not isinstance(item, str) or not item for item in _array(outline[field], f"$.fineOutline.{field}")):
raise ContractError(f"$.fineOutline.{field} 必须是非空字符串数组")
for index, fact in enumerate(_array(outline["declaredNewFacts"], "$.fineOutline.declaredNewFacts")):
item = _object(fact, f"$.fineOutline.declaredNewFacts[{index}]", frozenset({"factId", "text", "sourceRef"}))
_string(item["factId"], f"$.fineOutline.declaredNewFacts[{index}].factId")
_string(item["text"], f"$.fineOutline.declaredNewFacts[{index}].text")
_source_ref(item["sourceRef"], f"$.fineOutline.declaredNewFacts[{index}].sourceRef")
state = _object(context["narrativeState"], "$.narrativeState", frozenset({"time", "location", "characterPositions", "immediateSituation"}))
for field in ("time", "location", "immediateSituation"):
_string(state[field], f"$.narrativeState.{field}", nonempty=False)
positions = _object(state["characterPositions"], "$.narrativeState.characterPositions", frozenset(state["characterPositions"].keys()) if isinstance(state["characterPositions"], Mapping) else frozenset())
for key, item in positions.items():
_string(key, "$.narrativeState.characterPositions.<key>")
_string(item, f"$.narrativeState.characterPositions.{key}")
for index, evidence in enumerate(_array(context["factEvidence"], "$.factEvidence")):
item = _object(evidence, f"$.factEvidence[{index}]", frozenset({"evidenceId", "fact", "sourceType", "sourceRef", "contentSha256", "riskLevel"}))
for field in ("evidenceId", "fact", "sourceType"):
_string(item[field], f"$.factEvidence[{index}].{field}")
if item["sourceType"] not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}:
raise ContractError(f"$.factEvidence[{index}].sourceType 枚举非法")
_source_ref(item["sourceRef"], f"$.factEvidence[{index}].sourceRef")
if item["sourceType"] == "historical_prose":
required_location = {"chapter", "blockId", "startCodePoint", "endCodePoint"}
if not required_location.issubset(item["sourceRef"]):
raise ContractError(f"$.factEvidence[{index}] 历史事实必须回到章、块和字符区间")
_hash(item["contentSha256"], f"$.factEvidence[{index}].contentSha256")
if item["riskLevel"] not in {"low", "medium", "high"}:
raise ContractError(f"$.factEvidence[{index}].riskLevel 枚举非法")
for index, evidence in enumerate(_array(context["proseEvidence"], "$.proseEvidence")):
item = _object(evidence, f"$.proseEvidence[{index}]", frozenset({"evidenceId", "chapter", "sourceRef", "contentSha256", "purpose", "text", "isRecentBaseline"}))
_string(item["evidenceId"], f"$.proseEvidence[{index}].evidenceId")
chapter = _integer(item["chapter"], f"$.proseEvidence[{index}].chapter", minimum=1)
if chapter > as_of:
raise ContractError(f"$.proseEvidence[{index}] 超出冻结线")
_source_ref(item["sourceRef"], f"$.proseEvidence[{index}].sourceRef")
if not {"chapter", "blockId", "startCodePoint", "endCodePoint"}.issubset(item["sourceRef"]):
raise ContractError(f"$.proseEvidence[{index}] 必须带章、块和字符区间")
_hash(item["contentSha256"], f"$.proseEvidence[{index}].contentSha256")
_string(item["purpose"], f"$.proseEvidence[{index}].purpose")
text = _string(item["text"], f"$.proseEvidence[{index}].text")
_boolean(item["isRecentBaseline"], f"$.proseEvidence[{index}].isRecentBaseline")
expected = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
if item["contentSha256"] != expected:
raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配")
index_hints = _array(context.get("indexHints", []), "$.indexHints")
for index, hint in enumerate(index_hints):
item = _object(
hint,
f"$.indexHints[{index}]",
frozenset(
{
"cardId",
"name",
"type",
"content",
"sourceId",
"sourceVersion",
"asOf",
}
),
)
for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion"):
_string(item[field], f"$.indexHints[{index}].{field}")
hint_as_of = _integer(item["asOf"], f"$.indexHints[{index}].asOf", minimum=1)
if hint_as_of > as_of:
raise ContractError(f"$.indexHints[{index}] 超出冻结线")
if index_hints and not (
context["mode"] == "diagnostic_only"
and context["purpose"] in {"evaluation", "diagnostic"}
and context["acceptanceEligible"] is False
):
raise ContractError("indexHints 只允许用于不可接受的评测或诊断上下文")
if context["mode"] == "production" and index_hints:
raise ContractError("生产上下文禁止 indexHints")
if evidence_strategy == "card_index_only":
if context["mode"] != "diagnostic_only" or context["purpose"] not in {"evaluation", "diagnostic"}:
raise ContractError("card_index_only 只允许用于诊断上下文")
if context["factEvidence"] or context["proseEvidence"] or not index_hints:
raise ContractError("card_index_only 必须仅包含 indexHints")
elif evidence_strategy == "historical_prose_only":
if index_hints or context["factEvidence"] or not context["proseEvidence"]:
raise ContractError("historical_prose_only 必须仅包含历史原文")
elif evidence_strategy == "card_index_plus_prose":
if context["factEvidence"] or not index_hints or not context["proseEvidence"]:
raise ContractError("card_index_plus_prose 必须包含 indexHints 与历史原文")
# 生产正文必须携带截至冻结点的连续前四章。诊断 A/C 保持同一
# 原文基线,只有显式 card_index_only 策略获准跳过该生产要求。
if evidence_strategy != "card_index_only":
baseline_chapters = sorted(
item["chapter"]
for item in context["proseEvidence"]
if item["isRecentBaseline"] is True
)
expected_baseline = list(range(max(1, as_of - 3), as_of + 1))
if baseline_chapters != expected_baseline:
raise ContractError("原文证据必须包含截至冻结点的连续前四章基线")
for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")):
# 范式引用走放宽校验:来源指针仍严格,另允许 name/summary/writingPoints 内容字段。
_pattern_source_ref(reference, f"$.patternReferences[{index}]")
if "humanizationProvenance" in context:
provenance = _object(
context["humanizationProvenance"],
"$.humanizationProvenance",
frozenset({"schemaVersion", "ruleLibraryVersion", "voiceLedgerSha256", "constraintCount"}),
)
_string(provenance["schemaVersion"], "$.humanizationProvenance.schemaVersion")
_string(provenance["ruleLibraryVersion"], "$.humanizationProvenance.ruleLibraryVersion")
if provenance["voiceLedgerSha256"] is not None:
_hash(provenance["voiceLedgerSha256"], "$.humanizationProvenance.voiceLedgerSha256")
_integer(provenance["constraintCount"], "$.humanizationProvenance.constraintCount", minimum=0)
if "styleConstraints" in context:
# 文风约束(规划期选定的 style 画像投影):非空字符串数组;缺省时整个键省略,
# 上下文逐字节不变(不破坏既有冻结哈希),build 端按空数组投影。
style_rules = _array(context["styleConstraints"], "$.styleConstraints")
for index, rule in enumerate(style_rules):
_string(rule, f"$.styleConstraints[{index}]", nonempty=True)
context["styleConstraints"] = [str(rule) for rule in style_rules]
for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")):
item = _object(coverage, f"$.evidenceCoverage[{index}]", frozenset({"elementId", "elementType", "name", "status", "factEvidenceIds", "proseEvidenceIds", "gapReason"}))
for field in ("elementId", "elementType", "name", "gapReason"):
_string(item[field], f"$.evidenceCoverage[{index}].{field}", nonempty=field != "gapReason")
if item["status"] not in {"supported", "declared_new", "card_gap", "style_gap", "unsupported", "conflict"}:
raise ContractError(f"$.evidenceCoverage[{index}].status 枚举非法")
for field in ("factEvidenceIds", "proseEvidenceIds"):
if any(not isinstance(item_id, str) or not item_id for item_id in _array(item[field], f"$.evidenceCoverage[{index}].{field}")):
raise ContractError(f"$.evidenceCoverage[{index}].{field} 必须是字符串数组")
output = _object(context["outputContract"], "$.outputContract", frozenset({"targetChars", "minChars", "maxChars", "frontmatterRequired"}))
for field in ("targetChars", "minChars", "maxChars"):
_integer(output[field], f"$.outputContract.{field}", minimum=1)
if not output["minChars"] <= output["targetChars"] <= output["maxChars"]:
raise ContractError("$.outputContract 篇幅范围不包含目标值")
_boolean(output["frontmatterRequired"], "$.outputContract.frontmatterRequired")
budget = _object(context["tokenBudget"], "$.tokenBudget", frozenset({"maxContextChars", "usedContextChars"}))
for field in budget:
_integer(budget[field], f"$.tokenBudget.{field}", minimum=0)
if budget["usedContextChars"] > budget["maxContextChars"]:
raise ContractError("$.tokenBudget 已超预算")
for index, omitted in enumerate(_array(context["omittedSources"], "$.omittedSources")):
item = _object(omitted, f"$.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
_string(item["sourceId"], f"$.omittedSources[{index}].sourceId")
_string(item["reason"], f"$.omittedSources[{index}].reason")
eligible = _boolean(context["acceptanceEligible"], "$.acceptanceEligible")
if context["purpose"] in {"evaluation", "diagnostic"} or context["mode"] == "diagnostic_only":
if eligible:
raise ContractError("评测或诊断上下文必须 acceptanceEligible=false")
elif context["qualityPolicyVersion"] != "writer-production-v1":
raise ContractError("生产上下文必须绑定 writer-production-v1")
if context["contextSnapshot"]["contextSha256"] != retrieval_identity(context):
raise ContractError("$.contextSnapshot.contextSha256 与上下文内容不匹配")
return json.loads(canonical_json(context))
def build_writer_creative_input(value: Any) -> dict[str, Any]:
"""从完整冻结上下文投影 writer 唯一可见的创作输入。"""
context = validate_writer_context(value)
outline = context["fineOutline"]
fact_constraints = [
{
"constraintId": item["evidenceId"],
"text": item["fact"],
"kind": "established",
"riskLevel": item["riskLevel"],
}
for item in context["factEvidence"]
]
fact_constraints.extend(
{
"constraintId": item["factId"],
"text": item["text"],
"kind": "declared_new",
"riskLevel": "low",
}
for item in outline["declaredNewFacts"]
)
prose_excerpts = [
{
"excerptId": item["evidenceId"],
"chapter": item["chapter"],
"purpose": item["purpose"],
"text": item["text"],
"isRecentBaseline": item["isRecentBaseline"],
}
for item in context["proseEvidence"]
]
# SoT 变更:把范式卡内容投影给写手。WHY——此前只投影 referenceId+kind 两个标签,
# 写手看不到范式卡写什么,「范式指导」单变量实际为空;这里把名字、一句话摘要和写法
# 要点 surface 出来(均为可选,存在且非空才给)。来源指针(sourceId/sourceVersion)
# 一律不进写手输入,只留在冻结上下文供审计回读。
pattern_references = []
for index, item in enumerate(context["patternReferences"]):
reference: dict[str, Any] = {
"referenceId": f"pattern-{index + 1}",
"kind": item.get("sourceType", "authorized_pattern"),
}
if item.get("name"):
reference["name"] = item["name"]
if item.get("summary"):
reference["summary"] = item["summary"]
if item.get("writingPoints"):
reference["writingPoints"] = dict(item["writingPoints"])
pattern_references.append(reference)
output_contract = context["outputContract"]
creative_input = {
"fineOutline": {
"hardConstraints": list(outline["hardConstraints"]),
"adjustableBeats": list(outline["adjustableBeats"]),
"declaredNewFacts": [
{"factId": item["factId"], "text": item["text"]}
for item in outline["declaredNewFacts"]
],
},
"narrativeState": context["narrativeState"],
"factConstraints": fact_constraints,
"proseExcerpts": prose_excerpts,
"patternReferences": pattern_references,
"lengthContract": {
"targetChars": output_contract["targetChars"],
"minChars": output_contract["minChars"],
"maxChars": output_contract["maxChars"],
"frontmatterRequired": output_contract["frontmatterRequired"],
},
# 文风约束从冻结上下文投影(规划期选定的 style 画像);上下文未带则为空,
# 不再写死恒空——style 真注入的出口。
"styleConstraints": [str(rule) for rule in context.get("styleConstraints", [])],
}
return json.loads(canonical_json(creative_input))
def validate_writer_draft(value: Any) -> dict[str, Any]:
"""校验 writer 模型的单字段创作输出。"""
draft = _object(value, "$", frozenset({"candidateBody"}))
body = draft["candidateBody"]
if not isinstance(body, str) or not body.strip():
raise ContractError("$.candidateBody 必须是非空字符串")
return {"candidateBody": body}
def build_candidate_envelope(
context_value: Any,
draft_value: Any,
*,
candidate_version: int = 1,
) -> dict[str, Any]:
"""规范化模型正文并绑定可信候选身份。"""
context = validate_writer_context(context_value)
draft = validate_writer_draft(draft_value)
version = _integer(candidate_version, "$.candidateVersion", minimum=1)
body = normalize_text(draft["candidateBody"])
if not body.strip():
raise ContractError("$.candidateBody 规范化后不得为空")
candidate_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
envelope = {
"schemaVersion": OUTPUT_VERSION,
"runId": context["runId"],
"attempt": context["attempt"],
"mode": context["mode"],
"qualityPolicyVersion": context["qualityPolicyVersion"],
"contextSnapshotId": context["contextSnapshot"]["manifestId"],
"contextSnapshotSha256": context["contextSnapshot"]["contextSha256"],
"candidateVersion": version,
"candidateSha256": candidate_hash,
"acceptanceEligible": context["acceptanceEligible"],
"candidateBody": body,
}
return validate_writer_output(envelope)
def validate_writer_output(value: Any) -> dict[str, Any]:
"""校验 CandidateEnvelope v2 的信任绑定与正文哈希。"""
output = _object(
value,
"$",
frozenset(
{
"schemaVersion",
"runId",
"attempt",
"mode",
"qualityPolicyVersion",
"contextSnapshotId",
"contextSnapshotSha256",
"candidateVersion",
"candidateSha256",
"acceptanceEligible",
"candidateBody",
}
),
)
if output["schemaVersion"] != OUTPUT_VERSION:
raise ContractError("$.schemaVersion 版本不支持")
if not _RUN_ID_RE.fullmatch(_string(output["runId"], "$.runId")):
raise ContractError("$.runId 格式非法")
_integer(output["attempt"], "$.attempt", minimum=1)
if output["mode"] not in {"production", "diagnostic_only"}:
raise ContractError("$.mode 枚举非法")
_string(output["qualityPolicyVersion"], "$.qualityPolicyVersion")
_hash(output["contextSnapshotId"], "$.contextSnapshotId")
_hash(output["contextSnapshotSha256"], "$.contextSnapshotSha256")
_integer(output["candidateVersion"], "$.candidateVersion", minimum=1)
body = _string(output["candidateBody"], "$.candidateBody")
candidate_hash = _hash(output["candidateSha256"], "$.candidateSha256")
expected_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
if candidate_hash != expected_hash:
raise ContractError("$.candidateSha256 与规范正文不匹配")
eligible = _boolean(output["acceptanceEligible"], "$.acceptanceEligible")
if output["mode"] == "diagnostic_only" and eligible:
raise ContractError("诊断候选必须 acceptanceEligible=false")
return json.loads(canonical_json(output))
__all__ = [
"CONTEXT_VERSION", "DRAFT_VERSION", "OUTPUT_VERSION", "PLAN_VERSION", "MANIFEST_VERSION", "TIE_BREAK",
"PATTERN_CONTENT_FIELDS", "PATTERN_NAME_MAX_CHARS", "PATTERN_SUMMARY_MAX_CHARS",
"PATTERN_POINTS_MAX_FIELDS", "PATTERN_POINT_MAX_CHARS",
"ContractError", "normalize_text", "canonical_json", "retrieval_identity", "han_count",
"calculate_target_chars", "validate_writer_context", "build_writer_creative_input",
"validate_writer_draft", "build_candidate_envelope", "validate_writer_output",
"project_pattern_pointers",
]