修五个硬伤,让"前期准备"从文档纪律变成机械门禁,并补齐写作链上游注入: - 细纲合同合一:新建 meta/schemas/fine_outline.yaml 作唯一字段权威(必填9+推荐7), 统一此前 planning 产出形与 writer 装配消费形两套不相交字段;fine-outline 技能对齐。 - 落库字段覆盖门禁:persist_planning 按 schema 校验必填字段,缺必填失败关闭, 「字段存疑」逃生口;只对声明「字段覆盖门禁:强制」的型生效(fine_outline 已强制)。 - 文风真注入:WriterContext 增可选 styleConstraints(空则省键、上下文逐字节不变), 装配投影给 writer,不再写死为空。 - 范式规划期绑定:read-context load_confirmed_pattern_bindings 只读已确认 assembly 绑定 (实验仓承载,确认即绑定,不改主仓表);生产脚本注入已绑定范式。 - 取数端统一:read-context 三个一等取数端(细纲/范式/文风),生产编排 step2 接线, 已确认细纲不再人肉读库传参。 - harness:AGENTS.md 第9节新增「提交前必须独立子代理形而上四维审查」硬规则。 (writer_contract/retrieve 含上一会话 asOf=0 开篇冻结线改动,同属创作链修复,随本笔提交。) 测试:9 套全绿(合同6/门禁8/装配17/writer_contract21/检索11/统一3/细纲读取4/范式读取5/文风读取6)。
820 lines
40 KiB
Python
820 lines
40 KiB
Python
#!/usr/bin/env python3
|
||
"""正文写手上下文与输出的严格合同。
|
||
|
||
本模块只处理纯数据,不读取文件、数据库或网络。所有进入写手的文本先做
|
||
Unicode NFC 与换行归一化,所有身份哈希都来自同一份规范 JSON。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import hashlib
|
||
import json
|
||
import math
|
||
import re
|
||
import unicodedata
|
||
from decimal import Decimal, ROUND_HALF_UP
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
|
||
CONTEXT_VERSION = "writer-context-v1"
|
||
DRAFT_VERSION = "writer-draft-v2"
|
||
OUTPUT_VERSION = "candidate-envelope-v2"
|
||
PLAN_VERSION = "writer-retrieval-plan-v1"
|
||
MANIFEST_VERSION = "writer-retrieval-manifest-v1"
|
||
TIE_BREAK = "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC"
|
||
|
||
_HASH_RE = re.compile(r"^sha256:[0-9a-f]{64}$")
|
||
_RUN_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$")
|
||
_VOLATILE_IDENTITY_FIELDS = frozenset(
|
||
{"runId", "generatedAt", "timestamp", "executionNode", "contextSha256"}
|
||
)
|
||
|
||
# Unicode Script=Han 覆盖的标准区段。兼容表意文字也按“汉字”计数,
|
||
# 但标点、Markdown、拉丁字母和数字不会落入这些区段。
|
||
_HAN_RANGES = (
|
||
(0x3400, 0x4DBF),
|
||
(0x4E00, 0x9FFF),
|
||
(0xF900, 0xFAFF),
|
||
(0x20000, 0x2EBEF),
|
||
(0x2F800, 0x2FA1F),
|
||
(0x30000, 0x323AF),
|
||
)
|
||
|
||
|
||
class ContractError(ValueError):
|
||
"""输入不符合严格合同时抛出,调用方必须失败关闭。"""
|
||
|
||
|
||
def normalize_text(value: str) -> str:
|
||
"""把文本统一为 NFC 与 LF,供哈希和 Unicode 偏移共同使用。
|
||
|
||
模型在 JSON 输出里常把换行双重转义成字面 ``\\n``(反斜杠+n 两个字符),这里连同
|
||
真实的 CRLF/CR 一并还原为真正的换行符 LF,避免正文带着字面 ``\\n`` 显示异常、
|
||
以及检测/盲评的跨段引文因换行表示不同而匹配失败。
|
||
"""
|
||
|
||
if not isinstance(value, str):
|
||
raise ContractError("待归一化文本必须是字符串")
|
||
value = value.replace("\r\n", "\n").replace("\r", "\n") # 真实 CRLF/CR → LF
|
||
# 字面转义还原:先处理 \r\n(4 字符)再处理 \n / \r(2 字符),顺序避免半截替换
|
||
value = value.replace("\\r\\n", "\n").replace("\\n", "\n").replace("\\r", "\n")
|
||
return unicodedata.normalize("NFC", value)
|
||
|
||
|
||
def _normalize_json(value: Any) -> Any:
|
||
"""递归归一化 JSON 值,并拒绝 JSON 之外或不可复现的数值。"""
|
||
|
||
if isinstance(value, str):
|
||
return normalize_text(value)
|
||
if value is None or isinstance(value, (bool, int)):
|
||
return value
|
||
if isinstance(value, float):
|
||
if not math.isfinite(value):
|
||
raise ContractError("规范 JSON 不允许 NaN 或 Infinity")
|
||
return value
|
||
if isinstance(value, list):
|
||
return [_normalize_json(item) for item in value]
|
||
if isinstance(value, tuple):
|
||
return [_normalize_json(item) for item in value]
|
||
if isinstance(value, Mapping):
|
||
result: dict[str, Any] = {}
|
||
for raw_key, item in value.items():
|
||
if not isinstance(raw_key, str):
|
||
raise ContractError("规范 JSON 的对象键必须是字符串")
|
||
key = normalize_text(raw_key)
|
||
if key in result:
|
||
raise ContractError(f"对象键在 NFC 归一化后冲突: {key}")
|
||
result[key] = _normalize_json(item)
|
||
return result
|
||
raise ContractError(f"值不是 JSON 类型: {type(value).__name__}")
|
||
|
||
|
||
def canonical_json(value: Any) -> str:
|
||
"""输出 UTF-8 语义、键排序、无额外空白的规范 JSON 文本。"""
|
||
|
||
return json.dumps(
|
||
_normalize_json(value),
|
||
ensure_ascii=False,
|
||
sort_keys=True,
|
||
separators=(",", ":"),
|
||
allow_nan=False,
|
||
)
|
||
|
||
|
||
def _without_volatile_fields(value: Any) -> Any:
|
||
"""递归移除运行元数据,防止同一检索输入得到不同身份。"""
|
||
|
||
if isinstance(value, list):
|
||
return [_without_volatile_fields(item) for item in value]
|
||
if isinstance(value, Mapping):
|
||
return {
|
||
key: _without_volatile_fields(item)
|
||
for key, item in value.items()
|
||
if key not in _VOLATILE_IDENTITY_FIELDS
|
||
}
|
||
return value
|
||
|
||
|
||
def retrieval_identity(value: Any) -> str:
|
||
"""计算计划、manifest 或上下文的稳定 SHA-256 身份。"""
|
||
|
||
encoded = canonical_json(_without_volatile_fields(value)).encode("utf-8")
|
||
return "sha256:" + hashlib.sha256(encoded).hexdigest()
|
||
|
||
|
||
def han_count(value: str) -> int:
|
||
"""统计 Unicode Han code point,不把标点或 Markdown 算入正文长度。"""
|
||
|
||
text = normalize_text(value)
|
||
return sum(any(start <= ord(char) <= end for start, end in _HAN_RANGES) for char in text)
|
||
|
||
|
||
def _round_half_up(value: Decimal) -> int:
|
||
"""以十进制 ROUND_HALF_UP 规则取整,避免 Python 银行家舍入。"""
|
||
|
||
return int(value.quantize(Decimal("1"), rounding=ROUND_HALF_UP))
|
||
|
||
|
||
def _round_to_100(value: Decimal) -> int:
|
||
"""以百字为单位执行十进制半入取整。"""
|
||
|
||
return int((value / Decimal(100)).quantize(Decimal("1"), rounding=ROUND_HALF_UP) * 100)
|
||
|
||
|
||
def calculate_target_chars(
|
||
*,
|
||
explicit_target_chars: int | None = None,
|
||
recent_chapter_han_counts: Sequence[int] = (),
|
||
default_target_chars: int = 4000,
|
||
hard_event_count: int = 3,
|
||
foreshadowing_action_count: int = 0,
|
||
required_scene_count: int = 0,
|
||
min_chars: int = 2000,
|
||
max_chars: int = 10000,
|
||
) -> int:
|
||
"""按冻结历史中位数与细纲密度计算确定性目标汉字数。"""
|
||
|
||
counts = {
|
||
"default_target_chars": default_target_chars,
|
||
"hard_event_count": hard_event_count,
|
||
"foreshadowing_action_count": foreshadowing_action_count,
|
||
"required_scene_count": required_scene_count,
|
||
"min_chars": min_chars,
|
||
"max_chars": max_chars,
|
||
}
|
||
if any(isinstance(value, bool) or not isinstance(value, int) for value in counts.values()):
|
||
raise ContractError("篇幅参数必须是整数")
|
||
if default_target_chars <= 0 or min_chars <= 0 or max_chars < min_chars or any(value < 0 for key, value in counts.items() if "count" in key):
|
||
raise ContractError("篇幅边界或细纲计数非法")
|
||
if explicit_target_chars is not None:
|
||
if isinstance(explicit_target_chars, bool) or not isinstance(explicit_target_chars, int):
|
||
raise ContractError("显式 targetChars 必须是整数")
|
||
return min(max(explicit_target_chars, min_chars), max_chars)
|
||
|
||
valid_counts = list(recent_chapter_han_counts)[-20:]
|
||
if any(isinstance(value, bool) or not isinstance(value, int) or value < 500 for value in valid_counts):
|
||
raise ContractError("历史章汉字数必须是大于等于 500 的整数")
|
||
if len(valid_counts) >= 3:
|
||
ordered_counts = sorted(valid_counts)
|
||
midpoint = len(ordered_counts) // 2
|
||
if len(ordered_counts) % 2:
|
||
baseline = Decimal(ordered_counts[midpoint])
|
||
else:
|
||
baseline = (Decimal(ordered_counts[midpoint - 1]) + Decimal(ordered_counts[midpoint])) / 2
|
||
else:
|
||
baseline = Decimal(default_target_chars)
|
||
density = (
|
||
Decimal(hard_event_count)
|
||
+ Decimal("0.5") * foreshadowing_action_count
|
||
+ Decimal("0.5") * required_scene_count
|
||
)
|
||
factor = min(Decimal("1.15"), max(Decimal("0.85"), Decimal("0.85") + Decimal("0.05") * (density - 3)))
|
||
target = _round_to_100(Decimal(_round_half_up(baseline)) * factor)
|
||
return min(max(target, min_chars), max_chars)
|
||
|
||
|
||
def _object(
|
||
value: Any,
|
||
path: str,
|
||
required: set[str] | frozenset[str],
|
||
optional: set[str] | frozenset[str] = frozenset(),
|
||
) -> Mapping[str, Any]:
|
||
"""校验严格对象,任何缺字段或未知字段都立即失败。"""
|
||
|
||
if not isinstance(value, Mapping):
|
||
raise ContractError(f"{path} 必须是对象")
|
||
missing = sorted(required - set(value))
|
||
unknown = sorted(set(value) - required - optional)
|
||
if missing:
|
||
raise ContractError(f"{path} 缺少字段: {','.join(missing)}")
|
||
if unknown:
|
||
raise ContractError(f"{path} 包含未知字段: {','.join(unknown)}")
|
||
return value
|
||
|
||
|
||
def _string(value: Any, path: str, *, nonempty: bool = True) -> str:
|
||
"""校验字符串,并在需要时拒绝空值。"""
|
||
|
||
if not isinstance(value, str) or (nonempty and not value.strip()):
|
||
raise ContractError(f"{path} 必须是非空字符串")
|
||
if value != normalize_text(value):
|
||
raise ContractError(f"{path} 必须预先归一化为 Unicode NFC/LF")
|
||
return value
|
||
|
||
|
||
def _integer(value: Any, path: str, *, minimum: int = 0) -> int:
|
||
"""校验整数,显式排除 bool 这一 Python int 子类。"""
|
||
|
||
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
|
||
raise ContractError(f"{path} 必须是大于等于 {minimum} 的整数")
|
||
return value
|
||
|
||
|
||
def _boolean(value: Any, path: str) -> bool:
|
||
"""校验严格布尔值。"""
|
||
|
||
if not isinstance(value, bool):
|
||
raise ContractError(f"{path} 必须是布尔值")
|
||
return value
|
||
|
||
|
||
def _array(value: Any, path: str) -> list[Any]:
|
||
"""校验数组并返回原值,拒绝元组等隐式转换。"""
|
||
|
||
if not isinstance(value, list):
|
||
raise ContractError(f"{path} 必须是数组")
|
||
return value
|
||
|
||
|
||
def _hash(value: Any, path: str) -> str:
|
||
"""校验带算法前缀的 SHA-256。"""
|
||
|
||
text = _string(value, path)
|
||
if not _HASH_RE.fullmatch(text):
|
||
raise ContractError(f"{path} 必须是 sha256: 加 64 位小写十六进制")
|
||
return text
|
||
|
||
|
||
_SOURCE_REF_REQUIRED = frozenset({"sourceId", "sourceVersion"})
|
||
_SOURCE_REF_OPTIONAL = frozenset(
|
||
{"chapter", "blockId", "startCodePoint", "endCodePoint", "contentSha256", "sourceType"}
|
||
)
|
||
|
||
# 范式引用(patternReferences)在严格来源指针之外额外允许的内容字段。
|
||
# WHY(SoT 变更):此前 patternReferences 只能带来源指针,写手最终只看到一个空标签
|
||
# (referenceId+kind),读不到范式卡的名字/摘要/写法,「范式指导」这个实验单变量
|
||
# 形同虚设。放宽这三个内容字段只针对 patternReferences,其它来源指针不受影响。
|
||
PATTERN_CONTENT_FIELDS = frozenset({"name", "summary", "writingPoints"})
|
||
|
||
# 范式引用内容字段的体量硬上限(name/summary/写法要点值按 code point 计,字段数按个计)。
|
||
# WHY:范式卡原始字段可能长达数千字,直接灌给写手会撑爆上下文预算。合同侧按这些上限
|
||
# 失败关闭——无论检索端将来怎么换,超量内容都进不了写手输入;检索端投影时应先截断到
|
||
# 上限以内,合同复核是第二道闸。
|
||
PATTERN_NAME_MAX_CHARS = 40
|
||
PATTERN_SUMMARY_MAX_CHARS = 120
|
||
PATTERN_POINTS_MAX_FIELDS = 6
|
||
PATTERN_POINT_MAX_CHARS = 200
|
||
|
||
|
||
def _validate_source_ref_pointers(ref: Mapping[str, Any], path: str) -> None:
|
||
"""校验来源指针自身(sourceId/sourceVersion 必填及定位字段),供严格与放宽校验复用。"""
|
||
|
||
_string(ref["sourceId"], f"{path}.sourceId")
|
||
_string(ref["sourceVersion"], f"{path}.sourceVersion")
|
||
for field in ("chapter", "blockId", "startCodePoint", "endCodePoint"):
|
||
if field in ref:
|
||
_integer(ref[field], f"{path}.{field}", minimum=0 if "CodePoint" in field else 1)
|
||
if "startCodePoint" in ref and "endCodePoint" in ref and ref["endCodePoint"] <= ref["startCodePoint"]:
|
||
raise ContractError(f"{path} 字符区间必须是非空左闭右开区间")
|
||
if "contentSha256" in ref:
|
||
_hash(ref["contentSha256"], f"{path}.contentSha256")
|
||
if "sourceType" in ref:
|
||
_string(ref["sourceType"], f"{path}.sourceType")
|
||
|
||
|
||
def _source_ref(value: Any, path: str) -> None:
|
||
"""校验不可变来源引用;历史原文可额外携带块和字符区间。"""
|
||
|
||
ref = _object(value, path, _SOURCE_REF_REQUIRED, _SOURCE_REF_OPTIONAL)
|
||
_validate_source_ref_pointers(ref, path)
|
||
|
||
|
||
def _pattern_source_ref(value: Any, path: str) -> None:
|
||
"""校验范式引用:严格来源指针 + 放宽且限量的内容字段。
|
||
|
||
WHY(SoT 变更):让写手真正读到范式卡——名字、一句话摘要、写法要点——而不是
|
||
只看到一个来源标签。来源指针仍必填,保证可回读、可审计;内容字段全部限量并失败
|
||
关闭,防止撑爆写手上下文。放宽只针对 patternReferences:proseEvidence/factEvidence/
|
||
manifest 等其它来源指针继续走严格的 _source_ref,任何名字/摘要字段仍按「未知字段」拒收。
|
||
"""
|
||
|
||
ref = _object(value, path, _SOURCE_REF_REQUIRED, _SOURCE_REF_OPTIONAL | PATTERN_CONTENT_FIELDS)
|
||
_validate_source_ref_pointers(ref, path)
|
||
if "name" in ref and len(_string(ref["name"], f"{path}.name")) > PATTERN_NAME_MAX_CHARS:
|
||
raise ContractError(f"{path}.name 超出体量上限 {PATTERN_NAME_MAX_CHARS} 字")
|
||
if "summary" in ref and len(_string(ref["summary"], f"{path}.summary")) > PATTERN_SUMMARY_MAX_CHARS:
|
||
raise ContractError(f"{path}.summary 超出体量上限 {PATTERN_SUMMARY_MAX_CHARS} 字")
|
||
if "writingPoints" in ref:
|
||
points = ref["writingPoints"]
|
||
if not isinstance(points, Mapping):
|
||
raise ContractError(f"{path}.writingPoints 必须是对象")
|
||
if len(points) > PATTERN_POINTS_MAX_FIELDS:
|
||
raise ContractError(f"{path}.writingPoints 超出 {PATTERN_POINTS_MAX_FIELDS} 个字段上限")
|
||
for key, item in points.items():
|
||
key_text = _string(key, f"{path}.writingPoints.<key>")
|
||
if len(key_text) > PATTERN_NAME_MAX_CHARS:
|
||
raise ContractError(f"{path}.writingPoints 字段名超出体量上限 {PATTERN_NAME_MAX_CHARS} 字")
|
||
if len(_string(item, f"{path}.writingPoints.{key_text}")) > PATTERN_POINT_MAX_CHARS:
|
||
raise ContractError(
|
||
f"{path}.writingPoints.{key_text} 超出体量上限 {PATTERN_POINT_MAX_CHARS} 字"
|
||
)
|
||
|
||
|
||
def pattern_references_for_arm(
|
||
arm: str, c_references: Sequence[Mapping[str, Any]]
|
||
) -> list[dict[str, Any]]:
|
||
"""按臂分配范式引用的唯一事实源:A 臂恒空,其余臂(B/C)拿 C 臂候选范式卡。
|
||
|
||
WHY:Gate A 的唯一实验变量是「有无卡(含范式卡)」。A 臂是纯历史原文对照,必须
|
||
恒空,否则 A/C 单变量对照被破坏。范式卡的链路有两段独立 assemble:装配端
|
||
(load_writer_reference_work)检索出 C 臂候选并冻结进 config.json 的
|
||
``writerContextInput.patternReferences``;回放端(run_writer_replay)真写时再从
|
||
config.json 读出候选、重新 assemble 各臂上下文。两段必须按完全相同的规则分臂,
|
||
因此把规则收敛到合同模块这一处由两端复用——任一段各写一套,就会出现「C 臂真写
|
||
读不到范式卡(实验失效)」或「A 臂混入范式卡(对照破坏)」。返回深拷贝,避免
|
||
各臂上下文与冻结候选互相串改。
|
||
"""
|
||
|
||
if arm == "A":
|
||
return []
|
||
return [copy.deepcopy(dict(item)) for item in c_references]
|
||
|
||
|
||
def project_pattern_pointers(value: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""从可能携带内容字段的引用中投影出纯来源指针(供 manifest 等审计账本使用)。
|
||
|
||
WHY:manifest 记录「哪些来源入包」,只承载可回读指针,不承载范式卡正文;范式卡
|
||
内容只由上下文内的 patternReferences 承载(并计入上下文身份哈希)。
|
||
"""
|
||
|
||
return {key: value[key] for key in (_SOURCE_REF_REQUIRED | _SOURCE_REF_OPTIONAL) if key in value}
|
||
|
||
|
||
def _validate_plan(value: Any, path: str) -> None:
|
||
"""校验固定检索计划,不允许写手临场扩张查询。"""
|
||
|
||
plan = _object(
|
||
value,
|
||
path,
|
||
frozenset(
|
||
{"planVersion", "planId", "runId", "asOf", "queries", "cardIndexVersion", "proseIndexVersion", "filters", "tieBreak", "tokenBudget"}
|
||
),
|
||
)
|
||
if plan["planVersion"] != PLAN_VERSION:
|
||
raise ContractError(f"{path}.planVersion 版本不支持")
|
||
_hash(plan["planId"], f"{path}.planId")
|
||
if not _RUN_ID_RE.fullmatch(_string(plan["runId"], f"{path}.runId")):
|
||
raise ContractError(f"{path}.runId 格式非法")
|
||
# asOf=0 表示开篇前冻结线(第 1 章之前):此刻只有设定/大纲/细纲,无历史正文
|
||
_integer(plan["asOf"], f"{path}.asOf", minimum=0)
|
||
for index, query in enumerate(_array(plan["queries"], f"{path}.queries")):
|
||
item = _object(query, f"{path}.queries[{index}]", frozenset({"queryId", "text", "entityTypes", "purpose", "topK"}))
|
||
_string(item["queryId"], f"{path}.queries[{index}].queryId")
|
||
_string(item["text"], f"{path}.queries[{index}].text")
|
||
if any(not isinstance(kind, str) or not kind for kind in _array(item["entityTypes"], f"{path}.queries[{index}].entityTypes")):
|
||
raise ContractError(f"{path}.queries[{index}].entityTypes 必须是非空字符串数组")
|
||
_string(item["purpose"], f"{path}.queries[{index}].purpose")
|
||
_integer(item["topK"], f"{path}.queries[{index}].topK", minimum=1)
|
||
_string(plan["cardIndexVersion"], f"{path}.cardIndexVersion")
|
||
_string(plan["proseIndexVersion"], f"{path}.proseIndexVersion")
|
||
filters = _object(plan["filters"], f"{path}.filters", frozenset({"workId", "asOfChapter", "sourceStatus", "authorizationRequired"}))
|
||
_integer(filters["workId"], f"{path}.filters.workId", minimum=1)
|
||
_integer(filters["asOfChapter"], f"{path}.filters.asOfChapter", minimum=0)
|
||
_string(filters["sourceStatus"], f"{path}.filters.sourceStatus")
|
||
_boolean(filters["authorizationRequired"], f"{path}.filters.authorizationRequired")
|
||
if plan["tieBreak"] != TIE_BREAK:
|
||
raise ContractError(f"{path}.tieBreak 不符合稳定排序合同")
|
||
budget = _object(plan["tokenBudget"], f"{path}.tokenBudget", frozenset({"maxContextChars"}), frozenset({"cardChars", "recentProseChars", "historicalProseChars", "patternChars"}))
|
||
for key, item in budget.items():
|
||
_integer(item, f"{path}.tokenBudget.{key}", minimum=0)
|
||
identity_payload = {key: item for key, item in plan.items() if key != "planId"}
|
||
if plan["planId"] != retrieval_identity(identity_payload):
|
||
raise ContractError(f"{path}.planId 与计划内容不匹配")
|
||
|
||
|
||
def _validate_manifest(value: Any, path: str) -> None:
|
||
"""校验检索清单的来源与裁剪回显。"""
|
||
|
||
manifest = _object(value, path, frozenset({"manifestVersion", "manifestId", "planId", "sources", "omittedSources"}))
|
||
if manifest["manifestVersion"] != MANIFEST_VERSION:
|
||
raise ContractError(f"{path}.manifestVersion 版本不支持")
|
||
_hash(manifest["manifestId"], f"{path}.manifestId")
|
||
_hash(manifest["planId"], f"{path}.planId")
|
||
for index, source in enumerate(_array(manifest["sources"], f"{path}.sources")):
|
||
_source_ref(source, f"{path}.sources[{index}]")
|
||
for index, omitted in enumerate(_array(manifest["omittedSources"], f"{path}.omittedSources")):
|
||
item = _object(omitted, f"{path}.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
|
||
_string(item["sourceId"], f"{path}.omittedSources[{index}].sourceId")
|
||
_string(item["reason"], f"{path}.omittedSources[{index}].reason")
|
||
identity_payload = {key: item for key, item in manifest.items() if key != "manifestId"}
|
||
if manifest["manifestId"] != retrieval_identity(identity_payload):
|
||
raise ContractError(f"{path}.manifestId 与来源清单不匹配")
|
||
|
||
|
||
def validate_writer_context(value: Any) -> dict[str, Any]:
|
||
"""校验 WriterContext v1;成功时返回可安全复制的规范 JSON 对象。"""
|
||
|
||
required = frozenset(
|
||
{
|
||
"schemaVersion", "runId", "attempt", "mode", "purpose", "qualityPolicyVersion",
|
||
"workId", "targetChapter", "asOf", "contextSnapshot", "sourceVersion",
|
||
"authorizationSnapshot", "sourceStatus", "retrievalPlan", "retrievalManifest",
|
||
"fineOutline", "narrativeState", "factEvidence", "proseEvidence",
|
||
"patternReferences", "evidenceCoverage", "outputContract", "tokenBudget",
|
||
"omittedSources", "acceptanceEligible",
|
||
}
|
||
)
|
||
context = _object(
|
||
value,
|
||
"$",
|
||
required,
|
||
frozenset({"evidenceStrategy", "indexHints", "styleConstraints"}),
|
||
)
|
||
if context["schemaVersion"] != CONTEXT_VERSION:
|
||
raise ContractError("$.schemaVersion 版本不支持")
|
||
run_id = _string(context["runId"], "$.runId")
|
||
if not _RUN_ID_RE.fullmatch(run_id):
|
||
raise ContractError("$.runId 格式非法")
|
||
_integer(context["attempt"], "$.attempt", minimum=1)
|
||
if context["mode"] not in {"production", "diagnostic_only"}:
|
||
raise ContractError("$.mode 枚举非法")
|
||
if context["purpose"] not in {"production", "evaluation", "diagnostic"}:
|
||
raise ContractError("$.purpose 枚举非法")
|
||
evidence_strategy = context.get("evidenceStrategy", "production_dual_evidence")
|
||
if evidence_strategy not in {
|
||
"production_dual_evidence",
|
||
"historical_prose_only",
|
||
"card_index_only",
|
||
"card_index_plus_prose",
|
||
}:
|
||
raise ContractError("$.evidenceStrategy 枚举非法")
|
||
if context["mode"] == "production" and evidence_strategy != "production_dual_evidence":
|
||
raise ContractError("生产上下文必须使用 production_dual_evidence")
|
||
_string(context["qualityPolicyVersion"], "$.qualityPolicyVersion")
|
||
_integer(context["workId"], "$.workId", minimum=1)
|
||
target = _integer(context["targetChapter"], "$.targetChapter", minimum=1)
|
||
# asOf=0 合法:开篇前冻结线,正文基线必为空(任何历史正文都会被后续"超出冻结线"挡下)
|
||
as_of = _integer(context["asOf"], "$.asOf", minimum=0)
|
||
if as_of >= target:
|
||
raise ContractError("$.asOf 必须早于 targetChapter")
|
||
|
||
snapshot = _object(context["contextSnapshot"], "$.contextSnapshot", frozenset({"manifestId", "contextSha256", "generatedAt"}))
|
||
_hash(snapshot["manifestId"], "$.contextSnapshot.manifestId")
|
||
_hash(snapshot["contextSha256"], "$.contextSnapshot.contextSha256")
|
||
_string(snapshot["generatedAt"], "$.contextSnapshot.generatedAt")
|
||
_string(context["sourceVersion"], "$.sourceVersion")
|
||
authorization = _object(
|
||
context["authorizationSnapshot"],
|
||
"$.authorizationSnapshot",
|
||
frozenset({"snapshotId", "allowedPurpose", "verifiedAt"}),
|
||
frozenset({"expiresAt", "sourceVersion"}),
|
||
)
|
||
for key, item in authorization.items():
|
||
_string(item, f"$.authorizationSnapshot.{key}")
|
||
if context["sourceStatus"] not in {"active", "authorized", "frozen_authorized"}:
|
||
raise ContractError("$.sourceStatus 不允许生成")
|
||
|
||
_validate_plan(context["retrievalPlan"], "$.retrievalPlan")
|
||
_validate_manifest(context["retrievalManifest"], "$.retrievalManifest")
|
||
if context["retrievalPlan"]["runId"] != run_id or context["retrievalPlan"]["asOf"] != as_of:
|
||
raise ContractError("检索计划与上下文的 runId/asOf 不一致")
|
||
if context["retrievalManifest"]["planId"] != context["retrievalPlan"]["planId"]:
|
||
raise ContractError("检索 manifest 未绑定当前计划")
|
||
if context["contextSnapshot"]["manifestId"] != context["retrievalManifest"]["manifestId"]:
|
||
raise ContractError("上下文快照未绑定当前 manifest")
|
||
|
||
outline = _object(context["fineOutline"], "$.fineOutline", frozenset({"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"}))
|
||
_source_ref(outline["sourceRef"], "$.fineOutline.sourceRef")
|
||
for field in ("hardConstraints", "adjustableBeats"):
|
||
if any(not isinstance(item, str) or not item for item in _array(outline[field], f"$.fineOutline.{field}")):
|
||
raise ContractError(f"$.fineOutline.{field} 必须是非空字符串数组")
|
||
for index, fact in enumerate(_array(outline["declaredNewFacts"], "$.fineOutline.declaredNewFacts")):
|
||
item = _object(fact, f"$.fineOutline.declaredNewFacts[{index}]", frozenset({"factId", "text", "sourceRef"}))
|
||
_string(item["factId"], f"$.fineOutline.declaredNewFacts[{index}].factId")
|
||
_string(item["text"], f"$.fineOutline.declaredNewFacts[{index}].text")
|
||
_source_ref(item["sourceRef"], f"$.fineOutline.declaredNewFacts[{index}].sourceRef")
|
||
|
||
state = _object(context["narrativeState"], "$.narrativeState", frozenset({"time", "location", "characterPositions", "immediateSituation"}))
|
||
for field in ("time", "location", "immediateSituation"):
|
||
_string(state[field], f"$.narrativeState.{field}", nonempty=False)
|
||
positions = _object(state["characterPositions"], "$.narrativeState.characterPositions", frozenset(state["characterPositions"].keys()) if isinstance(state["characterPositions"], Mapping) else frozenset())
|
||
for key, item in positions.items():
|
||
_string(key, "$.narrativeState.characterPositions.<key>")
|
||
_string(item, f"$.narrativeState.characterPositions.{key}")
|
||
|
||
for index, evidence in enumerate(_array(context["factEvidence"], "$.factEvidence")):
|
||
item = _object(evidence, f"$.factEvidence[{index}]", frozenset({"evidenceId", "fact", "sourceType", "sourceRef", "contentSha256", "riskLevel"}))
|
||
for field in ("evidenceId", "fact", "sourceType"):
|
||
_string(item[field], f"$.factEvidence[{index}].{field}")
|
||
if item["sourceType"] not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}:
|
||
raise ContractError(f"$.factEvidence[{index}].sourceType 枚举非法")
|
||
_source_ref(item["sourceRef"], f"$.factEvidence[{index}].sourceRef")
|
||
if item["sourceType"] == "historical_prose":
|
||
required_location = {"chapter", "blockId", "startCodePoint", "endCodePoint"}
|
||
if not required_location.issubset(item["sourceRef"]):
|
||
raise ContractError(f"$.factEvidence[{index}] 历史事实必须回到章、块和字符区间")
|
||
_hash(item["contentSha256"], f"$.factEvidence[{index}].contentSha256")
|
||
if item["riskLevel"] not in {"low", "medium", "high"}:
|
||
raise ContractError(f"$.factEvidence[{index}].riskLevel 枚举非法")
|
||
|
||
for index, evidence in enumerate(_array(context["proseEvidence"], "$.proseEvidence")):
|
||
item = _object(evidence, f"$.proseEvidence[{index}]", frozenset({"evidenceId", "chapter", "sourceRef", "contentSha256", "purpose", "text", "isRecentBaseline"}))
|
||
_string(item["evidenceId"], f"$.proseEvidence[{index}].evidenceId")
|
||
chapter = _integer(item["chapter"], f"$.proseEvidence[{index}].chapter", minimum=1)
|
||
if chapter > as_of:
|
||
raise ContractError(f"$.proseEvidence[{index}] 超出冻结线")
|
||
_source_ref(item["sourceRef"], f"$.proseEvidence[{index}].sourceRef")
|
||
if not {"chapter", "blockId", "startCodePoint", "endCodePoint"}.issubset(item["sourceRef"]):
|
||
raise ContractError(f"$.proseEvidence[{index}] 必须带章、块和字符区间")
|
||
_hash(item["contentSha256"], f"$.proseEvidence[{index}].contentSha256")
|
||
_string(item["purpose"], f"$.proseEvidence[{index}].purpose")
|
||
text = _string(item["text"], f"$.proseEvidence[{index}].text")
|
||
_boolean(item["isRecentBaseline"], f"$.proseEvidence[{index}].isRecentBaseline")
|
||
expected = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||
if item["contentSha256"] != expected:
|
||
raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配")
|
||
|
||
index_hints = _array(context.get("indexHints", []), "$.indexHints")
|
||
for index, hint in enumerate(index_hints):
|
||
item = _object(
|
||
hint,
|
||
f"$.indexHints[{index}]",
|
||
frozenset(
|
||
{
|
||
"cardId",
|
||
"name",
|
||
"type",
|
||
"content",
|
||
"sourceId",
|
||
"sourceVersion",
|
||
"asOf",
|
||
}
|
||
),
|
||
)
|
||
for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion"):
|
||
_string(item[field], f"$.indexHints[{index}].{field}")
|
||
hint_as_of = _integer(item["asOf"], f"$.indexHints[{index}].asOf", minimum=1)
|
||
if hint_as_of > as_of:
|
||
raise ContractError(f"$.indexHints[{index}] 超出冻结线")
|
||
if index_hints and not (
|
||
context["mode"] == "diagnostic_only"
|
||
and context["purpose"] in {"evaluation", "diagnostic"}
|
||
and context["acceptanceEligible"] is False
|
||
):
|
||
raise ContractError("indexHints 只允许用于不可接受的评测或诊断上下文")
|
||
if context["mode"] == "production" and index_hints:
|
||
raise ContractError("生产上下文禁止 indexHints")
|
||
if evidence_strategy == "card_index_only":
|
||
if context["mode"] != "diagnostic_only" or context["purpose"] not in {"evaluation", "diagnostic"}:
|
||
raise ContractError("card_index_only 只允许用于诊断上下文")
|
||
if context["factEvidence"] or context["proseEvidence"] or not index_hints:
|
||
raise ContractError("card_index_only 必须仅包含 indexHints")
|
||
elif evidence_strategy == "historical_prose_only":
|
||
if index_hints or context["factEvidence"] or not context["proseEvidence"]:
|
||
raise ContractError("historical_prose_only 必须仅包含历史原文")
|
||
elif evidence_strategy == "card_index_plus_prose":
|
||
if context["factEvidence"] or not index_hints or not context["proseEvidence"]:
|
||
raise ContractError("card_index_plus_prose 必须包含 indexHints 与历史原文")
|
||
|
||
# 生产正文必须携带截至冻结点的连续前四章。诊断 A/C 保持同一
|
||
# 原文基线,只有显式 card_index_only 策略获准跳过该生产要求。
|
||
if evidence_strategy != "card_index_only":
|
||
baseline_chapters = sorted(
|
||
item["chapter"]
|
||
for item in context["proseEvidence"]
|
||
if item["isRecentBaseline"] is True
|
||
)
|
||
expected_baseline = list(range(max(1, as_of - 3), as_of + 1))
|
||
if baseline_chapters != expected_baseline:
|
||
raise ContractError("原文证据必须包含截至冻结点的连续前四章基线")
|
||
|
||
for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")):
|
||
# 范式引用走放宽校验:来源指针仍严格,另允许 name/summary/writingPoints 内容字段。
|
||
_pattern_source_ref(reference, f"$.patternReferences[{index}]")
|
||
if "styleConstraints" in context:
|
||
# 文风约束(规划期选定的 style 画像投影):非空字符串数组;缺省时整个键省略,
|
||
# 上下文逐字节不变(不破坏既有冻结哈希),build 端按空数组投影。
|
||
style_rules = _array(context["styleConstraints"], "$.styleConstraints")
|
||
for index, rule in enumerate(style_rules):
|
||
_string(rule, f"$.styleConstraints[{index}]", nonempty=True)
|
||
context["styleConstraints"] = [str(rule) for rule in style_rules]
|
||
for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")):
|
||
item = _object(coverage, f"$.evidenceCoverage[{index}]", frozenset({"elementId", "elementType", "name", "status", "factEvidenceIds", "proseEvidenceIds", "gapReason"}))
|
||
for field in ("elementId", "elementType", "name", "gapReason"):
|
||
_string(item[field], f"$.evidenceCoverage[{index}].{field}", nonempty=field != "gapReason")
|
||
if item["status"] not in {"supported", "declared_new", "card_gap", "style_gap", "unsupported", "conflict"}:
|
||
raise ContractError(f"$.evidenceCoverage[{index}].status 枚举非法")
|
||
for field in ("factEvidenceIds", "proseEvidenceIds"):
|
||
if any(not isinstance(item_id, str) or not item_id for item_id in _array(item[field], f"$.evidenceCoverage[{index}].{field}")):
|
||
raise ContractError(f"$.evidenceCoverage[{index}].{field} 必须是字符串数组")
|
||
|
||
output = _object(context["outputContract"], "$.outputContract", frozenset({"targetChars", "minChars", "maxChars", "frontmatterRequired"}))
|
||
for field in ("targetChars", "minChars", "maxChars"):
|
||
_integer(output[field], f"$.outputContract.{field}", minimum=1)
|
||
if not output["minChars"] <= output["targetChars"] <= output["maxChars"]:
|
||
raise ContractError("$.outputContract 篇幅范围不包含目标值")
|
||
_boolean(output["frontmatterRequired"], "$.outputContract.frontmatterRequired")
|
||
budget = _object(context["tokenBudget"], "$.tokenBudget", frozenset({"maxContextChars", "usedContextChars"}))
|
||
for field in budget:
|
||
_integer(budget[field], f"$.tokenBudget.{field}", minimum=0)
|
||
if budget["usedContextChars"] > budget["maxContextChars"]:
|
||
raise ContractError("$.tokenBudget 已超预算")
|
||
for index, omitted in enumerate(_array(context["omittedSources"], "$.omittedSources")):
|
||
item = _object(omitted, f"$.omittedSources[{index}]", frozenset({"sourceId", "reason"}))
|
||
_string(item["sourceId"], f"$.omittedSources[{index}].sourceId")
|
||
_string(item["reason"], f"$.omittedSources[{index}].reason")
|
||
|
||
eligible = _boolean(context["acceptanceEligible"], "$.acceptanceEligible")
|
||
if context["purpose"] in {"evaluation", "diagnostic"} or context["mode"] == "diagnostic_only":
|
||
if eligible:
|
||
raise ContractError("评测或诊断上下文必须 acceptanceEligible=false")
|
||
elif context["qualityPolicyVersion"] != "writer-production-v1":
|
||
raise ContractError("生产上下文必须绑定 writer-production-v1")
|
||
if context["contextSnapshot"]["contextSha256"] != retrieval_identity(context):
|
||
raise ContractError("$.contextSnapshot.contextSha256 与上下文内容不匹配")
|
||
return json.loads(canonical_json(context))
|
||
|
||
|
||
def build_writer_creative_input(value: Any) -> dict[str, Any]:
|
||
"""从完整冻结上下文投影 writer 唯一可见的创作输入。"""
|
||
|
||
context = validate_writer_context(value)
|
||
outline = context["fineOutline"]
|
||
fact_constraints = [
|
||
{
|
||
"constraintId": item["evidenceId"],
|
||
"text": item["fact"],
|
||
"kind": "established",
|
||
"riskLevel": item["riskLevel"],
|
||
}
|
||
for item in context["factEvidence"]
|
||
]
|
||
fact_constraints.extend(
|
||
{
|
||
"constraintId": item["factId"],
|
||
"text": item["text"],
|
||
"kind": "declared_new",
|
||
"riskLevel": "low",
|
||
}
|
||
for item in outline["declaredNewFacts"]
|
||
)
|
||
prose_excerpts = [
|
||
{
|
||
"excerptId": item["evidenceId"],
|
||
"chapter": item["chapter"],
|
||
"purpose": item["purpose"],
|
||
"text": item["text"],
|
||
"isRecentBaseline": item["isRecentBaseline"],
|
||
}
|
||
for item in context["proseEvidence"]
|
||
]
|
||
# SoT 变更:把范式卡内容投影给写手。WHY——此前只投影 referenceId+kind 两个标签,
|
||
# 写手看不到范式卡写什么,「范式指导」单变量实际为空;这里把名字、一句话摘要和写法
|
||
# 要点 surface 出来(均为可选,存在且非空才给)。来源指针(sourceId/sourceVersion)
|
||
# 一律不进写手输入,只留在冻结上下文供审计回读。
|
||
pattern_references = []
|
||
for index, item in enumerate(context["patternReferences"]):
|
||
reference: dict[str, Any] = {
|
||
"referenceId": f"pattern-{index + 1}",
|
||
"kind": item.get("sourceType", "authorized_pattern"),
|
||
}
|
||
if item.get("name"):
|
||
reference["name"] = item["name"]
|
||
if item.get("summary"):
|
||
reference["summary"] = item["summary"]
|
||
if item.get("writingPoints"):
|
||
reference["writingPoints"] = dict(item["writingPoints"])
|
||
pattern_references.append(reference)
|
||
output_contract = context["outputContract"]
|
||
creative_input = {
|
||
"fineOutline": {
|
||
"hardConstraints": list(outline["hardConstraints"]),
|
||
"adjustableBeats": list(outline["adjustableBeats"]),
|
||
"declaredNewFacts": [
|
||
{"factId": item["factId"], "text": item["text"]}
|
||
for item in outline["declaredNewFacts"]
|
||
],
|
||
},
|
||
"narrativeState": context["narrativeState"],
|
||
"factConstraints": fact_constraints,
|
||
"proseExcerpts": prose_excerpts,
|
||
"patternReferences": pattern_references,
|
||
"lengthContract": {
|
||
"targetChars": output_contract["targetChars"],
|
||
"minChars": output_contract["minChars"],
|
||
"maxChars": output_contract["maxChars"],
|
||
"frontmatterRequired": output_contract["frontmatterRequired"],
|
||
},
|
||
# 文风约束从冻结上下文投影(规划期选定的 style 画像);上下文未带则为空,
|
||
# 不再写死恒空——style 真注入的出口。
|
||
"styleConstraints": [str(rule) for rule in context.get("styleConstraints", [])],
|
||
}
|
||
return json.loads(canonical_json(creative_input))
|
||
|
||
|
||
def validate_writer_draft(value: Any) -> dict[str, Any]:
|
||
"""校验 writer 模型的单字段创作输出。"""
|
||
|
||
draft = _object(value, "$", frozenset({"candidateBody"}))
|
||
body = draft["candidateBody"]
|
||
if not isinstance(body, str) or not body.strip():
|
||
raise ContractError("$.candidateBody 必须是非空字符串")
|
||
return {"candidateBody": body}
|
||
|
||
|
||
def build_candidate_envelope(
|
||
context_value: Any,
|
||
draft_value: Any,
|
||
*,
|
||
candidate_version: int = 1,
|
||
) -> dict[str, Any]:
|
||
"""规范化模型正文并绑定可信候选身份。"""
|
||
|
||
context = validate_writer_context(context_value)
|
||
draft = validate_writer_draft(draft_value)
|
||
version = _integer(candidate_version, "$.candidateVersion", minimum=1)
|
||
body = normalize_text(draft["candidateBody"])
|
||
if not body.strip():
|
||
raise ContractError("$.candidateBody 规范化后不得为空")
|
||
candidate_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
|
||
envelope = {
|
||
"schemaVersion": OUTPUT_VERSION,
|
||
"runId": context["runId"],
|
||
"attempt": context["attempt"],
|
||
"mode": context["mode"],
|
||
"qualityPolicyVersion": context["qualityPolicyVersion"],
|
||
"contextSnapshotId": context["contextSnapshot"]["manifestId"],
|
||
"contextSnapshotSha256": context["contextSnapshot"]["contextSha256"],
|
||
"candidateVersion": version,
|
||
"candidateSha256": candidate_hash,
|
||
"acceptanceEligible": context["acceptanceEligible"],
|
||
"candidateBody": body,
|
||
}
|
||
return validate_writer_output(envelope)
|
||
|
||
|
||
def validate_writer_output(value: Any) -> dict[str, Any]:
|
||
"""校验 CandidateEnvelope v2 的信任绑定与正文哈希。"""
|
||
|
||
output = _object(
|
||
value,
|
||
"$",
|
||
frozenset(
|
||
{
|
||
"schemaVersion",
|
||
"runId",
|
||
"attempt",
|
||
"mode",
|
||
"qualityPolicyVersion",
|
||
"contextSnapshotId",
|
||
"contextSnapshotSha256",
|
||
"candidateVersion",
|
||
"candidateSha256",
|
||
"acceptanceEligible",
|
||
"candidateBody",
|
||
}
|
||
),
|
||
)
|
||
if output["schemaVersion"] != OUTPUT_VERSION:
|
||
raise ContractError("$.schemaVersion 版本不支持")
|
||
if not _RUN_ID_RE.fullmatch(_string(output["runId"], "$.runId")):
|
||
raise ContractError("$.runId 格式非法")
|
||
_integer(output["attempt"], "$.attempt", minimum=1)
|
||
if output["mode"] not in {"production", "diagnostic_only"}:
|
||
raise ContractError("$.mode 枚举非法")
|
||
_string(output["qualityPolicyVersion"], "$.qualityPolicyVersion")
|
||
_hash(output["contextSnapshotId"], "$.contextSnapshotId")
|
||
_hash(output["contextSnapshotSha256"], "$.contextSnapshotSha256")
|
||
_integer(output["candidateVersion"], "$.candidateVersion", minimum=1)
|
||
body = _string(output["candidateBody"], "$.candidateBody")
|
||
candidate_hash = _hash(output["candidateSha256"], "$.candidateSha256")
|
||
expected_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
|
||
if candidate_hash != expected_hash:
|
||
raise ContractError("$.candidateSha256 与规范正文不匹配")
|
||
eligible = _boolean(output["acceptanceEligible"], "$.acceptanceEligible")
|
||
if output["mode"] == "diagnostic_only" and eligible:
|
||
raise ContractError("诊断候选必须 acceptanceEligible=false")
|
||
return json.loads(canonical_json(output))
|
||
|
||
|
||
__all__ = [
|
||
"CONTEXT_VERSION", "DRAFT_VERSION", "OUTPUT_VERSION", "PLAN_VERSION", "MANIFEST_VERSION", "TIE_BREAK",
|
||
"PATTERN_CONTENT_FIELDS", "PATTERN_NAME_MAX_CHARS", "PATTERN_SUMMARY_MAX_CHARS",
|
||
"PATTERN_POINTS_MAX_FIELDS", "PATTERN_POINT_MAX_CHARS",
|
||
"ContractError", "normalize_text", "canonical_json", "retrieval_identity", "han_count",
|
||
"calculate_target_chars", "validate_writer_context", "build_writer_creative_input",
|
||
"validate_writer_draft", "build_candidate_envelope", "validate_writer_output",
|
||
"project_pattern_pointers",
|
||
]
|