一、技能重组(动作-对象命名) - 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为 clean-book-text/decide-candidate/write-next-chapter/access-database/ check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持) - agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名 二、先审后入创作闭环(本次核心) 正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交", DB 级兜底,编排层跳步即被硬拒。 - candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链 - fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量, 模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本 - projection_registry.py + example_projection_run(107):投影登记与恢复 - acceptance_state.py:接受前置实时状态重读 - lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格 - DDL 105:example_candidate 增 semantic_status/semantic_report_sha256 - write_canonical.accept:语义兜底+同事务合并增量+登记投影; run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链 - claude_runtime:兼容新 CLI modelUsage 信息字段 三、审查修复(独立子代理四维审查后) - 事实增量 propose→approve 翻态正道,不撞唯一键 - 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记 - 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理 测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。 创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
742 lines
32 KiB
Python
742 lines
32 KiB
Python
#!/usr/bin/env python3
|
||
"""确定性组装 WriterContext v1 与可审阅 RetrievalManifest。
|
||
|
||
组装器不访问数据库、不调用模型。调用方必须先完成检索计划、卡索引和
|
||
sourceRefs 原文回读,再把冻结结果交给本模块。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import hashlib
|
||
import json
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
from writer_contract import (
|
||
ContractError,
|
||
MANIFEST_VERSION,
|
||
build_writer_creative_input,
|
||
canonical_json,
|
||
normalize_text,
|
||
project_pattern_pointers,
|
||
retrieval_identity,
|
||
validate_writer_context,
|
||
)
|
||
|
||
|
||
class AssemblyError(ValueError):
|
||
"""上下文不连续、预算不足或证据合同非法时抛出。"""
|
||
|
||
|
||
EVALUATION_STRATEGY_ALIASES = {
|
||
"historical_prose_only": "generic_prose_retrieval",
|
||
"card_index_only": "card_only_diagnostic",
|
||
"card_index_plus_prose": "card_indexed_prose_retrieval",
|
||
"generic_prose_retrieval": "generic_prose_retrieval",
|
||
"card_only_diagnostic": "card_only_diagnostic",
|
||
"card_indexed_prose_retrieval": "card_indexed_prose_retrieval",
|
||
}
|
||
CONTRACT_STRATEGIES = {
|
||
"generic_prose_retrieval": "historical_prose_only",
|
||
"card_only_diagnostic": "card_index_only",
|
||
"card_indexed_prose_retrieval": "card_index_plus_prose",
|
||
}
|
||
def _hash_text(text: str) -> str:
|
||
"""对已经归一化的证据文本计算带算法前缀的哈希。"""
|
||
|
||
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||
|
||
|
||
def _hash_value(value: Any) -> str:
|
||
"""对规范 JSON 对象计算不泄露原文的稳定摘要。"""
|
||
|
||
return "sha256:" + hashlib.sha256(canonical_json(value).encode("utf-8")).hexdigest()
|
||
|
||
|
||
def _source_key(source_ref: Mapping[str, Any]) -> tuple[str, str, int]:
|
||
"""来源版本、来源 ID 与偏移共同定义片段身份。"""
|
||
|
||
return (
|
||
str(source_ref.get("sourceVersion") or ""),
|
||
str(source_ref.get("sourceId") or ""),
|
||
int(source_ref.get("startCodePoint") or 0),
|
||
)
|
||
|
||
|
||
def _whole_source_key(source_ref: Mapping[str, Any]) -> tuple[str, str]:
|
||
"""同一块全文已存在时,卡片子区间不再重复注入。"""
|
||
|
||
return (str(source_ref.get("sourceVersion") or ""), str(source_ref.get("sourceId") or ""))
|
||
|
||
|
||
def _normalize_prose(raw: Mapping[str, Any], *, recent: bool, purpose: str | None = None) -> dict[str, Any]:
|
||
"""规范化单条原文证据,并机械复核内容哈希和字符区间。"""
|
||
|
||
required = {"chapter", "sourceRef", "text"}
|
||
if not isinstance(raw, Mapping) or not required.issubset(raw):
|
||
raise AssemblyError("原文证据缺少 chapter/sourceRef/text")
|
||
chapter = raw["chapter"]
|
||
if isinstance(chapter, bool) or not isinstance(chapter, int) or chapter <= 0:
|
||
raise AssemblyError("原文证据 chapter 必须是正整数")
|
||
source_ref = copy.deepcopy(dict(raw["sourceRef"])) if isinstance(raw["sourceRef"], Mapping) else None
|
||
if source_ref is None or not source_ref.get("sourceId") or not source_ref.get("sourceVersion"):
|
||
raise AssemblyError("原文证据缺少不可变来源引用")
|
||
text = normalize_text(str(raw["text"]))
|
||
if not text:
|
||
raise AssemblyError("原文证据不能为空")
|
||
source_ref["chapter"] = chapter
|
||
source_ref.setdefault("startCodePoint", 0)
|
||
source_ref.setdefault("endCodePoint", len(text))
|
||
if source_ref["endCodePoint"] > len(text) and source_ref["startCodePoint"] == 0:
|
||
raise AssemblyError("原文来源字符区间超过文本长度")
|
||
return {
|
||
"evidenceId": str(raw.get("evidenceId") or f"prose:{source_ref['sourceId']}:{source_ref['startCodePoint']}"),
|
||
"chapter": chapter,
|
||
"sourceRef": source_ref,
|
||
"contentSha256": _hash_text(text),
|
||
"purpose": purpose or str(raw.get("purpose") or "card_source"),
|
||
"text": text,
|
||
"isRecentBaseline": recent,
|
||
}
|
||
|
||
|
||
def _normalize_fact(raw: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""规范化事实证据,并保留来源类型与风险优先级。"""
|
||
|
||
required = {"evidenceId", "fact", "sourceType", "sourceRef"}
|
||
if not isinstance(raw, Mapping) or not required.issubset(raw):
|
||
raise AssemblyError("事实证据缺少 evidenceId/fact/sourceType/sourceRef")
|
||
source_type = str(raw["sourceType"])
|
||
if source_type not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}:
|
||
raise AssemblyError("事实证据 sourceType 非法")
|
||
source_ref = copy.deepcopy(dict(raw["sourceRef"])) if isinstance(raw["sourceRef"], Mapping) else None
|
||
if source_ref is None or not source_ref.get("sourceId") or not source_ref.get("sourceVersion"):
|
||
raise AssemblyError("事实证据缺少不可变来源引用")
|
||
fact = normalize_text(str(raw["fact"]))
|
||
if not fact:
|
||
raise AssemblyError("事实证据内容不能为空")
|
||
risk = str(raw.get("riskLevel") or "medium")
|
||
if risk not in {"low", "medium", "high"}:
|
||
raise AssemblyError("事实证据 riskLevel 非法")
|
||
return {
|
||
"evidenceId": str(raw["evidenceId"]),
|
||
"fact": fact,
|
||
"sourceType": source_type,
|
||
"sourceRef": source_ref,
|
||
"contentSha256": _hash_text(fact),
|
||
"riskLevel": risk,
|
||
}
|
||
|
||
|
||
def _normalize_index_hint(raw: Mapping[str, Any], *, as_of: int) -> dict[str, Any]:
|
||
"""规范化冻结卡索引提示;它只用于诊断,绝不升级为事实证据。"""
|
||
|
||
required = {"cardId", "name", "type", "content", "sourceId", "sourceVersion", "asOf"}
|
||
if not isinstance(raw, Mapping) or set(raw) != required:
|
||
raise AssemblyError("indexHints 每项字段必须严格匹配 WriterContext v1")
|
||
hint_as_of = raw["asOf"]
|
||
if isinstance(hint_as_of, bool) or not isinstance(hint_as_of, int) or hint_as_of <= 0:
|
||
raise AssemblyError("indexHints.asOf 必须是正整数")
|
||
if hint_as_of > as_of:
|
||
raise AssemblyError("indexHints 包含目标章或未来章")
|
||
normalized = {
|
||
key: normalize_text(str(raw[key]))
|
||
for key in required - {"asOf"}
|
||
}
|
||
if any(not value.strip() for value in normalized.values()):
|
||
raise AssemblyError("indexHints 字符串字段不能为空")
|
||
return {**normalized, "asOf": hint_as_of}
|
||
|
||
|
||
def _select_recent_baseline(recent_chapters: Sequence[Mapping[str, Any]], *, as_of: int) -> list[dict[str, Any]]:
|
||
"""选择冻结章起连续四章全文;作品不足四章时从第一章开始。"""
|
||
|
||
by_chapter: dict[int, dict[str, Any]] = {}
|
||
for raw in recent_chapters:
|
||
normalized = _normalize_prose(raw, recent=True, purpose="recent_full_chapter")
|
||
chapter = normalized["chapter"]
|
||
if chapter > as_of:
|
||
raise AssemblyError("近章原文包含目标章或未来章")
|
||
if chapter in by_chapter:
|
||
raise AssemblyError(f"第 {chapter} 章完整原文重复")
|
||
by_chapter[chapter] = normalized
|
||
first = max(1, as_of - 3)
|
||
expected = list(range(first, as_of + 1))
|
||
missing = [chapter for chapter in expected if chapter not in by_chapter]
|
||
if missing:
|
||
raise AssemblyError(f"连续四章基线缺章: {','.join(map(str, missing))}")
|
||
return [by_chapter[chapter] for chapter in expected]
|
||
|
||
|
||
def _outline_contract(fine_outline: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""只保留 WriterContext 合同字段,检索用实体列表不会被倾倒给写手。"""
|
||
|
||
required = {"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"}
|
||
if not isinstance(fine_outline, Mapping) or not required.issubset(fine_outline):
|
||
raise AssemblyError("fineOutline 缺少严格合同字段")
|
||
return {key: copy.deepcopy(fine_outline[key]) for key in ("sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts")}
|
||
|
||
|
||
def _coverage_elements(fine_outline: Mapping[str, Any]) -> list[dict[str, str]]:
|
||
"""从细纲提取人物、关系、物品、地点和力量体系覆盖目标。"""
|
||
|
||
groups = (
|
||
("entities", None),
|
||
("relations", "character_relation"),
|
||
("items", "item"),
|
||
("locations", "location"),
|
||
("powerSystems", "power_system"),
|
||
)
|
||
result: list[dict[str, str]] = []
|
||
seen: set[str] = set()
|
||
for field, fallback_type in groups:
|
||
values = fine_outline.get(field, [])
|
||
if not isinstance(values, list):
|
||
raise AssemblyError(f"fineOutline.{field} 必须是数组")
|
||
for index, raw in enumerate(values):
|
||
if isinstance(raw, str):
|
||
item = {"id": f"{field}:{index}", "type": fallback_type or "unknown", "name": raw}
|
||
elif isinstance(raw, Mapping):
|
||
item = {
|
||
"id": str(raw.get("id") or f"{field}:{index}"),
|
||
"type": str(raw.get("type") or fallback_type or "unknown"),
|
||
"name": str(raw.get("name") or ""),
|
||
}
|
||
else:
|
||
raise AssemblyError(f"fineOutline.{field}[{index}] 类型非法")
|
||
if not item["name"]:
|
||
raise AssemblyError(f"fineOutline.{field}[{index}] 缺少名称")
|
||
if item["id"] not in seen:
|
||
seen.add(item["id"])
|
||
result.append(item)
|
||
return result
|
||
|
||
|
||
def _build_coverage(
|
||
fine_outline: Mapping[str, Any],
|
||
facts: Sequence[Mapping[str, Any]],
|
||
prose: Sequence[Mapping[str, Any]],
|
||
cards: Sequence[Mapping[str, Any]],
|
||
) -> list[dict[str, Any]]:
|
||
"""把五类细纲要素映射到最终保留的事实与原文证据。"""
|
||
|
||
prose_by_source = {
|
||
str(item["sourceRef"]["sourceId"]): item["evidenceId"]
|
||
for item in prose
|
||
}
|
||
declared = fine_outline.get("declaredNewFacts", [])
|
||
result: list[dict[str, Any]] = []
|
||
for element in _coverage_elements(fine_outline):
|
||
fact_ids = [item["evidenceId"] for item in facts if element["name"] in item["fact"]]
|
||
prose_ids: list[str] = []
|
||
for card in cards:
|
||
if str(card.get("name")) != element["name"] and str(card.get("type")) != element["type"]:
|
||
continue
|
||
for ref in card.get("sourceRefs") or []:
|
||
evidence_id = prose_by_source.get(str(ref.get("sourceId")))
|
||
if evidence_id and evidence_id not in prose_ids:
|
||
prose_ids.append(evidence_id)
|
||
is_declared = any(
|
||
isinstance(item, Mapping)
|
||
and (str(item.get("factId")) == element["id"] or element["name"] in str(item.get("text") or ""))
|
||
for item in declared
|
||
)
|
||
if is_declared:
|
||
status, reason = "declared_new", ""
|
||
elif fact_ids and prose_ids:
|
||
status, reason = "supported", ""
|
||
elif fact_ids:
|
||
status, reason = "style_gap", "缺少历史表现原文"
|
||
elif prose_ids:
|
||
status, reason = "card_gap", "原文可证但缺少冻结事实索引"
|
||
else:
|
||
status, reason = "unsupported", "没有可信事实证据"
|
||
result.append(
|
||
{
|
||
"elementId": element["id"],
|
||
"elementType": element["type"],
|
||
"name": element["name"],
|
||
"status": status,
|
||
"factEvidenceIds": sorted(fact_ids),
|
||
"proseEvidenceIds": sorted(prose_ids),
|
||
"gapReason": reason,
|
||
}
|
||
)
|
||
return result
|
||
|
||
|
||
def _manifest(
|
||
*,
|
||
plan_id: str,
|
||
facts: Sequence[Mapping[str, Any]],
|
||
prose: Sequence[Mapping[str, Any]],
|
||
pattern_references: Sequence[Mapping[str, Any]],
|
||
omitted: Sequence[Mapping[str, str]],
|
||
index_hints: Sequence[Mapping[str, Any]] = (),
|
||
) -> dict[str, Any]:
|
||
"""由最终入包来源集合计算稳定 manifest,不含 runId 或时间戳。"""
|
||
|
||
unique: dict[tuple[str, str, int], dict[str, Any]] = {}
|
||
for item in [*facts, *prose]:
|
||
ref = copy.deepcopy(dict(item["sourceRef"]))
|
||
unique[_source_key(ref)] = ref
|
||
for raw_ref in pattern_references:
|
||
# manifest 是来源账本(审计用),只登记可回读指针;范式卡的内容字段(name/
|
||
# summary/writingPoints)剥掉,不进 manifest——内容只由上下文内的 patternReferences
|
||
# 承载。否则 manifest 的 _source_ref 严格校验会以「未知字段」拒收内容字段。
|
||
ref = project_pattern_pointers(raw_ref)
|
||
unique[_source_key(ref)] = ref
|
||
for hint in index_hints:
|
||
# 这里记录的是全部冻结卡索引,并不代表卡缺少原文来源。
|
||
ref = {
|
||
"sourceId": hint["sourceId"],
|
||
"sourceVersion": hint["sourceVersion"],
|
||
"chapter": hint["asOf"],
|
||
"sourceType": "diagnostic_card_index",
|
||
}
|
||
unique[_source_key(ref)] = ref
|
||
sources = [unique[key] for key in sorted(unique)]
|
||
omitted_rows = sorted(
|
||
[copy.deepcopy(dict(item)) for item in omitted],
|
||
key=lambda item: (str(item.get("reason")), str(item.get("sourceId"))),
|
||
)
|
||
payload = {
|
||
"manifestVersion": MANIFEST_VERSION,
|
||
"planId": plan_id,
|
||
"sources": sources,
|
||
"omittedSources": omitted_rows,
|
||
}
|
||
return {**payload, "manifestId": retrieval_identity(payload)}
|
||
|
||
|
||
def _markdown_manifest(manifest: Mapping[str, Any]) -> str:
|
||
"""渲染不含运行时元数据的人类审阅清单,保证字节稳定。"""
|
||
|
||
lines = [
|
||
"# 正文检索清单",
|
||
"",
|
||
f"- 计划:`{manifest['planId']}`",
|
||
f"- 清单:`{manifest['manifestId']}`",
|
||
f"- 纳入来源:{len(manifest['sources'])}",
|
||
f"- 排除来源:{len(manifest['omittedSources'])}",
|
||
"",
|
||
"## 纳入来源",
|
||
"",
|
||
]
|
||
if manifest["sources"]:
|
||
for source in manifest["sources"]:
|
||
location = f"第{source['chapter']}章" if "chapter" in source else "权威版本"
|
||
lines.append(f"- `{source['sourceId']}` @ `{source['sourceVersion']}`,{location}")
|
||
else:
|
||
lines.append("- 无")
|
||
lines.extend(["", "## 排除来源", ""])
|
||
if manifest["omittedSources"]:
|
||
for source in manifest["omittedSources"]:
|
||
lines.append(f"- `{source['sourceId']}`:{source['reason']}")
|
||
else:
|
||
lines.append("- 无")
|
||
return "\n".join(lines) + "\n"
|
||
|
||
|
||
def _context_size(context: Mapping[str, Any]) -> int:
|
||
"""用最终规范 JSON 的 Unicode code point 数作为确定性预算单位。"""
|
||
|
||
return len(canonical_json(context))
|
||
|
||
|
||
def _select_exact_prose_chars(
|
||
prose: Sequence[Mapping[str, Any]], *, char_budget: int
|
||
) -> list[dict[str, Any]]:
|
||
"""按稳定顺序截取恰好指定代码点数,来源区间和内容哈希同步收窄。"""
|
||
|
||
if isinstance(char_budget, bool) or not isinstance(char_budget, int) or char_budget < 0:
|
||
raise AssemblyError("tokenBudget.proseCharBudget 必须是非负整数")
|
||
remaining = char_budget
|
||
selected: list[dict[str, Any]] = []
|
||
for raw in prose:
|
||
if remaining == 0:
|
||
break
|
||
item = copy.deepcopy(dict(raw))
|
||
text = str(item["text"])
|
||
if len(text) > remaining:
|
||
text = text[:remaining]
|
||
item["text"] = text
|
||
item["sourceRef"]["endCodePoint"] = (
|
||
int(item["sourceRef"].get("startCodePoint") or 0) + len(text)
|
||
)
|
||
item["contentSha256"] = _hash_text(text)
|
||
selected.append(item)
|
||
remaining -= len(text)
|
||
if remaining != 0:
|
||
raise AssemblyError(
|
||
f"历史原文不足 proseCharBudget: expected={char_budget}, missing={remaining}"
|
||
)
|
||
return selected
|
||
|
||
|
||
def _collect_diff_paths(left: Any, right: Any, path: str = "$") -> list[str]:
|
||
"""递归收集两个完整对象的所有差异路径,不把原始值写入回执。"""
|
||
|
||
if type(left) is not type(right):
|
||
return [path]
|
||
if isinstance(left, Mapping):
|
||
paths: list[str] = []
|
||
for key in sorted(set(left) | set(right)):
|
||
child = f"{path}.{key}"
|
||
if key not in left or key not in right:
|
||
paths.append(child)
|
||
else:
|
||
paths.extend(_collect_diff_paths(left[key], right[key], child))
|
||
return paths
|
||
if isinstance(left, list):
|
||
if len(left) != len(right):
|
||
return [f"{path}.length"]
|
||
paths = []
|
||
for index, (left_item, right_item) in enumerate(zip(left, right, strict=True)):
|
||
paths.extend(_collect_diff_paths(left_item, right_item, f"{path}[{index}]"))
|
||
return paths
|
||
return [] if left == right else [path]
|
||
|
||
|
||
def _is_allowed_ac_diff(path: str) -> bool:
|
||
"""只允许预注册证据策略改变模型可见的事实约束、原文摘录与范式引用。
|
||
|
||
WHY 含 patternReferences:Gate A 的唯一变量是「有无卡(含范式卡)」。C 臂注入公共
|
||
范式卡、A 臂恒空,因此两臂创意输入的 patternReferences 必然不同——这是实验设计本身,
|
||
不是越界改动。把它列入白名单,门禁才不会把预期的范式差异误判为非法字段漂移。
|
||
"""
|
||
|
||
return path.startswith(("$.factConstraints", "$.proseExcerpts", "$.patternReferences"))
|
||
|
||
|
||
def build_context_allowlist_diff_receipt(
|
||
*,
|
||
sample_id: str,
|
||
context_a: Mapping[str, Any],
|
||
context_c: Mapping[str, Any],
|
||
prose_char_budget: int,
|
||
require_nonempty: bool = False,
|
||
) -> dict[str, Any]:
|
||
"""在 WriterCreativeInput v2 上校验 A/C 单变量并输出脱敏回执。"""
|
||
|
||
if not str(sample_id).strip():
|
||
raise AssemblyError("allowlist diff 缺少 sampleId")
|
||
if isinstance(prose_char_budget, bool) or not isinstance(prose_char_budget, int) or prose_char_budget < 0:
|
||
raise AssemblyError("allowlist diff 的 proseCharBudget 必须是非负整数")
|
||
if not isinstance(require_nonempty, bool):
|
||
raise AssemblyError("allowlist diff 的 require_nonempty 必须是布尔值")
|
||
if require_nonempty and prose_char_budget <= 0:
|
||
raise AssemblyError("正式 A/C allowlist diff 的 proseCharBudget 必须是正整数")
|
||
try:
|
||
creative_inputs = {
|
||
"A": build_writer_creative_input(context_a),
|
||
"C": build_writer_creative_input(context_c),
|
||
}
|
||
except (ContractError, TypeError, ValueError) as error:
|
||
raise AssemblyError(f"allowlist diff 的 WriterContext 非法: {error}") from error
|
||
context_a = creative_inputs["A"]
|
||
context_c = creative_inputs["C"]
|
||
baseline_a = [item for item in context_a["proseExcerpts"] if item["isRecentBaseline"] is True]
|
||
baseline_c = [item for item in context_c["proseExcerpts"] if item["isRecentBaseline"] is True]
|
||
if baseline_a != baseline_c:
|
||
raise AssemblyError("allowlist diff 发现 A/C 连续基线不一致")
|
||
prose_counts = {
|
||
"A": sum(
|
||
len(str(item.get("text") or ""))
|
||
for item in context_a["proseExcerpts"]
|
||
if item["isRecentBaseline"] is False
|
||
),
|
||
"C": sum(
|
||
len(str(item.get("text") or ""))
|
||
for item in context_c["proseExcerpts"]
|
||
if item["isRecentBaseline"] is False
|
||
),
|
||
}
|
||
if set(prose_counts.values()) != {prose_char_budget}:
|
||
raise AssemblyError(
|
||
"allowlist diff 发现 A/C 原文字符预算不一致: "
|
||
f"expected={prose_char_budget}, actual={prose_counts}"
|
||
)
|
||
differences = _collect_diff_paths(context_a, context_c)
|
||
rejected = [path for path in differences if not _is_allowed_ac_diff(path)]
|
||
if rejected:
|
||
raise AssemblyError(f"allowlist diff 包含非原文字段差异: {rejected}")
|
||
context_hashes = {
|
||
"A": _hash_value(context_a),
|
||
"C": _hash_value(context_c),
|
||
}
|
||
if require_nonempty and (
|
||
not differences or context_hashes["A"] == context_hashes["C"]
|
||
):
|
||
raise AssemblyError("正式 A/C WriterCreativeInput 必须存在非空允许差异")
|
||
payload = {
|
||
"schemaVersion": "writer-creative-input-allowlist-diff-v2",
|
||
"sampleId": str(sample_id),
|
||
"arms": ["A", "C"],
|
||
"ok": True,
|
||
"proseCharBudget": prose_char_budget,
|
||
"proseCharCount": prose_counts,
|
||
"allowedDifferencePaths": sorted(differences),
|
||
"rejectedDifferencePaths": [],
|
||
"contextSha256": context_hashes,
|
||
}
|
||
return {**payload, "receiptSha256": _hash_value(payload)}
|
||
|
||
|
||
def assemble_context(
|
||
*,
|
||
run_id: str,
|
||
attempt: int,
|
||
mode: str,
|
||
purpose: str,
|
||
quality_policy_version: str,
|
||
work_id: int,
|
||
target_chapter: int,
|
||
as_of: int,
|
||
source_version: str,
|
||
authorization_snapshot: Mapping[str, Any],
|
||
source_status: str,
|
||
retrieval_plan: Mapping[str, Any],
|
||
retrieval_result: Mapping[str, Any],
|
||
fine_outline: Mapping[str, Any],
|
||
narrative_state: Mapping[str, Any],
|
||
recent_chapters: Sequence[Mapping[str, Any]],
|
||
output_contract: Mapping[str, Any],
|
||
token_budget: Mapping[str, int],
|
||
pattern_references: Sequence[Mapping[str, Any]] = (),
|
||
style_constraints: Sequence[str] = (),
|
||
generated_at: str,
|
||
evidence_strategy: str = "production_dual_evidence",
|
||
) -> dict[str, Any]:
|
||
"""组装稳定 WriterContext,并返回规范 JSON 与 Markdown manifest。"""
|
||
|
||
if retrieval_plan.get("runId") != run_id or retrieval_plan.get("asOf") != as_of:
|
||
raise AssemblyError("retrievalPlan 与当前 runId/asOf 不一致")
|
||
if retrieval_plan.get("filters", {}).get("workId") != work_id:
|
||
raise AssemblyError("retrievalPlan 与当前 workId 不一致")
|
||
max_chars = token_budget.get("maxContextChars")
|
||
if isinstance(max_chars, bool) or not isinstance(max_chars, int) or max_chars <= 0:
|
||
raise AssemblyError("tokenBudget.maxContextChars 必须是正整数")
|
||
|
||
if evidence_strategy == "production_dual_evidence":
|
||
orchestration_strategy = evidence_strategy
|
||
contract_strategy = evidence_strategy
|
||
elif evidence_strategy in EVALUATION_STRATEGY_ALIASES:
|
||
orchestration_strategy = EVALUATION_STRATEGY_ALIASES[evidence_strategy]
|
||
contract_strategy = CONTRACT_STRATEGIES[orchestration_strategy]
|
||
else:
|
||
raise AssemblyError("evidenceStrategy 非法")
|
||
if mode == "production" and orchestration_strategy != "production_dual_evidence":
|
||
raise AssemblyError("生产上下文必须使用 production_dual_evidence")
|
||
if orchestration_strategy != "production_dual_evidence" and not (
|
||
mode == "diagnostic_only" and purpose in {"evaluation", "diagnostic"}
|
||
):
|
||
raise AssemblyError("诊断证据策略只允许用于 diagnostic_only 评测或诊断")
|
||
|
||
baseline = (
|
||
[]
|
||
if orchestration_strategy == "card_only_diagnostic"
|
||
else _select_recent_baseline(recent_chapters, as_of=as_of)
|
||
)
|
||
baseline_blocks = {_whole_source_key(item["sourceRef"]) for item in baseline}
|
||
supplemental: list[dict[str, Any]] = []
|
||
seen_fragments = {_source_key(item["sourceRef"]) for item in baseline}
|
||
for raw in retrieval_result.get("proseEvidence", []):
|
||
retrieval_arm = raw.get("retrievalArm") if isinstance(raw, Mapping) else None
|
||
if retrieval_arm not in {None, "A", "C"}:
|
||
raise AssemblyError("proseEvidence.retrievalArm 只能是 A/C")
|
||
if orchestration_strategy == "generic_prose_retrieval" and retrieval_arm != "A":
|
||
continue
|
||
if orchestration_strategy == "card_indexed_prose_retrieval" and retrieval_arm == "A":
|
||
continue
|
||
item = _normalize_prose(raw, recent=False)
|
||
if item["chapter"] > as_of:
|
||
raise AssemblyError("卡来源原文包含目标章或未来章")
|
||
if _whole_source_key(item["sourceRef"]) in baseline_blocks:
|
||
continue
|
||
key = _source_key(item["sourceRef"])
|
||
if key not in seen_fragments:
|
||
seen_fragments.add(key)
|
||
supplemental.append(item)
|
||
supplemental.sort(key=lambda item: _source_key(item["sourceRef"]))
|
||
prose_char_budget = token_budget.get("proseCharBudget")
|
||
if orchestration_strategy in {
|
||
"generic_prose_retrieval",
|
||
"card_indexed_prose_retrieval",
|
||
} and prose_char_budget is not None:
|
||
supplemental = _select_exact_prose_chars(
|
||
supplemental,
|
||
char_budget=prose_char_budget,
|
||
)
|
||
facts = sorted(
|
||
[_normalize_fact(item) for item in retrieval_result.get("factEvidence", [])],
|
||
key=lambda item: ({"high": 0, "medium": 1, "low": 2}[item["riskLevel"]], item["evidenceId"]),
|
||
)
|
||
cards = [copy.deepcopy(dict(item)) for item in retrieval_result.get("cards", [])]
|
||
# 新合同由检索器显式给出覆盖全部冻结命中卡的 indexHints。旧字段只在
|
||
# 历史评测夹具尚未迁移时回退,不能再决定有 sourceRefs 的卡是否入包。
|
||
raw_index_hints = retrieval_result.get("indexHints")
|
||
if raw_index_hints is None:
|
||
raw_index_hints = retrieval_result.get("unverifiedIndexHints", [])
|
||
index_hints = sorted(
|
||
[
|
||
_normalize_index_hint(item, as_of=as_of)
|
||
for item in raw_index_hints
|
||
],
|
||
key=lambda item: (item["sourceVersion"], item["sourceId"], item["cardId"]),
|
||
)
|
||
if orchestration_strategy in {"card_only_diagnostic", "card_indexed_prose_retrieval"} and not index_hints:
|
||
raise AssemblyError("卡索引诊断策略缺少 indexHints")
|
||
if orchestration_strategy in {"production_dual_evidence", "generic_prose_retrieval"}:
|
||
index_hints = []
|
||
inherited_omitted = [
|
||
{"sourceId": str(item.get("sourceId") or "unknown"), "reason": str(item.get("reason") or "not_relevant")}
|
||
for item in retrieval_result.get("manifest", {}).get("omittedSources", [])
|
||
if isinstance(item, Mapping)
|
||
]
|
||
|
||
selected_facts: list[dict[str, Any]] = []
|
||
# 连续前四章全文是正文实验 v1 的不可裁剪基线。预算容不下时必须失败关闭,
|
||
# 不能静默退化成只保留最近一两章。
|
||
selected_prose: list[dict[str, Any]] = list(baseline)
|
||
omitted = list(inherited_omitted)
|
||
candidates: list[tuple[str, dict[str, Any]]] = []
|
||
if orchestration_strategy == "production_dual_evidence":
|
||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "high")
|
||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "medium")
|
||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "low")
|
||
candidates.extend(("prose", item) for item in supplemental)
|
||
elif orchestration_strategy in {
|
||
"generic_prose_retrieval",
|
||
"card_indexed_prose_retrieval",
|
||
}:
|
||
candidates.extend(("prose", item) for item in supplemental)
|
||
|
||
def make_context() -> dict[str, Any]:
|
||
"""用当前选择集生成完整上下文,供预算试算与最终冻结。"""
|
||
# 规划期选定的文风约束:归一化后仅在非空时入冻结上下文(空则省略整个键,
|
||
# 上下文逐字节不变,不破坏既有哈希与 A/C 单变量)。
|
||
style_rules = [normalize_text(str(rule)) for rule in style_constraints if str(rule).strip()]
|
||
ordered_prose = sorted(
|
||
selected_prose,
|
||
key=lambda item: (
|
||
0 if item["isRecentBaseline"] else 1,
|
||
item["chapter"] if item["isRecentBaseline"] else 0,
|
||
_source_key(item["sourceRef"]),
|
||
),
|
||
)
|
||
ordered_facts = sorted(selected_facts, key=lambda item: item["evidenceId"])
|
||
manifest = _manifest(
|
||
plan_id=str(retrieval_plan["planId"]),
|
||
facts=ordered_facts,
|
||
prose=ordered_prose,
|
||
pattern_references=pattern_references,
|
||
omitted=omitted,
|
||
index_hints=index_hints,
|
||
)
|
||
context = {
|
||
"schemaVersion": "writer-context-v1",
|
||
"runId": run_id,
|
||
"attempt": attempt,
|
||
"mode": mode,
|
||
"purpose": purpose,
|
||
"qualityPolicyVersion": quality_policy_version,
|
||
"workId": work_id,
|
||
"targetChapter": target_chapter,
|
||
"asOf": as_of,
|
||
"contextSnapshot": {
|
||
"manifestId": manifest["manifestId"],
|
||
"contextSha256": "sha256:" + "0" * 64,
|
||
"generatedAt": normalize_text(generated_at),
|
||
},
|
||
"sourceVersion": source_version,
|
||
"authorizationSnapshot": copy.deepcopy(dict(authorization_snapshot)),
|
||
"sourceStatus": source_status,
|
||
"retrievalPlan": copy.deepcopy(dict(retrieval_plan)),
|
||
"retrievalManifest": manifest,
|
||
"fineOutline": _outline_contract(fine_outline),
|
||
"narrativeState": copy.deepcopy(dict(narrative_state)),
|
||
"factEvidence": ordered_facts,
|
||
"proseEvidence": ordered_prose,
|
||
"evidenceStrategy": contract_strategy,
|
||
"indexHints": copy.deepcopy(index_hints),
|
||
"patternReferences": [copy.deepcopy(dict(item)) for item in pattern_references],
|
||
"evidenceCoverage": _build_coverage(fine_outline, ordered_facts, ordered_prose, cards),
|
||
"outputContract": copy.deepcopy(dict(output_contract)),
|
||
"tokenBudget": {"maxContextChars": max_chars, "usedContextChars": 0},
|
||
"omittedSources": sorted(copy.deepcopy(omitted), key=lambda item: (item["reason"], item["sourceId"])),
|
||
"acceptanceEligible": mode == "production" and purpose == "production",
|
||
}
|
||
if style_rules:
|
||
context["styleConstraints"] = list(style_rules)
|
||
# usedContextChars 自身位数会影响 JSON 长度,迭代到数值稳定。
|
||
for _ in range(8):
|
||
used = _context_size(context)
|
||
if context["tokenBudget"]["usedContextChars"] == used:
|
||
break
|
||
context["tokenBudget"]["usedContextChars"] = used
|
||
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
|
||
return context
|
||
|
||
baseline_context = make_context()
|
||
if _context_size(baseline_context) > max_chars:
|
||
raise AssemblyError("上下文预算不足以容纳细纲硬约束与连续前四章全文基线")
|
||
for kind, item in candidates:
|
||
target = selected_facts if kind == "fact" else selected_prose
|
||
target.append(item)
|
||
trial = make_context()
|
||
if _context_size(trial) > max_chars:
|
||
target.pop()
|
||
omitted.append({"sourceId": str(item["sourceRef"]["sourceId"]), "reason": "token_budget"})
|
||
|
||
context = make_context()
|
||
if prose_char_budget is not None and orchestration_strategy in {
|
||
"generic_prose_retrieval",
|
||
"card_indexed_prose_retrieval",
|
||
}:
|
||
selected_chars = sum(
|
||
len(item["text"])
|
||
for item in context["proseEvidence"]
|
||
if item["isRecentBaseline"] is False
|
||
)
|
||
if selected_chars != prose_char_budget:
|
||
raise AssemblyError(
|
||
"上下文总预算无法保留完整 proseCharBudget: "
|
||
f"expected={prose_char_budget}, actual={selected_chars}"
|
||
)
|
||
# 添加裁剪回显可能占用少量预算;若越界,按最低优先级继续移除。
|
||
while _context_size(context) > max_chars and (selected_prose or selected_facts):
|
||
removable_prose = [item for item in selected_prose if not item["isRecentBaseline"]]
|
||
if removable_prose:
|
||
removed = removable_prose[-1]
|
||
selected_prose.remove(removed)
|
||
elif selected_facts and selected_facts[-1]["riskLevel"] != "high":
|
||
removed = selected_facts.pop()
|
||
elif selected_facts:
|
||
removed = selected_facts.pop()
|
||
else:
|
||
raise AssemblyError("上下文预算不足以保留连续前四章全文基线")
|
||
omitted.append({"sourceId": str(removed["sourceRef"]["sourceId"]), "reason": "token_budget"})
|
||
context = make_context()
|
||
if _context_size(context) > max_chars:
|
||
raise AssemblyError("上下文预算不足以保留可用证据")
|
||
context["tokenBudget"]["usedContextChars"] = _context_size(context)
|
||
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
|
||
normalized_context = validate_writer_context(context)
|
||
writer_creative_input = build_writer_creative_input(normalized_context)
|
||
return {
|
||
"context": normalized_context,
|
||
"contextJson": canonical_json(normalized_context),
|
||
"writerCreativeInput": writer_creative_input,
|
||
"writerCreativeInputJson": canonical_json(writer_creative_input),
|
||
"orchestrationStrategy": orchestration_strategy,
|
||
"manifestMarkdown": _markdown_manifest(normalized_context["retrievalManifest"]),
|
||
}
|
||
|
||
|
||
__all__ = [
|
||
"AssemblyError",
|
||
"assemble_context",
|
||
"build_context_allowlist_diff_receipt",
|
||
]
|