#!/usr/bin/env python3 """确定性组装 WriterContext v1 与可审阅 RetrievalManifest。 组装器不访问数据库、不调用模型。调用方必须先完成检索计划、卡索引和 sourceRefs 原文回读,再把冻结结果交给本模块。 """ from __future__ import annotations import copy import hashlib import json from typing import Any, Mapping, Sequence from deai.schemas import validate as validate_humanization_contract from writer_contract import ( ContractError, MANIFEST_VERSION, build_writer_creative_input, canonical_json, normalize_text, project_pattern_pointers, retrieval_identity, validate_writer_context, ) class AssemblyError(ValueError): """上下文不连续、预算不足或证据合同非法时抛出。""" EVALUATION_STRATEGY_ALIASES = { "historical_prose_only": "generic_prose_retrieval", "card_index_only": "card_only_diagnostic", "card_index_plus_prose": "card_indexed_prose_retrieval", "generic_prose_retrieval": "generic_prose_retrieval", "card_only_diagnostic": "card_only_diagnostic", "card_indexed_prose_retrieval": "card_indexed_prose_retrieval", } CONTRACT_STRATEGIES = { "generic_prose_retrieval": "historical_prose_only", "card_only_diagnostic": "card_index_only", "card_indexed_prose_retrieval": "card_index_plus_prose", } def _hash_text(text: str) -> str: """对已经归一化的证据文本计算带算法前缀的哈希。""" return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() def _hash_value(value: Any) -> str: """对规范 JSON 对象计算不泄露原文的稳定摘要。""" return "sha256:" + hashlib.sha256(canonical_json(value).encode("utf-8")).hexdigest() def _source_key(source_ref: Mapping[str, Any]) -> tuple[str, str, int]: """来源版本、来源 ID 与偏移共同定义片段身份。""" return ( str(source_ref.get("sourceVersion") or ""), str(source_ref.get("sourceId") or ""), int(source_ref.get("startCodePoint") or 0), ) def _whole_source_key(source_ref: Mapping[str, Any]) -> tuple[str, str]: """同一块全文已存在时,卡片子区间不再重复注入。""" return (str(source_ref.get("sourceVersion") or ""), str(source_ref.get("sourceId") or "")) def _normalize_prose(raw: Mapping[str, Any], *, recent: bool, purpose: str | None = None) -> dict[str, Any]: """规范化单条原文证据,并机械复核内容哈希和字符区间。""" required = {"chapter", "sourceRef", "text"} if not isinstance(raw, Mapping) or not required.issubset(raw): raise AssemblyError("原文证据缺少 chapter/sourceRef/text") chapter = raw["chapter"] if isinstance(chapter, bool) or not isinstance(chapter, int) or chapter <= 0: raise AssemblyError("原文证据 chapter 必须是正整数") source_ref = copy.deepcopy(dict(raw["sourceRef"])) if isinstance(raw["sourceRef"], Mapping) else None if source_ref is None or not source_ref.get("sourceId") or not source_ref.get("sourceVersion"): raise AssemblyError("原文证据缺少不可变来源引用") text = normalize_text(str(raw["text"])) if not text: raise AssemblyError("原文证据不能为空") source_ref["chapter"] = chapter source_ref.setdefault("startCodePoint", 0) source_ref.setdefault("endCodePoint", len(text)) if source_ref["endCodePoint"] > len(text) and source_ref["startCodePoint"] == 0: raise AssemblyError("原文来源字符区间超过文本长度") return { "evidenceId": str(raw.get("evidenceId") or f"prose:{source_ref['sourceId']}:{source_ref['startCodePoint']}"), "chapter": chapter, "sourceRef": source_ref, "contentSha256": _hash_text(text), "purpose": purpose or str(raw.get("purpose") or "card_source"), "text": text, "isRecentBaseline": recent, } def _normalize_fact(raw: Mapping[str, Any]) -> dict[str, Any]: """规范化事实证据,并保留来源类型与风险优先级。""" required = {"evidenceId", "fact", "sourceType", "sourceRef"} if not isinstance(raw, Mapping) or not required.issubset(raw): raise AssemblyError("事实证据缺少 evidenceId/fact/sourceType/sourceRef") source_type = str(raw["sourceType"]) if source_type not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}: raise AssemblyError("事实证据 sourceType 非法") source_ref = copy.deepcopy(dict(raw["sourceRef"])) if isinstance(raw["sourceRef"], Mapping) else None if source_ref is None or not source_ref.get("sourceId") or not source_ref.get("sourceVersion"): raise AssemblyError("事实证据缺少不可变来源引用") fact = normalize_text(str(raw["fact"])) if not fact: raise AssemblyError("事实证据内容不能为空") risk = str(raw.get("riskLevel") or "medium") if risk not in {"low", "medium", "high"}: raise AssemblyError("事实证据 riskLevel 非法") return { "evidenceId": str(raw["evidenceId"]), "fact": fact, "sourceType": source_type, "sourceRef": source_ref, "contentSha256": _hash_text(fact), "riskLevel": risk, } def _normalize_index_hint(raw: Mapping[str, Any], *, as_of: int) -> dict[str, Any]: """规范化冻结卡索引提示;它只用于诊断,绝不升级为事实证据。""" required = {"cardId", "name", "type", "content", "sourceId", "sourceVersion", "asOf"} if not isinstance(raw, Mapping) or set(raw) != required: raise AssemblyError("indexHints 每项字段必须严格匹配 WriterContext v1") hint_as_of = raw["asOf"] if isinstance(hint_as_of, bool) or not isinstance(hint_as_of, int) or hint_as_of <= 0: raise AssemblyError("indexHints.asOf 必须是正整数") if hint_as_of > as_of: raise AssemblyError("indexHints 包含目标章或未来章") normalized = { key: normalize_text(str(raw[key])) for key in required - {"asOf"} } if any(not value.strip() for value in normalized.values()): raise AssemblyError("indexHints 字符串字段不能为空") return {**normalized, "asOf": hint_as_of} def _select_recent_baseline(recent_chapters: Sequence[Mapping[str, Any]], *, as_of: int) -> list[dict[str, Any]]: """选择冻结章起连续四章全文;作品不足四章时从第一章开始。""" by_chapter: dict[int, dict[str, Any]] = {} for raw in recent_chapters: normalized = _normalize_prose(raw, recent=True, purpose="recent_full_chapter") chapter = normalized["chapter"] if chapter > as_of: raise AssemblyError("近章原文包含目标章或未来章") if chapter in by_chapter: raise AssemblyError(f"第 {chapter} 章完整原文重复") by_chapter[chapter] = normalized first = max(1, as_of - 3) expected = list(range(first, as_of + 1)) missing = [chapter for chapter in expected if chapter not in by_chapter] if missing: raise AssemblyError(f"连续四章基线缺章: {','.join(map(str, missing))}") return [by_chapter[chapter] for chapter in expected] def _outline_contract(fine_outline: Mapping[str, Any]) -> dict[str, Any]: """只保留 WriterContext 合同字段,检索用实体列表不会被倾倒给写手。 支持全中文键名(来源引用、硬约束、可调整节拍、新事实宣告)及英文兼容键。 """ if not isinstance(fine_outline, Mapping): raise AssemblyError("fineOutline 必须是对象") source_ref = fine_outline.get("来源引用") or fine_outline.get("sourceRef") hard_constraints = fine_outline.get("硬约束") if "硬约束" in fine_outline else fine_outline.get("hardConstraints") adjustable_beats = fine_outline.get("可调整节拍") if "可调整节拍" in fine_outline else fine_outline.get("adjustableBeats") declared_new_facts = fine_outline.get("新事实宣告") if "新事实宣告" in fine_outline else fine_outline.get("declaredNewFacts") if source_ref is None or hard_constraints is None or adjustable_beats is None or declared_new_facts is None: raise AssemblyError("fineOutline 缺少严格合同字段 (来源引用/硬约束/可调整节拍/新事实宣告)") normalized_facts = [] for item in declared_new_facts: if isinstance(item, Mapping): f_id = str(item.get("事实ID") or item.get("factId") or "") f_text = str(item.get("文本") or item.get("text") or "") f_ref = item.get("来源引用") or item.get("sourceRef") or source_ref normalized_facts.append({ "factId": f_id, "text": f_text, "sourceRef": f_ref, }) else: normalized_facts.append(item) return { "sourceRef": copy.deepcopy(source_ref), "hardConstraints": copy.deepcopy(hard_constraints), "adjustableBeats": copy.deepcopy(adjustable_beats), "declaredNewFacts": copy.deepcopy(normalized_facts), } def _coverage_elements(fine_outline: Mapping[str, Any]) -> list[dict[str, str]]: """从细纲提取人物、关系、物品、地点和力量体系覆盖目标。""" groups = ( ("entities", None), ("relations", "character_relation"), ("items", "item"), ("locations", "location"), ("powerSystems", "power_system"), ) result: list[dict[str, str]] = [] seen: set[str] = set() for field, fallback_type in groups: values = fine_outline.get(field, []) if not isinstance(values, list): raise AssemblyError(f"fineOutline.{field} 必须是数组") for index, raw in enumerate(values): if isinstance(raw, str): item = {"id": f"{field}:{index}", "type": fallback_type or "unknown", "name": raw} elif isinstance(raw, Mapping): item = { "id": str(raw.get("id") or f"{field}:{index}"), "type": str(raw.get("type") or fallback_type or "unknown"), "name": str(raw.get("name") or ""), } else: raise AssemblyError(f"fineOutline.{field}[{index}] 类型非法") if not item["name"]: raise AssemblyError(f"fineOutline.{field}[{index}] 缺少名称") if item["id"] not in seen: seen.add(item["id"]) result.append(item) return result def _build_coverage( fine_outline: Mapping[str, Any], facts: Sequence[Mapping[str, Any]], prose: Sequence[Mapping[str, Any]], cards: Sequence[Mapping[str, Any]], ) -> list[dict[str, Any]]: """把五类细纲要素映射到最终保留的事实与原文证据。""" prose_by_source = { str(item["sourceRef"]["sourceId"]): item["evidenceId"] for item in prose } declared = fine_outline.get("declaredNewFacts", []) result: list[dict[str, Any]] = [] for element in _coverage_elements(fine_outline): fact_ids = [item["evidenceId"] for item in facts if element["name"] in item["fact"]] prose_ids: list[str] = [] for card in cards: if str(card.get("name")) != element["name"] and str(card.get("type")) != element["type"]: continue for ref in card.get("sourceRefs") or []: evidence_id = prose_by_source.get(str(ref.get("sourceId"))) if evidence_id and evidence_id not in prose_ids: prose_ids.append(evidence_id) is_declared = any( isinstance(item, Mapping) and (str(item.get("factId")) == element["id"] or element["name"] in str(item.get("text") or "")) for item in declared ) if is_declared: status, reason = "declared_new", "" elif fact_ids and prose_ids: status, reason = "supported", "" elif fact_ids: status, reason = "style_gap", "缺少历史表现原文" elif prose_ids: status, reason = "card_gap", "原文可证但缺少冻结事实索引" else: status, reason = "unsupported", "没有可信事实证据" result.append( { "elementId": element["id"], "elementType": element["type"], "name": element["name"], "status": status, "factEvidenceIds": sorted(fact_ids), "proseEvidenceIds": sorted(prose_ids), "gapReason": reason, } ) return result def _manifest( *, plan_id: str, facts: Sequence[Mapping[str, Any]], prose: Sequence[Mapping[str, Any]], pattern_references: Sequence[Mapping[str, Any]], omitted: Sequence[Mapping[str, str]], index_hints: Sequence[Mapping[str, Any]] = (), ) -> dict[str, Any]: """由最终入包来源集合计算稳定 manifest,不含 runId 或时间戳。""" unique: dict[tuple[str, str, int], dict[str, Any]] = {} for item in [*facts, *prose]: ref = copy.deepcopy(dict(item["sourceRef"])) unique[_source_key(ref)] = ref for raw_ref in pattern_references: # manifest 是来源账本(审计用),只登记可回读指针;范式卡的内容字段(name/ # summary/writingPoints)剥掉,不进 manifest——内容只由上下文内的 patternReferences # 承载。否则 manifest 的 _source_ref 严格校验会以「未知字段」拒收内容字段。 ref = project_pattern_pointers(raw_ref) unique[_source_key(ref)] = ref for hint in index_hints: # 这里记录的是全部冻结卡索引,并不代表卡缺少原文来源。 ref = { "sourceId": hint["sourceId"], "sourceVersion": hint["sourceVersion"], "chapter": hint["asOf"], "sourceType": "diagnostic_card_index", } unique[_source_key(ref)] = ref sources = [unique[key] for key in sorted(unique)] omitted_rows = sorted( [copy.deepcopy(dict(item)) for item in omitted], key=lambda item: (str(item.get("reason")), str(item.get("sourceId"))), ) payload = { "manifestVersion": MANIFEST_VERSION, "planId": plan_id, "sources": sources, "omittedSources": omitted_rows, } return {**payload, "manifestId": retrieval_identity(payload)} def _markdown_manifest(manifest: Mapping[str, Any]) -> str: """渲染不含运行时元数据的人类审阅清单,保证字节稳定。""" lines = [ "# 正文检索清单", "", f"- 计划:`{manifest['planId']}`", f"- 清单:`{manifest['manifestId']}`", f"- 纳入来源:{len(manifest['sources'])}", f"- 排除来源:{len(manifest['omittedSources'])}", "", "## 纳入来源", "", ] if manifest["sources"]: for source in manifest["sources"]: location = f"第{source['chapter']}章" if "chapter" in source else "权威版本" lines.append(f"- `{source['sourceId']}` @ `{source['sourceVersion']}`,{location}") else: lines.append("- 无") lines.extend(["", "## 排除来源", ""]) if manifest["omittedSources"]: for source in manifest["omittedSources"]: lines.append(f"- `{source['sourceId']}`:{source['reason']}") else: lines.append("- 无") return "\n".join(lines) + "\n" def _context_size(context: Mapping[str, Any]) -> int: """用最终规范 JSON 的 Unicode code point 数作为确定性预算单位。""" return len(canonical_json(context)) def _select_exact_prose_chars( prose: Sequence[Mapping[str, Any]], *, char_budget: int ) -> list[dict[str, Any]]: """按稳定顺序截取恰好指定代码点数,来源区间和内容哈希同步收窄。""" if isinstance(char_budget, bool) or not isinstance(char_budget, int) or char_budget < 0: raise AssemblyError("tokenBudget.proseCharBudget 必须是非负整数") remaining = char_budget selected: list[dict[str, Any]] = [] for raw in prose: if remaining == 0: break item = copy.deepcopy(dict(raw)) text = str(item["text"]) if len(text) > remaining: text = text[:remaining] item["text"] = text item["sourceRef"]["endCodePoint"] = ( int(item["sourceRef"].get("startCodePoint") or 0) + len(text) ) item["contentSha256"] = _hash_text(text) selected.append(item) remaining -= len(text) if remaining != 0: raise AssemblyError( f"历史原文不足 proseCharBudget: expected={char_budget}, missing={remaining}" ) return selected def _collect_diff_paths(left: Any, right: Any, path: str = "$") -> list[str]: """递归收集两个完整对象的所有差异路径,不把原始值写入回执。""" if type(left) is not type(right): return [path] if isinstance(left, Mapping): paths: list[str] = [] for key in sorted(set(left) | set(right)): child = f"{path}.{key}" if key not in left or key not in right: paths.append(child) else: paths.extend(_collect_diff_paths(left[key], right[key], child)) return paths if isinstance(left, list): if len(left) != len(right): return [f"{path}.length"] paths = [] for index, (left_item, right_item) in enumerate(zip(left, right, strict=True)): paths.extend(_collect_diff_paths(left_item, right_item, f"{path}[{index}]")) return paths return [] if left == right else [path] def _is_allowed_ac_diff(path: str) -> bool: """只允许预注册证据策略改变模型可见的事实约束、原文摘录与范式引用。 WHY 含 patternReferences:Gate A 的唯一变量是「有无卡(含范式卡)」。C 臂注入公共 范式卡、A 臂恒空,因此两臂创意输入的 patternReferences 必然不同——这是实验设计本身, 不是越界改动。把它列入白名单,门禁才不会把预期的范式差异误判为非法字段漂移。 """ return path.startswith(("$.factConstraints", "$.proseExcerpts", "$.patternReferences")) def build_context_allowlist_diff_receipt( *, sample_id: str, context_a: Mapping[str, Any], context_c: Mapping[str, Any], prose_char_budget: int, require_nonempty: bool = False, ) -> dict[str, Any]: """在 WriterCreativeInput v2 上校验 A/C 单变量并输出脱敏回执。""" if not str(sample_id).strip(): raise AssemblyError("allowlist diff 缺少 sampleId") if isinstance(prose_char_budget, bool) or not isinstance(prose_char_budget, int) or prose_char_budget < 0: raise AssemblyError("allowlist diff 的 proseCharBudget 必须是非负整数") if not isinstance(require_nonempty, bool): raise AssemblyError("allowlist diff 的 require_nonempty 必须是布尔值") if require_nonempty and prose_char_budget <= 0: raise AssemblyError("正式 A/C allowlist diff 的 proseCharBudget 必须是正整数") try: creative_inputs = { "A": build_writer_creative_input(context_a), "C": build_writer_creative_input(context_c), } except (ContractError, TypeError, ValueError) as error: raise AssemblyError(f"allowlist diff 的 WriterContext 非法: {error}") from error context_a = creative_inputs["A"] context_c = creative_inputs["C"] baseline_a = [item for item in context_a["proseExcerpts"] if item["isRecentBaseline"] is True] baseline_c = [item for item in context_c["proseExcerpts"] if item["isRecentBaseline"] is True] if baseline_a != baseline_c: raise AssemblyError("allowlist diff 发现 A/C 连续基线不一致") prose_counts = { "A": sum( len(str(item.get("text") or "")) for item in context_a["proseExcerpts"] if item["isRecentBaseline"] is False ), "C": sum( len(str(item.get("text") or "")) for item in context_c["proseExcerpts"] if item["isRecentBaseline"] is False ), } if set(prose_counts.values()) != {prose_char_budget}: raise AssemblyError( "allowlist diff 发现 A/C 原文字符预算不一致: " f"expected={prose_char_budget}, actual={prose_counts}" ) differences = _collect_diff_paths(context_a, context_c) rejected = [path for path in differences if not _is_allowed_ac_diff(path)] if rejected: raise AssemblyError(f"allowlist diff 包含非原文字段差异: {rejected}") context_hashes = { "A": _hash_value(context_a), "C": _hash_value(context_c), } if require_nonempty and ( not differences or context_hashes["A"] == context_hashes["C"] ): raise AssemblyError("正式 A/C WriterCreativeInput 必须存在非空允许差异") payload = { "schemaVersion": "writer-creative-input-allowlist-diff-v2", "sampleId": str(sample_id), "arms": ["A", "C"], "ok": True, "proseCharBudget": prose_char_budget, "proseCharCount": prose_counts, "allowedDifferencePaths": sorted(differences), "rejectedDifferencePaths": [], "contextSha256": context_hashes, } return {**payload, "receiptSha256": _hash_value(payload)} def assemble_context( *, run_id: str, attempt: int, mode: str, purpose: str, quality_policy_version: str, work_id: int, target_chapter: int, as_of: int, source_version: str, authorization_snapshot: Mapping[str, Any], source_status: str, retrieval_plan: Mapping[str, Any], retrieval_result: Mapping[str, Any], fine_outline: Mapping[str, Any], narrative_state: Mapping[str, Any], recent_chapters: Sequence[Mapping[str, Any]], output_contract: Mapping[str, Any], token_budget: Mapping[str, int], generation_length_contract: Mapping[str, Any] | None = None, pattern_references: Sequence[Mapping[str, Any]] = (), style_constraints: Sequence[str] = (), humanization_contract: Mapping[str, Any] | None = None, generated_at: str, evidence_strategy: str = "production_dual_evidence", ) -> dict[str, Any]: """组装稳定 WriterContext,并返回规范 JSON 与 Markdown manifest。""" if retrieval_plan.get("runId") != run_id or retrieval_plan.get("asOf") != as_of: raise AssemblyError("retrievalPlan 与当前 runId/asOf 不一致") humanization_guidance: list[str] = [] humanization_provenance = None if humanization_contract is not None: try: validate_humanization_contract(dict(humanization_contract), "prevention") except (TypeError, ValueError) as exc: raise AssemblyError(f"humanization_contract 不满足 prevention 合同: {exc}") from exc if humanization_contract.get("schema_version") not in {"ai-flavor-prevention-v1", "ai-flavor-prevention-v2"}: raise AssemblyError("humanization_contract schema_version 不支持") contract_work_ref = humanization_contract.get("work_ref") if contract_work_ref not in {None, f"work:{work_id}"}: raise AssemblyError("humanization_contract 与当前 work_id 不一致") raw_guidance = humanization_contract.get("writer_constraints", []) if not isinstance(raw_guidance, list) or any(not isinstance(item, str) or not item.strip() for item in raw_guidance): raise AssemblyError("humanization_contract.writer_constraints 必须是非空字符串数组") humanization_guidance = [normalize_text(item) for item in raw_guidance] basis = humanization_contract.get("built_from") or {} humanization_provenance = { "schemaVersion": humanization_contract["schema_version"], "ruleLibraryVersion": str(basis.get("rule_library_version") or "unknown"), "voiceLedgerSha256": ( "sha256:" + str(basis["voice_ledger_sha256"]) if basis.get("voice_ledger_sha256") and not str(basis["voice_ledger_sha256"]).startswith("sha256:") else basis.get("voice_ledger_sha256") ), "constraintCount": len(humanization_guidance), } if retrieval_plan.get("filters", {}).get("workId") != work_id: raise AssemblyError("retrievalPlan 与当前 workId 不一致") max_chars = token_budget.get("maxContextChars") if isinstance(max_chars, bool) or not isinstance(max_chars, int) or max_chars <= 0: raise AssemblyError("tokenBudget.maxContextChars 必须是正整数") if evidence_strategy == "production_dual_evidence": orchestration_strategy = evidence_strategy contract_strategy = evidence_strategy elif evidence_strategy in EVALUATION_STRATEGY_ALIASES: orchestration_strategy = EVALUATION_STRATEGY_ALIASES[evidence_strategy] contract_strategy = CONTRACT_STRATEGIES[orchestration_strategy] else: raise AssemblyError("evidenceStrategy 非法") if mode == "production" and orchestration_strategy != "production_dual_evidence": raise AssemblyError("生产上下文必须使用 production_dual_evidence") if orchestration_strategy != "production_dual_evidence" and not ( mode == "diagnostic_only" and purpose in {"evaluation", "diagnostic"} ): raise AssemblyError("诊断证据策略只允许用于 diagnostic_only 评测或诊断") baseline = ( [] if orchestration_strategy == "card_only_diagnostic" else _select_recent_baseline(recent_chapters, as_of=as_of) ) baseline_blocks = {_whole_source_key(item["sourceRef"]) for item in baseline} supplemental: list[dict[str, Any]] = [] seen_fragments = {_source_key(item["sourceRef"]) for item in baseline} for raw in retrieval_result.get("proseEvidence", []): retrieval_arm = raw.get("retrievalArm") if isinstance(raw, Mapping) else None if retrieval_arm not in {None, "A", "C"}: raise AssemblyError("proseEvidence.retrievalArm 只能是 A/C") if orchestration_strategy == "generic_prose_retrieval" and retrieval_arm != "A": continue if orchestration_strategy == "card_indexed_prose_retrieval" and retrieval_arm == "A": continue item = _normalize_prose(raw, recent=False) if item["chapter"] > as_of: raise AssemblyError("卡来源原文包含目标章或未来章") if _whole_source_key(item["sourceRef"]) in baseline_blocks: continue key = _source_key(item["sourceRef"]) if key not in seen_fragments: seen_fragments.add(key) supplemental.append(item) supplemental.sort(key=lambda item: _source_key(item["sourceRef"])) prose_char_budget = token_budget.get("proseCharBudget") if orchestration_strategy in { "generic_prose_retrieval", "card_indexed_prose_retrieval", } and prose_char_budget is not None: supplemental = _select_exact_prose_chars( supplemental, char_budget=prose_char_budget, ) facts = sorted( [_normalize_fact(item) for item in retrieval_result.get("factEvidence", [])], key=lambda item: ({"high": 0, "medium": 1, "low": 2}[item["riskLevel"]], item["evidenceId"]), ) cards = [copy.deepcopy(dict(item)) for item in retrieval_result.get("cards", [])] # 新合同由检索器显式给出覆盖全部冻结命中卡的 indexHints。旧字段只在 # 历史评测夹具尚未迁移时回退,不能再决定有 sourceRefs 的卡是否入包。 raw_index_hints = retrieval_result.get("indexHints") if raw_index_hints is None: raw_index_hints = retrieval_result.get("unverifiedIndexHints", []) index_hints = sorted( [ _normalize_index_hint(item, as_of=as_of) for item in raw_index_hints ], key=lambda item: (item["sourceVersion"], item["sourceId"], item["cardId"]), ) if orchestration_strategy in {"card_only_diagnostic", "card_indexed_prose_retrieval"} and not index_hints: raise AssemblyError("卡索引诊断策略缺少 indexHints") if orchestration_strategy in {"production_dual_evidence", "generic_prose_retrieval"}: index_hints = [] inherited_omitted = [ {"sourceId": str(item.get("sourceId") or "unknown"), "reason": str(item.get("reason") or "not_relevant")} for item in retrieval_result.get("manifest", {}).get("omittedSources", []) if isinstance(item, Mapping) ] selected_facts: list[dict[str, Any]] = [] # 连续前四章全文是正文实验 v1 的不可裁剪基线。预算容不下时必须失败关闭, # 不能静默退化成只保留最近一两章。 selected_prose: list[dict[str, Any]] = list(baseline) omitted = list(inherited_omitted) candidates: list[tuple[str, dict[str, Any]]] = [] if orchestration_strategy == "production_dual_evidence": candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "high") candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "medium") candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "low") candidates.extend(("prose", item) for item in supplemental) elif orchestration_strategy in { "generic_prose_retrieval", "card_indexed_prose_retrieval", }: candidates.extend(("prose", item) for item in supplemental) def make_context() -> dict[str, Any]: """用当前选择集生成完整上下文,供预算试算与最终冻结。""" # 规划期选定的文风约束:归一化后仅在非空时入冻结上下文(空则省略整个键, # 上下文逐字节不变,不破坏既有哈希与 A/C 单变量)。 style_rules = [normalize_text(str(rule)) for rule in style_constraints if str(rule).strip()] style_rules.extend(humanization_guidance) ordered_prose = sorted( selected_prose, key=lambda item: ( 0 if item["isRecentBaseline"] else 1, item["chapter"] if item["isRecentBaseline"] else 0, _source_key(item["sourceRef"]), ), ) ordered_facts = sorted(selected_facts, key=lambda item: item["evidenceId"]) manifest = _manifest( plan_id=str(retrieval_plan["planId"]), facts=ordered_facts, prose=ordered_prose, pattern_references=pattern_references, omitted=omitted, index_hints=index_hints, ) context = { "schemaVersion": "writer-context-v1", "runId": run_id, "attempt": attempt, "mode": mode, "purpose": purpose, "qualityPolicyVersion": quality_policy_version, "workId": work_id, "targetChapter": target_chapter, "asOf": as_of, "contextSnapshot": { "manifestId": manifest["manifestId"], "contextSha256": "sha256:" + "0" * 64, "generatedAt": normalize_text(generated_at), }, "sourceVersion": source_version, "authorizationSnapshot": copy.deepcopy(dict(authorization_snapshot)), "sourceStatus": source_status, "retrievalPlan": copy.deepcopy(dict(retrieval_plan)), "retrievalManifest": manifest, "fineOutline": _outline_contract(fine_outline), "narrativeState": copy.deepcopy(dict(narrative_state)), "factEvidence": ordered_facts, "proseEvidence": ordered_prose, "evidenceStrategy": contract_strategy, "indexHints": copy.deepcopy(index_hints), "patternReferences": [copy.deepcopy(dict(item)) for item in pattern_references], "evidenceCoverage": _build_coverage(fine_outline, ordered_facts, ordered_prose, cards), "outputContract": copy.deepcopy(dict(output_contract)), "tokenBudget": {"maxContextChars": max_chars, "usedContextChars": 0}, "omittedSources": sorted(copy.deepcopy(omitted), key=lambda item: (item["reason"], item["sourceId"])), "acceptanceEligible": mode == "production" and purpose == "production", } if humanization_provenance is not None: context["humanizationProvenance"] = copy.deepcopy(humanization_provenance) if generation_length_contract is not None: context["generationLengthContract"] = copy.deepcopy(dict(generation_length_contract)) if style_rules: context["styleConstraints"] = list(style_rules) # usedContextChars 自身位数会影响 JSON 长度,迭代到数值稳定。 for _ in range(8): used = _context_size(context) if context["tokenBudget"]["usedContextChars"] == used: break context["tokenBudget"]["usedContextChars"] = used context["contextSnapshot"]["contextSha256"] = retrieval_identity(context) return context baseline_context = make_context() if _context_size(baseline_context) > max_chars: raise AssemblyError("上下文预算不足以容纳细纲硬约束与连续前四章全文基线") for kind, item in candidates: target = selected_facts if kind == "fact" else selected_prose target.append(item) trial = make_context() if _context_size(trial) > max_chars: target.pop() omitted.append({"sourceId": str(item["sourceRef"]["sourceId"]), "reason": "token_budget"}) context = make_context() if prose_char_budget is not None and orchestration_strategy in { "generic_prose_retrieval", "card_indexed_prose_retrieval", }: selected_chars = sum( len(item["text"]) for item in context["proseEvidence"] if item["isRecentBaseline"] is False ) if selected_chars != prose_char_budget: raise AssemblyError( "上下文总预算无法保留完整 proseCharBudget: " f"expected={prose_char_budget}, actual={selected_chars}" ) # 添加裁剪回显可能占用少量预算;若越界,按最低优先级继续移除。 while _context_size(context) > max_chars and (selected_prose or selected_facts): removable_prose = [item for item in selected_prose if not item["isRecentBaseline"]] if removable_prose: removed = removable_prose[-1] selected_prose.remove(removed) elif selected_facts and selected_facts[-1]["riskLevel"] != "high": removed = selected_facts.pop() elif selected_facts: removed = selected_facts.pop() else: raise AssemblyError("上下文预算不足以保留连续前四章全文基线") omitted.append({"sourceId": str(removed["sourceRef"]["sourceId"]), "reason": "token_budget"}) context = make_context() if _context_size(context) > max_chars: raise AssemblyError("上下文预算不足以保留可用证据") context["tokenBudget"]["usedContextChars"] = _context_size(context) context["contextSnapshot"]["contextSha256"] = retrieval_identity(context) normalized_context = validate_writer_context(context) writer_creative_input = build_writer_creative_input(normalized_context) return { "context": normalized_context, "contextJson": canonical_json(normalized_context), "writerCreativeInput": writer_creative_input, "writerCreativeInputJson": canonical_json(writer_creative_input), "orchestrationStrategy": orchestration_strategy, "manifestMarkdown": _markdown_manifest(normalized_context["retrievalManifest"]), } __all__ = [ "AssemblyError", "assemble_context", "build_context_allowlist_diff_receipt", ]