zizi b0bc7a8745 框架: 技能按动作-对象重组 + 先审后入创作闭环
一、技能重组(动作-对象命名)
- 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为
  clean-book-text/decide-candidate/write-next-chapter/access-database/
  check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持)
- agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名

二、先审后入创作闭环(本次核心)
正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交",
DB 级兜底,编排层跳步即被硬拒。
- candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链
- fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量,
  模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本
- projection_registry.py + example_projection_run(107):投影登记与恢复
- acceptance_state.py:接受前置实时状态重读
- lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格
- DDL 105:example_candidate 增 semantic_status/semantic_report_sha256
- write_canonical.accept:语义兜底+同事务合并增量+登记投影;
  run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链
- claude_runtime:兼容新 CLI modelUsage 信息字段

三、审查修复(独立子代理四维审查后)
- 事实增量 propose→approve 翻态正道,不撞唯一键
- 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记
- 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理

测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。
创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
2026-08-14 10:24:08 +08:00

2188 lines
97 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""从实验库只读装配 Writer Gate A 五章真实临时配置。
本适配器只执行 SELECT,并把来源证明、授权、目标 scaffold、全量冻结历史、
冻结近章正文和预注册 upgrade_book 卡固定在同一个 REPEATABLE READ READ ONLY
事务中。目标章正文不查询;目标 scaffold 只拆成写手细纲和 evaluator-only 真值断言。
"""
from __future__ import annotations
import argparse
import copy
import hashlib
import json
import os
import sys
import uuid
from datetime import datetime, timedelta, timezone
from pathlib import Path
from typing import Any, Callable, Mapping, Sequence
import psycopg
from psycopg.rows import dict_row
SCRIPT_DIR = Path(__file__).resolve().parent
READ_CONTEXT_SCRIPTS = SCRIPT_DIR.parents[1] / "assemble-context" / "scripts"
SNAPSHOT_SCRIPTS = SCRIPT_DIR.parents[1] / "freeze-context" / "scripts"
sys.path.insert(0, str(READ_CONTEXT_SCRIPTS))
sys.path.insert(0, str(SNAPSHOT_SCRIPTS))
from build_snapshot import normalize_chapter_range # noqa: E402
from assemble_writer_context import ( # noqa: E402
AssemblyError,
assemble_context,
build_context_allowlist_diff_receipt,
)
from load_reference_work import ( # noqa: E402
DSN,
TENANT_ID,
AdapterError,
_target_facts_from_scaffold,
begin_read_snapshot,
project_authorization,
project_card,
validate_source_records,
)
from retrieve_writer_sources import ( # noqa: E402
ReplayCardIndexRepository,
RetrievalError,
build_retrieval_plan,
retrieve_writer_sources,
)
from writer_contract import ( # noqa: E402
PATTERN_NAME_MAX_CHARS,
PATTERN_POINTS_MAX_FIELDS,
PATTERN_POINT_MAX_CHARS,
PATTERN_SUMMARY_MAX_CHARS,
han_count,
normalize_text,
pattern_references_for_arm,
)
from writer_eval_preregister import build_balanced_preregistration # noqa: E402
PRIVATE_TMP = Path("/private/tmp").resolve()
DEFAULT_BASE_CONFIG = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
DEFAULT_SELECTOR_CONFIG = (
SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-card-selectors-v1.json"
)
CANONICAL_CHAPTER_STATUSES = frozenset({"published", "confirmed", "canonical"})
EXPECTED_WRITER_INPUT_PROVENANCE = "preregistered_fine_outline"
EXPECTED_ORACLE_INPUT_PROVENANCE = "oracle_reference_scaffold"
PREREGISTERED_MAX_CONTEXT_CHARS = 140_000
RUNTIME_ADAPTER_VERSION_PREFIX = "writer-runtime-v1"
BUDGET_ROLES = ("writer", "semantic_detector", "blind_judge")
# raw vault 对租约的硬上限是 24 小时。五章三臂会串行执行多角色长调用,因此装配时
# 使用接近硬上限但留有时钟余量的窗口;execute 仍会在启动和每笔调用前复检。
RAW_RETENTION_WINDOW = timedelta(hours=23)
# 公共范式库五型(muse_knowledge_draft.draft_payload->>'型'):
# 套路 / 通用桥段 / 叙事技法 / 情感桥段 / 打斗桥段。C 臂按型各召回若干张。
PATTERN_CARD_TYPES = ("trope", "scene_pattern", "craft", "emotion", "combat")
# 每型最多取 2 张:五型合计 ≤ 10,严格低于总量硬上限,避免撑爆写手上下文预算。
PATTERN_TOP_PER_TYPE = 2
# 范式引用总量硬上限;即使提高每型 top,也不会超过这个数。
PATTERN_TOTAL_CAP = 12
class WriterReferenceWorkError(AdapterError):
"""真实正文装配输入缺失、歧义、漂移或越过冻结线时抛出。"""
def _safe_json(value: Any) -> str:
"""稳定序列化临时配置,便于重复运行后比较哈希。"""
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
def _sha256_value(value: Any) -> str:
"""对规范 JSON 对象计算带算法前缀的稳定哈希。"""
return "sha256:" + hashlib.sha256(_safe_json(value).encode("utf-8")).hexdigest()
def _is_full_sha256(value: Any) -> bool:
"""只接受不带前缀的 64 位小写 SHA-256。"""
return isinstance(value, str) and len(value) == 64 and all(
char in "0123456789abcdef" for char in value
)
def _validate_self_hash(value: Mapping[str, Any], field: str) -> None:
"""校验对象的 receiptSha256 是否等于其规范 JSON 自哈希。"""
receipt = value.get("receiptSha256")
if not isinstance(receipt, str) or not receipt.startswith("sha256:") or not _is_full_sha256(
receipt.removeprefix("sha256:")
):
raise WriterReferenceWorkError(f"{field}.receiptSha256 必须是 canonical SHA-256")
expected = _sha256_value(
{key: item for key, item in value.items() if key != "receiptSha256"}
)
if receipt != expected:
raise WriterReferenceWorkError(f"{field}.receiptSha256 canonical 自哈希无效")
def _stamp_raw_retention(config: dict[str, Any]) -> None:
"""为输出配置的 rawRetention 盖运行期租约到期时间戳并重签自哈希。
base 是仓内静态文件,存不了未来时间戳;retainUntil 是运行期值,必须由装配器
在生成仓外临时配置时盖戳。重签使用与 execute 端校验
(run_writer_replay._validate_authorization_records)同源的规范自哈希算法
(_sha256_value,即 canonical_sha256):授权本身(approvedBy/authorizationId/
status 等)一律不变,只补租约到期时间戳并刷新 receiptSha256。
"""
raw = config["executionAuthorization"]["rawRetention"]
raw["retainUntil"] = (datetime.now(timezone.utc) + RAW_RETENTION_WINDOW).isoformat()
raw["receiptSha256"] = _sha256_value(
{key: value for key, value in raw.items() if key != "receiptSha256"}
)
def _positive_chapter(value: Any, field: str) -> int:
"""严格接受正整数章号,不把 bool、浮点或模糊文本猜成章号。"""
if isinstance(value, bool):
raise WriterReferenceWorkError(f"{field} 必须是正整数章号")
if isinstance(value, str) and value.isdigit():
value = int(value)
if not isinstance(value, int) or value <= 0:
raise WriterReferenceWorkError(f"{field} 必须是正整数章号")
return value
def _card_payload(row: Mapping[str, Any]) -> Mapping[str, Any]:
"""读取卡 payload;选择器只信任结构化 type/name/alias。"""
payload = row.get("draft_payload")
if not isinstance(payload, Mapping):
raise WriterReferenceWorkError(f"卡 {row.get('id')} 缺少 draft_payload 对象")
return payload
def _selector_samples(selector_config: Mapping[str, Any]) -> list[Mapping[str, Any]]:
"""校验稳定选择器顶层结构和样本唯一性。"""
if not isinstance(selector_config, Mapping):
raise WriterReferenceWorkError("卡选择器配置必须是对象")
samples = selector_config.get("samples")
if not isinstance(samples, list) or not samples or any(
not isinstance(item, Mapping) for item in samples
):
raise WriterReferenceWorkError("卡选择器 samples 必须是非空对象数组")
sample_ids = [str(item.get("sampleId") or "") for item in samples]
if any(not item for item in sample_ids) or len(sample_ids) != len(set(sample_ids)):
raise WriterReferenceWorkError("卡选择器 sampleId 必须非空且唯一")
targets = [
_positive_chapter(item.get("targetChapter"), f"{item['sampleId']}.targetChapter")
for item in samples
]
if len(targets) != len(set(targets)):
raise WriterReferenceWorkError("卡选择器 targetChapter 必须唯一")
return samples
def selector_sha256(content: bytes) -> str:
"""对选择器规范文件原始字节计算带算法前缀的 SHA-256。"""
if not isinstance(content, bytes) or not content:
raise WriterReferenceWorkError("卡选择器规范文件不能为空")
return "sha256:" + hashlib.sha256(content).hexdigest()
def _validate_loader_controls(
base_config: Mapping[str, Any],
selector_config: Mapping[str, Any],
*,
selector_digest: str,
) -> Mapping[str, Any]:
"""在读取数据库前校验预注册选择器、输入来源和统一上下文上限。"""
if not isinstance(base_config, Mapping):
raise WriterReferenceWorkError("基础配置必须是对象")
common = base_config.get("commonControls")
if not isinstance(common, Mapping):
raise WriterReferenceWorkError("基础配置缺少 commonControls")
selector_version = str(selector_config.get("selectorVersion") or "")
if not selector_version or common.get("selectorVersion") != selector_version:
raise WriterReferenceWorkError("selectorVersion 未与预注册公共控制绑定")
if common.get("selectorSha256") != selector_digest:
raise WriterReferenceWorkError("选择器规范 JSON 的 SHA-256 与预注册公共控制不一致")
if common.get("writerInputProvenance") != EXPECTED_WRITER_INPUT_PROVENANCE:
raise WriterReferenceWorkError(
"writerInputProvenance 必须固定为 preregistered_fine_outline"
)
if common.get("oracleInputProvenance") != EXPECTED_ORACLE_INPUT_PROVENANCE:
raise WriterReferenceWorkError(
"oracleInputProvenance 必须固定为 oracle_reference_scaffold"
)
max_context_chars = common.get("maxContextChars")
if max_context_chars != PREREGISTERED_MAX_CONTEXT_CHARS:
raise WriterReferenceWorkError(
"commonControls.maxContextChars 必须严格等于预注册固定值 140000"
)
samples = base_config.get("samples")
if not isinstance(samples, list) or not samples:
raise WriterReferenceWorkError("基础配置 samples 不能为空")
execution_authorization = base_config.get("executionAuthorization")
if not isinstance(execution_authorization, Mapping):
raise WriterReferenceWorkError("基础配置缺少 executionAuthorization")
runtime_probe = execution_authorization.get("runtimeProbe")
if not isinstance(runtime_probe, Mapping):
raise WriterReferenceWorkError("executionAuthorization.runtimeProbe 必须是对象")
if runtime_probe.get("status") != "successful":
raise WriterReferenceWorkError("executionAuthorization.runtimeProbe 必须为 successful")
_validate_self_hash(runtime_probe, "executionAuthorization.runtimeProbe")
cli_version = runtime_probe.get("claudeCliVersion")
executable_sha256 = runtime_probe.get("claudeExecutableSha256")
if not isinstance(cli_version, str) or not cli_version.strip():
raise WriterReferenceWorkError("runtimeProbe.claudeCliVersion 不能为空")
if not _is_full_sha256(executable_sha256):
raise WriterReferenceWorkError(
"runtimeProbe.claudeExecutableSha256 必须是完整 64 位小写 SHA-256"
)
resolved_model_id = runtime_probe.get("resolvedModelId")
if not isinstance(resolved_model_id, str) or not resolved_model_id.strip():
raise WriterReferenceWorkError("runtimeProbe.resolvedModelId 不能为空")
if common.get("modelVersion") != resolved_model_id:
raise WriterReferenceWorkError(
"commonControls.modelVersion 必须精确等于 runtimeProbe.resolvedModelId"
)
expected_adapter_version = (
f"{RUNTIME_ADAPTER_VERSION_PREFIX}|claude-cli-{cli_version}|"
f"binary-sha256-{executable_sha256}"
)
if common.get("adapterVersion") != expected_adapter_version:
raise WriterReferenceWorkError(
"commonControls.adapterVersion 未绑定 runtimeProbe 的 CLI version 与 executable hash"
)
for control_name in ("budget", "rawRetention"):
control = execution_authorization.get(control_name)
if not isinstance(control, Mapping):
raise WriterReferenceWorkError(
f"executionAuthorization.{control_name} 必须是对象"
)
allowed_statuses = {"pending", "approved"}
if control.get("status") not in allowed_statuses:
raise WriterReferenceWorkError(
f"executionAuthorization.{control_name} 状态必须为 {sorted(allowed_statuses)}"
)
_validate_self_hash(control, f"executionAuthorization.{control_name}")
budget = execution_authorization["budget"]
planned_calls = budget.get("plannedCalls")
max_calls = budget.get("maxCalls")
if not isinstance(planned_calls, Mapping) or set(planned_calls) != set(BUDGET_ROLES):
raise WriterReferenceWorkError(
"executionAuthorization.budget.plannedCalls 必须完整覆盖三角色"
)
if not isinstance(max_calls, Mapping) or set(max_calls) != set(BUDGET_ROLES):
raise WriterReferenceWorkError(
"executionAuthorization.budget.maxCalls 必须完整覆盖三角色"
)
required_calls = len(samples) * 3
for role in BUDGET_ROLES:
planned = planned_calls[role]
maximum = max_calls[role]
if (
isinstance(planned, bool)
or not isinstance(planned, int)
or planned < required_calls
):
raise WriterReferenceWorkError(
f"executionAuthorization.budget.plannedCalls.{role} 必须是不低于 {required_calls} 的整数"
)
if isinstance(maximum, bool) or not isinstance(maximum, int) or maximum < planned:
raise WriterReferenceWorkError(
f"executionAuthorization.budget.maxCalls.{role} 必须是不低于 plannedCalls 的整数"
)
if common.get("sampling") != {
"temperature": "unsupported",
"topP": "unsupported",
"seed": "unsupported",
"reproducibilityClaim": "not_claimed",
}:
raise WriterReferenceWorkError("sampling 必须固定为 unsupported/not_claimed")
if base_config.get("strategyVersion") != "writer-abc-single-variable-v2":
raise WriterReferenceWorkError("strategyVersion 必须升级为 writer-abc-single-variable-v2")
if base_config.get("armPolicies") != {
"A": {
"evidenceStrategy": "generic_prose_retrieval",
"gateThresholdEligible": True,
"acceptanceEligible": False,
"writerCardVisibility": False,
},
"B": {
"evidenceStrategy": "card_only_diagnostic",
"gateThresholdEligible": False,
"acceptanceEligible": False,
"writerCardVisibility": True,
},
"C": {
"evidenceStrategy": "card_indexed_prose_retrieval",
"gateThresholdEligible": True,
"acceptanceEligible": False,
"writerCardVisibility": False,
},
}:
raise WriterReferenceWorkError("armPolicies 未固定 A/C Gate 与 B 负对照边界")
sample_ids = [str(sample.get("sampleId") or "") for sample in samples if isinstance(sample, Mapping)]
expected_preregistration = build_balanced_preregistration(
evaluation_set_version=str(base_config.get("evaluationSetVersion") or ""),
sample_ids=sample_ids,
)
if base_config.get("evaluationPreregistration") != expected_preregistration:
raise WriterReferenceWorkError("evaluationPreregistration 顺序、平衡或哈希不匹配")
for index, sample in enumerate(samples):
if not isinstance(sample, Mapping):
raise WriterReferenceWorkError(f"samples[{index}] 必须是对象")
configured = sample.get("writerContextInput", {}).get("tokenBudget", {})
if not isinstance(configured, Mapping):
raise WriterReferenceWorkError(f"samples[{index}] tokenBudget 必须是对象")
if configured.get("maxContextChars") != max_context_chars:
raise WriterReferenceWorkError(
f"samples[{index}] maxContextChars 必须原样使用预注册公共控制值"
)
return common
def _required_character_probes(base_config: Mapping[str, Any]) -> list[dict[str, Any]]:
"""只从冻结细纲要求提取具名角色;卡内角色状态不参与判定。"""
probes: list[dict[str, Any]] = []
for raw_sample in base_config.get("samples", []):
sample_id = str(raw_sample.get("sampleId") or "")
as_of = _positive_chapter(raw_sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
requirements = raw_sample.get("writerContextInput", {}).get("requirements", {})
basis = raw_sample.get("newCharacterBasis")
if not isinstance(requirements, Mapping) or not isinstance(basis, Mapping):
raise WriterReferenceWorkError(f"{sample_id} 缺少角色要求或冻结判定基线")
required = requirements.get("requiredCharacters")
generic = basis.get("genericRoles")
if (
not isinstance(required, list)
or not required
or any(not isinstance(name, str) or not name.strip() for name in required)
or len(required) != len(set(required))
):
raise WriterReferenceWorkError(f"{sample_id}.requiredCharacters 必须是无重复非空字符串数组")
if (
not isinstance(generic, list)
or any(not isinstance(name, str) or not name.strip() for name in generic)
or not set(generic).issubset(required)
):
raise WriterReferenceWorkError(f"{sample_id}.genericRoles 必须是 requiredCharacters 子集")
named = [name for name in required if name not in set(generic)]
if generic and named:
raise WriterReferenceWorkError(f"{sample_id} 暂不允许具名角色与泛称角色混合计算比例")
probes.extend(
{"sampleId": sample_id, "name": name, "asOfChapter": as_of}
for name in named
)
identities = [(item["sampleId"], item["name"]) for item in probes]
if len(identities) != len(set(identities)):
raise WriterReferenceWorkError("具名角色 Canonical 查询包含重复项")
return probes
def _canonical_character_index(
rows: Sequence[Mapping[str, Any]],
probes: Sequence[Mapping[str, Any]],
) -> dict[str, dict[str, dict[str, Any]]]:
"""校验 Canonical 正文命中结果,并按样本和角色名建立只含章号的索引。"""
expected = {
(str(item["sampleId"]), str(item["name"])): int(item["asOfChapter"])
for item in probes
}
indexed: dict[str, dict[str, dict[str, Any]]] = {}
seen: set[tuple[str, str]] = set()
for index, raw in enumerate(rows):
if not isinstance(raw, Mapping):
raise WriterReferenceWorkError(f"canonical_character_mentions[{index}] 不是对象")
sample_id = str(raw.get("sample_id") or raw.get("sampleId") or "")
name = str(raw.get("name") or "")
identity = (sample_id, name)
if identity not in expected or identity in seen:
raise WriterReferenceWorkError("Canonical 角色命中结果含额外项或重复项")
seen.add(identity)
as_of = _positive_chapter(
raw.get("as_of_chapter", raw.get("asOfChapter")),
f"{sample_id}.{name}.asOfChapter",
)
if as_of != expected[identity]:
raise WriterReferenceWorkError("Canonical 角色命中结果冻结点漂移")
raw_hits = raw.get("hit_chapters", raw.get("hitChapters")) or []
if not isinstance(raw_hits, list):
raise WriterReferenceWorkError("Canonical 角色命中章必须是数组")
hits = sorted({_positive_chapter(item, f"{sample_id}.{name}.hitChapters") for item in raw_hits})
if any(chapter > as_of for chapter in hits):
raise WriterReferenceWorkError("Canonical 角色命中结果越过冻结点")
raw_first = raw.get("first_chapter", raw.get("firstChapter"))
first = None if raw_first is None else _positive_chapter(raw_first, f"{sample_id}.{name}.firstChapter")
if first != (hits[0] if hits else None):
raise WriterReferenceWorkError("Canonical 角色首次命中章与命中章集合不一致")
indexed.setdefault(sample_id, {})[name] = {
"firstChapter": first,
"hitChapters": hits,
}
if seen != set(expected):
raise WriterReferenceWorkError("Canonical 角色命中结果缺少预注册具名角色")
return indexed
def _recompute_new_character_ratio(
sample: dict[str, Any],
character_index: Mapping[str, Mapping[str, Mapping[str, Any]]],
) -> None:
"""依据冻结 Canonical 正文重写具名角色分区;不读取或信任卡内容。"""
sample_id = str(sample.get("sampleId") or "")
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
requirements = sample.get("writerContextInput", {}).get("requirements", {})
basis = sample.get("newCharacterBasis")
if not isinstance(requirements, Mapping) or not isinstance(basis, Mapping):
raise WriterReferenceWorkError(f"{sample_id} 缺少角色比例输入")
required = list(requirements.get("requiredCharacters") or [])
generic = list(basis.get("genericRoles") or [])
if generic:
sample["newCharacterRatio"] = None
sample["newCharacterRatioStatus"] = "unresolved_generic_role"
sample["newCharacterBasis"] = {
"definition": "named_required_characters_absent_before_as_of_ratio",
"asOfChapter": as_of,
"requiredCharacters": required,
"knownBeforeAsOf": [],
"absentBeforeAsOf": [],
"genericRoles": generic,
}
return
mentions = character_index.get(sample_id, {})
known = [name for name in required if mentions.get(name, {}).get("hitChapters")]
absent = [name for name in required if name not in known]
sample["newCharacterRatio"] = len(absent) / len(required)
sample["newCharacterRatioStatus"] = "resolved"
sample["newCharacterBasis"] = {
"definition": "named_required_characters_absent_before_as_of_ratio",
"asOfChapter": as_of,
"requiredCharacters": required,
"knownBeforeAsOf": known,
"absentBeforeAsOf": absent,
"genericRoles": [],
}
def resolve_card_selectors(
card_rows: Sequence[Mapping[str, Any]],
selector_config: Mapping[str, Any],
) -> dict[str, list[dict[str, Any]]]:
"""按 type + canonical name/alias 精确唯一解析预注册卡。
这里只读取稳定配置,不接收候选分数、回放结果或目标章正文,因此运行结果
不可能反向改变选卡。canonical name 和 alias 都是全字符串相等匹配。
"""
if not isinstance(card_rows, Sequence) or isinstance(card_rows, (str, bytes)):
raise WriterReferenceWorkError("卡查询结果必须是数组")
result: dict[str, list[dict[str, Any]]] = {}
used_card_ids: set[str] = set()
for sample in _selector_samples(selector_config):
sample_id = str(sample["sampleId"])
selectors = sample.get("cards")
if not isinstance(selectors, list) or not selectors or any(
not isinstance(item, Mapping) for item in selectors
):
raise WriterReferenceWorkError(f"{sample_id}.cards 必须是非空对象数组")
identities: list[tuple[str, str]] = []
selected: list[dict[str, Any]] = []
for index, selector in enumerate(selectors):
card_type = str(selector.get("type") or "").strip()
requested_name = str(selector.get("name") or "").strip()
if not card_type or not requested_name:
raise WriterReferenceWorkError(f"{sample_id}.cards[{index}] 缺少 type/name")
identity = (card_type, requested_name)
if identity in identities:
raise WriterReferenceWorkError(f"{sample_id} 含重复卡选择器 {identity}")
identities.append(identity)
matches: list[dict[str, Any]] = []
for raw_row in card_rows:
if not isinstance(raw_row, Mapping):
raise WriterReferenceWorkError("卡查询结果含非对象行")
payload = _card_payload(raw_row)
if str(payload.get("type") or "") != card_type:
continue
canonical_name = str(payload.get("名称") or "")
raw_aliases = payload.get("别名")
if raw_aliases is None:
aliases: list[str] = []
elif isinstance(raw_aliases, list) and all(
isinstance(alias, str) for alias in raw_aliases
):
aliases = raw_aliases
else:
raise WriterReferenceWorkError(f"卡 {raw_row.get('id')} 的别名不是字符串数组")
if canonical_name == requested_name or requested_name in aliases:
matches.append(copy.deepcopy(dict(raw_row)))
if len(matches) != 1:
raise WriterReferenceWorkError(
f"{sample_id} 选择器 ({card_type},{requested_name}) 必须唯一匹配,实际 {len(matches)} 张"
)
card_id = str(matches[0].get("id") or "")
if not card_id or card_id in used_card_ids:
raise WriterReferenceWorkError(f"卡 {card_id or '<empty>'} 被重复选择")
used_card_ids.add(card_id)
selected.append(matches[0])
result[sample_id] = selected
return result
def milestone_reference_chapters(
milestones: Sequence[Mapping[str, Any]],
*,
as_of: int,
limit: int = 3,
) -> list[int]:
"""展开明确里程碑区间,去重后返回冻结线内最近最多三章。"""
freeze = _positive_chapter(as_of, "as_of")
if isinstance(limit, bool) or not isinstance(limit, int) or limit <= 0:
raise WriterReferenceWorkError("里程碑引用上限必须是正整数")
chapters: set[int] = set()
for index, milestone in enumerate(milestones):
if not isinstance(milestone, Mapping):
raise WriterReferenceWorkError(f"milestones[{index}] 不是对象")
raw_chapter = next(
(
milestone[key]
for key in ("chapter", "chapter_no", "order_no", "章", "章号")
if key in milestone
),
None,
)
bounds = normalize_chapter_range(raw_chapter)
if bounds is None:
raise WriterReferenceWorkError(f"milestones[{index}] 缺少明确绝对章号边界")
# 卡可包含未来演变,但来源定位只展开完整落在冻结线内的里程碑。
if bounds[1] > freeze:
continue
chapters.update(range(bounds[0], bounds[1] + 1))
if not chapters:
raise WriterReferenceWorkError("卡在冻结线内没有可定位 Canonical 原文的里程碑")
return sorted(chapters)[-limit:]
def _normalize_card_milestones(card: dict[str, Any], *, as_of: int) -> None:
"""把区间里程碑状态归一到明确结束章,供既有冻结器严格消费。"""
normalized: list[dict[str, Any]] = []
for index, raw in enumerate(card.get("milestones") or []):
if not isinstance(raw, Mapping):
raise WriterReferenceWorkError(f"卡 {card.get('cardId')} 里程碑[{index}] 非法")
raw_chapter = next(
(raw[key] for key in ("chapter", "chapter_no", "order_no", "章", "章号") if key in raw),
None,
)
bounds = normalize_chapter_range(raw_chapter)
if bounds is None or bounds[1] > as_of:
raise WriterReferenceWorkError(f"卡 {card.get('cardId')} 含未冻结或无边界里程碑")
item = copy.deepcopy(dict(raw))
for alias in ("chapter_no", "order_no", "章", "章号"):
item.pop(alias, None)
item["chapter"] = bounds[1]
normalized.append(item)
normalized.sort(key=lambda item: (item["chapter"], str(item.get("id") or "")))
card["milestones"] = normalized
card["stateAsOf"] = copy.deepcopy(normalized)
def index_unique_canonical_blocks(
block_rows: Sequence[Mapping[str, Any]],
*,
allowed_chapters: set[int],
) -> dict[int, dict[str, Any]]:
"""校验每个请求章恰好一个 Canonical block,并拒绝额外未来章。"""
if not allowed_chapters:
raise WriterReferenceWorkError("Canonical block 请求章集合不能为空")
grouped: dict[int, list[dict[str, Any]]] = {}
for index, raw in enumerate(block_rows):
if not isinstance(raw, Mapping):
raise WriterReferenceWorkError(f"block_rows[{index}] 不是对象")
chapter = _positive_chapter(raw.get("chapter"), f"block_rows[{index}].chapter")
if chapter not in allowed_chapters:
raise WriterReferenceWorkError(f"读取到未请求或目标/未来章正文: {chapter}")
if str(raw.get("chapter_status") or "") not in CANONICAL_CHAPTER_STATUSES:
raise WriterReferenceWorkError(f"第 {chapter} 章不是 Canonical 状态")
text = raw.get("content_text")
if not isinstance(text, str) or not text:
raise WriterReferenceWorkError(f"第 {chapter} 章 Canonical block 正文为空")
grouped.setdefault(chapter, []).append(copy.deepcopy(dict(raw)))
missing = sorted(allowed_chapters - set(grouped))
duplicates = sorted(chapter for chapter, rows in grouped.items() if len(rows) != 1)
if missing or duplicates:
raise WriterReferenceWorkError(
f"Canonical block 必须逐章唯一: missing={missing}, non_unique={duplicates}"
)
return {chapter: rows[0] for chapter, rows in grouped.items()}
def select_recent_canonical_blocks(
block_rows: Sequence[Mapping[str, Any]],
*,
as_of: int,
) -> list[dict[str, Any]]:
"""按冻结点选择连续四章,缺章或重复 block 均失败关闭。"""
freeze = _positive_chapter(as_of, "as_of")
expected = list(range(max(1, freeze - 3), freeze + 1))
relevant = [row for row in block_rows if row.get("chapter") in expected]
indexed = index_unique_canonical_blocks(relevant, allowed_chapters=set(expected))
return [indexed[chapter] for chapter in expected]
def _block_source_ref(row: Mapping[str, Any]) -> dict[str, Any]:
"""为唯一 Canonical block 生成完整代码点区间来源引用。"""
chapter = _positive_chapter(row.get("chapter"), "block.chapter")
block_id = row.get("block_id")
if isinstance(block_id, bool) or not isinstance(block_id, int) or block_id <= 0:
raise WriterReferenceWorkError(f"第 {chapter} 章 block_id 非法")
text = str(row.get("content_text") or "")
revision = row.get("revision") or 0
source_version = f"chapter:{chapter}:block:{block_id}:revision:{revision}"
return {
"sourceId": f"chapter:{chapter}:block:{block_id}",
"sourceVersion": source_version,
"chapter": chapter,
"blockId": block_id,
"startCodePoint": 0,
"endCodePoint": len(text),
}
def _oracle_authorization(
authorization: Mapping[str, Any], *, source_version: str
) -> dict[str, Any]:
"""提取 evaluator-only 授权投影并严格绑定来源版本与重验时间。"""
snapshot = authorization.get("authorizationSnapshot")
if not isinstance(snapshot, Mapping):
raise WriterReferenceWorkError("oracleTruthPack 缺少不可变授权快照")
if authorization.get("allowedPurpose") != ["offline_evaluation"]:
raise WriterReferenceWorkError("oracleTruthPack 只允许 offline_evaluation")
if snapshot.get("allowedPurpose") != ["offline_evaluation"]:
raise WriterReferenceWorkError("oracleTruthPack 授权快照用途不唯一")
if str(snapshot.get("sourceVersion") or "") != source_version:
raise WriterReferenceWorkError("oracleTruthPack 授权来源版本不匹配")
source_status = str(snapshot.get("sourceStatus") or "")
if source_status not in {"active", "authorized", "frozen_authorized"}:
raise WriterReferenceWorkError("oracleTruthPack 授权来源状态不可用")
revalidation_at = str(snapshot.get("revalidationAt") or "")
if not revalidation_at:
raise WriterReferenceWorkError("oracleTruthPack 授权缺少重验时间")
snapshot_id = str(snapshot.get("id") or "")
if not snapshot_id:
raise WriterReferenceWorkError("oracleTruthPack 授权快照 ID 为空")
return {
"allowedPurpose": "offline_evaluation",
"sourceStatus": source_status,
"sourceVersion": source_version,
"revalidationAt": revalidation_at,
"evaluatorOnly": True,
"snapshotId": snapshot_id,
}
def _historical_oracle_assertions(
oracle_block_rows: Sequence[Mapping[str, Any]], *, as_of: int
) -> list[dict[str, Any]]:
"""把冻结线内每章 Canonical 正文投影为稳定、可追溯的历史断言。"""
expected = set(range(1, as_of + 1))
relevant = [row for row in oracle_block_rows if row.get("chapter") in expected]
try:
indexed = index_unique_canonical_blocks(relevant, allowed_chapters=expected)
except WriterReferenceWorkError as error:
raise WriterReferenceWorkError(f"oracle Canonical 历史缺章或不唯一: {error}") from error
assertions: list[dict[str, Any]] = []
for chapter in range(1, as_of + 1):
row = indexed[chapter]
statement = normalize_text(str(row["content_text"]))
source_ref = _block_source_ref(row)
assertions.append(
{
"assertionId": f"historical:chapter:{chapter}:block:{row['block_id']}",
"assertionType": "canonical_history",
"statement": statement,
"sourceVersion": source_ref["sourceVersion"],
"chapterBoundary": {"minChapter": chapter, "maxChapter": chapter},
"contentSha256": "sha256:"
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
}
)
return assertions
def _target_oracle_assertions(
scaffold: Mapping[str, Any], *, target_chapter: int, source_version: str
) -> list[dict[str, Any]]:
"""只从预注册目标 scaffold 拆出目标断言,不读取目标章正文。"""
target_facts = _target_facts_from_scaffold(scaffold, target_chapter)
assertions: list[dict[str, Any]] = []
for raw in target_facts["forbiddenFacts"]:
statement = normalize_text(str(raw["text"]))
assertions.append(
{
"assertionId": str(raw["id"]),
"assertionType": "target_reference_scaffold",
"statement": statement,
"sourceVersion": source_version,
"chapterBoundary": {
"minChapter": target_chapter,
"maxChapter": target_chapter,
},
"contentSha256": "sha256:"
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
}
)
return assertions
def _validate_oracle_assertion(
assertion: Mapping[str, Any], *, expected_type: str, as_of: int, target_chapter: int
) -> None:
"""严格校验 oracle 断言字段、内容哈希和章号边界。"""
required = {
"assertionId",
"assertionType",
"statement",
"sourceVersion",
"chapterBoundary",
"contentSha256",
}
if not isinstance(assertion, Mapping) or set(assertion) != required:
raise WriterReferenceWorkError("oracleTruthPack assertion 字段不严格")
if assertion["assertionType"] != expected_type:
raise WriterReferenceWorkError("oracleTruthPack assertionType 非法")
boundary = assertion["chapterBoundary"]
if not isinstance(boundary, Mapping) or set(boundary) != {"minChapter", "maxChapter"}:
raise WriterReferenceWorkError("oracleTruthPack chapterBoundary 非法")
minimum = _positive_chapter(boundary["minChapter"], "oracle.chapterBoundary.minChapter")
maximum = _positive_chapter(boundary["maxChapter"], "oracle.chapterBoundary.maxChapter")
if minimum != maximum:
raise WriterReferenceWorkError("oracleTruthPack 断言必须绑定唯一章")
if expected_type == "canonical_history" and maximum > as_of:
raise WriterReferenceWorkError("oracleTruthPack 历史断言越过冻结线")
if expected_type == "target_reference_scaffold" and maximum != target_chapter:
raise WriterReferenceWorkError("oracleTruthPack 目标断言未绑定目标章")
statement = str(assertion["statement"])
expected_hash = "sha256:" + hashlib.sha256(statement.encode("utf-8")).hexdigest()
if assertion["contentSha256"] != expected_hash:
raise WriterReferenceWorkError("oracleTruthPack 断言内容哈希不一致")
def build_oracle_truth_pack(
*,
evaluation_set_version: str,
sample_id: str,
work_id: int,
as_of: int,
target_chapter: int,
source_version: str,
authorization: Mapping[str, Any],
oracle_block_rows: Sequence[Mapping[str, Any]],
scaffold: Mapping[str, Any],
) -> dict[str, Any]:
"""构建严格 oracle-truth-pack-v1,仅返回临时评测配置所需对象。"""
auth = _oracle_authorization(authorization, source_version=source_version)
historical = _historical_oracle_assertions(oracle_block_rows, as_of=as_of)
targets = _target_oracle_assertions(
scaffold,
target_chapter=target_chapter,
source_version=source_version,
)
source_snapshot = {
"sourceVersion": source_version,
"asOf": as_of,
"historicalAssertionHashes": [item["contentSha256"] for item in historical],
"targetAssertionHashes": [item["contentSha256"] for item in targets],
}
payload = {
"schemaVersion": "oracle-truth-pack-v1",
"evaluationSetVersion": str(evaluation_set_version),
"sampleId": str(sample_id),
"workId": work_id,
"asOf": as_of,
"sourceSnapshotSha256": _sha256_value(source_snapshot),
"authorizationSnapshotId": auth["snapshotId"],
"authorization": auth,
"historicalAssertions": historical,
"targetAssertions": targets,
}
if len({item["assertionId"] for item in [*historical, *targets]}) != len(
[*historical, *targets]
):
raise WriterReferenceWorkError("oracleTruthPack assertionId 不唯一")
for assertion in historical:
_validate_oracle_assertion(
assertion,
expected_type="canonical_history",
as_of=as_of,
target_chapter=target_chapter,
)
for assertion in targets:
_validate_oracle_assertion(
assertion,
expected_type="target_reference_scaffold",
as_of=as_of,
target_chapter=target_chapter,
)
return validate_oracle_truth_pack(
{**payload, "packSha256": _sha256_value(payload)}
)
def validate_oracle_truth_pack(pack: Mapping[str, Any]) -> dict[str, Any]:
"""严格复核 oracle-truth-pack-v1 的 schema、授权、章界和整体哈希。"""
required = {
"schemaVersion",
"evaluationSetVersion",
"sampleId",
"workId",
"asOf",
"sourceSnapshotSha256",
"authorizationSnapshotId",
"authorization",
"historicalAssertions",
"targetAssertions",
"packSha256",
}
if not isinstance(pack, Mapping) or set(pack) != required:
raise WriterReferenceWorkError("oracleTruthPack 顶层字段不严格")
if pack["schemaVersion"] != "oracle-truth-pack-v1":
raise WriterReferenceWorkError("oracleTruthPack schemaVersion 不支持")
as_of = _positive_chapter(pack["asOf"], "oracleTruthPack.asOf")
target_chapter = as_of + 1
if not str(pack["evaluationSetVersion"]).strip() or not str(pack["sampleId"]).strip():
raise WriterReferenceWorkError("oracleTruthPack 评测集或样本绑定为空")
_positive_chapter(pack["workId"], "oracleTruthPack.workId")
source_snapshot_hash = str(pack["sourceSnapshotSha256"])
if not source_snapshot_hash.startswith("sha256:") or len(source_snapshot_hash) != 71:
raise WriterReferenceWorkError("oracleTruthPack sourceSnapshotSha256 非法")
authorization = pack["authorization"]
if not isinstance(authorization, Mapping) or set(authorization) != {
"allowedPurpose",
"sourceStatus",
"sourceVersion",
"revalidationAt",
"evaluatorOnly",
"snapshotId",
}:
raise WriterReferenceWorkError("oracleTruthPack authorization 字段不严格")
if (
authorization["allowedPurpose"] != "offline_evaluation"
or authorization["evaluatorOnly"] is not True
or authorization["snapshotId"] != pack["authorizationSnapshotId"]
or not str(authorization["revalidationAt"]).strip()
):
raise WriterReferenceWorkError("oracleTruthPack evaluator-only 授权非法")
historical = pack["historicalAssertions"]
targets = pack["targetAssertions"]
if not isinstance(historical, list) or not isinstance(targets, list) or not targets:
raise WriterReferenceWorkError("oracleTruthPack 断言数组非法")
for assertion in historical:
_validate_oracle_assertion(
assertion,
expected_type="canonical_history",
as_of=as_of,
target_chapter=target_chapter,
)
for assertion in targets:
_validate_oracle_assertion(
assertion,
expected_type="target_reference_scaffold",
as_of=as_of,
target_chapter=target_chapter,
)
if {
assertion["chapterBoundary"]["maxChapter"] for assertion in historical
} != set(range(1, as_of + 1)):
raise WriterReferenceWorkError("oracleTruthPack Canonical 历史断言缺章")
all_assertions = [*historical, *targets]
if len({assertion["assertionId"] for assertion in all_assertions}) != len(all_assertions):
raise WriterReferenceWorkError("oracleTruthPack assertionId 不唯一")
if any(
assertion["sourceVersion"] != authorization["sourceVersion"]
and assertion["assertionType"] == "target_reference_scaffold"
for assertion in all_assertions
):
raise WriterReferenceWorkError("oracleTruthPack 目标断言来源版本不匹配")
payload = {key: copy.deepcopy(value) for key, value in pack.items() if key != "packSha256"}
if pack["packSha256"] != _sha256_value(payload):
raise WriterReferenceWorkError("oracleTruthPack packSha256 不一致")
return copy.deepcopy(dict(pack))
class SnapshotProseRepository:
"""只从同一数据库事务已冻结的 block 行展开来源引用。"""
def __init__(self, block_index: Mapping[int, Mapping[str, Any]]):
self._by_chapter = {
int(chapter): copy.deepcopy(dict(row)) for chapter, row in block_index.items()
}
self._by_block = {int(row["block_id"]): row for row in self._by_chapter.values()}
def read_source_refs(
self,
*,
work_id: int,
as_of: int,
source_refs: Sequence[Mapping[str, Any]],
) -> list[dict[str, Any]]:
"""逐引用校验章、块与代码点区间,不建立第二个数据库连接。"""
if work_id != 8:
raise WriterReferenceWorkError("Writer Gate A 只允许预注册 work=8")
freeze = _positive_chapter(as_of, "as_of")
result: list[dict[str, Any]] = []
for index, ref in enumerate(source_refs):
if not isinstance(ref, Mapping):
raise WriterReferenceWorkError(f"source_refs[{index}] 不是对象")
chapter = _positive_chapter(ref.get("chapter"), f"source_refs[{index}].chapter")
if chapter > freeze:
raise WriterReferenceWorkError(f"source_refs[{index}] 包含目标章或未来章")
block_id = ref.get("blockId")
row = self._by_block.get(block_id) if isinstance(block_id, int) else None
if row is None or int(row["chapter"]) != chapter:
raise WriterReferenceWorkError(f"source_refs[{index}] 不能定位唯一 Canonical block")
text = str(row["content_text"])
start = ref.get("startCodePoint")
end = ref.get("endCodePoint")
if (
isinstance(start, bool)
or isinstance(end, bool)
or not isinstance(start, int)
or not isinstance(end, int)
or start < 0
or end <= start
or end > len(text)
):
raise WriterReferenceWorkError(f"source_refs[{index}] 代码点区间越界")
fragment = text[start:end]
result.append(
{
"chapter": chapter,
"blockId": block_id,
"blockOrder": int(row.get("block_order") or 0),
"sourceRef": copy.deepcopy(dict(ref)),
"text": fragment,
"contentSha256": "sha256:"
+ hashlib.sha256(fragment.encode("utf-8")).hexdigest(),
"purpose": str(ref.get("sourceType") or "card_source"),
}
)
return result
def _target_scaffold_index(
rows: Sequence[Mapping[str, Any]], targets: set[int]
) -> dict[int, dict[str, Any]]:
"""目标 scaffold 也要求逐章唯一,禁止 ORDER BY/LIMIT 猜选。"""
grouped: dict[int, list[dict[str, Any]]] = {}
for index, raw in enumerate(rows):
if not isinstance(raw, Mapping):
raise WriterReferenceWorkError(f"target_scaffolds[{index}] 不是对象")
chapter = _positive_chapter(raw.get("chapter"), f"target_scaffolds[{index}].chapter")
if chapter not in targets:
raise WriterReferenceWorkError(f"读取到未预注册目标 scaffold: {chapter}")
grouped.setdefault(chapter, []).append(copy.deepcopy(dict(raw)))
missing = sorted(targets - set(grouped))
duplicates = sorted(chapter for chapter, values in grouped.items() if len(values) != 1)
if missing or duplicates:
raise WriterReferenceWorkError(
f"目标 scaffold 必须逐章唯一: missing={missing}, non_unique={duplicates}"
)
return {chapter: values[0] for chapter, values in grouped.items()}
def _requirement_hard_constraints(requirements: Any) -> list[str]:
"""把门禁会机械检查的要求清单 surface 为写手可见的硬约束字符串。
WHY:正文 Gate A 的机械门(check_writer_candidate)用这份 requirements 卡候选——
章末钩子锚点、伏笔动作锚点、硬事件锚点、必须出场角色。写手若看不到要查什么,
就可能漏写(如章末钩子锚点不在大纲文字里),机械门必挂 CHAPTER_END_HOOK_MISSING。
因此 loader 在装配写手细纲时,把这份清单转成硬约束并进 hardConstraints,让写手
知道门禁会查什么。本函数只读 requirements、不改写它。
确定性:追加顺序固定为 章末钩子→伏笔动作→硬事件→必须出场角色,同样输入产出
同样字符串。这些条目对 A/B/C 三臂逐样本完全相同(共享同一份 requirements),
不引入 A/C 单变量差异,因此不会撞差异回执冻结合同。
"""
if not isinstance(requirements, Mapping):
return []
constraints: list[str] = []
def _join_anchors(anchors: Any) -> str:
# 锚点必须是列表;列表内再逐项做类型保护(非字符串/空串跳过),保证拼接稳定。
if not isinstance(anchors, list):
return ""
return "、".join(anchor for anchor in anchors if isinstance(anchor, str) and anchor)
# 1. 章末钩子:门禁查正文结尾 maxDistanceFromEnd 字内是否出现锚点之一。
hook = requirements.get("chapterEndHook")
if isinstance(hook, Mapping):
anchors_text = _join_anchors(hook.get("anchors"))
max_distance = hook.get("maxDistanceFromEnd")
if (
anchors_text
and not isinstance(max_distance, bool)
and isinstance(max_distance, int)
and max_distance > 0
):
constraints.append(
f"章末钩子(硬要求):正文结尾 {max_distance} 字内必须出现以下锚点之一:{anchors_text}"
)
# 2. 伏笔动作:门禁查正文是否自然埋入锚点之一。
for item in requirements.get("foreshadowingActions") or []:
if not isinstance(item, Mapping):
continue
anchors_text = _join_anchors(item.get("anchors"))
if anchors_text:
constraints.append(
f"伏笔动作(硬要求):正文必须自然埋入以下锚点之一:{anchors_text}"
)
# 3. 硬事件:门禁查正文是否命中锚点之一。
for item in requirements.get("requiredEvents") or []:
if not isinstance(item, Mapping):
continue
anchors_text = _join_anchors(item.get("anchors"))
if anchors_text:
constraints.append(f"硬事件(硬要求):正文必须命中以下锚点之一:{anchors_text}")
# 4. 必须出场角色:门禁查正文是否出现角色名。
characters = [
character
for character in (requirements.get("requiredCharacters") or [])
if isinstance(character, str) and character
]
if characters:
constraints.append(
f"必须出场角色(硬要求):正文必须出现以下角色:{'、'.join(characters)}"
)
return constraints
def _selected_card_entities(cards: Sequence[Mapping[str, Any]]) -> list[dict[str, str]]:
"""把稳定选择器结果转换为检索计划实体,不从运行结果追加查询。"""
return [
{
"id": f"selected-card:{card['cardId']}",
"type": str(card["type"]),
"name": str(card["name"]),
}
for card in cards
]
def _context_token_budget(
configured: Mapping[str, Any],
*,
preregistered_max: int,
) -> dict[str, int]:
"""原样使用预注册上限;基线超限由组装器失败,补充证据由组装器裁剪。"""
configured_max = configured.get("maxContextChars")
if (
isinstance(configured_max, bool)
or not isinstance(configured_max, int)
or configured_max <= 0
):
raise WriterReferenceWorkError("tokenBudget.maxContextChars 必须是正整数")
if configured_max != preregistered_max:
raise WriterReferenceWorkError("loader 禁止改写或扩张预注册 maxContextChars")
return {"maxContextChars": preregistered_max}
def _context_source_status(authorization: Mapping[str, Any]) -> str:
"""按既有 WriterContext 授权绑定规则投影运行期来源状态。"""
source_status = str(authorization.get("sourceStatus") or "").lower()
if source_status in {"active", "approved", "licensed"}:
return "active"
if source_status == "authorized":
return "authorized"
raise WriterReferenceWorkError(f"授权来源状态不能进入 WriterContext: {source_status}")
def _project_card_for_sample(
row: Mapping[str, Any],
*,
as_of: int,
source_version: str,
block_index: Mapping[int, Mapping[str, Any]],
source_aliases: Sequence[str] = (),
) -> dict[str, Any]:
"""冻结卡;缺精确引用时只生成可识别且显式降级的整章代理。"""
card_id = str(row.get("id") or "")
card_version = f"{source_version}:card-{card_id}-rev-{row.get('revision') or 0}"
projected = project_card(row, as_of=as_of, source_version=card_version)
milestones = copy.deepcopy(projected.get("milestones") or [])
uses_chapter_proxy = not projected.get("sourceRefs")
if uses_chapter_proxy:
chapters = milestone_reference_chapters(milestones, as_of=as_of)
try:
projected["sourceRefs"] = []
payload = _card_payload(row)
raw_aliases = payload.get("别名") or []
if not isinstance(raw_aliases, list) or any(
not isinstance(alias, str) for alias in raw_aliases
):
raise WriterReferenceWorkError(f"卡 {card_id} 的别名不是字符串数组")
legal_names = {
str(projected.get("name") or "").strip(),
*(alias.strip() for alias in raw_aliases),
*(str(alias).strip() for alias in source_aliases),
}
legal_names.discard("")
for chapter in chapters:
row_text = normalize_text(str(block_index[chapter]["content_text"]))
if not any(name in row_text for name in legal_names):
raise WriterReferenceWorkError(
f"卡 {card_id} 的整章代理第 {chapter} 章未出现规范名或合法别名"
)
ref = _block_source_ref(block_index[chapter])
ref["sourceType"] = "card_chapter_proxy"
projected["sourceRefs"].append(ref)
except KeyError as error:
raise WriterReferenceWorkError(
f"卡 {card_id} 的里程碑章 {error.args[0]} 缺少唯一 Canonical block"
) from error
for index, ref in enumerate(projected.get("sourceRefs") or []):
chapter = _positive_chapter(ref.get("chapter"), f"卡 {card_id}.sourceRefs[{index}].chapter")
if chapter > as_of:
raise WriterReferenceWorkError(f"卡 {card_id} sourceRef 包含目标章或未来章")
if ref.get("blockId") not in {row["block_id"] for row in block_index.values()}:
raise WriterReferenceWorkError(f"卡 {card_id} sourceRef 未绑定本次 Canonical 快照")
if uses_chapter_proxy and ref.get("sourceType") != "card_chapter_proxy":
raise WriterReferenceWorkError(f"卡 {card_id} 整章代理缺少降级来源类型")
_normalize_card_milestones(projected, as_of=as_of)
return projected
def _sample_sources(
recent_rows: Sequence[Mapping[str, Any]],
cards: Sequence[Mapping[str, Any]],
*,
as_of: int,
) -> list[dict[str, Any]]:
"""生成只含冻结历史的可验证来源目录,不登记目标 scaffold 为历史来源。"""
sources = [_block_source_ref(row) for row in recent_rows]
for card in cards:
sources.append(
{
"sourceId": str(card["sourceId"]),
"sourceVersion": str(card["sourceVersion"]),
"chapterRange": f"1-{as_of}",
"scope": "card_projection",
}
)
return sources
def _recent_chapter_input(rows: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
"""把连续四章唯一 block 投影成 WriterContext 的完整历史基线。"""
return [
{
"chapter": int(row["chapter"]),
"sourceRef": _block_source_ref(row),
"text": normalize_text(str(row["content_text"])),
}
for row in rows
]
def _generic_historical_prose(
block_index: Mapping[int, Mapping[str, Any]],
*,
as_of: int,
work_id: int,
char_budget: int,
) -> list[dict[str, Any]]:
"""不使用卡,按冻结线向前选择通用 Canonical 历史原文候选。"""
if isinstance(char_budget, bool) or not isinstance(char_budget, int) or char_budget < 0:
raise WriterReferenceWorkError("proseCharBudget 必须是非负整数")
if char_budget == 0:
return []
baseline = set(range(max(1, as_of - 3), as_of + 1))
selected_refs: list[dict[str, Any]] = []
available_chars = 0
# 最近历史优先是通用时间邻近策略,不读取卡、里程碑或卡派生实体。
for chapter in range(as_of, 0, -1):
if chapter in baseline:
continue
row = block_index.get(chapter)
if row is None:
raise WriterReferenceWorkError(f"A 臂通用历史检索缺少第 {chapter} 章")
selected_refs.append(_block_source_ref(row))
available_chars += len(str(row["content_text"]))
if available_chars >= char_budget:
break
if available_chars < char_budget:
raise WriterReferenceWorkError(
f"A 臂通用历史原文不足: expected={char_budget}, actual={available_chars}"
)
repository = SnapshotProseRepository(block_index)
prose = repository.read_source_refs(
work_id=work_id,
as_of=as_of,
source_refs=selected_refs,
)
for item in prose:
item["purpose"] = "generic_historical_prose"
item["retrievalArm"] = "A"
return prose
def _required_block_chapters(
base_config: Mapping[str, Any],
selected: Mapping[str, Sequence[Mapping[str, Any]]],
) -> set[int]:
"""在正文查询前计算固定章集合,保证 SQL 不会读取目标章。"""
required: set[int] = set()
for sample in base_config.get("samples", []):
sample_id = str(sample.get("sampleId") or "")
target = _positive_chapter(sample.get("targetChapter"), f"{sample_id}.targetChapter")
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
if target != as_of + 1:
raise WriterReferenceWorkError(f"{sample_id} targetChapter 必须等于 asOfChapter+1")
required.update(range(max(1, as_of - 3), as_of + 1))
for row in selected.get(sample_id, []):
projected = project_card(
row,
as_of=as_of,
source_version=f"prequery:card-{row.get('id')}",
)
refs = projected.get("sourceRefs") or []
if refs:
for index, ref in enumerate(refs):
chapter = _positive_chapter(
ref.get("chapter"), f"{sample_id}.sourceRefs[{index}].chapter"
)
if chapter > as_of:
raise WriterReferenceWorkError(f"{sample_id} 卡引用包含目标章或未来章")
required.add(chapter)
else:
required.update(
milestone_reference_chapters(projected.get("milestones") or [], as_of=as_of)
)
return required
def load_writer_reference_rows(
*,
dsn: str,
tenant_id: int,
work_id: int,
targets: Sequence[int],
base_config: Mapping[str, Any],
selector_config: Mapping[str, Any],
selector_digest: str,
) -> dict[str, Any]:
"""在同一只读可重复读事务读取五章装配所需全部数据。"""
_validate_loader_controls(
base_config,
selector_config,
selector_digest=selector_digest,
)
character_probes = _required_character_probes(base_config)
normalized_targets = sorted({_positive_chapter(item, "targets[]") for item in targets})
if work_id != 8 or int(selector_config.get("workId") or 0) != work_id:
raise WriterReferenceWorkError("Writer Gate A 只允许预注册 work=8")
if not normalized_targets:
raise WriterReferenceWorkError("目标章集合不能为空")
selector_targets = sorted(
_positive_chapter(item.get("targetChapter"), f"{item['sampleId']}.targetChapter")
for item in _selector_samples(selector_config)
)
if selector_targets != normalized_targets:
raise WriterReferenceWorkError("调用目标章必须与稳定选择器完全一致")
selector_names = sorted(
{
str(card["name"])
for sample in _selector_samples(selector_config)
for card in sample["cards"]
}
)
selector_types = sorted(
{
str(card["type"])
for sample in _selector_samples(selector_config)
for card in sample["cards"]
}
)
with psycopg.connect(dsn, row_factory=dict_row) as conn:
begin_read_snapshot(conn)
work = conn.execute(
"""
SELECT id,title,revision,chapter_count,parse_status,import_status
FROM muse_content_work
WHERE tenant_id=%s AND id=%s AND deleted=FALSE
""",
(tenant_id, work_id),
).fetchone()
reference_rows = conn.execute(
"""
SELECT id,work_id,declared_chapter_count,imported_chapter_count,
parse_scope,parse_status,source_file,notes,update_time,deleted
FROM example_reference_work
WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE
ORDER BY id
""",
(tenant_id, work_id),
).fetchall()
if work is None or len(reference_rows) != 1:
raise WriterReferenceWorkError("作品或唯一参考作品登记不存在")
reference = reference_rows[0]
import_task_rows = conn.execute(
"""
SELECT id,status,command_id,source_snapshot,deleted
FROM muse_content_import_task
WHERE tenant_id=%s AND work_id=%s AND status='succeeded' AND deleted=FALSE
AND source_snapshot->>'file'=%s
ORDER BY id
""",
(tenant_id, work_id, reference.get("source_file")),
).fetchall()
document_rows = conn.execute(
"""
SELECT id,file_name,file_hash,deleted
FROM muse_knowledge_document
WHERE tenant_id=%s AND file_name=%s AND deleted=FALSE
ORDER BY id
""",
(tenant_id, reference.get("source_file")),
).fetchall()
source = validate_source_records(reference_rows, import_task_rows, document_rows)
authorization_row = conn.execute(
"""
SELECT id,snapshot_version,source_hash,source_version,copyright_status,source_status,
allowed_purpose,forbidden_purpose,authorization_basis,authorized_by,
display_summary,checked_at,expires_at,revalidation_at
FROM example_reference_authorization_snapshot
WHERE tenant_id=%s AND work_id=%s AND source_version=%s
ORDER BY checked_at DESC,id DESC
LIMIT 1
""",
(tenant_id, work_id, source["sourceVersion"]),
).fetchone()
authorization = project_authorization(authorization_row, source)
target_scaffolds = conn.execute(
"""
SELECT s.id,s.chapter_id,ch.order_no AS chapter,ch.title,s.outline_text,
s.entities,s.pattern_hints
FROM example_parse_scaffold s
JOIN muse_content_chapter ch ON ch.id=s.chapter_id
WHERE s.tenant_id=%s AND s.work_id=%s AND s.deleted=FALSE
AND ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
AND ch.order_no=ANY(%s)
ORDER BY ch.order_no,s.id
""",
(tenant_id, work_id, tenant_id, work_id, normalized_targets),
).fetchall()
card_rows = conn.execute(
"""
SELECT id,status,source_type,source_id,revision,draft_payload,deleted
FROM muse_knowledge_draft
WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE
AND source_type='upgrade_book'
AND draft_payload->>'type'=ANY(%s)
AND (
draft_payload->>'名称'=ANY(%s)
OR COALESCE(draft_payload->'别名','[]'::jsonb) ?| %s
)
ORDER BY id
""",
(tenant_id, work_id, selector_types, selector_names, selector_names),
).fetchall()
selected = resolve_card_selectors(card_rows, selector_config)
character_mentions = conn.execute(
"""
WITH probes AS (
SELECT *
FROM unnest(%s::text[],%s::text[],%s::integer[])
AS probe(sample_id,name,as_of_chapter)
)
SELECT probe.sample_id,probe.name,probe.as_of_chapter,
MIN(ch.order_no) FILTER (WHERE b.id IS NOT NULL) AS first_chapter,
COALESCE(
ARRAY_AGG(DISTINCT ch.order_no ORDER BY ch.order_no)
FILTER (WHERE b.id IS NOT NULL),
ARRAY[]::integer[]
) AS hit_chapters
FROM probes probe
LEFT JOIN muse_content_chapter ch
ON ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
AND ch.status IN ('published','confirmed','canonical')
AND ch.order_no<=probe.as_of_chapter
LEFT JOIN muse_content_block b
ON b.chapter_id=ch.id AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
AND POSITION(probe.name IN b.content_text)>0
GROUP BY probe.sample_id,probe.name,probe.as_of_chapter
ORDER BY probe.sample_id,probe.name
""",
(
[str(item["sampleId"]) for item in character_probes],
[str(item["name"]) for item in character_probes],
[int(item["asOfChapter"]) for item in character_probes],
tenant_id,
work_id,
tenant_id,
work_id,
),
).fetchall()
required_chapters: set[int] = set()
for target in normalized_targets:
as_of = target - 1
required_chapters.update(range(max(1, as_of - 3), as_of + 1))
target_by_sample = {
str(item["sampleId"]): _positive_chapter(
item.get("targetChapter"), f"{item['sampleId']}.targetChapter"
)
for item in selector_config["samples"]
}
for sample_id, rows in selected.items():
as_of = target_by_sample[sample_id] - 1
for row in rows:
projected = project_card(
row,
as_of=as_of,
source_version=f"prequery:card-{row.get('id')}",
)
refs = projected.get("sourceRefs") or []
if refs:
required_chapters.update(
_positive_chapter(ref.get("chapter"), "card.sourceRef.chapter")
for ref in refs
)
else:
required_chapters.update(
milestone_reference_chapters(projected.get("milestones") or [], as_of=as_of)
)
if any(chapter in normalized_targets for chapter in required_chapters):
raise WriterReferenceWorkError("正文读取集合包含目标章")
block_rows = conn.execute(
"""
SELECT ch.order_no AS chapter,ch.id AS chapter_id,ch.status AS chapter_status,
b.id AS block_id,b.order_no AS block_order,b.revision,b.content_text
FROM muse_content_chapter ch
JOIN muse_content_block b ON b.chapter_id=ch.id
WHERE ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
AND ch.status IN ('published','confirmed','canonical')
AND ch.order_no=ANY(%s)
ORDER BY ch.order_no,b.order_no,b.id
""",
(tenant_id, work_id, tenant_id, work_id, sorted(required_chapters)),
).fetchall()
# oracle 必须覆盖每个样本冻结线内的全部 Canonical 历史,因此单独读取
# 1..max(asOf),但仍复用当前连接和同一个只读冻结事务。
oracle_block_rows = conn.execute(
"""
SELECT ch.order_no AS chapter,ch.id AS chapter_id,ch.status AS chapter_status,
b.id AS block_id,b.order_no AS block_order,b.revision,b.content_text
FROM muse_content_chapter ch
JOIN muse_content_block b ON b.chapter_id=ch.id
WHERE ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
AND ch.status IN ('published','confirmed','canonical')
AND ch.order_no<=%s
ORDER BY ch.order_no,b.order_no,b.id
""",
(tenant_id, work_id, tenant_id, work_id, max(normalized_targets) - 1),
).fetchall()
index_unique_canonical_blocks(block_rows, allowed_chapters=required_chapters)
try:
index_unique_canonical_blocks(
oracle_block_rows,
allowed_chapters=set(range(1, max(normalized_targets))),
)
except WriterReferenceWorkError as error:
raise WriterReferenceWorkError(f"oracle Canonical 历史缺章或不唯一: {error}") from error
_target_scaffold_index(target_scaffolds, set(normalized_targets))
_canonical_character_index(character_mentions, character_probes)
return {
"work": work,
"reference": reference,
"source": source,
"authorization": authorization,
"target_scaffolds": target_scaffolds,
"card_rows": [row for rows in selected.values() for row in rows],
"block_rows": block_rows,
"oracle_block_rows": oracle_block_rows,
"canonical_character_mentions": character_mentions,
}
def _default_pattern_card_searcher(
*, dsn: str, tenant_id: int
) -> Callable[..., list[dict[str, Any]]]:
"""惰性导入公共范式库检索器,返回签名 ``(intent, *, ttype, top)`` 的调用体。
WHY 惰性:离线测试与 dry-run 不应被迫加载数据库/嵌入依赖,也不能在装配时
真连库;只有生产入口 ``main`` 才显式取用本函数,把真实检索接入 C 臂。
"""
search_scripts = SCRIPT_DIR.parents[1] / "search-knowledge" / "scripts"
sys.path.insert(0, str(search_scripts))
from search import search_cards # noqa: E402 惰性导入,避免模块级副作用
def _searcher(intent: str, *, ttype: str, top: int) -> list[dict[str, Any]]:
# 公共范式还在 draft 双轨,但必须走专用检索面;dsn/tenant 显式绑定本次 loader,
# 防止真实正文来自一套快照、范式却被默认常量带到另一库或另一租户。
return search_cards(
intent,
scope="public_pattern",
ttype=ttype,
purpose="generation",
top=top,
dsn=dsn,
tenant_id=tenant_id,
)
return _searcher
def _flatten_pattern_point(value: Any) -> str:
"""把范式卡字段值拍平成文本。WHY:writingPoints 合同是「字符串→字符串」,而
search_cards 的 visibleFields 值可能是列表/对象,统一拍平后才能过合同。"""
if isinstance(value, str):
return value
return json.dumps(value, ensure_ascii=False, sort_keys=True)
def _truncate_for_writer(text: str, max_chars: int) -> str:
"""按 code point 截断到上限以内,超长补一个省略号并重新 NFC 归一化。
WHY:截断可能落在组合字符边界、导致结果不再是 NFC,而合同 _string 会复核
value == NFC(value);因此截断后必须再归一化一次,保证产出永远过得了合同。
"""
if len(text) <= max_chars:
return text
return normalize_text(text[: max(0, max_chars - 1)] + "…")
def _pattern_content_projection(card: Mapping[str, Any]) -> dict[str, Any]:
"""把 search_cards 的 name/summary/visibleFields 投影为合同允许的限量内容字段。
WHY(SoT 变更):写手要真正读到范式卡的名字、一句话摘要和写法要点,而不只是一个
来源标签;但 visibleFields 原始字段可能长达数千字,直接灌入会撑爆写手上下文预算,
因此逐字段截断、只取前若干个字段。上限与合同(writer_contract._pattern_source_ref)
共用同一组常量,合同侧再失败关闭复核,双重保证体量受控。
"""
content: dict[str, Any] = {}
name = normalize_text(str(card.get("name") or "")).strip()
if name:
content["name"] = _truncate_for_writer(name, PATTERN_NAME_MAX_CHARS)
summary = normalize_text(str(card.get("summary") or "")).strip()
if summary:
content["summary"] = _truncate_for_writer(summary, PATTERN_SUMMARY_MAX_CHARS)
visible = card.get("visibleFields")
if isinstance(visible, Mapping):
points: dict[str, str] = {}
# visibleFields 来自库内 jsonb,键序确定;按序取前 N 个非空字段作为写法要点。
for key, value in visible.items():
if len(points) >= PATTERN_POINTS_MAX_FIELDS:
break
point_key = normalize_text(str(key)).strip()
point_value = normalize_text(_flatten_pattern_point(value)).strip()
if not point_key or not point_value:
continue
points[point_key] = _truncate_for_writer(point_value, PATTERN_POINT_MAX_CHARS)
if points:
content["writingPoints"] = points
return content
def _retrieve_pattern_references(
intent: str,
*,
card_searcher: Callable[..., list[dict[str, Any]]],
types: Sequence[str] = PATTERN_CARD_TYPES,
top_per_type: int = PATTERN_TOP_PER_TYPE,
total_cap: int = PATTERN_TOTAL_CAP,
) -> list[dict[str, Any]]:
"""按本章检索意图,从公共范式库五型各召回 top-k 卡,投影为写手合同 patternReferences。
WHY:Writer Gate A 的 C 臂要验证「范式指导是否提升质量」,需要把公共范式卡接入
写手输入。每张卡投影成 WriterContext v1 ``patternReferences``:来源指针
(sourceId / sourceVersion / sourceType,保证可回读可审计)**外加内容字段**
(name/summary/writingPoints,保证写手真正读到范式卡的名字、摘要与写法要点)。
内容字段经 ``_pattern_content_projection`` 截断到合同上限以内,确保通过
``validate_writer_context`` 的 ``_pattern_source_ref`` 校验。search_cards 已直接给出
稳定的 ``sourceId``(draft:{id})与 ``sourceVersion``(draft-revision:{n}),正好复用。
总量受控:每型最多 top_per_type 张,且累计不超过 total_cap,避免撑爆上下文预算。
"""
intent_text = normalize_text(str(intent or "")).strip()
if not intent_text:
# 没有检索意图(细纲为空)就不召回,失败关闭而非注入空引用。
return []
references: list[dict[str, Any]] = []
seen: set[tuple[str, str]] = set()
for card_type in types:
if len(references) >= total_cap:
break
# 剩余名额决定本型实际 top,保证累计严格不超过 total_cap。
top = min(top_per_type, total_cap - len(references))
if top <= 0:
break
for card in card_searcher(intent_text, ttype=card_type, top=top):
# WHY: SQL 是第一道范围门,loader 仍只接受专用公共范式面标记为可用于生产
# 检索的行;fake/未来替换实现若漏做范围过滤,也不能把治理草稿注入写手。
if (
card.get("retrievalScope") != "public_pattern"
or card.get("productionRetrievalEligible") is not True
or card.get("sourceKind") != "draft"
):
continue
source_id = normalize_text(str(card.get("sourceId") or "")).strip()
source_version = normalize_text(str(card.get("sourceVersion") or "")).strip()
if not source_id or not source_version:
# 缺稳定来源指针的卡不能进冻结上下文,跳过而非混入空引用。
continue
key = (source_version, source_id)
if key in seen:
# 跨型去重:同一张卡只注入一次。
continue
seen.add(key)
card_kind = normalize_text(str(card.get("type") or card_type)).strip() or card_type
references.append(
{
"sourceId": source_id,
"sourceVersion": source_version,
# sourceType 会成为写手最终看到的 kind;用范式卡的型作标识。
"sourceType": card_kind,
# SoT 变更:内容字段(名字/摘要/写法要点)随来源指针一起注入,写手
# 才能真正读到范式卡;此前只有上面三个指针字段,写手只见一个空标签。
**_pattern_content_projection(card),
}
)
if len(references) >= total_cap:
break
return references
def _pattern_references_for_arm(arm: str, c_references: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
"""装配端分臂的薄封装:语义唯一事实源在 writer_contract.pattern_references_for_arm。
WHY:装配与回放是两段独立 assemble 的链路,必须按完全相同的规则分臂(A 恒空 /
其余臂拿 C 候选),否则 C 臂真写读不到范式卡或 A 臂混入范式卡。判定逻辑一律走
合同模块,不在装配端另写一套;保留这个私有入口只为兼容既有离线测试的导入面。
"""
return pattern_references_for_arm(arm, c_references)
def assemble_writer_gate_config(
*,
base_config: Mapping[str, Any],
selector_config: Mapping[str, Any],
selector_digest: str,
rows: Mapping[str, Any],
pattern_card_searcher: Callable[..., list[dict[str, Any]]] | None = None,
) -> dict[str, Any]:
"""把同一事务快照装配成 canonical_frozen_prose 五样本配置。
``pattern_card_searcher`` 为 None 时不注入范式卡(A/C 两臂 patternReferences 均空),
保持历史行为与离线测试的零数据库依赖;生产入口显式传入真实检索器才启用 C 臂注入。
"""
common_controls = _validate_loader_controls(
base_config,
selector_config,
selector_digest=selector_digest,
)
if not isinstance(base_config, Mapping) or base_config.get("profile") != "writer_replay":
raise WriterReferenceWorkError("基础配置 profile 必须是 writer_replay")
samples = base_config.get("samples")
if not isinstance(samples, list) or not samples:
raise WriterReferenceWorkError("基础配置 samples 不能为空")
sample_ids = [str(item.get("sampleId") or "") for item in samples]
selector_samples = _selector_samples(selector_config)
selector_ids = [str(item["sampleId"]) for item in selector_samples]
if sample_ids != selector_ids:
raise WriterReferenceWorkError("稳定卡选择器样本顺序必须与预注册配置完全一致")
for sample, selector in zip(samples, selector_samples, strict=True):
if sample.get("targetChapter") != selector.get("targetChapter"):
raise WriterReferenceWorkError(
f"{sample.get('sampleId')} targetChapter 未绑定稳定选择器"
)
if selector_config.get("evaluationSetVersion") != base_config.get("evaluationSetVersion"):
raise WriterReferenceWorkError("卡选择器 evaluationSetVersion 未绑定预注册配置")
work_id = int(selector_config.get("workId") or 0)
if work_id != 8 or base_config.get("referenceWork", {}).get("id") != work_id:
raise WriterReferenceWorkError("基础配置与卡选择器必须共同绑定 work=8")
selected = resolve_card_selectors(rows.get("card_rows", []), selector_config)
selector_by_sample = {str(item["sampleId"]): item for item in selector_samples}
targets = {_positive_chapter(item.get("targetChapter"), "sample.targetChapter") for item in samples}
scaffolds = _target_scaffold_index(rows.get("target_scaffolds", []), targets)
required_chapters = _required_block_chapters(base_config, selected)
block_index = index_unique_canonical_blocks(
rows.get("block_rows", []), allowed_chapters=required_chapters
)
source = rows.get("source")
authorization = rows.get("authorization")
work = rows.get("work")
if not all(isinstance(item, Mapping) for item in (source, authorization, work)):
raise WriterReferenceWorkError("作品、来源或授权投影缺失")
source_version = str(source.get("sourceVersion") or "")
if not source_version.startswith("raw-file-v1:sha256:"):
raise WriterReferenceWorkError("真实 Writer 配置必须绑定原文件版本")
config = copy.deepcopy(dict(base_config))
# 盖戳只作用于深拷贝出的输出配置:补 rawRetention 的运行期租约到期时间戳,
# base_config(及其 _validate_loader_controls 已校验的自哈希)不受影响。
_stamp_raw_retention(config)
config["referenceWork"] = {
"id": work_id,
"title": str(work.get("title") or ""),
"version": source_version,
}
config["authorization"] = copy.deepcopy(dict(authorization))
config["evaluationPreregistration"] = build_balanced_preregistration(
evaluation_set_version=str(base_config.get("evaluationSetVersion") or ""),
sample_ids=sample_ids,
)
assembled_samples: list[dict[str, Any]] = []
oracle_packs: dict[str, dict[str, Any]] = {}
diff_receipts: dict[str, dict[str, Any]] = {}
oracle_block_rows = rows.get("oracle_block_rows")
if not isinstance(oracle_block_rows, list):
raise WriterReferenceWorkError("oracle Canonical 全量历史缺失")
max_as_of = max(targets) - 1
try:
oracle_index = index_unique_canonical_blocks(
oracle_block_rows,
allowed_chapters=set(range(1, max_as_of + 1)),
)
except WriterReferenceWorkError as error:
raise WriterReferenceWorkError(f"oracle Canonical 历史缺章或不唯一: {error}") from error
prose_repository = SnapshotProseRepository(block_index)
top_snapshot = authorization.get("authorizationSnapshot")
if not isinstance(top_snapshot, Mapping):
raise WriterReferenceWorkError("授权缺少不可变 authorizationSnapshot")
character_index = _canonical_character_index(
rows.get("canonical_character_mentions", []),
_required_character_probes(base_config),
)
for raw_sample in samples:
sample = copy.deepcopy(dict(raw_sample))
sample_id = str(sample["sampleId"])
target = _positive_chapter(sample.get("targetChapter"), f"{sample_id}.targetChapter")
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
if target != as_of + 1:
raise WriterReferenceWorkError(f"{sample_id} targetChapter 必须等于 asOfChapter+1")
scaffold = scaffolds[target]
outline_text = str(scaffold.get("outline_text") or "").strip()
if not outline_text:
raise WriterReferenceWorkError(f"{sample_id} 目标 scaffold 为空")
oracle_packs[sample_id] = build_oracle_truth_pack(
evaluation_set_version=str(base_config.get("evaluationSetVersion") or ""),
sample_id=sample_id,
work_id=work_id,
as_of=as_of,
target_chapter=target,
source_version=source_version,
authorization=authorization,
oracle_block_rows=oracle_block_rows,
scaffold=scaffold,
)
recent_rows = [block_index[chapter] for chapter in range(max(1, as_of - 3), as_of + 1)]
actual_counts = [han_count(str(row["content_text"])) for row in recent_rows]
expected_counts = sample.get("frozenRecentHanCounts")
if actual_counts != expected_counts:
raise WriterReferenceWorkError(
f"{sample_id} 冻结近章 Han 计数漂移: expected={expected_counts}, actual={actual_counts}"
)
projected_cards = [
_project_card_for_sample(
row,
as_of=as_of,
source_version=source_version,
block_index=block_index,
source_aliases=selector_by_sample[sample_id]["cards"][index].get(
"sourceAliases", []
),
)
for index, row in enumerate(selected[sample_id])
]
fine_outline = {
"sourceRef": {
"sourceId": f"scaffold:{scaffold.get('id')}",
"sourceVersion": source_version,
"chapter": target,
},
# 大纲文字之外,把门禁会查的要求清单 surface 为硬约束,让写手看到门禁查什么。
"hardConstraints": [outline_text]
+ _requirement_hard_constraints(
sample.get("writerContextInput", {}).get("requirements", {})
),
"adjustableBeats": copy.deepcopy(
sample.get("writerContextInput", {}).get("fineOutline", {}).get(
"adjustableBeats", []
)
),
"declaredNewFacts": [],
"entities": _selected_card_entities(projected_cards),
}
token_budget = _context_token_budget(
sample.get("writerContextInput", {}).get("tokenBudget", {}),
preregistered_max=int(common_controls["maxContextChars"]),
)
prose_char_budget = sample.get("proseCharBudget")
if (
isinstance(prose_char_budget, bool)
or not isinstance(prose_char_budget, int)
or prose_char_budget <= 0
):
raise WriterReferenceWorkError(
f"{sample_id}.proseCharBudget 必须是预注册正整数"
)
plan = build_retrieval_plan(
run_id=f"writer-loader:{sample_id}",
work_id=work_id,
target_chapter=target,
as_of=as_of,
fine_outline=fine_outline,
card_index_version=f"upgrade-book:{source_version}",
prose_index_version=source_version,
token_budget=token_budget,
)
sources = _sample_sources(recent_rows, projected_cards, as_of=as_of)
leakage = {
"method": "target-scaffold-proxy-and-chapter-bound-audit",
"targetFacts": _target_facts_from_scaffold(scaffold, target),
}
replay_repository = ReplayCardIndexRepository.from_replay_config(
{
"targetChapter": target,
"snapshot": {
"asOfChapter": as_of,
"snapshotVersion": f"writer-gate-a-canonical-{target}-v1",
"data": {
"chapters": [
{
"chapter": int(row["chapter"]),
"sourceId": _block_source_ref(row)["sourceId"],
"contentSha256": "sha256:"
+ hashlib.sha256(
str(row["content_text"]).encode("utf-8")
).hexdigest(),
}
for row in recent_rows
],
"cards": [],
},
},
"authorization": authorization,
"sources": sources,
"leakageAudit": leakage,
},
cards=projected_cards,
preregistered_card_ids=[str(card["cardId"]) for card in projected_cards],
)
retrieval_result = retrieve_writer_sources(
plan=plan,
card_repository=replay_repository,
prose_repository=prose_repository,
)
# 既有检索器会为带 sourceRefs 的卡附加卡摘要事实。Writer Gate A 明确把卡
# 限定为索引,因此真实配置只保留索引提示和回读原文,不把摘要升级为权威事实。
retrieval_result["factEvidence"] = []
card_prose = []
for item in retrieval_result.get("proseEvidence", []):
projected = copy.deepcopy(dict(item))
projected["retrievalArm"] = "C"
card_prose.append(projected)
generic_prose = _generic_historical_prose(
oracle_index,
as_of=as_of,
work_id=work_id,
char_budget=prose_char_budget,
)
retrieval_result["proseEvidence"] = [*generic_prose, *card_prose]
assembly_token_budget = {
**token_budget,
"proseCharBudget": prose_char_budget,
}
context_input = copy.deepcopy(dict(sample.get("writerContextInput") or {}))
context_input.update(
{
"contentMode": "canonical_frozen_prose",
"sourceVersion": source_version,
"sourceStatus": _context_source_status(authorization),
"authorizationSnapshot": {
"snapshotId": str(top_snapshot.get("id") or ""),
"allowedPurpose": "offline_evaluation",
"verifiedAt": str(top_snapshot.get("checkedAt") or ""),
"sourceVersion": source_version,
},
"generatedAt": str(top_snapshot.get("checkedAt") or ""),
"fineOutline": fine_outline,
"narrativeState": {
"time": "",
"location": "",
"characterPositions": {},
"immediateSituation": "",
},
"recentChapters": _recent_chapter_input(recent_rows),
"retrievalResult": retrieval_result,
"retrievalQueries": copy.deepcopy(plan["queries"]),
"cardIndexVersion": plan["cardIndexVersion"],
"proseIndexVersion": plan["proseIndexVersion"],
"tokenBudget": assembly_token_budget,
}
)
# 检索意图取自写手细纲:硬约束(含大纲文字与门禁要求)+ 可调节拍。
# WHY:这两段是本章创作意图的最稠密表达,用它做向量检索能召回最贴合的范式卡。
pattern_intent = "\n".join(
[str(item) for item in fine_outline["hardConstraints"]]
+ [str(item) for item in fine_outline["adjustableBeats"]]
)
# 未提供检索器时为空,保持历史行为;提供时仅 C 臂经 _pattern_references_for_arm 取用。
c_pattern_references = (
_retrieve_pattern_references(pattern_intent, card_searcher=pattern_card_searcher)
if pattern_card_searcher is not None
else []
)
# 把 C 臂候选范式卡冻结进 writerContextInput.patternReferences,随 config.json 序列化。
# WHY:回放端(run_writer_replay)真写时会从 config.json 重新 assemble 各臂上下文;
# 若候选不写进 writerContextInput,C 臂真写就拿不到范式卡,实验失效。这里冻结全量
# 候选(含来源指针,只留在冻结上下文供审计回读),回放端读出后再经同一事实源
# pattern_references_for_arm 按臂分配——A 恒空,单变量规则两端只有一处定义。
context_input["patternReferences"] = _pattern_references_for_arm("C", c_pattern_references)
writer_contexts: dict[str, dict[str, Any]] = {}
try:
for arm, strategy in (
("A", "generic_prose_retrieval"),
("C", "card_indexed_prose_retrieval"),
):
writer_contexts[arm] = assemble_context(
run_id=str(plan["runId"]),
attempt=1,
mode="diagnostic_only",
purpose="evaluation",
quality_policy_version="writer-eval-v1",
work_id=work_id,
target_chapter=target,
as_of=as_of,
source_version=source_version,
authorization_snapshot=context_input["authorizationSnapshot"],
source_status=context_input["sourceStatus"],
retrieval_plan=plan,
retrieval_result=retrieval_result,
fine_outline=fine_outline,
narrative_state=context_input["narrativeState"],
recent_chapters=context_input["recentChapters"],
output_contract=context_input["outputContract"],
token_budget=assembly_token_budget,
pattern_references=_pattern_references_for_arm(arm, c_pattern_references),
generated_at=context_input["generatedAt"],
evidence_strategy=strategy,
)["context"]
diff_receipts[sample_id] = build_context_allowlist_diff_receipt(
sample_id=sample_id,
context_a=writer_contexts["A"],
context_c=writer_contexts["C"],
prose_char_budget=prose_char_budget,
require_nonempty=True,
)
except AssemblyError as error:
raise WriterReferenceWorkError(
f"{sample_id} A/C WriterContext 单变量校验失败: {error}"
) from error
sample.update(
{
"workId": work_id,
"targetTitle": str(scaffold.get("title") or sample.get("targetTitle") or ""),
"snapshotVersion": f"writer-gate-a-canonical-{target}-v1",
"snapshotData": {
"chapters": [
{
"chapter": int(row["chapter"]),
"sourceId": _block_source_ref(row)["sourceId"],
"contentSha256": "sha256:"
+ hashlib.sha256(
str(row["content_text"]).encode("utf-8")
).hexdigest(),
}
for row in recent_rows
],
"cards": [],
},
"sources": sources,
"outlineSource": f"scaffold:{scaffold.get('id')}",
"fineOutlineSource": f"scaffold:{scaffold.get('id')}",
"writerContextInput": context_input,
"leakageAudit": leakage,
}
)
_recompute_new_character_ratio(sample, character_index)
assembled_samples.append(sample)
config["samples"] = assembled_samples
config["oracleTruthPacks"] = oracle_packs
config["writerContextDiffReceipts"] = diff_receipts
return config
def write_temporary_config(config: Mapping[str, Any], output_dir: Path) -> Path:
"""排他创建 /private/tmp 独立子目录并写入唯一完整配置。"""
resolved = output_dir.expanduser().resolve()
if resolved == PRIVATE_TMP or not resolved.is_relative_to(PRIVATE_TMP):
raise WriterReferenceWorkError("输出目录必须位于 /private/tmp 的独立子目录")
try:
resolved.mkdir(mode=0o700, parents=False, exist_ok=False)
except FileExistsError as error:
raise WriterReferenceWorkError("输出目录必须是尚不存在的独立子目录") from error
except FileNotFoundError as error:
raise WriterReferenceWorkError("输出目录父目录必须已存在") from error
config_path = resolved / "config.json"
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL
if hasattr(os, "O_NOFOLLOW"):
flags |= os.O_NOFOLLOW
descriptor = os.open(config_path, flags, 0o600)
try:
with os.fdopen(descriptor, "w", encoding="utf-8") as stream:
stream.write(_safe_json(config) + "\n")
except BaseException:
# fdopen 接管 descriptor;异常时只删除本次新建文件,不碰调用方目录。
config_path.unlink(missing_ok=True)
raise
return config_path
def _parse_args() -> argparse.Namespace:
"""解析真实 Writer Gate A 装配器参数。"""
parser = argparse.ArgumentParser(description="装配 Writer Gate A 五章真实临时配置")
parser.add_argument("--dsn", default=DSN)
parser.add_argument("--tenant-id", type=int, default=TENANT_ID)
parser.add_argument("--base-config", type=Path, default=DEFAULT_BASE_CONFIG)
parser.add_argument("--card-selectors", type=Path, default=DEFAULT_SELECTOR_CONFIG)
parser.add_argument("--output-dir", type=Path)
return parser.parse_args()
def main() -> int:
"""执行单事务只读装配并只回显安全摘要与临时配置路径。"""
args = _parse_args()
base_config = json.loads(args.base_config.read_text(encoding="utf-8"))
selector_bytes = args.card_selectors.read_bytes()
selectors = json.loads(selector_bytes.decode("utf-8"))
selector_digest = selector_sha256(selector_bytes)
samples = base_config.get("samples")
if not isinstance(samples, list):
raise WriterReferenceWorkError("基础配置 samples 非法")
targets = [_positive_chapter(item.get("targetChapter"), "sample.targetChapter") for item in samples]
rows = load_writer_reference_rows(
dsn=args.dsn,
tenant_id=args.tenant_id,
work_id=int(selectors.get("workId") or 0),
targets=targets,
base_config=base_config,
selector_config=selectors,
selector_digest=selector_digest,
)
config = assemble_writer_gate_config(
base_config=base_config,
selector_config=selectors,
selector_digest=selector_digest,
rows=rows,
# 生产装配才真连公共范式库:C 臂注入范式卡,A 臂保持空对照。
pattern_card_searcher=_default_pattern_card_searcher(
dsn=args.dsn,
tenant_id=args.tenant_id,
),
)
output_dir = args.output_dir or (
PRIVATE_TMP / f"writer-gate-a-{uuid.uuid4().hex}"
)
config_path = write_temporary_config(config, output_dir)
summary = {
"status": "ready_for_writer_dry_run",
"workId": config["referenceWork"]["id"],
"sampleCount": len(config["samples"]),
"contentMode": "canonical_frozen_prose",
"configPath": str(config_path),
}
print(json.dumps(summary, ensure_ascii=False, sort_keys=True))
return 0
if __name__ == "__main__":
try:
raise SystemExit(main())
except (
WriterReferenceWorkError,
RetrievalError,
OSError,
json.JSONDecodeError,
psycopg.Error,
) as error:
print(
json.dumps(
{"status": "blocked_writer_reference_adapter", "error": str(error)},
ensure_ascii=False,
)
)
raise SystemExit(2)
__all__ = [
"WriterReferenceWorkError",
"SnapshotProseRepository",
"assemble_writer_gate_config",
"build_oracle_truth_pack",
"index_unique_canonical_blocks",
"load_writer_reference_rows",
"milestone_reference_chapters",
"resolve_card_selectors",
"selector_sha256",
"select_recent_canonical_blocks",
"validate_oracle_truth_pack",
"write_temporary_config",
]