一、技能重组(动作-对象命名) - 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为 clean-book-text/decide-candidate/write-next-chapter/access-database/ check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持) - agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名 二、先审后入创作闭环(本次核心) 正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交", DB 级兜底,编排层跳步即被硬拒。 - candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链 - fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量, 模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本 - projection_registry.py + example_projection_run(107):投影登记与恢复 - acceptance_state.py:接受前置实时状态重读 - lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格 - DDL 105:example_candidate 增 semantic_status/semantic_report_sha256 - write_canonical.accept:语义兜底+同事务合并增量+登记投影; run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链 - claude_runtime:兼容新 CLI modelUsage 信息字段 三、审查修复(独立子代理四维审查后) - 事实增量 propose→approve 翻态正道,不撞唯一键 - 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记 - 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理 测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。 创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
2188 lines
97 KiB
Python
2188 lines
97 KiB
Python
#!/usr/bin/env python3
|
||
"""从实验库只读装配 Writer Gate A 五章真实临时配置。
|
||
|
||
本适配器只执行 SELECT,并把来源证明、授权、目标 scaffold、全量冻结历史、
|
||
冻结近章正文和预注册 upgrade_book 卡固定在同一个 REPEATABLE READ READ ONLY
|
||
事务中。目标章正文不查询;目标 scaffold 只拆成写手细纲和 evaluator-only 真值断言。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import copy
|
||
import hashlib
|
||
import json
|
||
import os
|
||
import sys
|
||
import uuid
|
||
from datetime import datetime, timedelta, timezone
|
||
from pathlib import Path
|
||
from typing import Any, Callable, Mapping, Sequence
|
||
|
||
import psycopg
|
||
from psycopg.rows import dict_row
|
||
|
||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||
READ_CONTEXT_SCRIPTS = SCRIPT_DIR.parents[1] / "assemble-context" / "scripts"
|
||
SNAPSHOT_SCRIPTS = SCRIPT_DIR.parents[1] / "freeze-context" / "scripts"
|
||
sys.path.insert(0, str(READ_CONTEXT_SCRIPTS))
|
||
sys.path.insert(0, str(SNAPSHOT_SCRIPTS))
|
||
|
||
from build_snapshot import normalize_chapter_range # noqa: E402
|
||
from assemble_writer_context import ( # noqa: E402
|
||
AssemblyError,
|
||
assemble_context,
|
||
build_context_allowlist_diff_receipt,
|
||
)
|
||
from load_reference_work import ( # noqa: E402
|
||
DSN,
|
||
TENANT_ID,
|
||
AdapterError,
|
||
_target_facts_from_scaffold,
|
||
begin_read_snapshot,
|
||
project_authorization,
|
||
project_card,
|
||
validate_source_records,
|
||
)
|
||
from retrieve_writer_sources import ( # noqa: E402
|
||
ReplayCardIndexRepository,
|
||
RetrievalError,
|
||
build_retrieval_plan,
|
||
retrieve_writer_sources,
|
||
)
|
||
from writer_contract import ( # noqa: E402
|
||
PATTERN_NAME_MAX_CHARS,
|
||
PATTERN_POINTS_MAX_FIELDS,
|
||
PATTERN_POINT_MAX_CHARS,
|
||
PATTERN_SUMMARY_MAX_CHARS,
|
||
han_count,
|
||
normalize_text,
|
||
pattern_references_for_arm,
|
||
)
|
||
from writer_eval_preregister import build_balanced_preregistration # noqa: E402
|
||
|
||
|
||
PRIVATE_TMP = Path("/private/tmp").resolve()
|
||
DEFAULT_BASE_CONFIG = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
DEFAULT_SELECTOR_CONFIG = (
|
||
SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-card-selectors-v1.json"
|
||
)
|
||
CANONICAL_CHAPTER_STATUSES = frozenset({"published", "confirmed", "canonical"})
|
||
EXPECTED_WRITER_INPUT_PROVENANCE = "preregistered_fine_outline"
|
||
EXPECTED_ORACLE_INPUT_PROVENANCE = "oracle_reference_scaffold"
|
||
PREREGISTERED_MAX_CONTEXT_CHARS = 140_000
|
||
RUNTIME_ADAPTER_VERSION_PREFIX = "writer-runtime-v1"
|
||
BUDGET_ROLES = ("writer", "semantic_detector", "blind_judge")
|
||
# raw vault 对租约的硬上限是 24 小时。五章三臂会串行执行多角色长调用,因此装配时
|
||
# 使用接近硬上限但留有时钟余量的窗口;execute 仍会在启动和每笔调用前复检。
|
||
RAW_RETENTION_WINDOW = timedelta(hours=23)
|
||
|
||
# 公共范式库五型(muse_knowledge_draft.draft_payload->>'型'):
|
||
# 套路 / 通用桥段 / 叙事技法 / 情感桥段 / 打斗桥段。C 臂按型各召回若干张。
|
||
PATTERN_CARD_TYPES = ("trope", "scene_pattern", "craft", "emotion", "combat")
|
||
# 每型最多取 2 张:五型合计 ≤ 10,严格低于总量硬上限,避免撑爆写手上下文预算。
|
||
PATTERN_TOP_PER_TYPE = 2
|
||
# 范式引用总量硬上限;即使提高每型 top,也不会超过这个数。
|
||
PATTERN_TOTAL_CAP = 12
|
||
|
||
|
||
class WriterReferenceWorkError(AdapterError):
|
||
"""真实正文装配输入缺失、歧义、漂移或越过冻结线时抛出。"""
|
||
|
||
|
||
def _safe_json(value: Any) -> str:
|
||
"""稳定序列化临时配置,便于重复运行后比较哈希。"""
|
||
|
||
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
||
|
||
|
||
def _sha256_value(value: Any) -> str:
|
||
"""对规范 JSON 对象计算带算法前缀的稳定哈希。"""
|
||
|
||
return "sha256:" + hashlib.sha256(_safe_json(value).encode("utf-8")).hexdigest()
|
||
|
||
|
||
def _is_full_sha256(value: Any) -> bool:
|
||
"""只接受不带前缀的 64 位小写 SHA-256。"""
|
||
|
||
return isinstance(value, str) and len(value) == 64 and all(
|
||
char in "0123456789abcdef" for char in value
|
||
)
|
||
|
||
|
||
def _validate_self_hash(value: Mapping[str, Any], field: str) -> None:
|
||
"""校验对象的 receiptSha256 是否等于其规范 JSON 自哈希。"""
|
||
|
||
receipt = value.get("receiptSha256")
|
||
if not isinstance(receipt, str) or not receipt.startswith("sha256:") or not _is_full_sha256(
|
||
receipt.removeprefix("sha256:")
|
||
):
|
||
raise WriterReferenceWorkError(f"{field}.receiptSha256 必须是 canonical SHA-256")
|
||
expected = _sha256_value(
|
||
{key: item for key, item in value.items() if key != "receiptSha256"}
|
||
)
|
||
if receipt != expected:
|
||
raise WriterReferenceWorkError(f"{field}.receiptSha256 canonical 自哈希无效")
|
||
|
||
|
||
def _stamp_raw_retention(config: dict[str, Any]) -> None:
|
||
"""为输出配置的 rawRetention 盖运行期租约到期时间戳并重签自哈希。
|
||
|
||
base 是仓内静态文件,存不了未来时间戳;retainUntil 是运行期值,必须由装配器
|
||
在生成仓外临时配置时盖戳。重签使用与 execute 端校验
|
||
(run_writer_replay._validate_authorization_records)同源的规范自哈希算法
|
||
(_sha256_value,即 canonical_sha256):授权本身(approvedBy/authorizationId/
|
||
status 等)一律不变,只补租约到期时间戳并刷新 receiptSha256。
|
||
"""
|
||
|
||
raw = config["executionAuthorization"]["rawRetention"]
|
||
raw["retainUntil"] = (datetime.now(timezone.utc) + RAW_RETENTION_WINDOW).isoformat()
|
||
raw["receiptSha256"] = _sha256_value(
|
||
{key: value for key, value in raw.items() if key != "receiptSha256"}
|
||
)
|
||
|
||
|
||
def _positive_chapter(value: Any, field: str) -> int:
|
||
"""严格接受正整数章号,不把 bool、浮点或模糊文本猜成章号。"""
|
||
|
||
if isinstance(value, bool):
|
||
raise WriterReferenceWorkError(f"{field} 必须是正整数章号")
|
||
if isinstance(value, str) and value.isdigit():
|
||
value = int(value)
|
||
if not isinstance(value, int) or value <= 0:
|
||
raise WriterReferenceWorkError(f"{field} 必须是正整数章号")
|
||
return value
|
||
|
||
|
||
def _card_payload(row: Mapping[str, Any]) -> Mapping[str, Any]:
|
||
"""读取卡 payload;选择器只信任结构化 type/name/alias。"""
|
||
|
||
payload = row.get("draft_payload")
|
||
if not isinstance(payload, Mapping):
|
||
raise WriterReferenceWorkError(f"卡 {row.get('id')} 缺少 draft_payload 对象")
|
||
return payload
|
||
|
||
|
||
def _selector_samples(selector_config: Mapping[str, Any]) -> list[Mapping[str, Any]]:
|
||
"""校验稳定选择器顶层结构和样本唯一性。"""
|
||
|
||
if not isinstance(selector_config, Mapping):
|
||
raise WriterReferenceWorkError("卡选择器配置必须是对象")
|
||
samples = selector_config.get("samples")
|
||
if not isinstance(samples, list) or not samples or any(
|
||
not isinstance(item, Mapping) for item in samples
|
||
):
|
||
raise WriterReferenceWorkError("卡选择器 samples 必须是非空对象数组")
|
||
sample_ids = [str(item.get("sampleId") or "") for item in samples]
|
||
if any(not item for item in sample_ids) or len(sample_ids) != len(set(sample_ids)):
|
||
raise WriterReferenceWorkError("卡选择器 sampleId 必须非空且唯一")
|
||
targets = [
|
||
_positive_chapter(item.get("targetChapter"), f"{item['sampleId']}.targetChapter")
|
||
for item in samples
|
||
]
|
||
if len(targets) != len(set(targets)):
|
||
raise WriterReferenceWorkError("卡选择器 targetChapter 必须唯一")
|
||
return samples
|
||
|
||
|
||
def selector_sha256(content: bytes) -> str:
|
||
"""对选择器规范文件原始字节计算带算法前缀的 SHA-256。"""
|
||
|
||
if not isinstance(content, bytes) or not content:
|
||
raise WriterReferenceWorkError("卡选择器规范文件不能为空")
|
||
return "sha256:" + hashlib.sha256(content).hexdigest()
|
||
|
||
|
||
def _validate_loader_controls(
|
||
base_config: Mapping[str, Any],
|
||
selector_config: Mapping[str, Any],
|
||
*,
|
||
selector_digest: str,
|
||
) -> Mapping[str, Any]:
|
||
"""在读取数据库前校验预注册选择器、输入来源和统一上下文上限。"""
|
||
|
||
if not isinstance(base_config, Mapping):
|
||
raise WriterReferenceWorkError("基础配置必须是对象")
|
||
common = base_config.get("commonControls")
|
||
if not isinstance(common, Mapping):
|
||
raise WriterReferenceWorkError("基础配置缺少 commonControls")
|
||
selector_version = str(selector_config.get("selectorVersion") or "")
|
||
if not selector_version or common.get("selectorVersion") != selector_version:
|
||
raise WriterReferenceWorkError("selectorVersion 未与预注册公共控制绑定")
|
||
if common.get("selectorSha256") != selector_digest:
|
||
raise WriterReferenceWorkError("选择器规范 JSON 的 SHA-256 与预注册公共控制不一致")
|
||
if common.get("writerInputProvenance") != EXPECTED_WRITER_INPUT_PROVENANCE:
|
||
raise WriterReferenceWorkError(
|
||
"writerInputProvenance 必须固定为 preregistered_fine_outline"
|
||
)
|
||
if common.get("oracleInputProvenance") != EXPECTED_ORACLE_INPUT_PROVENANCE:
|
||
raise WriterReferenceWorkError(
|
||
"oracleInputProvenance 必须固定为 oracle_reference_scaffold"
|
||
)
|
||
max_context_chars = common.get("maxContextChars")
|
||
if max_context_chars != PREREGISTERED_MAX_CONTEXT_CHARS:
|
||
raise WriterReferenceWorkError(
|
||
"commonControls.maxContextChars 必须严格等于预注册固定值 140000"
|
||
)
|
||
samples = base_config.get("samples")
|
||
if not isinstance(samples, list) or not samples:
|
||
raise WriterReferenceWorkError("基础配置 samples 不能为空")
|
||
execution_authorization = base_config.get("executionAuthorization")
|
||
if not isinstance(execution_authorization, Mapping):
|
||
raise WriterReferenceWorkError("基础配置缺少 executionAuthorization")
|
||
runtime_probe = execution_authorization.get("runtimeProbe")
|
||
if not isinstance(runtime_probe, Mapping):
|
||
raise WriterReferenceWorkError("executionAuthorization.runtimeProbe 必须是对象")
|
||
if runtime_probe.get("status") != "successful":
|
||
raise WriterReferenceWorkError("executionAuthorization.runtimeProbe 必须为 successful")
|
||
_validate_self_hash(runtime_probe, "executionAuthorization.runtimeProbe")
|
||
cli_version = runtime_probe.get("claudeCliVersion")
|
||
executable_sha256 = runtime_probe.get("claudeExecutableSha256")
|
||
if not isinstance(cli_version, str) or not cli_version.strip():
|
||
raise WriterReferenceWorkError("runtimeProbe.claudeCliVersion 不能为空")
|
||
if not _is_full_sha256(executable_sha256):
|
||
raise WriterReferenceWorkError(
|
||
"runtimeProbe.claudeExecutableSha256 必须是完整 64 位小写 SHA-256"
|
||
)
|
||
resolved_model_id = runtime_probe.get("resolvedModelId")
|
||
if not isinstance(resolved_model_id, str) or not resolved_model_id.strip():
|
||
raise WriterReferenceWorkError("runtimeProbe.resolvedModelId 不能为空")
|
||
if common.get("modelVersion") != resolved_model_id:
|
||
raise WriterReferenceWorkError(
|
||
"commonControls.modelVersion 必须精确等于 runtimeProbe.resolvedModelId"
|
||
)
|
||
expected_adapter_version = (
|
||
f"{RUNTIME_ADAPTER_VERSION_PREFIX}|claude-cli-{cli_version}|"
|
||
f"binary-sha256-{executable_sha256}"
|
||
)
|
||
if common.get("adapterVersion") != expected_adapter_version:
|
||
raise WriterReferenceWorkError(
|
||
"commonControls.adapterVersion 未绑定 runtimeProbe 的 CLI version 与 executable hash"
|
||
)
|
||
for control_name in ("budget", "rawRetention"):
|
||
control = execution_authorization.get(control_name)
|
||
if not isinstance(control, Mapping):
|
||
raise WriterReferenceWorkError(
|
||
f"executionAuthorization.{control_name} 必须是对象"
|
||
)
|
||
allowed_statuses = {"pending", "approved"}
|
||
if control.get("status") not in allowed_statuses:
|
||
raise WriterReferenceWorkError(
|
||
f"executionAuthorization.{control_name} 状态必须为 {sorted(allowed_statuses)}"
|
||
)
|
||
_validate_self_hash(control, f"executionAuthorization.{control_name}")
|
||
budget = execution_authorization["budget"]
|
||
planned_calls = budget.get("plannedCalls")
|
||
max_calls = budget.get("maxCalls")
|
||
if not isinstance(planned_calls, Mapping) or set(planned_calls) != set(BUDGET_ROLES):
|
||
raise WriterReferenceWorkError(
|
||
"executionAuthorization.budget.plannedCalls 必须完整覆盖三角色"
|
||
)
|
||
if not isinstance(max_calls, Mapping) or set(max_calls) != set(BUDGET_ROLES):
|
||
raise WriterReferenceWorkError(
|
||
"executionAuthorization.budget.maxCalls 必须完整覆盖三角色"
|
||
)
|
||
required_calls = len(samples) * 3
|
||
for role in BUDGET_ROLES:
|
||
planned = planned_calls[role]
|
||
maximum = max_calls[role]
|
||
if (
|
||
isinstance(planned, bool)
|
||
or not isinstance(planned, int)
|
||
or planned < required_calls
|
||
):
|
||
raise WriterReferenceWorkError(
|
||
f"executionAuthorization.budget.plannedCalls.{role} 必须是不低于 {required_calls} 的整数"
|
||
)
|
||
if isinstance(maximum, bool) or not isinstance(maximum, int) or maximum < planned:
|
||
raise WriterReferenceWorkError(
|
||
f"executionAuthorization.budget.maxCalls.{role} 必须是不低于 plannedCalls 的整数"
|
||
)
|
||
if common.get("sampling") != {
|
||
"temperature": "unsupported",
|
||
"topP": "unsupported",
|
||
"seed": "unsupported",
|
||
"reproducibilityClaim": "not_claimed",
|
||
}:
|
||
raise WriterReferenceWorkError("sampling 必须固定为 unsupported/not_claimed")
|
||
if base_config.get("strategyVersion") != "writer-abc-single-variable-v2":
|
||
raise WriterReferenceWorkError("strategyVersion 必须升级为 writer-abc-single-variable-v2")
|
||
if base_config.get("armPolicies") != {
|
||
"A": {
|
||
"evidenceStrategy": "generic_prose_retrieval",
|
||
"gateThresholdEligible": True,
|
||
"acceptanceEligible": False,
|
||
"writerCardVisibility": False,
|
||
},
|
||
"B": {
|
||
"evidenceStrategy": "card_only_diagnostic",
|
||
"gateThresholdEligible": False,
|
||
"acceptanceEligible": False,
|
||
"writerCardVisibility": True,
|
||
},
|
||
"C": {
|
||
"evidenceStrategy": "card_indexed_prose_retrieval",
|
||
"gateThresholdEligible": True,
|
||
"acceptanceEligible": False,
|
||
"writerCardVisibility": False,
|
||
},
|
||
}:
|
||
raise WriterReferenceWorkError("armPolicies 未固定 A/C Gate 与 B 负对照边界")
|
||
sample_ids = [str(sample.get("sampleId") or "") for sample in samples if isinstance(sample, Mapping)]
|
||
expected_preregistration = build_balanced_preregistration(
|
||
evaluation_set_version=str(base_config.get("evaluationSetVersion") or ""),
|
||
sample_ids=sample_ids,
|
||
)
|
||
if base_config.get("evaluationPreregistration") != expected_preregistration:
|
||
raise WriterReferenceWorkError("evaluationPreregistration 顺序、平衡或哈希不匹配")
|
||
for index, sample in enumerate(samples):
|
||
if not isinstance(sample, Mapping):
|
||
raise WriterReferenceWorkError(f"samples[{index}] 必须是对象")
|
||
configured = sample.get("writerContextInput", {}).get("tokenBudget", {})
|
||
if not isinstance(configured, Mapping):
|
||
raise WriterReferenceWorkError(f"samples[{index}] tokenBudget 必须是对象")
|
||
if configured.get("maxContextChars") != max_context_chars:
|
||
raise WriterReferenceWorkError(
|
||
f"samples[{index}] maxContextChars 必须原样使用预注册公共控制值"
|
||
)
|
||
return common
|
||
|
||
|
||
def _required_character_probes(base_config: Mapping[str, Any]) -> list[dict[str, Any]]:
|
||
"""只从冻结细纲要求提取具名角色;卡内角色状态不参与判定。"""
|
||
|
||
probes: list[dict[str, Any]] = []
|
||
for raw_sample in base_config.get("samples", []):
|
||
sample_id = str(raw_sample.get("sampleId") or "")
|
||
as_of = _positive_chapter(raw_sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
requirements = raw_sample.get("writerContextInput", {}).get("requirements", {})
|
||
basis = raw_sample.get("newCharacterBasis")
|
||
if not isinstance(requirements, Mapping) or not isinstance(basis, Mapping):
|
||
raise WriterReferenceWorkError(f"{sample_id} 缺少角色要求或冻结判定基线")
|
||
required = requirements.get("requiredCharacters")
|
||
generic = basis.get("genericRoles")
|
||
if (
|
||
not isinstance(required, list)
|
||
or not required
|
||
or any(not isinstance(name, str) or not name.strip() for name in required)
|
||
or len(required) != len(set(required))
|
||
):
|
||
raise WriterReferenceWorkError(f"{sample_id}.requiredCharacters 必须是无重复非空字符串数组")
|
||
if (
|
||
not isinstance(generic, list)
|
||
or any(not isinstance(name, str) or not name.strip() for name in generic)
|
||
or not set(generic).issubset(required)
|
||
):
|
||
raise WriterReferenceWorkError(f"{sample_id}.genericRoles 必须是 requiredCharacters 子集")
|
||
named = [name for name in required if name not in set(generic)]
|
||
if generic and named:
|
||
raise WriterReferenceWorkError(f"{sample_id} 暂不允许具名角色与泛称角色混合计算比例")
|
||
probes.extend(
|
||
{"sampleId": sample_id, "name": name, "asOfChapter": as_of}
|
||
for name in named
|
||
)
|
||
identities = [(item["sampleId"], item["name"]) for item in probes]
|
||
if len(identities) != len(set(identities)):
|
||
raise WriterReferenceWorkError("具名角色 Canonical 查询包含重复项")
|
||
return probes
|
||
|
||
|
||
def _canonical_character_index(
|
||
rows: Sequence[Mapping[str, Any]],
|
||
probes: Sequence[Mapping[str, Any]],
|
||
) -> dict[str, dict[str, dict[str, Any]]]:
|
||
"""校验 Canonical 正文命中结果,并按样本和角色名建立只含章号的索引。"""
|
||
|
||
expected = {
|
||
(str(item["sampleId"]), str(item["name"])): int(item["asOfChapter"])
|
||
for item in probes
|
||
}
|
||
indexed: dict[str, dict[str, dict[str, Any]]] = {}
|
||
seen: set[tuple[str, str]] = set()
|
||
for index, raw in enumerate(rows):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"canonical_character_mentions[{index}] 不是对象")
|
||
sample_id = str(raw.get("sample_id") or raw.get("sampleId") or "")
|
||
name = str(raw.get("name") or "")
|
||
identity = (sample_id, name)
|
||
if identity not in expected or identity in seen:
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果含额外项或重复项")
|
||
seen.add(identity)
|
||
as_of = _positive_chapter(
|
||
raw.get("as_of_chapter", raw.get("asOfChapter")),
|
||
f"{sample_id}.{name}.asOfChapter",
|
||
)
|
||
if as_of != expected[identity]:
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果冻结点漂移")
|
||
raw_hits = raw.get("hit_chapters", raw.get("hitChapters")) or []
|
||
if not isinstance(raw_hits, list):
|
||
raise WriterReferenceWorkError("Canonical 角色命中章必须是数组")
|
||
hits = sorted({_positive_chapter(item, f"{sample_id}.{name}.hitChapters") for item in raw_hits})
|
||
if any(chapter > as_of for chapter in hits):
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果越过冻结点")
|
||
raw_first = raw.get("first_chapter", raw.get("firstChapter"))
|
||
first = None if raw_first is None else _positive_chapter(raw_first, f"{sample_id}.{name}.firstChapter")
|
||
if first != (hits[0] if hits else None):
|
||
raise WriterReferenceWorkError("Canonical 角色首次命中章与命中章集合不一致")
|
||
indexed.setdefault(sample_id, {})[name] = {
|
||
"firstChapter": first,
|
||
"hitChapters": hits,
|
||
}
|
||
if seen != set(expected):
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果缺少预注册具名角色")
|
||
return indexed
|
||
|
||
|
||
def _recompute_new_character_ratio(
|
||
sample: dict[str, Any],
|
||
character_index: Mapping[str, Mapping[str, Mapping[str, Any]]],
|
||
) -> None:
|
||
"""依据冻结 Canonical 正文重写具名角色分区;不读取或信任卡内容。"""
|
||
|
||
sample_id = str(sample.get("sampleId") or "")
|
||
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
requirements = sample.get("writerContextInput", {}).get("requirements", {})
|
||
basis = sample.get("newCharacterBasis")
|
||
if not isinstance(requirements, Mapping) or not isinstance(basis, Mapping):
|
||
raise WriterReferenceWorkError(f"{sample_id} 缺少角色比例输入")
|
||
required = list(requirements.get("requiredCharacters") or [])
|
||
generic = list(basis.get("genericRoles") or [])
|
||
if generic:
|
||
sample["newCharacterRatio"] = None
|
||
sample["newCharacterRatioStatus"] = "unresolved_generic_role"
|
||
sample["newCharacterBasis"] = {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": as_of,
|
||
"requiredCharacters": required,
|
||
"knownBeforeAsOf": [],
|
||
"absentBeforeAsOf": [],
|
||
"genericRoles": generic,
|
||
}
|
||
return
|
||
mentions = character_index.get(sample_id, {})
|
||
known = [name for name in required if mentions.get(name, {}).get("hitChapters")]
|
||
absent = [name for name in required if name not in known]
|
||
sample["newCharacterRatio"] = len(absent) / len(required)
|
||
sample["newCharacterRatioStatus"] = "resolved"
|
||
sample["newCharacterBasis"] = {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": as_of,
|
||
"requiredCharacters": required,
|
||
"knownBeforeAsOf": known,
|
||
"absentBeforeAsOf": absent,
|
||
"genericRoles": [],
|
||
}
|
||
|
||
|
||
def resolve_card_selectors(
|
||
card_rows: Sequence[Mapping[str, Any]],
|
||
selector_config: Mapping[str, Any],
|
||
) -> dict[str, list[dict[str, Any]]]:
|
||
"""按 type + canonical name/alias 精确唯一解析预注册卡。
|
||
|
||
这里只读取稳定配置,不接收候选分数、回放结果或目标章正文,因此运行结果
|
||
不可能反向改变选卡。canonical name 和 alias 都是全字符串相等匹配。
|
||
"""
|
||
|
||
if not isinstance(card_rows, Sequence) or isinstance(card_rows, (str, bytes)):
|
||
raise WriterReferenceWorkError("卡查询结果必须是数组")
|
||
result: dict[str, list[dict[str, Any]]] = {}
|
||
used_card_ids: set[str] = set()
|
||
for sample in _selector_samples(selector_config):
|
||
sample_id = str(sample["sampleId"])
|
||
selectors = sample.get("cards")
|
||
if not isinstance(selectors, list) or not selectors or any(
|
||
not isinstance(item, Mapping) for item in selectors
|
||
):
|
||
raise WriterReferenceWorkError(f"{sample_id}.cards 必须是非空对象数组")
|
||
identities: list[tuple[str, str]] = []
|
||
selected: list[dict[str, Any]] = []
|
||
for index, selector in enumerate(selectors):
|
||
card_type = str(selector.get("type") or "").strip()
|
||
requested_name = str(selector.get("name") or "").strip()
|
||
if not card_type or not requested_name:
|
||
raise WriterReferenceWorkError(f"{sample_id}.cards[{index}] 缺少 type/name")
|
||
identity = (card_type, requested_name)
|
||
if identity in identities:
|
||
raise WriterReferenceWorkError(f"{sample_id} 含重复卡选择器 {identity}")
|
||
identities.append(identity)
|
||
|
||
matches: list[dict[str, Any]] = []
|
||
for raw_row in card_rows:
|
||
if not isinstance(raw_row, Mapping):
|
||
raise WriterReferenceWorkError("卡查询结果含非对象行")
|
||
payload = _card_payload(raw_row)
|
||
if str(payload.get("type") or "") != card_type:
|
||
continue
|
||
canonical_name = str(payload.get("名称") or "")
|
||
raw_aliases = payload.get("别名")
|
||
if raw_aliases is None:
|
||
aliases: list[str] = []
|
||
elif isinstance(raw_aliases, list) and all(
|
||
isinstance(alias, str) for alias in raw_aliases
|
||
):
|
||
aliases = raw_aliases
|
||
else:
|
||
raise WriterReferenceWorkError(f"卡 {raw_row.get('id')} 的别名不是字符串数组")
|
||
if canonical_name == requested_name or requested_name in aliases:
|
||
matches.append(copy.deepcopy(dict(raw_row)))
|
||
if len(matches) != 1:
|
||
raise WriterReferenceWorkError(
|
||
f"{sample_id} 选择器 ({card_type},{requested_name}) 必须唯一匹配,实际 {len(matches)} 张"
|
||
)
|
||
card_id = str(matches[0].get("id") or "")
|
||
if not card_id or card_id in used_card_ids:
|
||
raise WriterReferenceWorkError(f"卡 {card_id or '<empty>'} 被重复选择")
|
||
used_card_ids.add(card_id)
|
||
selected.append(matches[0])
|
||
result[sample_id] = selected
|
||
return result
|
||
|
||
|
||
def milestone_reference_chapters(
|
||
milestones: Sequence[Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
limit: int = 3,
|
||
) -> list[int]:
|
||
"""展开明确里程碑区间,去重后返回冻结线内最近最多三章。"""
|
||
|
||
freeze = _positive_chapter(as_of, "as_of")
|
||
if isinstance(limit, bool) or not isinstance(limit, int) or limit <= 0:
|
||
raise WriterReferenceWorkError("里程碑引用上限必须是正整数")
|
||
chapters: set[int] = set()
|
||
for index, milestone in enumerate(milestones):
|
||
if not isinstance(milestone, Mapping):
|
||
raise WriterReferenceWorkError(f"milestones[{index}] 不是对象")
|
||
raw_chapter = next(
|
||
(
|
||
milestone[key]
|
||
for key in ("chapter", "chapter_no", "order_no", "章", "章号")
|
||
if key in milestone
|
||
),
|
||
None,
|
||
)
|
||
bounds = normalize_chapter_range(raw_chapter)
|
||
if bounds is None:
|
||
raise WriterReferenceWorkError(f"milestones[{index}] 缺少明确绝对章号边界")
|
||
# 卡可包含未来演变,但来源定位只展开完整落在冻结线内的里程碑。
|
||
if bounds[1] > freeze:
|
||
continue
|
||
chapters.update(range(bounds[0], bounds[1] + 1))
|
||
if not chapters:
|
||
raise WriterReferenceWorkError("卡在冻结线内没有可定位 Canonical 原文的里程碑")
|
||
return sorted(chapters)[-limit:]
|
||
|
||
|
||
def _normalize_card_milestones(card: dict[str, Any], *, as_of: int) -> None:
|
||
"""把区间里程碑状态归一到明确结束章,供既有冻结器严格消费。"""
|
||
|
||
normalized: list[dict[str, Any]] = []
|
||
for index, raw in enumerate(card.get("milestones") or []):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"卡 {card.get('cardId')} 里程碑[{index}] 非法")
|
||
raw_chapter = next(
|
||
(raw[key] for key in ("chapter", "chapter_no", "order_no", "章", "章号") if key in raw),
|
||
None,
|
||
)
|
||
bounds = normalize_chapter_range(raw_chapter)
|
||
if bounds is None or bounds[1] > as_of:
|
||
raise WriterReferenceWorkError(f"卡 {card.get('cardId')} 含未冻结或无边界里程碑")
|
||
item = copy.deepcopy(dict(raw))
|
||
for alias in ("chapter_no", "order_no", "章", "章号"):
|
||
item.pop(alias, None)
|
||
item["chapter"] = bounds[1]
|
||
normalized.append(item)
|
||
normalized.sort(key=lambda item: (item["chapter"], str(item.get("id") or "")))
|
||
card["milestones"] = normalized
|
||
card["stateAsOf"] = copy.deepcopy(normalized)
|
||
|
||
|
||
def index_unique_canonical_blocks(
|
||
block_rows: Sequence[Mapping[str, Any]],
|
||
*,
|
||
allowed_chapters: set[int],
|
||
) -> dict[int, dict[str, Any]]:
|
||
"""校验每个请求章恰好一个 Canonical block,并拒绝额外未来章。"""
|
||
|
||
if not allowed_chapters:
|
||
raise WriterReferenceWorkError("Canonical block 请求章集合不能为空")
|
||
grouped: dict[int, list[dict[str, Any]]] = {}
|
||
for index, raw in enumerate(block_rows):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"block_rows[{index}] 不是对象")
|
||
chapter = _positive_chapter(raw.get("chapter"), f"block_rows[{index}].chapter")
|
||
if chapter not in allowed_chapters:
|
||
raise WriterReferenceWorkError(f"读取到未请求或目标/未来章正文: {chapter}")
|
||
if str(raw.get("chapter_status") or "") not in CANONICAL_CHAPTER_STATUSES:
|
||
raise WriterReferenceWorkError(f"第 {chapter} 章不是 Canonical 状态")
|
||
text = raw.get("content_text")
|
||
if not isinstance(text, str) or not text:
|
||
raise WriterReferenceWorkError(f"第 {chapter} 章 Canonical block 正文为空")
|
||
grouped.setdefault(chapter, []).append(copy.deepcopy(dict(raw)))
|
||
missing = sorted(allowed_chapters - set(grouped))
|
||
duplicates = sorted(chapter for chapter, rows in grouped.items() if len(rows) != 1)
|
||
if missing or duplicates:
|
||
raise WriterReferenceWorkError(
|
||
f"Canonical block 必须逐章唯一: missing={missing}, non_unique={duplicates}"
|
||
)
|
||
return {chapter: rows[0] for chapter, rows in grouped.items()}
|
||
|
||
|
||
def select_recent_canonical_blocks(
|
||
block_rows: Sequence[Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
) -> list[dict[str, Any]]:
|
||
"""按冻结点选择连续四章,缺章或重复 block 均失败关闭。"""
|
||
|
||
freeze = _positive_chapter(as_of, "as_of")
|
||
expected = list(range(max(1, freeze - 3), freeze + 1))
|
||
relevant = [row for row in block_rows if row.get("chapter") in expected]
|
||
indexed = index_unique_canonical_blocks(relevant, allowed_chapters=set(expected))
|
||
return [indexed[chapter] for chapter in expected]
|
||
|
||
|
||
def _block_source_ref(row: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""为唯一 Canonical block 生成完整代码点区间来源引用。"""
|
||
|
||
chapter = _positive_chapter(row.get("chapter"), "block.chapter")
|
||
block_id = row.get("block_id")
|
||
if isinstance(block_id, bool) or not isinstance(block_id, int) or block_id <= 0:
|
||
raise WriterReferenceWorkError(f"第 {chapter} 章 block_id 非法")
|
||
text = str(row.get("content_text") or "")
|
||
revision = row.get("revision") or 0
|
||
source_version = f"chapter:{chapter}:block:{block_id}:revision:{revision}"
|
||
return {
|
||
"sourceId": f"chapter:{chapter}:block:{block_id}",
|
||
"sourceVersion": source_version,
|
||
"chapter": chapter,
|
||
"blockId": block_id,
|
||
"startCodePoint": 0,
|
||
"endCodePoint": len(text),
|
||
}
|
||
|
||
|
||
def _oracle_authorization(
|
||
authorization: Mapping[str, Any], *, source_version: str
|
||
) -> dict[str, Any]:
|
||
"""提取 evaluator-only 授权投影并严格绑定来源版本与重验时间。"""
|
||
|
||
snapshot = authorization.get("authorizationSnapshot")
|
||
if not isinstance(snapshot, Mapping):
|
||
raise WriterReferenceWorkError("oracleTruthPack 缺少不可变授权快照")
|
||
if authorization.get("allowedPurpose") != ["offline_evaluation"]:
|
||
raise WriterReferenceWorkError("oracleTruthPack 只允许 offline_evaluation")
|
||
if snapshot.get("allowedPurpose") != ["offline_evaluation"]:
|
||
raise WriterReferenceWorkError("oracleTruthPack 授权快照用途不唯一")
|
||
if str(snapshot.get("sourceVersion") or "") != source_version:
|
||
raise WriterReferenceWorkError("oracleTruthPack 授权来源版本不匹配")
|
||
source_status = str(snapshot.get("sourceStatus") or "")
|
||
if source_status not in {"active", "authorized", "frozen_authorized"}:
|
||
raise WriterReferenceWorkError("oracleTruthPack 授权来源状态不可用")
|
||
revalidation_at = str(snapshot.get("revalidationAt") or "")
|
||
if not revalidation_at:
|
||
raise WriterReferenceWorkError("oracleTruthPack 授权缺少重验时间")
|
||
snapshot_id = str(snapshot.get("id") or "")
|
||
if not snapshot_id:
|
||
raise WriterReferenceWorkError("oracleTruthPack 授权快照 ID 为空")
|
||
return {
|
||
"allowedPurpose": "offline_evaluation",
|
||
"sourceStatus": source_status,
|
||
"sourceVersion": source_version,
|
||
"revalidationAt": revalidation_at,
|
||
"evaluatorOnly": True,
|
||
"snapshotId": snapshot_id,
|
||
}
|
||
|
||
|
||
def _historical_oracle_assertions(
|
||
oracle_block_rows: Sequence[Mapping[str, Any]], *, as_of: int
|
||
) -> list[dict[str, Any]]:
|
||
"""把冻结线内每章 Canonical 正文投影为稳定、可追溯的历史断言。"""
|
||
|
||
expected = set(range(1, as_of + 1))
|
||
relevant = [row for row in oracle_block_rows if row.get("chapter") in expected]
|
||
try:
|
||
indexed = index_unique_canonical_blocks(relevant, allowed_chapters=expected)
|
||
except WriterReferenceWorkError as error:
|
||
raise WriterReferenceWorkError(f"oracle Canonical 历史缺章或不唯一: {error}") from error
|
||
assertions: list[dict[str, Any]] = []
|
||
for chapter in range(1, as_of + 1):
|
||
row = indexed[chapter]
|
||
statement = normalize_text(str(row["content_text"]))
|
||
source_ref = _block_source_ref(row)
|
||
assertions.append(
|
||
{
|
||
"assertionId": f"historical:chapter:{chapter}:block:{row['block_id']}",
|
||
"assertionType": "canonical_history",
|
||
"statement": statement,
|
||
"sourceVersion": source_ref["sourceVersion"],
|
||
"chapterBoundary": {"minChapter": chapter, "maxChapter": chapter},
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
|
||
}
|
||
)
|
||
return assertions
|
||
|
||
|
||
def _target_oracle_assertions(
|
||
scaffold: Mapping[str, Any], *, target_chapter: int, source_version: str
|
||
) -> list[dict[str, Any]]:
|
||
"""只从预注册目标 scaffold 拆出目标断言,不读取目标章正文。"""
|
||
|
||
target_facts = _target_facts_from_scaffold(scaffold, target_chapter)
|
||
assertions: list[dict[str, Any]] = []
|
||
for raw in target_facts["forbiddenFacts"]:
|
||
statement = normalize_text(str(raw["text"]))
|
||
assertions.append(
|
||
{
|
||
"assertionId": str(raw["id"]),
|
||
"assertionType": "target_reference_scaffold",
|
||
"statement": statement,
|
||
"sourceVersion": source_version,
|
||
"chapterBoundary": {
|
||
"minChapter": target_chapter,
|
||
"maxChapter": target_chapter,
|
||
},
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(statement.encode("utf-8")).hexdigest(),
|
||
}
|
||
)
|
||
return assertions
|
||
|
||
|
||
def _validate_oracle_assertion(
|
||
assertion: Mapping[str, Any], *, expected_type: str, as_of: int, target_chapter: int
|
||
) -> None:
|
||
"""严格校验 oracle 断言字段、内容哈希和章号边界。"""
|
||
|
||
required = {
|
||
"assertionId",
|
||
"assertionType",
|
||
"statement",
|
||
"sourceVersion",
|
||
"chapterBoundary",
|
||
"contentSha256",
|
||
}
|
||
if not isinstance(assertion, Mapping) or set(assertion) != required:
|
||
raise WriterReferenceWorkError("oracleTruthPack assertion 字段不严格")
|
||
if assertion["assertionType"] != expected_type:
|
||
raise WriterReferenceWorkError("oracleTruthPack assertionType 非法")
|
||
boundary = assertion["chapterBoundary"]
|
||
if not isinstance(boundary, Mapping) or set(boundary) != {"minChapter", "maxChapter"}:
|
||
raise WriterReferenceWorkError("oracleTruthPack chapterBoundary 非法")
|
||
minimum = _positive_chapter(boundary["minChapter"], "oracle.chapterBoundary.minChapter")
|
||
maximum = _positive_chapter(boundary["maxChapter"], "oracle.chapterBoundary.maxChapter")
|
||
if minimum != maximum:
|
||
raise WriterReferenceWorkError("oracleTruthPack 断言必须绑定唯一章")
|
||
if expected_type == "canonical_history" and maximum > as_of:
|
||
raise WriterReferenceWorkError("oracleTruthPack 历史断言越过冻结线")
|
||
if expected_type == "target_reference_scaffold" and maximum != target_chapter:
|
||
raise WriterReferenceWorkError("oracleTruthPack 目标断言未绑定目标章")
|
||
statement = str(assertion["statement"])
|
||
expected_hash = "sha256:" + hashlib.sha256(statement.encode("utf-8")).hexdigest()
|
||
if assertion["contentSha256"] != expected_hash:
|
||
raise WriterReferenceWorkError("oracleTruthPack 断言内容哈希不一致")
|
||
|
||
|
||
def build_oracle_truth_pack(
|
||
*,
|
||
evaluation_set_version: str,
|
||
sample_id: str,
|
||
work_id: int,
|
||
as_of: int,
|
||
target_chapter: int,
|
||
source_version: str,
|
||
authorization: Mapping[str, Any],
|
||
oracle_block_rows: Sequence[Mapping[str, Any]],
|
||
scaffold: Mapping[str, Any],
|
||
) -> dict[str, Any]:
|
||
"""构建严格 oracle-truth-pack-v1,仅返回临时评测配置所需对象。"""
|
||
|
||
auth = _oracle_authorization(authorization, source_version=source_version)
|
||
historical = _historical_oracle_assertions(oracle_block_rows, as_of=as_of)
|
||
targets = _target_oracle_assertions(
|
||
scaffold,
|
||
target_chapter=target_chapter,
|
||
source_version=source_version,
|
||
)
|
||
source_snapshot = {
|
||
"sourceVersion": source_version,
|
||
"asOf": as_of,
|
||
"historicalAssertionHashes": [item["contentSha256"] for item in historical],
|
||
"targetAssertionHashes": [item["contentSha256"] for item in targets],
|
||
}
|
||
payload = {
|
||
"schemaVersion": "oracle-truth-pack-v1",
|
||
"evaluationSetVersion": str(evaluation_set_version),
|
||
"sampleId": str(sample_id),
|
||
"workId": work_id,
|
||
"asOf": as_of,
|
||
"sourceSnapshotSha256": _sha256_value(source_snapshot),
|
||
"authorizationSnapshotId": auth["snapshotId"],
|
||
"authorization": auth,
|
||
"historicalAssertions": historical,
|
||
"targetAssertions": targets,
|
||
}
|
||
if len({item["assertionId"] for item in [*historical, *targets]}) != len(
|
||
[*historical, *targets]
|
||
):
|
||
raise WriterReferenceWorkError("oracleTruthPack assertionId 不唯一")
|
||
for assertion in historical:
|
||
_validate_oracle_assertion(
|
||
assertion,
|
||
expected_type="canonical_history",
|
||
as_of=as_of,
|
||
target_chapter=target_chapter,
|
||
)
|
||
for assertion in targets:
|
||
_validate_oracle_assertion(
|
||
assertion,
|
||
expected_type="target_reference_scaffold",
|
||
as_of=as_of,
|
||
target_chapter=target_chapter,
|
||
)
|
||
return validate_oracle_truth_pack(
|
||
{**payload, "packSha256": _sha256_value(payload)}
|
||
)
|
||
|
||
|
||
def validate_oracle_truth_pack(pack: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""严格复核 oracle-truth-pack-v1 的 schema、授权、章界和整体哈希。"""
|
||
|
||
required = {
|
||
"schemaVersion",
|
||
"evaluationSetVersion",
|
||
"sampleId",
|
||
"workId",
|
||
"asOf",
|
||
"sourceSnapshotSha256",
|
||
"authorizationSnapshotId",
|
||
"authorization",
|
||
"historicalAssertions",
|
||
"targetAssertions",
|
||
"packSha256",
|
||
}
|
||
if not isinstance(pack, Mapping) or set(pack) != required:
|
||
raise WriterReferenceWorkError("oracleTruthPack 顶层字段不严格")
|
||
if pack["schemaVersion"] != "oracle-truth-pack-v1":
|
||
raise WriterReferenceWorkError("oracleTruthPack schemaVersion 不支持")
|
||
as_of = _positive_chapter(pack["asOf"], "oracleTruthPack.asOf")
|
||
target_chapter = as_of + 1
|
||
if not str(pack["evaluationSetVersion"]).strip() or not str(pack["sampleId"]).strip():
|
||
raise WriterReferenceWorkError("oracleTruthPack 评测集或样本绑定为空")
|
||
_positive_chapter(pack["workId"], "oracleTruthPack.workId")
|
||
source_snapshot_hash = str(pack["sourceSnapshotSha256"])
|
||
if not source_snapshot_hash.startswith("sha256:") or len(source_snapshot_hash) != 71:
|
||
raise WriterReferenceWorkError("oracleTruthPack sourceSnapshotSha256 非法")
|
||
authorization = pack["authorization"]
|
||
if not isinstance(authorization, Mapping) or set(authorization) != {
|
||
"allowedPurpose",
|
||
"sourceStatus",
|
||
"sourceVersion",
|
||
"revalidationAt",
|
||
"evaluatorOnly",
|
||
"snapshotId",
|
||
}:
|
||
raise WriterReferenceWorkError("oracleTruthPack authorization 字段不严格")
|
||
if (
|
||
authorization["allowedPurpose"] != "offline_evaluation"
|
||
or authorization["evaluatorOnly"] is not True
|
||
or authorization["snapshotId"] != pack["authorizationSnapshotId"]
|
||
or not str(authorization["revalidationAt"]).strip()
|
||
):
|
||
raise WriterReferenceWorkError("oracleTruthPack evaluator-only 授权非法")
|
||
historical = pack["historicalAssertions"]
|
||
targets = pack["targetAssertions"]
|
||
if not isinstance(historical, list) or not isinstance(targets, list) or not targets:
|
||
raise WriterReferenceWorkError("oracleTruthPack 断言数组非法")
|
||
for assertion in historical:
|
||
_validate_oracle_assertion(
|
||
assertion,
|
||
expected_type="canonical_history",
|
||
as_of=as_of,
|
||
target_chapter=target_chapter,
|
||
)
|
||
for assertion in targets:
|
||
_validate_oracle_assertion(
|
||
assertion,
|
||
expected_type="target_reference_scaffold",
|
||
as_of=as_of,
|
||
target_chapter=target_chapter,
|
||
)
|
||
if {
|
||
assertion["chapterBoundary"]["maxChapter"] for assertion in historical
|
||
} != set(range(1, as_of + 1)):
|
||
raise WriterReferenceWorkError("oracleTruthPack Canonical 历史断言缺章")
|
||
all_assertions = [*historical, *targets]
|
||
if len({assertion["assertionId"] for assertion in all_assertions}) != len(all_assertions):
|
||
raise WriterReferenceWorkError("oracleTruthPack assertionId 不唯一")
|
||
if any(
|
||
assertion["sourceVersion"] != authorization["sourceVersion"]
|
||
and assertion["assertionType"] == "target_reference_scaffold"
|
||
for assertion in all_assertions
|
||
):
|
||
raise WriterReferenceWorkError("oracleTruthPack 目标断言来源版本不匹配")
|
||
payload = {key: copy.deepcopy(value) for key, value in pack.items() if key != "packSha256"}
|
||
if pack["packSha256"] != _sha256_value(payload):
|
||
raise WriterReferenceWorkError("oracleTruthPack packSha256 不一致")
|
||
return copy.deepcopy(dict(pack))
|
||
|
||
|
||
class SnapshotProseRepository:
|
||
"""只从同一数据库事务已冻结的 block 行展开来源引用。"""
|
||
|
||
def __init__(self, block_index: Mapping[int, Mapping[str, Any]]):
|
||
self._by_chapter = {
|
||
int(chapter): copy.deepcopy(dict(row)) for chapter, row in block_index.items()
|
||
}
|
||
self._by_block = {int(row["block_id"]): row for row in self._by_chapter.values()}
|
||
|
||
def read_source_refs(
|
||
self,
|
||
*,
|
||
work_id: int,
|
||
as_of: int,
|
||
source_refs: Sequence[Mapping[str, Any]],
|
||
) -> list[dict[str, Any]]:
|
||
"""逐引用校验章、块与代码点区间,不建立第二个数据库连接。"""
|
||
|
||
if work_id != 8:
|
||
raise WriterReferenceWorkError("Writer Gate A 只允许预注册 work=8")
|
||
freeze = _positive_chapter(as_of, "as_of")
|
||
result: list[dict[str, Any]] = []
|
||
for index, ref in enumerate(source_refs):
|
||
if not isinstance(ref, Mapping):
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 不是对象")
|
||
chapter = _positive_chapter(ref.get("chapter"), f"source_refs[{index}].chapter")
|
||
if chapter > freeze:
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 包含目标章或未来章")
|
||
block_id = ref.get("blockId")
|
||
row = self._by_block.get(block_id) if isinstance(block_id, int) else None
|
||
if row is None or int(row["chapter"]) != chapter:
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 不能定位唯一 Canonical block")
|
||
text = str(row["content_text"])
|
||
start = ref.get("startCodePoint")
|
||
end = ref.get("endCodePoint")
|
||
if (
|
||
isinstance(start, bool)
|
||
or isinstance(end, bool)
|
||
or not isinstance(start, int)
|
||
or not isinstance(end, int)
|
||
or start < 0
|
||
or end <= start
|
||
or end > len(text)
|
||
):
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 代码点区间越界")
|
||
fragment = text[start:end]
|
||
result.append(
|
||
{
|
||
"chapter": chapter,
|
||
"blockId": block_id,
|
||
"blockOrder": int(row.get("block_order") or 0),
|
||
"sourceRef": copy.deepcopy(dict(ref)),
|
||
"text": fragment,
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(fragment.encode("utf-8")).hexdigest(),
|
||
"purpose": str(ref.get("sourceType") or "card_source"),
|
||
}
|
||
)
|
||
return result
|
||
|
||
|
||
def _target_scaffold_index(
|
||
rows: Sequence[Mapping[str, Any]], targets: set[int]
|
||
) -> dict[int, dict[str, Any]]:
|
||
"""目标 scaffold 也要求逐章唯一,禁止 ORDER BY/LIMIT 猜选。"""
|
||
|
||
grouped: dict[int, list[dict[str, Any]]] = {}
|
||
for index, raw in enumerate(rows):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"target_scaffolds[{index}] 不是对象")
|
||
chapter = _positive_chapter(raw.get("chapter"), f"target_scaffolds[{index}].chapter")
|
||
if chapter not in targets:
|
||
raise WriterReferenceWorkError(f"读取到未预注册目标 scaffold: {chapter}")
|
||
grouped.setdefault(chapter, []).append(copy.deepcopy(dict(raw)))
|
||
missing = sorted(targets - set(grouped))
|
||
duplicates = sorted(chapter for chapter, values in grouped.items() if len(values) != 1)
|
||
if missing or duplicates:
|
||
raise WriterReferenceWorkError(
|
||
f"目标 scaffold 必须逐章唯一: missing={missing}, non_unique={duplicates}"
|
||
)
|
||
return {chapter: values[0] for chapter, values in grouped.items()}
|
||
|
||
|
||
def _requirement_hard_constraints(requirements: Any) -> list[str]:
|
||
"""把门禁会机械检查的要求清单 surface 为写手可见的硬约束字符串。
|
||
|
||
WHY:正文 Gate A 的机械门(check_writer_candidate)用这份 requirements 卡候选——
|
||
章末钩子锚点、伏笔动作锚点、硬事件锚点、必须出场角色。写手若看不到要查什么,
|
||
就可能漏写(如章末钩子锚点不在大纲文字里),机械门必挂 CHAPTER_END_HOOK_MISSING。
|
||
因此 loader 在装配写手细纲时,把这份清单转成硬约束并进 hardConstraints,让写手
|
||
知道门禁会查什么。本函数只读 requirements、不改写它。
|
||
|
||
确定性:追加顺序固定为 章末钩子→伏笔动作→硬事件→必须出场角色,同样输入产出
|
||
同样字符串。这些条目对 A/B/C 三臂逐样本完全相同(共享同一份 requirements),
|
||
不引入 A/C 单变量差异,因此不会撞差异回执冻结合同。
|
||
"""
|
||
|
||
if not isinstance(requirements, Mapping):
|
||
return []
|
||
|
||
constraints: list[str] = []
|
||
|
||
def _join_anchors(anchors: Any) -> str:
|
||
# 锚点必须是列表;列表内再逐项做类型保护(非字符串/空串跳过),保证拼接稳定。
|
||
if not isinstance(anchors, list):
|
||
return ""
|
||
return "、".join(anchor for anchor in anchors if isinstance(anchor, str) and anchor)
|
||
|
||
# 1. 章末钩子:门禁查正文结尾 maxDistanceFromEnd 字内是否出现锚点之一。
|
||
hook = requirements.get("chapterEndHook")
|
||
if isinstance(hook, Mapping):
|
||
anchors_text = _join_anchors(hook.get("anchors"))
|
||
max_distance = hook.get("maxDistanceFromEnd")
|
||
if (
|
||
anchors_text
|
||
and not isinstance(max_distance, bool)
|
||
and isinstance(max_distance, int)
|
||
and max_distance > 0
|
||
):
|
||
constraints.append(
|
||
f"章末钩子(硬要求):正文结尾 {max_distance} 字内必须出现以下锚点之一:{anchors_text}"
|
||
)
|
||
|
||
# 2. 伏笔动作:门禁查正文是否自然埋入锚点之一。
|
||
for item in requirements.get("foreshadowingActions") or []:
|
||
if not isinstance(item, Mapping):
|
||
continue
|
||
anchors_text = _join_anchors(item.get("anchors"))
|
||
if anchors_text:
|
||
constraints.append(
|
||
f"伏笔动作(硬要求):正文必须自然埋入以下锚点之一:{anchors_text}"
|
||
)
|
||
|
||
# 3. 硬事件:门禁查正文是否命中锚点之一。
|
||
for item in requirements.get("requiredEvents") or []:
|
||
if not isinstance(item, Mapping):
|
||
continue
|
||
anchors_text = _join_anchors(item.get("anchors"))
|
||
if anchors_text:
|
||
constraints.append(f"硬事件(硬要求):正文必须命中以下锚点之一:{anchors_text}")
|
||
|
||
# 4. 必须出场角色:门禁查正文是否出现角色名。
|
||
characters = [
|
||
character
|
||
for character in (requirements.get("requiredCharacters") or [])
|
||
if isinstance(character, str) and character
|
||
]
|
||
if characters:
|
||
constraints.append(
|
||
f"必须出场角色(硬要求):正文必须出现以下角色:{'、'.join(characters)}"
|
||
)
|
||
|
||
return constraints
|
||
|
||
|
||
def _selected_card_entities(cards: Sequence[Mapping[str, Any]]) -> list[dict[str, str]]:
|
||
"""把稳定选择器结果转换为检索计划实体,不从运行结果追加查询。"""
|
||
|
||
return [
|
||
{
|
||
"id": f"selected-card:{card['cardId']}",
|
||
"type": str(card["type"]),
|
||
"name": str(card["name"]),
|
||
}
|
||
for card in cards
|
||
]
|
||
|
||
|
||
def _context_token_budget(
|
||
configured: Mapping[str, Any],
|
||
*,
|
||
preregistered_max: int,
|
||
) -> dict[str, int]:
|
||
"""原样使用预注册上限;基线超限由组装器失败,补充证据由组装器裁剪。"""
|
||
|
||
configured_max = configured.get("maxContextChars")
|
||
if (
|
||
isinstance(configured_max, bool)
|
||
or not isinstance(configured_max, int)
|
||
or configured_max <= 0
|
||
):
|
||
raise WriterReferenceWorkError("tokenBudget.maxContextChars 必须是正整数")
|
||
if configured_max != preregistered_max:
|
||
raise WriterReferenceWorkError("loader 禁止改写或扩张预注册 maxContextChars")
|
||
return {"maxContextChars": preregistered_max}
|
||
|
||
|
||
def _context_source_status(authorization: Mapping[str, Any]) -> str:
|
||
"""按既有 WriterContext 授权绑定规则投影运行期来源状态。"""
|
||
|
||
source_status = str(authorization.get("sourceStatus") or "").lower()
|
||
if source_status in {"active", "approved", "licensed"}:
|
||
return "active"
|
||
if source_status == "authorized":
|
||
return "authorized"
|
||
raise WriterReferenceWorkError(f"授权来源状态不能进入 WriterContext: {source_status}")
|
||
|
||
|
||
def _project_card_for_sample(
|
||
row: Mapping[str, Any],
|
||
*,
|
||
as_of: int,
|
||
source_version: str,
|
||
block_index: Mapping[int, Mapping[str, Any]],
|
||
source_aliases: Sequence[str] = (),
|
||
) -> dict[str, Any]:
|
||
"""冻结卡;缺精确引用时只生成可识别且显式降级的整章代理。"""
|
||
|
||
card_id = str(row.get("id") or "")
|
||
card_version = f"{source_version}:card-{card_id}-rev-{row.get('revision') or 0}"
|
||
projected = project_card(row, as_of=as_of, source_version=card_version)
|
||
milestones = copy.deepcopy(projected.get("milestones") or [])
|
||
uses_chapter_proxy = not projected.get("sourceRefs")
|
||
if uses_chapter_proxy:
|
||
chapters = milestone_reference_chapters(milestones, as_of=as_of)
|
||
try:
|
||
projected["sourceRefs"] = []
|
||
payload = _card_payload(row)
|
||
raw_aliases = payload.get("别名") or []
|
||
if not isinstance(raw_aliases, list) or any(
|
||
not isinstance(alias, str) for alias in raw_aliases
|
||
):
|
||
raise WriterReferenceWorkError(f"卡 {card_id} 的别名不是字符串数组")
|
||
legal_names = {
|
||
str(projected.get("name") or "").strip(),
|
||
*(alias.strip() for alias in raw_aliases),
|
||
*(str(alias).strip() for alias in source_aliases),
|
||
}
|
||
legal_names.discard("")
|
||
for chapter in chapters:
|
||
row_text = normalize_text(str(block_index[chapter]["content_text"]))
|
||
if not any(name in row_text for name in legal_names):
|
||
raise WriterReferenceWorkError(
|
||
f"卡 {card_id} 的整章代理第 {chapter} 章未出现规范名或合法别名"
|
||
)
|
||
ref = _block_source_ref(block_index[chapter])
|
||
ref["sourceType"] = "card_chapter_proxy"
|
||
projected["sourceRefs"].append(ref)
|
||
except KeyError as error:
|
||
raise WriterReferenceWorkError(
|
||
f"卡 {card_id} 的里程碑章 {error.args[0]} 缺少唯一 Canonical block"
|
||
) from error
|
||
for index, ref in enumerate(projected.get("sourceRefs") or []):
|
||
chapter = _positive_chapter(ref.get("chapter"), f"卡 {card_id}.sourceRefs[{index}].chapter")
|
||
if chapter > as_of:
|
||
raise WriterReferenceWorkError(f"卡 {card_id} sourceRef 包含目标章或未来章")
|
||
if ref.get("blockId") not in {row["block_id"] for row in block_index.values()}:
|
||
raise WriterReferenceWorkError(f"卡 {card_id} sourceRef 未绑定本次 Canonical 快照")
|
||
if uses_chapter_proxy and ref.get("sourceType") != "card_chapter_proxy":
|
||
raise WriterReferenceWorkError(f"卡 {card_id} 整章代理缺少降级来源类型")
|
||
_normalize_card_milestones(projected, as_of=as_of)
|
||
return projected
|
||
|
||
|
||
def _sample_sources(
|
||
recent_rows: Sequence[Mapping[str, Any]],
|
||
cards: Sequence[Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
) -> list[dict[str, Any]]:
|
||
"""生成只含冻结历史的可验证来源目录,不登记目标 scaffold 为历史来源。"""
|
||
|
||
sources = [_block_source_ref(row) for row in recent_rows]
|
||
for card in cards:
|
||
sources.append(
|
||
{
|
||
"sourceId": str(card["sourceId"]),
|
||
"sourceVersion": str(card["sourceVersion"]),
|
||
"chapterRange": f"1-{as_of}",
|
||
"scope": "card_projection",
|
||
}
|
||
)
|
||
return sources
|
||
|
||
|
||
def _recent_chapter_input(rows: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
||
"""把连续四章唯一 block 投影成 WriterContext 的完整历史基线。"""
|
||
|
||
return [
|
||
{
|
||
"chapter": int(row["chapter"]),
|
||
"sourceRef": _block_source_ref(row),
|
||
"text": normalize_text(str(row["content_text"])),
|
||
}
|
||
for row in rows
|
||
]
|
||
|
||
|
||
def _generic_historical_prose(
|
||
block_index: Mapping[int, Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
work_id: int,
|
||
char_budget: int,
|
||
) -> list[dict[str, Any]]:
|
||
"""不使用卡,按冻结线向前选择通用 Canonical 历史原文候选。"""
|
||
|
||
if isinstance(char_budget, bool) or not isinstance(char_budget, int) or char_budget < 0:
|
||
raise WriterReferenceWorkError("proseCharBudget 必须是非负整数")
|
||
if char_budget == 0:
|
||
return []
|
||
baseline = set(range(max(1, as_of - 3), as_of + 1))
|
||
selected_refs: list[dict[str, Any]] = []
|
||
available_chars = 0
|
||
# 最近历史优先是通用时间邻近策略,不读取卡、里程碑或卡派生实体。
|
||
for chapter in range(as_of, 0, -1):
|
||
if chapter in baseline:
|
||
continue
|
||
row = block_index.get(chapter)
|
||
if row is None:
|
||
raise WriterReferenceWorkError(f"A 臂通用历史检索缺少第 {chapter} 章")
|
||
selected_refs.append(_block_source_ref(row))
|
||
available_chars += len(str(row["content_text"]))
|
||
if available_chars >= char_budget:
|
||
break
|
||
if available_chars < char_budget:
|
||
raise WriterReferenceWorkError(
|
||
f"A 臂通用历史原文不足: expected={char_budget}, actual={available_chars}"
|
||
)
|
||
repository = SnapshotProseRepository(block_index)
|
||
prose = repository.read_source_refs(
|
||
work_id=work_id,
|
||
as_of=as_of,
|
||
source_refs=selected_refs,
|
||
)
|
||
for item in prose:
|
||
item["purpose"] = "generic_historical_prose"
|
||
item["retrievalArm"] = "A"
|
||
return prose
|
||
|
||
|
||
def _required_block_chapters(
|
||
base_config: Mapping[str, Any],
|
||
selected: Mapping[str, Sequence[Mapping[str, Any]]],
|
||
) -> set[int]:
|
||
"""在正文查询前计算固定章集合,保证 SQL 不会读取目标章。"""
|
||
|
||
required: set[int] = set()
|
||
for sample in base_config.get("samples", []):
|
||
sample_id = str(sample.get("sampleId") or "")
|
||
target = _positive_chapter(sample.get("targetChapter"), f"{sample_id}.targetChapter")
|
||
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
if target != as_of + 1:
|
||
raise WriterReferenceWorkError(f"{sample_id} targetChapter 必须等于 asOfChapter+1")
|
||
required.update(range(max(1, as_of - 3), as_of + 1))
|
||
for row in selected.get(sample_id, []):
|
||
projected = project_card(
|
||
row,
|
||
as_of=as_of,
|
||
source_version=f"prequery:card-{row.get('id')}",
|
||
)
|
||
refs = projected.get("sourceRefs") or []
|
||
if refs:
|
||
for index, ref in enumerate(refs):
|
||
chapter = _positive_chapter(
|
||
ref.get("chapter"), f"{sample_id}.sourceRefs[{index}].chapter"
|
||
)
|
||
if chapter > as_of:
|
||
raise WriterReferenceWorkError(f"{sample_id} 卡引用包含目标章或未来章")
|
||
required.add(chapter)
|
||
else:
|
||
required.update(
|
||
milestone_reference_chapters(projected.get("milestones") or [], as_of=as_of)
|
||
)
|
||
return required
|
||
|
||
|
||
def load_writer_reference_rows(
|
||
*,
|
||
dsn: str,
|
||
tenant_id: int,
|
||
work_id: int,
|
||
targets: Sequence[int],
|
||
base_config: Mapping[str, Any],
|
||
selector_config: Mapping[str, Any],
|
||
selector_digest: str,
|
||
) -> dict[str, Any]:
|
||
"""在同一只读可重复读事务读取五章装配所需全部数据。"""
|
||
|
||
_validate_loader_controls(
|
||
base_config,
|
||
selector_config,
|
||
selector_digest=selector_digest,
|
||
)
|
||
character_probes = _required_character_probes(base_config)
|
||
normalized_targets = sorted({_positive_chapter(item, "targets[]") for item in targets})
|
||
if work_id != 8 or int(selector_config.get("workId") or 0) != work_id:
|
||
raise WriterReferenceWorkError("Writer Gate A 只允许预注册 work=8")
|
||
if not normalized_targets:
|
||
raise WriterReferenceWorkError("目标章集合不能为空")
|
||
selector_targets = sorted(
|
||
_positive_chapter(item.get("targetChapter"), f"{item['sampleId']}.targetChapter")
|
||
for item in _selector_samples(selector_config)
|
||
)
|
||
if selector_targets != normalized_targets:
|
||
raise WriterReferenceWorkError("调用目标章必须与稳定选择器完全一致")
|
||
selector_names = sorted(
|
||
{
|
||
str(card["name"])
|
||
for sample in _selector_samples(selector_config)
|
||
for card in sample["cards"]
|
||
}
|
||
)
|
||
selector_types = sorted(
|
||
{
|
||
str(card["type"])
|
||
for sample in _selector_samples(selector_config)
|
||
for card in sample["cards"]
|
||
}
|
||
)
|
||
|
||
with psycopg.connect(dsn, row_factory=dict_row) as conn:
|
||
begin_read_snapshot(conn)
|
||
work = conn.execute(
|
||
"""
|
||
SELECT id,title,revision,chapter_count,parse_status,import_status
|
||
FROM muse_content_work
|
||
WHERE tenant_id=%s AND id=%s AND deleted=FALSE
|
||
""",
|
||
(tenant_id, work_id),
|
||
).fetchone()
|
||
reference_rows = conn.execute(
|
||
"""
|
||
SELECT id,work_id,declared_chapter_count,imported_chapter_count,
|
||
parse_scope,parse_status,source_file,notes,update_time,deleted
|
||
FROM example_reference_work
|
||
WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, work_id),
|
||
).fetchall()
|
||
if work is None or len(reference_rows) != 1:
|
||
raise WriterReferenceWorkError("作品或唯一参考作品登记不存在")
|
||
reference = reference_rows[0]
|
||
import_task_rows = conn.execute(
|
||
"""
|
||
SELECT id,status,command_id,source_snapshot,deleted
|
||
FROM muse_content_import_task
|
||
WHERE tenant_id=%s AND work_id=%s AND status='succeeded' AND deleted=FALSE
|
||
AND source_snapshot->>'file'=%s
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, work_id, reference.get("source_file")),
|
||
).fetchall()
|
||
document_rows = conn.execute(
|
||
"""
|
||
SELECT id,file_name,file_hash,deleted
|
||
FROM muse_knowledge_document
|
||
WHERE tenant_id=%s AND file_name=%s AND deleted=FALSE
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, reference.get("source_file")),
|
||
).fetchall()
|
||
source = validate_source_records(reference_rows, import_task_rows, document_rows)
|
||
authorization_row = conn.execute(
|
||
"""
|
||
SELECT id,snapshot_version,source_hash,source_version,copyright_status,source_status,
|
||
allowed_purpose,forbidden_purpose,authorization_basis,authorized_by,
|
||
display_summary,checked_at,expires_at,revalidation_at
|
||
FROM example_reference_authorization_snapshot
|
||
WHERE tenant_id=%s AND work_id=%s AND source_version=%s
|
||
ORDER BY checked_at DESC,id DESC
|
||
LIMIT 1
|
||
""",
|
||
(tenant_id, work_id, source["sourceVersion"]),
|
||
).fetchone()
|
||
authorization = project_authorization(authorization_row, source)
|
||
|
||
target_scaffolds = conn.execute(
|
||
"""
|
||
SELECT s.id,s.chapter_id,ch.order_no AS chapter,ch.title,s.outline_text,
|
||
s.entities,s.pattern_hints
|
||
FROM example_parse_scaffold s
|
||
JOIN muse_content_chapter ch ON ch.id=s.chapter_id
|
||
WHERE s.tenant_id=%s AND s.work_id=%s AND s.deleted=FALSE
|
||
AND ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND ch.order_no=ANY(%s)
|
||
ORDER BY ch.order_no,s.id
|
||
""",
|
||
(tenant_id, work_id, tenant_id, work_id, normalized_targets),
|
||
).fetchall()
|
||
card_rows = conn.execute(
|
||
"""
|
||
SELECT id,status,source_type,source_id,revision,draft_payload,deleted
|
||
FROM muse_knowledge_draft
|
||
WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE
|
||
AND source_type='upgrade_book'
|
||
AND draft_payload->>'type'=ANY(%s)
|
||
AND (
|
||
draft_payload->>'名称'=ANY(%s)
|
||
OR COALESCE(draft_payload->'别名','[]'::jsonb) ?| %s
|
||
)
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, work_id, selector_types, selector_names, selector_names),
|
||
).fetchall()
|
||
selected = resolve_card_selectors(card_rows, selector_config)
|
||
character_mentions = conn.execute(
|
||
"""
|
||
WITH probes AS (
|
||
SELECT *
|
||
FROM unnest(%s::text[],%s::text[],%s::integer[])
|
||
AS probe(sample_id,name,as_of_chapter)
|
||
)
|
||
SELECT probe.sample_id,probe.name,probe.as_of_chapter,
|
||
MIN(ch.order_no) FILTER (WHERE b.id IS NOT NULL) AS first_chapter,
|
||
COALESCE(
|
||
ARRAY_AGG(DISTINCT ch.order_no ORDER BY ch.order_no)
|
||
FILTER (WHERE b.id IS NOT NULL),
|
||
ARRAY[]::integer[]
|
||
) AS hit_chapters
|
||
FROM probes probe
|
||
LEFT JOIN muse_content_chapter ch
|
||
ON ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND ch.status IN ('published','confirmed','canonical')
|
||
AND ch.order_no<=probe.as_of_chapter
|
||
LEFT JOIN muse_content_block b
|
||
ON b.chapter_id=ch.id AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
|
||
AND POSITION(probe.name IN b.content_text)>0
|
||
GROUP BY probe.sample_id,probe.name,probe.as_of_chapter
|
||
ORDER BY probe.sample_id,probe.name
|
||
""",
|
||
(
|
||
[str(item["sampleId"]) for item in character_probes],
|
||
[str(item["name"]) for item in character_probes],
|
||
[int(item["asOfChapter"]) for item in character_probes],
|
||
tenant_id,
|
||
work_id,
|
||
tenant_id,
|
||
work_id,
|
||
),
|
||
).fetchall()
|
||
required_chapters: set[int] = set()
|
||
for target in normalized_targets:
|
||
as_of = target - 1
|
||
required_chapters.update(range(max(1, as_of - 3), as_of + 1))
|
||
target_by_sample = {
|
||
str(item["sampleId"]): _positive_chapter(
|
||
item.get("targetChapter"), f"{item['sampleId']}.targetChapter"
|
||
)
|
||
for item in selector_config["samples"]
|
||
}
|
||
for sample_id, rows in selected.items():
|
||
as_of = target_by_sample[sample_id] - 1
|
||
for row in rows:
|
||
projected = project_card(
|
||
row,
|
||
as_of=as_of,
|
||
source_version=f"prequery:card-{row.get('id')}",
|
||
)
|
||
refs = projected.get("sourceRefs") or []
|
||
if refs:
|
||
required_chapters.update(
|
||
_positive_chapter(ref.get("chapter"), "card.sourceRef.chapter")
|
||
for ref in refs
|
||
)
|
||
else:
|
||
required_chapters.update(
|
||
milestone_reference_chapters(projected.get("milestones") or [], as_of=as_of)
|
||
)
|
||
if any(chapter in normalized_targets for chapter in required_chapters):
|
||
raise WriterReferenceWorkError("正文读取集合包含目标章")
|
||
block_rows = conn.execute(
|
||
"""
|
||
SELECT ch.order_no AS chapter,ch.id AS chapter_id,ch.status AS chapter_status,
|
||
b.id AS block_id,b.order_no AS block_order,b.revision,b.content_text
|
||
FROM muse_content_chapter ch
|
||
JOIN muse_content_block b ON b.chapter_id=ch.id
|
||
WHERE ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
|
||
AND ch.status IN ('published','confirmed','canonical')
|
||
AND ch.order_no=ANY(%s)
|
||
ORDER BY ch.order_no,b.order_no,b.id
|
||
""",
|
||
(tenant_id, work_id, tenant_id, work_id, sorted(required_chapters)),
|
||
).fetchall()
|
||
# oracle 必须覆盖每个样本冻结线内的全部 Canonical 历史,因此单独读取
|
||
# 1..max(asOf),但仍复用当前连接和同一个只读冻结事务。
|
||
oracle_block_rows = conn.execute(
|
||
"""
|
||
SELECT ch.order_no AS chapter,ch.id AS chapter_id,ch.status AS chapter_status,
|
||
b.id AS block_id,b.order_no AS block_order,b.revision,b.content_text
|
||
FROM muse_content_chapter ch
|
||
JOIN muse_content_block b ON b.chapter_id=ch.id
|
||
WHERE ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
|
||
AND ch.status IN ('published','confirmed','canonical')
|
||
AND ch.order_no<=%s
|
||
ORDER BY ch.order_no,b.order_no,b.id
|
||
""",
|
||
(tenant_id, work_id, tenant_id, work_id, max(normalized_targets) - 1),
|
||
).fetchall()
|
||
|
||
index_unique_canonical_blocks(block_rows, allowed_chapters=required_chapters)
|
||
try:
|
||
index_unique_canonical_blocks(
|
||
oracle_block_rows,
|
||
allowed_chapters=set(range(1, max(normalized_targets))),
|
||
)
|
||
except WriterReferenceWorkError as error:
|
||
raise WriterReferenceWorkError(f"oracle Canonical 历史缺章或不唯一: {error}") from error
|
||
_target_scaffold_index(target_scaffolds, set(normalized_targets))
|
||
_canonical_character_index(character_mentions, character_probes)
|
||
return {
|
||
"work": work,
|
||
"reference": reference,
|
||
"source": source,
|
||
"authorization": authorization,
|
||
"target_scaffolds": target_scaffolds,
|
||
"card_rows": [row for rows in selected.values() for row in rows],
|
||
"block_rows": block_rows,
|
||
"oracle_block_rows": oracle_block_rows,
|
||
"canonical_character_mentions": character_mentions,
|
||
}
|
||
|
||
|
||
def _default_pattern_card_searcher(
|
||
*, dsn: str, tenant_id: int
|
||
) -> Callable[..., list[dict[str, Any]]]:
|
||
"""惰性导入公共范式库检索器,返回签名 ``(intent, *, ttype, top)`` 的调用体。
|
||
|
||
WHY 惰性:离线测试与 dry-run 不应被迫加载数据库/嵌入依赖,也不能在装配时
|
||
真连库;只有生产入口 ``main`` 才显式取用本函数,把真实检索接入 C 臂。
|
||
"""
|
||
|
||
search_scripts = SCRIPT_DIR.parents[1] / "search-knowledge" / "scripts"
|
||
sys.path.insert(0, str(search_scripts))
|
||
from search import search_cards # noqa: E402 惰性导入,避免模块级副作用
|
||
|
||
def _searcher(intent: str, *, ttype: str, top: int) -> list[dict[str, Any]]:
|
||
# 公共范式还在 draft 双轨,但必须走专用检索面;dsn/tenant 显式绑定本次 loader,
|
||
# 防止真实正文来自一套快照、范式却被默认常量带到另一库或另一租户。
|
||
return search_cards(
|
||
intent,
|
||
scope="public_pattern",
|
||
ttype=ttype,
|
||
purpose="generation",
|
||
top=top,
|
||
dsn=dsn,
|
||
tenant_id=tenant_id,
|
||
)
|
||
|
||
return _searcher
|
||
|
||
|
||
def _flatten_pattern_point(value: Any) -> str:
|
||
"""把范式卡字段值拍平成文本。WHY:writingPoints 合同是「字符串→字符串」,而
|
||
search_cards 的 visibleFields 值可能是列表/对象,统一拍平后才能过合同。"""
|
||
|
||
if isinstance(value, str):
|
||
return value
|
||
return json.dumps(value, ensure_ascii=False, sort_keys=True)
|
||
|
||
|
||
def _truncate_for_writer(text: str, max_chars: int) -> str:
|
||
"""按 code point 截断到上限以内,超长补一个省略号并重新 NFC 归一化。
|
||
|
||
WHY:截断可能落在组合字符边界、导致结果不再是 NFC,而合同 _string 会复核
|
||
value == NFC(value);因此截断后必须再归一化一次,保证产出永远过得了合同。
|
||
"""
|
||
|
||
if len(text) <= max_chars:
|
||
return text
|
||
return normalize_text(text[: max(0, max_chars - 1)] + "…")
|
||
|
||
|
||
def _pattern_content_projection(card: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""把 search_cards 的 name/summary/visibleFields 投影为合同允许的限量内容字段。
|
||
|
||
WHY(SoT 变更):写手要真正读到范式卡的名字、一句话摘要和写法要点,而不只是一个
|
||
来源标签;但 visibleFields 原始字段可能长达数千字,直接灌入会撑爆写手上下文预算,
|
||
因此逐字段截断、只取前若干个字段。上限与合同(writer_contract._pattern_source_ref)
|
||
共用同一组常量,合同侧再失败关闭复核,双重保证体量受控。
|
||
"""
|
||
|
||
content: dict[str, Any] = {}
|
||
name = normalize_text(str(card.get("name") or "")).strip()
|
||
if name:
|
||
content["name"] = _truncate_for_writer(name, PATTERN_NAME_MAX_CHARS)
|
||
summary = normalize_text(str(card.get("summary") or "")).strip()
|
||
if summary:
|
||
content["summary"] = _truncate_for_writer(summary, PATTERN_SUMMARY_MAX_CHARS)
|
||
visible = card.get("visibleFields")
|
||
if isinstance(visible, Mapping):
|
||
points: dict[str, str] = {}
|
||
# visibleFields 来自库内 jsonb,键序确定;按序取前 N 个非空字段作为写法要点。
|
||
for key, value in visible.items():
|
||
if len(points) >= PATTERN_POINTS_MAX_FIELDS:
|
||
break
|
||
point_key = normalize_text(str(key)).strip()
|
||
point_value = normalize_text(_flatten_pattern_point(value)).strip()
|
||
if not point_key or not point_value:
|
||
continue
|
||
points[point_key] = _truncate_for_writer(point_value, PATTERN_POINT_MAX_CHARS)
|
||
if points:
|
||
content["writingPoints"] = points
|
||
return content
|
||
|
||
|
||
def _retrieve_pattern_references(
|
||
intent: str,
|
||
*,
|
||
card_searcher: Callable[..., list[dict[str, Any]]],
|
||
types: Sequence[str] = PATTERN_CARD_TYPES,
|
||
top_per_type: int = PATTERN_TOP_PER_TYPE,
|
||
total_cap: int = PATTERN_TOTAL_CAP,
|
||
) -> list[dict[str, Any]]:
|
||
"""按本章检索意图,从公共范式库五型各召回 top-k 卡,投影为写手合同 patternReferences。
|
||
|
||
WHY:Writer Gate A 的 C 臂要验证「范式指导是否提升质量」,需要把公共范式卡接入
|
||
写手输入。每张卡投影成 WriterContext v1 ``patternReferences``:来源指针
|
||
(sourceId / sourceVersion / sourceType,保证可回读可审计)**外加内容字段**
|
||
(name/summary/writingPoints,保证写手真正读到范式卡的名字、摘要与写法要点)。
|
||
内容字段经 ``_pattern_content_projection`` 截断到合同上限以内,确保通过
|
||
``validate_writer_context`` 的 ``_pattern_source_ref`` 校验。search_cards 已直接给出
|
||
稳定的 ``sourceId``(draft:{id})与 ``sourceVersion``(draft-revision:{n}),正好复用。
|
||
|
||
总量受控:每型最多 top_per_type 张,且累计不超过 total_cap,避免撑爆上下文预算。
|
||
"""
|
||
|
||
intent_text = normalize_text(str(intent or "")).strip()
|
||
if not intent_text:
|
||
# 没有检索意图(细纲为空)就不召回,失败关闭而非注入空引用。
|
||
return []
|
||
references: list[dict[str, Any]] = []
|
||
seen: set[tuple[str, str]] = set()
|
||
for card_type in types:
|
||
if len(references) >= total_cap:
|
||
break
|
||
# 剩余名额决定本型实际 top,保证累计严格不超过 total_cap。
|
||
top = min(top_per_type, total_cap - len(references))
|
||
if top <= 0:
|
||
break
|
||
for card in card_searcher(intent_text, ttype=card_type, top=top):
|
||
# WHY: SQL 是第一道范围门,loader 仍只接受专用公共范式面标记为可用于生产
|
||
# 检索的行;fake/未来替换实现若漏做范围过滤,也不能把治理草稿注入写手。
|
||
if (
|
||
card.get("retrievalScope") != "public_pattern"
|
||
or card.get("productionRetrievalEligible") is not True
|
||
or card.get("sourceKind") != "draft"
|
||
):
|
||
continue
|
||
source_id = normalize_text(str(card.get("sourceId") or "")).strip()
|
||
source_version = normalize_text(str(card.get("sourceVersion") or "")).strip()
|
||
if not source_id or not source_version:
|
||
# 缺稳定来源指针的卡不能进冻结上下文,跳过而非混入空引用。
|
||
continue
|
||
key = (source_version, source_id)
|
||
if key in seen:
|
||
# 跨型去重:同一张卡只注入一次。
|
||
continue
|
||
seen.add(key)
|
||
card_kind = normalize_text(str(card.get("type") or card_type)).strip() or card_type
|
||
references.append(
|
||
{
|
||
"sourceId": source_id,
|
||
"sourceVersion": source_version,
|
||
# sourceType 会成为写手最终看到的 kind;用范式卡的型作标识。
|
||
"sourceType": card_kind,
|
||
# SoT 变更:内容字段(名字/摘要/写法要点)随来源指针一起注入,写手
|
||
# 才能真正读到范式卡;此前只有上面三个指针字段,写手只见一个空标签。
|
||
**_pattern_content_projection(card),
|
||
}
|
||
)
|
||
if len(references) >= total_cap:
|
||
break
|
||
return references
|
||
|
||
|
||
def _pattern_references_for_arm(arm: str, c_references: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
||
"""装配端分臂的薄封装:语义唯一事实源在 writer_contract.pattern_references_for_arm。
|
||
|
||
WHY:装配与回放是两段独立 assemble 的链路,必须按完全相同的规则分臂(A 恒空 /
|
||
其余臂拿 C 候选),否则 C 臂真写读不到范式卡或 A 臂混入范式卡。判定逻辑一律走
|
||
合同模块,不在装配端另写一套;保留这个私有入口只为兼容既有离线测试的导入面。
|
||
"""
|
||
|
||
return pattern_references_for_arm(arm, c_references)
|
||
|
||
|
||
def assemble_writer_gate_config(
|
||
*,
|
||
base_config: Mapping[str, Any],
|
||
selector_config: Mapping[str, Any],
|
||
selector_digest: str,
|
||
rows: Mapping[str, Any],
|
||
pattern_card_searcher: Callable[..., list[dict[str, Any]]] | None = None,
|
||
) -> dict[str, Any]:
|
||
"""把同一事务快照装配成 canonical_frozen_prose 五样本配置。
|
||
|
||
``pattern_card_searcher`` 为 None 时不注入范式卡(A/C 两臂 patternReferences 均空),
|
||
保持历史行为与离线测试的零数据库依赖;生产入口显式传入真实检索器才启用 C 臂注入。
|
||
"""
|
||
|
||
common_controls = _validate_loader_controls(
|
||
base_config,
|
||
selector_config,
|
||
selector_digest=selector_digest,
|
||
)
|
||
if not isinstance(base_config, Mapping) or base_config.get("profile") != "writer_replay":
|
||
raise WriterReferenceWorkError("基础配置 profile 必须是 writer_replay")
|
||
samples = base_config.get("samples")
|
||
if not isinstance(samples, list) or not samples:
|
||
raise WriterReferenceWorkError("基础配置 samples 不能为空")
|
||
sample_ids = [str(item.get("sampleId") or "") for item in samples]
|
||
selector_samples = _selector_samples(selector_config)
|
||
selector_ids = [str(item["sampleId"]) for item in selector_samples]
|
||
if sample_ids != selector_ids:
|
||
raise WriterReferenceWorkError("稳定卡选择器样本顺序必须与预注册配置完全一致")
|
||
for sample, selector in zip(samples, selector_samples, strict=True):
|
||
if sample.get("targetChapter") != selector.get("targetChapter"):
|
||
raise WriterReferenceWorkError(
|
||
f"{sample.get('sampleId')} targetChapter 未绑定稳定选择器"
|
||
)
|
||
if selector_config.get("evaluationSetVersion") != base_config.get("evaluationSetVersion"):
|
||
raise WriterReferenceWorkError("卡选择器 evaluationSetVersion 未绑定预注册配置")
|
||
work_id = int(selector_config.get("workId") or 0)
|
||
if work_id != 8 or base_config.get("referenceWork", {}).get("id") != work_id:
|
||
raise WriterReferenceWorkError("基础配置与卡选择器必须共同绑定 work=8")
|
||
|
||
selected = resolve_card_selectors(rows.get("card_rows", []), selector_config)
|
||
selector_by_sample = {str(item["sampleId"]): item for item in selector_samples}
|
||
targets = {_positive_chapter(item.get("targetChapter"), "sample.targetChapter") for item in samples}
|
||
scaffolds = _target_scaffold_index(rows.get("target_scaffolds", []), targets)
|
||
required_chapters = _required_block_chapters(base_config, selected)
|
||
block_index = index_unique_canonical_blocks(
|
||
rows.get("block_rows", []), allowed_chapters=required_chapters
|
||
)
|
||
source = rows.get("source")
|
||
authorization = rows.get("authorization")
|
||
work = rows.get("work")
|
||
if not all(isinstance(item, Mapping) for item in (source, authorization, work)):
|
||
raise WriterReferenceWorkError("作品、来源或授权投影缺失")
|
||
source_version = str(source.get("sourceVersion") or "")
|
||
if not source_version.startswith("raw-file-v1:sha256:"):
|
||
raise WriterReferenceWorkError("真实 Writer 配置必须绑定原文件版本")
|
||
|
||
config = copy.deepcopy(dict(base_config))
|
||
# 盖戳只作用于深拷贝出的输出配置:补 rawRetention 的运行期租约到期时间戳,
|
||
# base_config(及其 _validate_loader_controls 已校验的自哈希)不受影响。
|
||
_stamp_raw_retention(config)
|
||
config["referenceWork"] = {
|
||
"id": work_id,
|
||
"title": str(work.get("title") or ""),
|
||
"version": source_version,
|
||
}
|
||
config["authorization"] = copy.deepcopy(dict(authorization))
|
||
config["evaluationPreregistration"] = build_balanced_preregistration(
|
||
evaluation_set_version=str(base_config.get("evaluationSetVersion") or ""),
|
||
sample_ids=sample_ids,
|
||
)
|
||
assembled_samples: list[dict[str, Any]] = []
|
||
oracle_packs: dict[str, dict[str, Any]] = {}
|
||
diff_receipts: dict[str, dict[str, Any]] = {}
|
||
oracle_block_rows = rows.get("oracle_block_rows")
|
||
if not isinstance(oracle_block_rows, list):
|
||
raise WriterReferenceWorkError("oracle Canonical 全量历史缺失")
|
||
max_as_of = max(targets) - 1
|
||
try:
|
||
oracle_index = index_unique_canonical_blocks(
|
||
oracle_block_rows,
|
||
allowed_chapters=set(range(1, max_as_of + 1)),
|
||
)
|
||
except WriterReferenceWorkError as error:
|
||
raise WriterReferenceWorkError(f"oracle Canonical 历史缺章或不唯一: {error}") from error
|
||
prose_repository = SnapshotProseRepository(block_index)
|
||
top_snapshot = authorization.get("authorizationSnapshot")
|
||
if not isinstance(top_snapshot, Mapping):
|
||
raise WriterReferenceWorkError("授权缺少不可变 authorizationSnapshot")
|
||
character_index = _canonical_character_index(
|
||
rows.get("canonical_character_mentions", []),
|
||
_required_character_probes(base_config),
|
||
)
|
||
|
||
for raw_sample in samples:
|
||
sample = copy.deepcopy(dict(raw_sample))
|
||
sample_id = str(sample["sampleId"])
|
||
target = _positive_chapter(sample.get("targetChapter"), f"{sample_id}.targetChapter")
|
||
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
if target != as_of + 1:
|
||
raise WriterReferenceWorkError(f"{sample_id} targetChapter 必须等于 asOfChapter+1")
|
||
scaffold = scaffolds[target]
|
||
outline_text = str(scaffold.get("outline_text") or "").strip()
|
||
if not outline_text:
|
||
raise WriterReferenceWorkError(f"{sample_id} 目标 scaffold 为空")
|
||
oracle_packs[sample_id] = build_oracle_truth_pack(
|
||
evaluation_set_version=str(base_config.get("evaluationSetVersion") or ""),
|
||
sample_id=sample_id,
|
||
work_id=work_id,
|
||
as_of=as_of,
|
||
target_chapter=target,
|
||
source_version=source_version,
|
||
authorization=authorization,
|
||
oracle_block_rows=oracle_block_rows,
|
||
scaffold=scaffold,
|
||
)
|
||
recent_rows = [block_index[chapter] for chapter in range(max(1, as_of - 3), as_of + 1)]
|
||
actual_counts = [han_count(str(row["content_text"])) for row in recent_rows]
|
||
expected_counts = sample.get("frozenRecentHanCounts")
|
||
if actual_counts != expected_counts:
|
||
raise WriterReferenceWorkError(
|
||
f"{sample_id} 冻结近章 Han 计数漂移: expected={expected_counts}, actual={actual_counts}"
|
||
)
|
||
|
||
projected_cards = [
|
||
_project_card_for_sample(
|
||
row,
|
||
as_of=as_of,
|
||
source_version=source_version,
|
||
block_index=block_index,
|
||
source_aliases=selector_by_sample[sample_id]["cards"][index].get(
|
||
"sourceAliases", []
|
||
),
|
||
)
|
||
for index, row in enumerate(selected[sample_id])
|
||
]
|
||
fine_outline = {
|
||
"sourceRef": {
|
||
"sourceId": f"scaffold:{scaffold.get('id')}",
|
||
"sourceVersion": source_version,
|
||
"chapter": target,
|
||
},
|
||
# 大纲文字之外,把门禁会查的要求清单 surface 为硬约束,让写手看到门禁查什么。
|
||
"hardConstraints": [outline_text]
|
||
+ _requirement_hard_constraints(
|
||
sample.get("writerContextInput", {}).get("requirements", {})
|
||
),
|
||
"adjustableBeats": copy.deepcopy(
|
||
sample.get("writerContextInput", {}).get("fineOutline", {}).get(
|
||
"adjustableBeats", []
|
||
)
|
||
),
|
||
"declaredNewFacts": [],
|
||
"entities": _selected_card_entities(projected_cards),
|
||
}
|
||
token_budget = _context_token_budget(
|
||
sample.get("writerContextInput", {}).get("tokenBudget", {}),
|
||
preregistered_max=int(common_controls["maxContextChars"]),
|
||
)
|
||
prose_char_budget = sample.get("proseCharBudget")
|
||
if (
|
||
isinstance(prose_char_budget, bool)
|
||
or not isinstance(prose_char_budget, int)
|
||
or prose_char_budget <= 0
|
||
):
|
||
raise WriterReferenceWorkError(
|
||
f"{sample_id}.proseCharBudget 必须是预注册正整数"
|
||
)
|
||
plan = build_retrieval_plan(
|
||
run_id=f"writer-loader:{sample_id}",
|
||
work_id=work_id,
|
||
target_chapter=target,
|
||
as_of=as_of,
|
||
fine_outline=fine_outline,
|
||
card_index_version=f"upgrade-book:{source_version}",
|
||
prose_index_version=source_version,
|
||
token_budget=token_budget,
|
||
)
|
||
sources = _sample_sources(recent_rows, projected_cards, as_of=as_of)
|
||
leakage = {
|
||
"method": "target-scaffold-proxy-and-chapter-bound-audit",
|
||
"targetFacts": _target_facts_from_scaffold(scaffold, target),
|
||
}
|
||
replay_repository = ReplayCardIndexRepository.from_replay_config(
|
||
{
|
||
"targetChapter": target,
|
||
"snapshot": {
|
||
"asOfChapter": as_of,
|
||
"snapshotVersion": f"writer-gate-a-canonical-{target}-v1",
|
||
"data": {
|
||
"chapters": [
|
||
{
|
||
"chapter": int(row["chapter"]),
|
||
"sourceId": _block_source_ref(row)["sourceId"],
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(
|
||
str(row["content_text"]).encode("utf-8")
|
||
).hexdigest(),
|
||
}
|
||
for row in recent_rows
|
||
],
|
||
"cards": [],
|
||
},
|
||
},
|
||
"authorization": authorization,
|
||
"sources": sources,
|
||
"leakageAudit": leakage,
|
||
},
|
||
cards=projected_cards,
|
||
preregistered_card_ids=[str(card["cardId"]) for card in projected_cards],
|
||
)
|
||
retrieval_result = retrieve_writer_sources(
|
||
plan=plan,
|
||
card_repository=replay_repository,
|
||
prose_repository=prose_repository,
|
||
)
|
||
# 既有检索器会为带 sourceRefs 的卡附加卡摘要事实。Writer Gate A 明确把卡
|
||
# 限定为索引,因此真实配置只保留索引提示和回读原文,不把摘要升级为权威事实。
|
||
retrieval_result["factEvidence"] = []
|
||
card_prose = []
|
||
for item in retrieval_result.get("proseEvidence", []):
|
||
projected = copy.deepcopy(dict(item))
|
||
projected["retrievalArm"] = "C"
|
||
card_prose.append(projected)
|
||
generic_prose = _generic_historical_prose(
|
||
oracle_index,
|
||
as_of=as_of,
|
||
work_id=work_id,
|
||
char_budget=prose_char_budget,
|
||
)
|
||
retrieval_result["proseEvidence"] = [*generic_prose, *card_prose]
|
||
assembly_token_budget = {
|
||
**token_budget,
|
||
"proseCharBudget": prose_char_budget,
|
||
}
|
||
|
||
context_input = copy.deepcopy(dict(sample.get("writerContextInput") or {}))
|
||
context_input.update(
|
||
{
|
||
"contentMode": "canonical_frozen_prose",
|
||
"sourceVersion": source_version,
|
||
"sourceStatus": _context_source_status(authorization),
|
||
"authorizationSnapshot": {
|
||
"snapshotId": str(top_snapshot.get("id") or ""),
|
||
"allowedPurpose": "offline_evaluation",
|
||
"verifiedAt": str(top_snapshot.get("checkedAt") or ""),
|
||
"sourceVersion": source_version,
|
||
},
|
||
"generatedAt": str(top_snapshot.get("checkedAt") or ""),
|
||
"fineOutline": fine_outline,
|
||
"narrativeState": {
|
||
"time": "",
|
||
"location": "",
|
||
"characterPositions": {},
|
||
"immediateSituation": "",
|
||
},
|
||
"recentChapters": _recent_chapter_input(recent_rows),
|
||
"retrievalResult": retrieval_result,
|
||
"retrievalQueries": copy.deepcopy(plan["queries"]),
|
||
"cardIndexVersion": plan["cardIndexVersion"],
|
||
"proseIndexVersion": plan["proseIndexVersion"],
|
||
"tokenBudget": assembly_token_budget,
|
||
}
|
||
)
|
||
# 检索意图取自写手细纲:硬约束(含大纲文字与门禁要求)+ 可调节拍。
|
||
# WHY:这两段是本章创作意图的最稠密表达,用它做向量检索能召回最贴合的范式卡。
|
||
pattern_intent = "\n".join(
|
||
[str(item) for item in fine_outline["hardConstraints"]]
|
||
+ [str(item) for item in fine_outline["adjustableBeats"]]
|
||
)
|
||
# 未提供检索器时为空,保持历史行为;提供时仅 C 臂经 _pattern_references_for_arm 取用。
|
||
c_pattern_references = (
|
||
_retrieve_pattern_references(pattern_intent, card_searcher=pattern_card_searcher)
|
||
if pattern_card_searcher is not None
|
||
else []
|
||
)
|
||
# 把 C 臂候选范式卡冻结进 writerContextInput.patternReferences,随 config.json 序列化。
|
||
# WHY:回放端(run_writer_replay)真写时会从 config.json 重新 assemble 各臂上下文;
|
||
# 若候选不写进 writerContextInput,C 臂真写就拿不到范式卡,实验失效。这里冻结全量
|
||
# 候选(含来源指针,只留在冻结上下文供审计回读),回放端读出后再经同一事实源
|
||
# pattern_references_for_arm 按臂分配——A 恒空,单变量规则两端只有一处定义。
|
||
context_input["patternReferences"] = _pattern_references_for_arm("C", c_pattern_references)
|
||
writer_contexts: dict[str, dict[str, Any]] = {}
|
||
try:
|
||
for arm, strategy in (
|
||
("A", "generic_prose_retrieval"),
|
||
("C", "card_indexed_prose_retrieval"),
|
||
):
|
||
writer_contexts[arm] = assemble_context(
|
||
run_id=str(plan["runId"]),
|
||
attempt=1,
|
||
mode="diagnostic_only",
|
||
purpose="evaluation",
|
||
quality_policy_version="writer-eval-v1",
|
||
work_id=work_id,
|
||
target_chapter=target,
|
||
as_of=as_of,
|
||
source_version=source_version,
|
||
authorization_snapshot=context_input["authorizationSnapshot"],
|
||
source_status=context_input["sourceStatus"],
|
||
retrieval_plan=plan,
|
||
retrieval_result=retrieval_result,
|
||
fine_outline=fine_outline,
|
||
narrative_state=context_input["narrativeState"],
|
||
recent_chapters=context_input["recentChapters"],
|
||
output_contract=context_input["outputContract"],
|
||
token_budget=assembly_token_budget,
|
||
pattern_references=_pattern_references_for_arm(arm, c_pattern_references),
|
||
generated_at=context_input["generatedAt"],
|
||
evidence_strategy=strategy,
|
||
)["context"]
|
||
diff_receipts[sample_id] = build_context_allowlist_diff_receipt(
|
||
sample_id=sample_id,
|
||
context_a=writer_contexts["A"],
|
||
context_c=writer_contexts["C"],
|
||
prose_char_budget=prose_char_budget,
|
||
require_nonempty=True,
|
||
)
|
||
except AssemblyError as error:
|
||
raise WriterReferenceWorkError(
|
||
f"{sample_id} A/C WriterContext 单变量校验失败: {error}"
|
||
) from error
|
||
sample.update(
|
||
{
|
||
"workId": work_id,
|
||
"targetTitle": str(scaffold.get("title") or sample.get("targetTitle") or ""),
|
||
"snapshotVersion": f"writer-gate-a-canonical-{target}-v1",
|
||
"snapshotData": {
|
||
"chapters": [
|
||
{
|
||
"chapter": int(row["chapter"]),
|
||
"sourceId": _block_source_ref(row)["sourceId"],
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(
|
||
str(row["content_text"]).encode("utf-8")
|
||
).hexdigest(),
|
||
}
|
||
for row in recent_rows
|
||
],
|
||
"cards": [],
|
||
},
|
||
"sources": sources,
|
||
"outlineSource": f"scaffold:{scaffold.get('id')}",
|
||
"fineOutlineSource": f"scaffold:{scaffold.get('id')}",
|
||
"writerContextInput": context_input,
|
||
"leakageAudit": leakage,
|
||
}
|
||
)
|
||
_recompute_new_character_ratio(sample, character_index)
|
||
assembled_samples.append(sample)
|
||
config["samples"] = assembled_samples
|
||
config["oracleTruthPacks"] = oracle_packs
|
||
config["writerContextDiffReceipts"] = diff_receipts
|
||
return config
|
||
|
||
|
||
def write_temporary_config(config: Mapping[str, Any], output_dir: Path) -> Path:
|
||
"""排他创建 /private/tmp 独立子目录并写入唯一完整配置。"""
|
||
|
||
resolved = output_dir.expanduser().resolve()
|
||
if resolved == PRIVATE_TMP or not resolved.is_relative_to(PRIVATE_TMP):
|
||
raise WriterReferenceWorkError("输出目录必须位于 /private/tmp 的独立子目录")
|
||
try:
|
||
resolved.mkdir(mode=0o700, parents=False, exist_ok=False)
|
||
except FileExistsError as error:
|
||
raise WriterReferenceWorkError("输出目录必须是尚不存在的独立子目录") from error
|
||
except FileNotFoundError as error:
|
||
raise WriterReferenceWorkError("输出目录父目录必须已存在") from error
|
||
config_path = resolved / "config.json"
|
||
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL
|
||
if hasattr(os, "O_NOFOLLOW"):
|
||
flags |= os.O_NOFOLLOW
|
||
descriptor = os.open(config_path, flags, 0o600)
|
||
try:
|
||
with os.fdopen(descriptor, "w", encoding="utf-8") as stream:
|
||
stream.write(_safe_json(config) + "\n")
|
||
except BaseException:
|
||
# fdopen 接管 descriptor;异常时只删除本次新建文件,不碰调用方目录。
|
||
config_path.unlink(missing_ok=True)
|
||
raise
|
||
return config_path
|
||
|
||
|
||
def _parse_args() -> argparse.Namespace:
|
||
"""解析真实 Writer Gate A 装配器参数。"""
|
||
|
||
parser = argparse.ArgumentParser(description="装配 Writer Gate A 五章真实临时配置")
|
||
parser.add_argument("--dsn", default=DSN)
|
||
parser.add_argument("--tenant-id", type=int, default=TENANT_ID)
|
||
parser.add_argument("--base-config", type=Path, default=DEFAULT_BASE_CONFIG)
|
||
parser.add_argument("--card-selectors", type=Path, default=DEFAULT_SELECTOR_CONFIG)
|
||
parser.add_argument("--output-dir", type=Path)
|
||
return parser.parse_args()
|
||
|
||
|
||
def main() -> int:
|
||
"""执行单事务只读装配并只回显安全摘要与临时配置路径。"""
|
||
|
||
args = _parse_args()
|
||
base_config = json.loads(args.base_config.read_text(encoding="utf-8"))
|
||
selector_bytes = args.card_selectors.read_bytes()
|
||
selectors = json.loads(selector_bytes.decode("utf-8"))
|
||
selector_digest = selector_sha256(selector_bytes)
|
||
samples = base_config.get("samples")
|
||
if not isinstance(samples, list):
|
||
raise WriterReferenceWorkError("基础配置 samples 非法")
|
||
targets = [_positive_chapter(item.get("targetChapter"), "sample.targetChapter") for item in samples]
|
||
rows = load_writer_reference_rows(
|
||
dsn=args.dsn,
|
||
tenant_id=args.tenant_id,
|
||
work_id=int(selectors.get("workId") or 0),
|
||
targets=targets,
|
||
base_config=base_config,
|
||
selector_config=selectors,
|
||
selector_digest=selector_digest,
|
||
)
|
||
config = assemble_writer_gate_config(
|
||
base_config=base_config,
|
||
selector_config=selectors,
|
||
selector_digest=selector_digest,
|
||
rows=rows,
|
||
# 生产装配才真连公共范式库:C 臂注入范式卡,A 臂保持空对照。
|
||
pattern_card_searcher=_default_pattern_card_searcher(
|
||
dsn=args.dsn,
|
||
tenant_id=args.tenant_id,
|
||
),
|
||
)
|
||
output_dir = args.output_dir or (
|
||
PRIVATE_TMP / f"writer-gate-a-{uuid.uuid4().hex}"
|
||
)
|
||
config_path = write_temporary_config(config, output_dir)
|
||
summary = {
|
||
"status": "ready_for_writer_dry_run",
|
||
"workId": config["referenceWork"]["id"],
|
||
"sampleCount": len(config["samples"]),
|
||
"contentMode": "canonical_frozen_prose",
|
||
"configPath": str(config_path),
|
||
}
|
||
print(json.dumps(summary, ensure_ascii=False, sort_keys=True))
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
raise SystemExit(main())
|
||
except (
|
||
WriterReferenceWorkError,
|
||
RetrievalError,
|
||
OSError,
|
||
json.JSONDecodeError,
|
||
psycopg.Error,
|
||
) as error:
|
||
print(
|
||
json.dumps(
|
||
{"status": "blocked_writer_reference_adapter", "error": str(error)},
|
||
ensure_ascii=False,
|
||
)
|
||
)
|
||
raise SystemExit(2)
|
||
|
||
|
||
__all__ = [
|
||
"WriterReferenceWorkError",
|
||
"SnapshotProseRepository",
|
||
"assemble_writer_gate_config",
|
||
"build_oracle_truth_pack",
|
||
"index_unique_canonical_blocks",
|
||
"load_writer_reference_rows",
|
||
"milestone_reference_chapters",
|
||
"resolve_card_selectors",
|
||
"selector_sha256",
|
||
"select_recent_canonical_blocks",
|
||
"validate_oracle_truth_pack",
|
||
"write_temporary_config",
|
||
]
|