1284 lines
55 KiB
Python
1284 lines
55 KiB
Python
#!/usr/bin/env python3
|
||
"""从实验库只读装配 Writer Gate A 五章真实临时配置。
|
||
|
||
本适配器只执行 SELECT,并把来源证明、授权、目标 scaffold、冻结近章正文和
|
||
预注册 upgrade_book 卡固定在同一个 REPEATABLE READ READ ONLY 事务中。
|
||
目标章正文不查询;目标 scaffold 只作为本层合法细纲和泄漏审计 proxy 使用。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import copy
|
||
import hashlib
|
||
import json
|
||
import sys
|
||
import uuid
|
||
from pathlib import Path
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
import psycopg
|
||
from psycopg.rows import dict_row
|
||
|
||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||
READ_CONTEXT_SCRIPTS = SCRIPT_DIR.parents[1] / "read-context" / "scripts"
|
||
sys.path.insert(0, str(READ_CONTEXT_SCRIPTS))
|
||
|
||
from build_snapshot import normalize_chapter_range # noqa: E402
|
||
from load_reference_work import ( # noqa: E402
|
||
DSN,
|
||
TENANT_ID,
|
||
AdapterError,
|
||
_target_facts_from_scaffold,
|
||
begin_read_snapshot,
|
||
project_authorization,
|
||
project_card,
|
||
validate_source_records,
|
||
)
|
||
from retrieve_writer_sources import ( # noqa: E402
|
||
ReplayCardIndexRepository,
|
||
RetrievalError,
|
||
build_retrieval_plan,
|
||
retrieve_writer_sources,
|
||
)
|
||
from writer_contract import han_count, normalize_text # noqa: E402
|
||
|
||
|
||
PRIVATE_TMP = Path("/private/tmp").resolve()
|
||
DEFAULT_BASE_CONFIG = SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-v1.json"
|
||
DEFAULT_SELECTOR_CONFIG = (
|
||
SCRIPT_DIR.parent / "configs" / "writer-gate-a-deep-space-card-selectors-v1.json"
|
||
)
|
||
CANONICAL_CHAPTER_STATUSES = frozenset({"published", "confirmed", "canonical"})
|
||
EXPECTED_INPUT_PROVENANCE = "oracle_reference_scaffold"
|
||
PREREGISTERED_MAX_CONTEXT_CHARS = 140_000
|
||
|
||
|
||
class WriterReferenceWorkError(AdapterError):
|
||
"""真实正文装配输入缺失、歧义、漂移或越过冻结线时抛出。"""
|
||
|
||
|
||
def _safe_json(value: Any) -> str:
|
||
"""稳定序列化临时配置,便于重复运行后比较哈希。"""
|
||
|
||
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
||
|
||
|
||
def _positive_chapter(value: Any, field: str) -> int:
|
||
"""严格接受正整数章号,不把 bool、浮点或模糊文本猜成章号。"""
|
||
|
||
if isinstance(value, bool):
|
||
raise WriterReferenceWorkError(f"{field} 必须是正整数章号")
|
||
if isinstance(value, str) and value.isdigit():
|
||
value = int(value)
|
||
if not isinstance(value, int) or value <= 0:
|
||
raise WriterReferenceWorkError(f"{field} 必须是正整数章号")
|
||
return value
|
||
|
||
|
||
def _card_payload(row: Mapping[str, Any]) -> Mapping[str, Any]:
|
||
"""读取卡 payload;选择器只信任结构化 type/name/alias。"""
|
||
|
||
payload = row.get("draft_payload")
|
||
if not isinstance(payload, Mapping):
|
||
raise WriterReferenceWorkError(f"卡 {row.get('id')} 缺少 draft_payload 对象")
|
||
return payload
|
||
|
||
|
||
def _selector_samples(selector_config: Mapping[str, Any]) -> list[Mapping[str, Any]]:
|
||
"""校验稳定选择器顶层结构和样本唯一性。"""
|
||
|
||
if not isinstance(selector_config, Mapping):
|
||
raise WriterReferenceWorkError("卡选择器配置必须是对象")
|
||
samples = selector_config.get("samples")
|
||
if not isinstance(samples, list) or not samples or any(
|
||
not isinstance(item, Mapping) for item in samples
|
||
):
|
||
raise WriterReferenceWorkError("卡选择器 samples 必须是非空对象数组")
|
||
sample_ids = [str(item.get("sampleId") or "") for item in samples]
|
||
if any(not item for item in sample_ids) or len(sample_ids) != len(set(sample_ids)):
|
||
raise WriterReferenceWorkError("卡选择器 sampleId 必须非空且唯一")
|
||
targets = [
|
||
_positive_chapter(item.get("targetChapter"), f"{item['sampleId']}.targetChapter")
|
||
for item in samples
|
||
]
|
||
if len(targets) != len(set(targets)):
|
||
raise WriterReferenceWorkError("卡选择器 targetChapter 必须唯一")
|
||
return samples
|
||
|
||
|
||
def selector_sha256(content: bytes) -> str:
|
||
"""对选择器规范文件原始字节计算带算法前缀的 SHA-256。"""
|
||
|
||
if not isinstance(content, bytes) or not content:
|
||
raise WriterReferenceWorkError("卡选择器规范文件不能为空")
|
||
return "sha256:" + hashlib.sha256(content).hexdigest()
|
||
|
||
|
||
def _validate_loader_controls(
|
||
base_config: Mapping[str, Any],
|
||
selector_config: Mapping[str, Any],
|
||
*,
|
||
selector_digest: str,
|
||
) -> Mapping[str, Any]:
|
||
"""在读取数据库前校验预注册选择器、输入来源和统一上下文上限。"""
|
||
|
||
if not isinstance(base_config, Mapping):
|
||
raise WriterReferenceWorkError("基础配置必须是对象")
|
||
common = base_config.get("commonControls")
|
||
if not isinstance(common, Mapping):
|
||
raise WriterReferenceWorkError("基础配置缺少 commonControls")
|
||
selector_version = str(selector_config.get("selectorVersion") or "")
|
||
if not selector_version or common.get("selectorVersion") != selector_version:
|
||
raise WriterReferenceWorkError("selectorVersion 未与预注册公共控制绑定")
|
||
if common.get("selectorSha256") != selector_digest:
|
||
raise WriterReferenceWorkError("选择器规范 JSON 的 SHA-256 与预注册公共控制不一致")
|
||
if common.get("inputProvenance") != EXPECTED_INPUT_PROVENANCE:
|
||
raise WriterReferenceWorkError("inputProvenance 必须固定为 oracle_reference_scaffold")
|
||
max_context_chars = common.get("maxContextChars")
|
||
if max_context_chars != PREREGISTERED_MAX_CONTEXT_CHARS:
|
||
raise WriterReferenceWorkError(
|
||
"commonControls.maxContextChars 必须严格等于预注册固定值 140000"
|
||
)
|
||
samples = base_config.get("samples")
|
||
if not isinstance(samples, list) or not samples:
|
||
raise WriterReferenceWorkError("基础配置 samples 不能为空")
|
||
for index, sample in enumerate(samples):
|
||
if not isinstance(sample, Mapping):
|
||
raise WriterReferenceWorkError(f"samples[{index}] 必须是对象")
|
||
configured = sample.get("writerContextInput", {}).get("tokenBudget", {})
|
||
if not isinstance(configured, Mapping):
|
||
raise WriterReferenceWorkError(f"samples[{index}] tokenBudget 必须是对象")
|
||
if configured.get("maxContextChars") != max_context_chars:
|
||
raise WriterReferenceWorkError(
|
||
f"samples[{index}] maxContextChars 必须原样使用预注册公共控制值"
|
||
)
|
||
return common
|
||
|
||
|
||
def _required_character_probes(base_config: Mapping[str, Any]) -> list[dict[str, Any]]:
|
||
"""只从冻结细纲要求提取具名角色;卡内角色状态不参与判定。"""
|
||
|
||
probes: list[dict[str, Any]] = []
|
||
for raw_sample in base_config.get("samples", []):
|
||
sample_id = str(raw_sample.get("sampleId") or "")
|
||
as_of = _positive_chapter(raw_sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
requirements = raw_sample.get("writerContextInput", {}).get("requirements", {})
|
||
basis = raw_sample.get("newCharacterBasis")
|
||
if not isinstance(requirements, Mapping) or not isinstance(basis, Mapping):
|
||
raise WriterReferenceWorkError(f"{sample_id} 缺少角色要求或冻结判定基线")
|
||
required = requirements.get("requiredCharacters")
|
||
generic = basis.get("genericRoles")
|
||
if (
|
||
not isinstance(required, list)
|
||
or not required
|
||
or any(not isinstance(name, str) or not name.strip() for name in required)
|
||
or len(required) != len(set(required))
|
||
):
|
||
raise WriterReferenceWorkError(f"{sample_id}.requiredCharacters 必须是无重复非空字符串数组")
|
||
if (
|
||
not isinstance(generic, list)
|
||
or any(not isinstance(name, str) or not name.strip() for name in generic)
|
||
or not set(generic).issubset(required)
|
||
):
|
||
raise WriterReferenceWorkError(f"{sample_id}.genericRoles 必须是 requiredCharacters 子集")
|
||
named = [name for name in required if name not in set(generic)]
|
||
if generic and named:
|
||
raise WriterReferenceWorkError(f"{sample_id} 暂不允许具名角色与泛称角色混合计算比例")
|
||
probes.extend(
|
||
{"sampleId": sample_id, "name": name, "asOfChapter": as_of}
|
||
for name in named
|
||
)
|
||
identities = [(item["sampleId"], item["name"]) for item in probes]
|
||
if len(identities) != len(set(identities)):
|
||
raise WriterReferenceWorkError("具名角色 Canonical 查询包含重复项")
|
||
return probes
|
||
|
||
|
||
def _canonical_character_index(
|
||
rows: Sequence[Mapping[str, Any]],
|
||
probes: Sequence[Mapping[str, Any]],
|
||
) -> dict[str, dict[str, dict[str, Any]]]:
|
||
"""校验 Canonical 正文命中结果,并按样本和角色名建立只含章号的索引。"""
|
||
|
||
expected = {
|
||
(str(item["sampleId"]), str(item["name"])): int(item["asOfChapter"])
|
||
for item in probes
|
||
}
|
||
indexed: dict[str, dict[str, dict[str, Any]]] = {}
|
||
seen: set[tuple[str, str]] = set()
|
||
for index, raw in enumerate(rows):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"canonical_character_mentions[{index}] 不是对象")
|
||
sample_id = str(raw.get("sample_id") or raw.get("sampleId") or "")
|
||
name = str(raw.get("name") or "")
|
||
identity = (sample_id, name)
|
||
if identity not in expected or identity in seen:
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果含额外项或重复项")
|
||
seen.add(identity)
|
||
as_of = _positive_chapter(
|
||
raw.get("as_of_chapter", raw.get("asOfChapter")),
|
||
f"{sample_id}.{name}.asOfChapter",
|
||
)
|
||
if as_of != expected[identity]:
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果冻结点漂移")
|
||
raw_hits = raw.get("hit_chapters", raw.get("hitChapters")) or []
|
||
if not isinstance(raw_hits, list):
|
||
raise WriterReferenceWorkError("Canonical 角色命中章必须是数组")
|
||
hits = sorted({_positive_chapter(item, f"{sample_id}.{name}.hitChapters") for item in raw_hits})
|
||
if any(chapter > as_of for chapter in hits):
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果越过冻结点")
|
||
raw_first = raw.get("first_chapter", raw.get("firstChapter"))
|
||
first = None if raw_first is None else _positive_chapter(raw_first, f"{sample_id}.{name}.firstChapter")
|
||
if first != (hits[0] if hits else None):
|
||
raise WriterReferenceWorkError("Canonical 角色首次命中章与命中章集合不一致")
|
||
indexed.setdefault(sample_id, {})[name] = {
|
||
"firstChapter": first,
|
||
"hitChapters": hits,
|
||
}
|
||
if seen != set(expected):
|
||
raise WriterReferenceWorkError("Canonical 角色命中结果缺少预注册具名角色")
|
||
return indexed
|
||
|
||
|
||
def _recompute_new_character_ratio(
|
||
sample: dict[str, Any],
|
||
character_index: Mapping[str, Mapping[str, Mapping[str, Any]]],
|
||
) -> None:
|
||
"""依据冻结 Canonical 正文重写具名角色分区;不读取或信任卡内容。"""
|
||
|
||
sample_id = str(sample.get("sampleId") or "")
|
||
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
requirements = sample.get("writerContextInput", {}).get("requirements", {})
|
||
basis = sample.get("newCharacterBasis")
|
||
if not isinstance(requirements, Mapping) or not isinstance(basis, Mapping):
|
||
raise WriterReferenceWorkError(f"{sample_id} 缺少角色比例输入")
|
||
required = list(requirements.get("requiredCharacters") or [])
|
||
generic = list(basis.get("genericRoles") or [])
|
||
if generic:
|
||
sample["newCharacterRatio"] = None
|
||
sample["newCharacterRatioStatus"] = "unresolved_generic_role"
|
||
sample["newCharacterBasis"] = {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": as_of,
|
||
"requiredCharacters": required,
|
||
"knownBeforeAsOf": [],
|
||
"absentBeforeAsOf": [],
|
||
"genericRoles": generic,
|
||
}
|
||
return
|
||
mentions = character_index.get(sample_id, {})
|
||
known = [name for name in required if mentions.get(name, {}).get("hitChapters")]
|
||
absent = [name for name in required if name not in known]
|
||
sample["newCharacterRatio"] = len(absent) / len(required)
|
||
sample["newCharacterRatioStatus"] = "resolved"
|
||
sample["newCharacterBasis"] = {
|
||
"definition": "named_required_characters_absent_before_as_of_ratio",
|
||
"asOfChapter": as_of,
|
||
"requiredCharacters": required,
|
||
"knownBeforeAsOf": known,
|
||
"absentBeforeAsOf": absent,
|
||
"genericRoles": [],
|
||
}
|
||
|
||
|
||
def resolve_card_selectors(
|
||
card_rows: Sequence[Mapping[str, Any]],
|
||
selector_config: Mapping[str, Any],
|
||
) -> dict[str, list[dict[str, Any]]]:
|
||
"""按 type + canonical name/alias 精确唯一解析预注册卡。
|
||
|
||
这里只读取稳定配置,不接收候选分数、回放结果或目标章正文,因此运行结果
|
||
不可能反向改变选卡。canonical name 和 alias 都是全字符串相等匹配。
|
||
"""
|
||
|
||
if not isinstance(card_rows, Sequence) or isinstance(card_rows, (str, bytes)):
|
||
raise WriterReferenceWorkError("卡查询结果必须是数组")
|
||
result: dict[str, list[dict[str, Any]]] = {}
|
||
used_card_ids: set[str] = set()
|
||
for sample in _selector_samples(selector_config):
|
||
sample_id = str(sample["sampleId"])
|
||
selectors = sample.get("cards")
|
||
if not isinstance(selectors, list) or not selectors or any(
|
||
not isinstance(item, Mapping) for item in selectors
|
||
):
|
||
raise WriterReferenceWorkError(f"{sample_id}.cards 必须是非空对象数组")
|
||
identities: list[tuple[str, str]] = []
|
||
selected: list[dict[str, Any]] = []
|
||
for index, selector in enumerate(selectors):
|
||
card_type = str(selector.get("type") or "").strip()
|
||
requested_name = str(selector.get("name") or "").strip()
|
||
if not card_type or not requested_name:
|
||
raise WriterReferenceWorkError(f"{sample_id}.cards[{index}] 缺少 type/name")
|
||
identity = (card_type, requested_name)
|
||
if identity in identities:
|
||
raise WriterReferenceWorkError(f"{sample_id} 含重复卡选择器 {identity}")
|
||
identities.append(identity)
|
||
|
||
matches: list[dict[str, Any]] = []
|
||
for raw_row in card_rows:
|
||
if not isinstance(raw_row, Mapping):
|
||
raise WriterReferenceWorkError("卡查询结果含非对象行")
|
||
payload = _card_payload(raw_row)
|
||
if str(payload.get("type") or "") != card_type:
|
||
continue
|
||
canonical_name = str(payload.get("名称") or "")
|
||
raw_aliases = payload.get("别名")
|
||
if raw_aliases is None:
|
||
aliases: list[str] = []
|
||
elif isinstance(raw_aliases, list) and all(
|
||
isinstance(alias, str) for alias in raw_aliases
|
||
):
|
||
aliases = raw_aliases
|
||
else:
|
||
raise WriterReferenceWorkError(f"卡 {raw_row.get('id')} 的别名不是字符串数组")
|
||
if canonical_name == requested_name or requested_name in aliases:
|
||
matches.append(copy.deepcopy(dict(raw_row)))
|
||
if len(matches) != 1:
|
||
raise WriterReferenceWorkError(
|
||
f"{sample_id} 选择器 ({card_type},{requested_name}) 必须唯一匹配,实际 {len(matches)} 张"
|
||
)
|
||
card_id = str(matches[0].get("id") or "")
|
||
if not card_id or card_id in used_card_ids:
|
||
raise WriterReferenceWorkError(f"卡 {card_id or '<empty>'} 被重复选择")
|
||
used_card_ids.add(card_id)
|
||
selected.append(matches[0])
|
||
result[sample_id] = selected
|
||
return result
|
||
|
||
|
||
def milestone_reference_chapters(
|
||
milestones: Sequence[Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
limit: int = 3,
|
||
) -> list[int]:
|
||
"""展开明确里程碑区间,去重后返回冻结线内最近最多三章。"""
|
||
|
||
freeze = _positive_chapter(as_of, "as_of")
|
||
if isinstance(limit, bool) or not isinstance(limit, int) or limit <= 0:
|
||
raise WriterReferenceWorkError("里程碑引用上限必须是正整数")
|
||
chapters: set[int] = set()
|
||
for index, milestone in enumerate(milestones):
|
||
if not isinstance(milestone, Mapping):
|
||
raise WriterReferenceWorkError(f"milestones[{index}] 不是对象")
|
||
raw_chapter = next(
|
||
(
|
||
milestone[key]
|
||
for key in ("chapter", "chapter_no", "order_no", "章", "章号")
|
||
if key in milestone
|
||
),
|
||
None,
|
||
)
|
||
bounds = normalize_chapter_range(raw_chapter)
|
||
if bounds is None:
|
||
raise WriterReferenceWorkError(f"milestones[{index}] 缺少明确绝对章号边界")
|
||
# 卡可包含未来演变,但来源定位只展开完整落在冻结线内的里程碑。
|
||
if bounds[1] > freeze:
|
||
continue
|
||
chapters.update(range(bounds[0], bounds[1] + 1))
|
||
if not chapters:
|
||
raise WriterReferenceWorkError("卡在冻结线内没有可定位 Canonical 原文的里程碑")
|
||
return sorted(chapters)[-limit:]
|
||
|
||
|
||
def _normalize_card_milestones(card: dict[str, Any], *, as_of: int) -> None:
|
||
"""把区间里程碑状态归一到明确结束章,供既有冻结器严格消费。"""
|
||
|
||
normalized: list[dict[str, Any]] = []
|
||
for index, raw in enumerate(card.get("milestones") or []):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"卡 {card.get('cardId')} 里程碑[{index}] 非法")
|
||
raw_chapter = next(
|
||
(raw[key] for key in ("chapter", "chapter_no", "order_no", "章", "章号") if key in raw),
|
||
None,
|
||
)
|
||
bounds = normalize_chapter_range(raw_chapter)
|
||
if bounds is None or bounds[1] > as_of:
|
||
raise WriterReferenceWorkError(f"卡 {card.get('cardId')} 含未冻结或无边界里程碑")
|
||
item = copy.deepcopy(dict(raw))
|
||
for alias in ("chapter_no", "order_no", "章", "章号"):
|
||
item.pop(alias, None)
|
||
item["chapter"] = bounds[1]
|
||
normalized.append(item)
|
||
normalized.sort(key=lambda item: (item["chapter"], str(item.get("id") or "")))
|
||
card["milestones"] = normalized
|
||
card["stateAsOf"] = copy.deepcopy(normalized)
|
||
|
||
|
||
def index_unique_canonical_blocks(
|
||
block_rows: Sequence[Mapping[str, Any]],
|
||
*,
|
||
allowed_chapters: set[int],
|
||
) -> dict[int, dict[str, Any]]:
|
||
"""校验每个请求章恰好一个 Canonical block,并拒绝额外未来章。"""
|
||
|
||
if not allowed_chapters:
|
||
raise WriterReferenceWorkError("Canonical block 请求章集合不能为空")
|
||
grouped: dict[int, list[dict[str, Any]]] = {}
|
||
for index, raw in enumerate(block_rows):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"block_rows[{index}] 不是对象")
|
||
chapter = _positive_chapter(raw.get("chapter"), f"block_rows[{index}].chapter")
|
||
if chapter not in allowed_chapters:
|
||
raise WriterReferenceWorkError(f"读取到未请求或目标/未来章正文: {chapter}")
|
||
if str(raw.get("chapter_status") or "") not in CANONICAL_CHAPTER_STATUSES:
|
||
raise WriterReferenceWorkError(f"第 {chapter} 章不是 Canonical 状态")
|
||
text = raw.get("content_text")
|
||
if not isinstance(text, str) or not text:
|
||
raise WriterReferenceWorkError(f"第 {chapter} 章 Canonical block 正文为空")
|
||
grouped.setdefault(chapter, []).append(copy.deepcopy(dict(raw)))
|
||
missing = sorted(allowed_chapters - set(grouped))
|
||
duplicates = sorted(chapter for chapter, rows in grouped.items() if len(rows) != 1)
|
||
if missing or duplicates:
|
||
raise WriterReferenceWorkError(
|
||
f"Canonical block 必须逐章唯一: missing={missing}, non_unique={duplicates}"
|
||
)
|
||
return {chapter: rows[0] for chapter, rows in grouped.items()}
|
||
|
||
|
||
def select_recent_canonical_blocks(
|
||
block_rows: Sequence[Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
) -> list[dict[str, Any]]:
|
||
"""按冻结点选择连续四章,缺章或重复 block 均失败关闭。"""
|
||
|
||
freeze = _positive_chapter(as_of, "as_of")
|
||
expected = list(range(max(1, freeze - 3), freeze + 1))
|
||
relevant = [row for row in block_rows if row.get("chapter") in expected]
|
||
indexed = index_unique_canonical_blocks(relevant, allowed_chapters=set(expected))
|
||
return [indexed[chapter] for chapter in expected]
|
||
|
||
|
||
def _block_source_ref(row: Mapping[str, Any]) -> dict[str, Any]:
|
||
"""为唯一 Canonical block 生成完整代码点区间来源引用。"""
|
||
|
||
chapter = _positive_chapter(row.get("chapter"), "block.chapter")
|
||
block_id = row.get("block_id")
|
||
if isinstance(block_id, bool) or not isinstance(block_id, int) or block_id <= 0:
|
||
raise WriterReferenceWorkError(f"第 {chapter} 章 block_id 非法")
|
||
text = str(row.get("content_text") or "")
|
||
revision = row.get("revision") or 0
|
||
source_version = f"chapter:{chapter}:block:{block_id}:revision:{revision}"
|
||
return {
|
||
"sourceId": f"chapter:{chapter}:block:{block_id}",
|
||
"sourceVersion": source_version,
|
||
"chapter": chapter,
|
||
"blockId": block_id,
|
||
"startCodePoint": 0,
|
||
"endCodePoint": len(text),
|
||
}
|
||
|
||
|
||
class SnapshotProseRepository:
|
||
"""只从同一数据库事务已冻结的 block 行展开来源引用。"""
|
||
|
||
def __init__(self, block_index: Mapping[int, Mapping[str, Any]]):
|
||
self._by_chapter = {
|
||
int(chapter): copy.deepcopy(dict(row)) for chapter, row in block_index.items()
|
||
}
|
||
self._by_block = {int(row["block_id"]): row for row in self._by_chapter.values()}
|
||
|
||
def read_source_refs(
|
||
self,
|
||
*,
|
||
work_id: int,
|
||
as_of: int,
|
||
source_refs: Sequence[Mapping[str, Any]],
|
||
) -> list[dict[str, Any]]:
|
||
"""逐引用校验章、块与代码点区间,不建立第二个数据库连接。"""
|
||
|
||
if work_id != 8:
|
||
raise WriterReferenceWorkError("Writer Gate A 只允许预注册 work=8")
|
||
freeze = _positive_chapter(as_of, "as_of")
|
||
result: list[dict[str, Any]] = []
|
||
for index, ref in enumerate(source_refs):
|
||
if not isinstance(ref, Mapping):
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 不是对象")
|
||
chapter = _positive_chapter(ref.get("chapter"), f"source_refs[{index}].chapter")
|
||
if chapter > freeze:
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 包含目标章或未来章")
|
||
block_id = ref.get("blockId")
|
||
row = self._by_block.get(block_id) if isinstance(block_id, int) else None
|
||
if row is None or int(row["chapter"]) != chapter:
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 不能定位唯一 Canonical block")
|
||
text = str(row["content_text"])
|
||
start = ref.get("startCodePoint")
|
||
end = ref.get("endCodePoint")
|
||
if (
|
||
isinstance(start, bool)
|
||
or isinstance(end, bool)
|
||
or not isinstance(start, int)
|
||
or not isinstance(end, int)
|
||
or start < 0
|
||
or end <= start
|
||
or end > len(text)
|
||
):
|
||
raise WriterReferenceWorkError(f"source_refs[{index}] 代码点区间越界")
|
||
fragment = text[start:end]
|
||
result.append(
|
||
{
|
||
"chapter": chapter,
|
||
"blockId": block_id,
|
||
"blockOrder": int(row.get("block_order") or 0),
|
||
"sourceRef": copy.deepcopy(dict(ref)),
|
||
"text": fragment,
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(fragment.encode("utf-8")).hexdigest(),
|
||
"purpose": str(ref.get("sourceType") or "card_source"),
|
||
}
|
||
)
|
||
return result
|
||
|
||
|
||
def _target_scaffold_index(
|
||
rows: Sequence[Mapping[str, Any]], targets: set[int]
|
||
) -> dict[int, dict[str, Any]]:
|
||
"""目标 scaffold 也要求逐章唯一,禁止 ORDER BY/LIMIT 猜选。"""
|
||
|
||
grouped: dict[int, list[dict[str, Any]]] = {}
|
||
for index, raw in enumerate(rows):
|
||
if not isinstance(raw, Mapping):
|
||
raise WriterReferenceWorkError(f"target_scaffolds[{index}] 不是对象")
|
||
chapter = _positive_chapter(raw.get("chapter"), f"target_scaffolds[{index}].chapter")
|
||
if chapter not in targets:
|
||
raise WriterReferenceWorkError(f"读取到未预注册目标 scaffold: {chapter}")
|
||
grouped.setdefault(chapter, []).append(copy.deepcopy(dict(raw)))
|
||
missing = sorted(targets - set(grouped))
|
||
duplicates = sorted(chapter for chapter, values in grouped.items() if len(values) != 1)
|
||
if missing or duplicates:
|
||
raise WriterReferenceWorkError(
|
||
f"目标 scaffold 必须逐章唯一: missing={missing}, non_unique={duplicates}"
|
||
)
|
||
return {chapter: values[0] for chapter, values in grouped.items()}
|
||
|
||
|
||
def _selected_card_entities(cards: Sequence[Mapping[str, Any]]) -> list[dict[str, str]]:
|
||
"""把稳定选择器结果转换为检索计划实体,不从运行结果追加查询。"""
|
||
|
||
return [
|
||
{
|
||
"id": f"selected-card:{card['cardId']}",
|
||
"type": str(card["type"]),
|
||
"name": str(card["name"]),
|
||
}
|
||
for card in cards
|
||
]
|
||
|
||
|
||
def _context_token_budget(
|
||
configured: Mapping[str, Any],
|
||
*,
|
||
preregistered_max: int,
|
||
) -> dict[str, int]:
|
||
"""原样使用预注册上限;基线超限由组装器失败,补充证据由组装器裁剪。"""
|
||
|
||
configured_max = configured.get("maxContextChars")
|
||
if (
|
||
isinstance(configured_max, bool)
|
||
or not isinstance(configured_max, int)
|
||
or configured_max <= 0
|
||
):
|
||
raise WriterReferenceWorkError("tokenBudget.maxContextChars 必须是正整数")
|
||
if configured_max != preregistered_max:
|
||
raise WriterReferenceWorkError("loader 禁止改写或扩张预注册 maxContextChars")
|
||
return {"maxContextChars": preregistered_max}
|
||
|
||
|
||
def _context_source_status(authorization: Mapping[str, Any]) -> str:
|
||
"""按既有 WriterContext 授权绑定规则投影运行期来源状态。"""
|
||
|
||
source_status = str(authorization.get("sourceStatus") or "").lower()
|
||
if source_status in {"active", "approved", "licensed"}:
|
||
return "active"
|
||
if source_status == "authorized":
|
||
return "authorized"
|
||
raise WriterReferenceWorkError(f"授权来源状态不能进入 WriterContext: {source_status}")
|
||
|
||
|
||
def _project_card_for_sample(
|
||
row: Mapping[str, Any],
|
||
*,
|
||
as_of: int,
|
||
source_version: str,
|
||
block_index: Mapping[int, Mapping[str, Any]],
|
||
source_aliases: Sequence[str] = (),
|
||
) -> dict[str, Any]:
|
||
"""冻结卡;缺精确引用时只生成可识别且显式降级的整章代理。"""
|
||
|
||
card_id = str(row.get("id") or "")
|
||
card_version = f"{source_version}:card-{card_id}-rev-{row.get('revision') or 0}"
|
||
projected = project_card(row, as_of=as_of, source_version=card_version)
|
||
milestones = copy.deepcopy(projected.get("milestones") or [])
|
||
uses_chapter_proxy = not projected.get("sourceRefs")
|
||
if uses_chapter_proxy:
|
||
chapters = milestone_reference_chapters(milestones, as_of=as_of)
|
||
try:
|
||
projected["sourceRefs"] = []
|
||
payload = _card_payload(row)
|
||
raw_aliases = payload.get("别名") or []
|
||
if not isinstance(raw_aliases, list) or any(
|
||
not isinstance(alias, str) for alias in raw_aliases
|
||
):
|
||
raise WriterReferenceWorkError(f"卡 {card_id} 的别名不是字符串数组")
|
||
legal_names = {
|
||
str(projected.get("name") or "").strip(),
|
||
*(alias.strip() for alias in raw_aliases),
|
||
*(str(alias).strip() for alias in source_aliases),
|
||
}
|
||
legal_names.discard("")
|
||
for chapter in chapters:
|
||
row_text = normalize_text(str(block_index[chapter]["content_text"]))
|
||
if not any(name in row_text for name in legal_names):
|
||
raise WriterReferenceWorkError(
|
||
f"卡 {card_id} 的整章代理第 {chapter} 章未出现规范名或合法别名"
|
||
)
|
||
ref = _block_source_ref(block_index[chapter])
|
||
ref["sourceType"] = "card_chapter_proxy"
|
||
projected["sourceRefs"].append(ref)
|
||
except KeyError as error:
|
||
raise WriterReferenceWorkError(
|
||
f"卡 {card_id} 的里程碑章 {error.args[0]} 缺少唯一 Canonical block"
|
||
) from error
|
||
for index, ref in enumerate(projected.get("sourceRefs") or []):
|
||
chapter = _positive_chapter(ref.get("chapter"), f"卡 {card_id}.sourceRefs[{index}].chapter")
|
||
if chapter > as_of:
|
||
raise WriterReferenceWorkError(f"卡 {card_id} sourceRef 包含目标章或未来章")
|
||
if ref.get("blockId") not in {row["block_id"] for row in block_index.values()}:
|
||
raise WriterReferenceWorkError(f"卡 {card_id} sourceRef 未绑定本次 Canonical 快照")
|
||
if uses_chapter_proxy and ref.get("sourceType") != "card_chapter_proxy":
|
||
raise WriterReferenceWorkError(f"卡 {card_id} 整章代理缺少降级来源类型")
|
||
_normalize_card_milestones(projected, as_of=as_of)
|
||
return projected
|
||
|
||
|
||
def _sample_sources(
|
||
recent_rows: Sequence[Mapping[str, Any]],
|
||
cards: Sequence[Mapping[str, Any]],
|
||
*,
|
||
as_of: int,
|
||
) -> list[dict[str, Any]]:
|
||
"""生成只含冻结历史的可验证来源目录,不登记目标 scaffold 为历史来源。"""
|
||
|
||
sources = [_block_source_ref(row) for row in recent_rows]
|
||
for card in cards:
|
||
sources.append(
|
||
{
|
||
"sourceId": str(card["sourceId"]),
|
||
"sourceVersion": str(card["sourceVersion"]),
|
||
"chapterRange": f"1-{as_of}",
|
||
"scope": "card_projection",
|
||
}
|
||
)
|
||
return sources
|
||
|
||
|
||
def _recent_chapter_input(rows: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
||
"""把连续四章唯一 block 投影成 WriterContext 的完整历史基线。"""
|
||
|
||
return [
|
||
{
|
||
"chapter": int(row["chapter"]),
|
||
"sourceRef": _block_source_ref(row),
|
||
"text": normalize_text(str(row["content_text"])),
|
||
}
|
||
for row in rows
|
||
]
|
||
|
||
|
||
def _required_block_chapters(
|
||
base_config: Mapping[str, Any],
|
||
selected: Mapping[str, Sequence[Mapping[str, Any]]],
|
||
) -> set[int]:
|
||
"""在正文查询前计算固定章集合,保证 SQL 不会读取目标章。"""
|
||
|
||
required: set[int] = set()
|
||
for sample in base_config.get("samples", []):
|
||
sample_id = str(sample.get("sampleId") or "")
|
||
target = _positive_chapter(sample.get("targetChapter"), f"{sample_id}.targetChapter")
|
||
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
if target != as_of + 1:
|
||
raise WriterReferenceWorkError(f"{sample_id} targetChapter 必须等于 asOfChapter+1")
|
||
required.update(range(max(1, as_of - 3), as_of + 1))
|
||
for row in selected.get(sample_id, []):
|
||
projected = project_card(
|
||
row,
|
||
as_of=as_of,
|
||
source_version=f"prequery:card-{row.get('id')}",
|
||
)
|
||
refs = projected.get("sourceRefs") or []
|
||
if refs:
|
||
for index, ref in enumerate(refs):
|
||
chapter = _positive_chapter(
|
||
ref.get("chapter"), f"{sample_id}.sourceRefs[{index}].chapter"
|
||
)
|
||
if chapter > as_of:
|
||
raise WriterReferenceWorkError(f"{sample_id} 卡引用包含目标章或未来章")
|
||
required.add(chapter)
|
||
else:
|
||
required.update(
|
||
milestone_reference_chapters(projected.get("milestones") or [], as_of=as_of)
|
||
)
|
||
return required
|
||
|
||
|
||
def load_writer_reference_rows(
|
||
*,
|
||
dsn: str,
|
||
tenant_id: int,
|
||
work_id: int,
|
||
targets: Sequence[int],
|
||
base_config: Mapping[str, Any],
|
||
selector_config: Mapping[str, Any],
|
||
selector_digest: str,
|
||
) -> dict[str, Any]:
|
||
"""在同一只读可重复读事务读取五章装配所需全部数据。"""
|
||
|
||
_validate_loader_controls(
|
||
base_config,
|
||
selector_config,
|
||
selector_digest=selector_digest,
|
||
)
|
||
character_probes = _required_character_probes(base_config)
|
||
normalized_targets = sorted({_positive_chapter(item, "targets[]") for item in targets})
|
||
if work_id != 8 or int(selector_config.get("workId") or 0) != work_id:
|
||
raise WriterReferenceWorkError("Writer Gate A 只允许预注册 work=8")
|
||
if not normalized_targets:
|
||
raise WriterReferenceWorkError("目标章集合不能为空")
|
||
selector_targets = sorted(
|
||
_positive_chapter(item.get("targetChapter"), f"{item['sampleId']}.targetChapter")
|
||
for item in _selector_samples(selector_config)
|
||
)
|
||
if selector_targets != normalized_targets:
|
||
raise WriterReferenceWorkError("调用目标章必须与稳定选择器完全一致")
|
||
selector_names = sorted(
|
||
{
|
||
str(card["name"])
|
||
for sample in _selector_samples(selector_config)
|
||
for card in sample["cards"]
|
||
}
|
||
)
|
||
selector_types = sorted(
|
||
{
|
||
str(card["type"])
|
||
for sample in _selector_samples(selector_config)
|
||
for card in sample["cards"]
|
||
}
|
||
)
|
||
|
||
with psycopg.connect(dsn, row_factory=dict_row) as conn:
|
||
begin_read_snapshot(conn)
|
||
work = conn.execute(
|
||
"""
|
||
SELECT id,title,revision,chapter_count,parse_status,import_status
|
||
FROM muse_content_work
|
||
WHERE tenant_id=%s AND id=%s AND deleted=FALSE
|
||
""",
|
||
(tenant_id, work_id),
|
||
).fetchone()
|
||
reference_rows = conn.execute(
|
||
"""
|
||
SELECT id,work_id,declared_chapter_count,imported_chapter_count,
|
||
parse_scope,parse_status,source_file,notes,update_time,deleted
|
||
FROM example_reference_work
|
||
WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, work_id),
|
||
).fetchall()
|
||
if work is None or len(reference_rows) != 1:
|
||
raise WriterReferenceWorkError("作品或唯一参考作品登记不存在")
|
||
reference = reference_rows[0]
|
||
import_task_rows = conn.execute(
|
||
"""
|
||
SELECT id,status,command_id,source_snapshot,deleted
|
||
FROM muse_content_import_task
|
||
WHERE tenant_id=%s AND work_id=%s AND status='succeeded' AND deleted=FALSE
|
||
AND source_snapshot->>'file'=%s
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, work_id, reference.get("source_file")),
|
||
).fetchall()
|
||
document_rows = conn.execute(
|
||
"""
|
||
SELECT id,file_name,file_hash,deleted
|
||
FROM muse_knowledge_document
|
||
WHERE tenant_id=%s AND file_name=%s AND deleted=FALSE
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, reference.get("source_file")),
|
||
).fetchall()
|
||
source = validate_source_records(reference_rows, import_task_rows, document_rows)
|
||
authorization_row = conn.execute(
|
||
"""
|
||
SELECT id,snapshot_version,source_hash,source_version,copyright_status,source_status,
|
||
allowed_purpose,forbidden_purpose,authorization_basis,authorized_by,
|
||
display_summary,checked_at,expires_at,revalidation_at
|
||
FROM example_reference_authorization_snapshot
|
||
WHERE tenant_id=%s AND work_id=%s AND source_version=%s
|
||
ORDER BY checked_at DESC,id DESC
|
||
LIMIT 1
|
||
""",
|
||
(tenant_id, work_id, source["sourceVersion"]),
|
||
).fetchone()
|
||
authorization = project_authorization(authorization_row, source)
|
||
|
||
target_scaffolds = conn.execute(
|
||
"""
|
||
SELECT s.id,s.chapter_id,ch.order_no AS chapter,ch.title,s.outline_text,
|
||
s.entities,s.pattern_hints
|
||
FROM example_parse_scaffold s
|
||
JOIN muse_content_chapter ch ON ch.id=s.chapter_id
|
||
WHERE s.tenant_id=%s AND s.work_id=%s AND s.deleted=FALSE
|
||
AND ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND ch.order_no=ANY(%s)
|
||
ORDER BY ch.order_no,s.id
|
||
""",
|
||
(tenant_id, work_id, tenant_id, work_id, normalized_targets),
|
||
).fetchall()
|
||
card_rows = conn.execute(
|
||
"""
|
||
SELECT id,status,source_type,source_id,revision,draft_payload,deleted
|
||
FROM muse_knowledge_draft
|
||
WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE
|
||
AND source_type='upgrade_book'
|
||
AND draft_payload->>'type'=ANY(%s)
|
||
AND (
|
||
draft_payload->>'名称'=ANY(%s)
|
||
OR COALESCE(draft_payload->'别名','[]'::jsonb) ?| %s
|
||
)
|
||
ORDER BY id
|
||
""",
|
||
(tenant_id, work_id, selector_types, selector_names, selector_names),
|
||
).fetchall()
|
||
selected = resolve_card_selectors(card_rows, selector_config)
|
||
character_mentions = conn.execute(
|
||
"""
|
||
WITH probes AS (
|
||
SELECT *
|
||
FROM unnest(%s::text[],%s::text[],%s::integer[])
|
||
AS probe(sample_id,name,as_of_chapter)
|
||
)
|
||
SELECT probe.sample_id,probe.name,probe.as_of_chapter,
|
||
MIN(ch.order_no) FILTER (WHERE b.id IS NOT NULL) AS first_chapter,
|
||
COALESCE(
|
||
ARRAY_AGG(DISTINCT ch.order_no ORDER BY ch.order_no)
|
||
FILTER (WHERE b.id IS NOT NULL),
|
||
ARRAY[]::integer[]
|
||
) AS hit_chapters
|
||
FROM probes probe
|
||
LEFT JOIN muse_content_chapter ch
|
||
ON ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND ch.status IN ('published','confirmed','canonical')
|
||
AND ch.order_no<=probe.as_of_chapter
|
||
LEFT JOIN muse_content_block b
|
||
ON b.chapter_id=ch.id AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
|
||
AND POSITION(probe.name IN b.content_text)>0
|
||
GROUP BY probe.sample_id,probe.name,probe.as_of_chapter
|
||
ORDER BY probe.sample_id,probe.name
|
||
""",
|
||
(
|
||
[str(item["sampleId"]) for item in character_probes],
|
||
[str(item["name"]) for item in character_probes],
|
||
[int(item["asOfChapter"]) for item in character_probes],
|
||
tenant_id,
|
||
work_id,
|
||
tenant_id,
|
||
work_id,
|
||
),
|
||
).fetchall()
|
||
required_chapters: set[int] = set()
|
||
for target in normalized_targets:
|
||
as_of = target - 1
|
||
required_chapters.update(range(max(1, as_of - 3), as_of + 1))
|
||
target_by_sample = {
|
||
str(item["sampleId"]): _positive_chapter(
|
||
item.get("targetChapter"), f"{item['sampleId']}.targetChapter"
|
||
)
|
||
for item in selector_config["samples"]
|
||
}
|
||
for sample_id, rows in selected.items():
|
||
as_of = target_by_sample[sample_id] - 1
|
||
for row in rows:
|
||
projected = project_card(
|
||
row,
|
||
as_of=as_of,
|
||
source_version=f"prequery:card-{row.get('id')}",
|
||
)
|
||
refs = projected.get("sourceRefs") or []
|
||
if refs:
|
||
required_chapters.update(
|
||
_positive_chapter(ref.get("chapter"), "card.sourceRef.chapter")
|
||
for ref in refs
|
||
)
|
||
else:
|
||
required_chapters.update(
|
||
milestone_reference_chapters(projected.get("milestones") or [], as_of=as_of)
|
||
)
|
||
if any(chapter in normalized_targets for chapter in required_chapters):
|
||
raise WriterReferenceWorkError("正文读取集合包含目标章")
|
||
block_rows = conn.execute(
|
||
"""
|
||
SELECT ch.order_no AS chapter,ch.id AS chapter_id,ch.status AS chapter_status,
|
||
b.id AS block_id,b.order_no AS block_order,b.revision,b.content_text
|
||
FROM muse_content_chapter ch
|
||
JOIN muse_content_block b ON b.chapter_id=ch.id
|
||
WHERE ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE
|
||
AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE
|
||
AND ch.status IN ('published','confirmed','canonical')
|
||
AND ch.order_no=ANY(%s)
|
||
ORDER BY ch.order_no,b.order_no,b.id
|
||
""",
|
||
(tenant_id, work_id, tenant_id, work_id, sorted(required_chapters)),
|
||
).fetchall()
|
||
|
||
index_unique_canonical_blocks(block_rows, allowed_chapters=required_chapters)
|
||
_target_scaffold_index(target_scaffolds, set(normalized_targets))
|
||
_canonical_character_index(character_mentions, character_probes)
|
||
return {
|
||
"work": work,
|
||
"reference": reference,
|
||
"source": source,
|
||
"authorization": authorization,
|
||
"target_scaffolds": target_scaffolds,
|
||
"card_rows": [row for rows in selected.values() for row in rows],
|
||
"block_rows": block_rows,
|
||
"canonical_character_mentions": character_mentions,
|
||
}
|
||
|
||
|
||
def assemble_writer_gate_config(
|
||
*,
|
||
base_config: Mapping[str, Any],
|
||
selector_config: Mapping[str, Any],
|
||
selector_digest: str,
|
||
rows: Mapping[str, Any],
|
||
) -> dict[str, Any]:
|
||
"""把同一事务快照装配成 canonical_frozen_prose 五样本配置。"""
|
||
|
||
common_controls = _validate_loader_controls(
|
||
base_config,
|
||
selector_config,
|
||
selector_digest=selector_digest,
|
||
)
|
||
if not isinstance(base_config, Mapping) or base_config.get("profile") != "writer_replay":
|
||
raise WriterReferenceWorkError("基础配置 profile 必须是 writer_replay")
|
||
samples = base_config.get("samples")
|
||
if not isinstance(samples, list) or not samples:
|
||
raise WriterReferenceWorkError("基础配置 samples 不能为空")
|
||
sample_ids = [str(item.get("sampleId") or "") for item in samples]
|
||
selector_samples = _selector_samples(selector_config)
|
||
selector_ids = [str(item["sampleId"]) for item in selector_samples]
|
||
if sample_ids != selector_ids:
|
||
raise WriterReferenceWorkError("稳定卡选择器样本顺序必须与预注册配置完全一致")
|
||
for sample, selector in zip(samples, selector_samples, strict=True):
|
||
if sample.get("targetChapter") != selector.get("targetChapter"):
|
||
raise WriterReferenceWorkError(
|
||
f"{sample.get('sampleId')} targetChapter 未绑定稳定选择器"
|
||
)
|
||
if selector_config.get("evaluationSetVersion") != base_config.get("evaluationSetVersion"):
|
||
raise WriterReferenceWorkError("卡选择器 evaluationSetVersion 未绑定预注册配置")
|
||
work_id = int(selector_config.get("workId") or 0)
|
||
if work_id != 8 or base_config.get("referenceWork", {}).get("id") != work_id:
|
||
raise WriterReferenceWorkError("基础配置与卡选择器必须共同绑定 work=8")
|
||
|
||
selected = resolve_card_selectors(rows.get("card_rows", []), selector_config)
|
||
selector_by_sample = {str(item["sampleId"]): item for item in selector_samples}
|
||
targets = {_positive_chapter(item.get("targetChapter"), "sample.targetChapter") for item in samples}
|
||
scaffolds = _target_scaffold_index(rows.get("target_scaffolds", []), targets)
|
||
required_chapters = _required_block_chapters(base_config, selected)
|
||
block_index = index_unique_canonical_blocks(
|
||
rows.get("block_rows", []), allowed_chapters=required_chapters
|
||
)
|
||
source = rows.get("source")
|
||
authorization = rows.get("authorization")
|
||
work = rows.get("work")
|
||
if not all(isinstance(item, Mapping) for item in (source, authorization, work)):
|
||
raise WriterReferenceWorkError("作品、来源或授权投影缺失")
|
||
source_version = str(source.get("sourceVersion") or "")
|
||
if not source_version.startswith("raw-file-v1:sha256:"):
|
||
raise WriterReferenceWorkError("真实 Writer 配置必须绑定原文件版本")
|
||
|
||
config = copy.deepcopy(dict(base_config))
|
||
config["referenceWork"] = {
|
||
"id": work_id,
|
||
"title": str(work.get("title") or ""),
|
||
"version": source_version,
|
||
}
|
||
config["authorization"] = copy.deepcopy(dict(authorization))
|
||
assembled_samples: list[dict[str, Any]] = []
|
||
prose_repository = SnapshotProseRepository(block_index)
|
||
top_snapshot = authorization.get("authorizationSnapshot")
|
||
if not isinstance(top_snapshot, Mapping):
|
||
raise WriterReferenceWorkError("授权缺少不可变 authorizationSnapshot")
|
||
character_index = _canonical_character_index(
|
||
rows.get("canonical_character_mentions", []),
|
||
_required_character_probes(base_config),
|
||
)
|
||
|
||
for raw_sample in samples:
|
||
sample = copy.deepcopy(dict(raw_sample))
|
||
sample_id = str(sample["sampleId"])
|
||
target = _positive_chapter(sample.get("targetChapter"), f"{sample_id}.targetChapter")
|
||
as_of = _positive_chapter(sample.get("asOfChapter"), f"{sample_id}.asOfChapter")
|
||
if target != as_of + 1:
|
||
raise WriterReferenceWorkError(f"{sample_id} targetChapter 必须等于 asOfChapter+1")
|
||
scaffold = scaffolds[target]
|
||
outline_text = str(scaffold.get("outline_text") or "").strip()
|
||
if not outline_text:
|
||
raise WriterReferenceWorkError(f"{sample_id} 目标 scaffold 为空")
|
||
recent_rows = [block_index[chapter] for chapter in range(max(1, as_of - 3), as_of + 1)]
|
||
actual_counts = [han_count(str(row["content_text"])) for row in recent_rows]
|
||
expected_counts = sample.get("frozenRecentHanCounts")
|
||
if actual_counts != expected_counts:
|
||
raise WriterReferenceWorkError(
|
||
f"{sample_id} 冻结近章 Han 计数漂移: expected={expected_counts}, actual={actual_counts}"
|
||
)
|
||
|
||
projected_cards = [
|
||
_project_card_for_sample(
|
||
row,
|
||
as_of=as_of,
|
||
source_version=source_version,
|
||
block_index=block_index,
|
||
source_aliases=selector_by_sample[sample_id]["cards"][index].get(
|
||
"sourceAliases", []
|
||
),
|
||
)
|
||
for index, row in enumerate(selected[sample_id])
|
||
]
|
||
fine_outline = {
|
||
"sourceRef": {
|
||
"sourceId": f"scaffold:{scaffold.get('id')}",
|
||
"sourceVersion": source_version,
|
||
"chapter": target,
|
||
},
|
||
"hardConstraints": [outline_text],
|
||
"adjustableBeats": copy.deepcopy(
|
||
sample.get("writerContextInput", {}).get("fineOutline", {}).get(
|
||
"adjustableBeats", []
|
||
)
|
||
),
|
||
"declaredNewFacts": [],
|
||
"entities": _selected_card_entities(projected_cards),
|
||
}
|
||
token_budget = _context_token_budget(
|
||
sample.get("writerContextInput", {}).get("tokenBudget", {}),
|
||
preregistered_max=int(common_controls["maxContextChars"]),
|
||
)
|
||
plan = build_retrieval_plan(
|
||
run_id=f"writer-loader:{sample_id}",
|
||
work_id=work_id,
|
||
target_chapter=target,
|
||
as_of=as_of,
|
||
fine_outline=fine_outline,
|
||
card_index_version=f"upgrade-book:{source_version}",
|
||
prose_index_version=source_version,
|
||
token_budget=token_budget,
|
||
)
|
||
sources = _sample_sources(recent_rows, projected_cards, as_of=as_of)
|
||
leakage = {
|
||
"method": "target-scaffold-proxy-and-chapter-bound-audit",
|
||
"targetFacts": _target_facts_from_scaffold(scaffold, target),
|
||
}
|
||
replay_repository = ReplayCardIndexRepository.from_replay_config(
|
||
{
|
||
"targetChapter": target,
|
||
"snapshot": {
|
||
"asOfChapter": as_of,
|
||
"snapshotVersion": f"writer-gate-a-canonical-{target}-v1",
|
||
"data": {
|
||
"chapters": [
|
||
{
|
||
"chapter": int(row["chapter"]),
|
||
"sourceId": _block_source_ref(row)["sourceId"],
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(
|
||
str(row["content_text"]).encode("utf-8")
|
||
).hexdigest(),
|
||
}
|
||
for row in recent_rows
|
||
],
|
||
"cards": [],
|
||
},
|
||
},
|
||
"authorization": authorization,
|
||
"sources": sources,
|
||
"leakageAudit": leakage,
|
||
},
|
||
cards=projected_cards,
|
||
preregistered_card_ids=[str(card["cardId"]) for card in projected_cards],
|
||
)
|
||
retrieval_result = retrieve_writer_sources(
|
||
plan=plan,
|
||
card_repository=replay_repository,
|
||
prose_repository=prose_repository,
|
||
)
|
||
# 既有检索器会为带 sourceRefs 的卡附加卡摘要事实。Writer Gate A 明确把卡
|
||
# 限定为索引,因此真实配置只保留索引提示和回读原文,不把摘要升级为权威事实。
|
||
retrieval_result["factEvidence"] = []
|
||
|
||
context_input = copy.deepcopy(dict(sample.get("writerContextInput") or {}))
|
||
context_input.update(
|
||
{
|
||
"contentMode": "canonical_frozen_prose",
|
||
"sourceVersion": source_version,
|
||
"sourceStatus": _context_source_status(authorization),
|
||
"authorizationSnapshot": {
|
||
"snapshotId": str(top_snapshot.get("id") or ""),
|
||
"allowedPurpose": "offline_evaluation",
|
||
"verifiedAt": str(top_snapshot.get("checkedAt") or ""),
|
||
"sourceVersion": source_version,
|
||
},
|
||
"generatedAt": str(top_snapshot.get("checkedAt") or ""),
|
||
"fineOutline": fine_outline,
|
||
"narrativeState": {
|
||
"time": "",
|
||
"location": "",
|
||
"characterPositions": {},
|
||
"immediateSituation": "",
|
||
},
|
||
"recentChapters": _recent_chapter_input(recent_rows),
|
||
"retrievalResult": retrieval_result,
|
||
"retrievalQueries": copy.deepcopy(plan["queries"]),
|
||
"cardIndexVersion": plan["cardIndexVersion"],
|
||
"proseIndexVersion": plan["proseIndexVersion"],
|
||
"tokenBudget": token_budget,
|
||
}
|
||
)
|
||
sample.update(
|
||
{
|
||
"workId": work_id,
|
||
"targetTitle": str(scaffold.get("title") or sample.get("targetTitle") or ""),
|
||
"snapshotVersion": f"writer-gate-a-canonical-{target}-v1",
|
||
"snapshotData": {
|
||
"chapters": [
|
||
{
|
||
"chapter": int(row["chapter"]),
|
||
"sourceId": _block_source_ref(row)["sourceId"],
|
||
"contentSha256": "sha256:"
|
||
+ hashlib.sha256(
|
||
str(row["content_text"]).encode("utf-8")
|
||
).hexdigest(),
|
||
}
|
||
for row in recent_rows
|
||
],
|
||
"cards": [],
|
||
},
|
||
"sources": sources,
|
||
"outlineSource": f"scaffold:{scaffold.get('id')}",
|
||
"fineOutlineSource": f"scaffold:{scaffold.get('id')}",
|
||
"writerContextInput": context_input,
|
||
"leakageAudit": leakage,
|
||
}
|
||
)
|
||
_recompute_new_character_ratio(sample, character_index)
|
||
assembled_samples.append(sample)
|
||
config["samples"] = assembled_samples
|
||
return config
|
||
|
||
|
||
def write_temporary_config(config: Mapping[str, Any], output_dir: Path) -> Path:
|
||
"""排他创建 /private/tmp 独立子目录并写入唯一完整配置。"""
|
||
|
||
resolved = output_dir.expanduser().resolve()
|
||
if resolved == PRIVATE_TMP or not resolved.is_relative_to(PRIVATE_TMP):
|
||
raise WriterReferenceWorkError("输出目录必须位于 /private/tmp 的独立子目录")
|
||
try:
|
||
resolved.mkdir(parents=False, exist_ok=False)
|
||
except FileExistsError as error:
|
||
raise WriterReferenceWorkError("输出目录必须是尚不存在的独立子目录") from error
|
||
except FileNotFoundError as error:
|
||
raise WriterReferenceWorkError("输出目录父目录必须已存在") from error
|
||
config_path = resolved / "config.json"
|
||
config_path.write_text(_safe_json(config) + "\n", encoding="utf-8")
|
||
return config_path
|
||
|
||
|
||
def _parse_args() -> argparse.Namespace:
|
||
"""解析真实 Writer Gate A 装配器参数。"""
|
||
|
||
parser = argparse.ArgumentParser(description="装配 Writer Gate A 五章真实临时配置")
|
||
parser.add_argument("--dsn", default=DSN)
|
||
parser.add_argument("--tenant-id", type=int, default=TENANT_ID)
|
||
parser.add_argument("--base-config", type=Path, default=DEFAULT_BASE_CONFIG)
|
||
parser.add_argument("--card-selectors", type=Path, default=DEFAULT_SELECTOR_CONFIG)
|
||
parser.add_argument("--output-dir", type=Path)
|
||
return parser.parse_args()
|
||
|
||
|
||
def main() -> int:
|
||
"""执行单事务只读装配并只回显安全摘要与临时配置路径。"""
|
||
|
||
args = _parse_args()
|
||
base_config = json.loads(args.base_config.read_text(encoding="utf-8"))
|
||
selector_bytes = args.card_selectors.read_bytes()
|
||
selectors = json.loads(selector_bytes.decode("utf-8"))
|
||
selector_digest = selector_sha256(selector_bytes)
|
||
samples = base_config.get("samples")
|
||
if not isinstance(samples, list):
|
||
raise WriterReferenceWorkError("基础配置 samples 非法")
|
||
targets = [_positive_chapter(item.get("targetChapter"), "sample.targetChapter") for item in samples]
|
||
rows = load_writer_reference_rows(
|
||
dsn=args.dsn,
|
||
tenant_id=args.tenant_id,
|
||
work_id=int(selectors.get("workId") or 0),
|
||
targets=targets,
|
||
base_config=base_config,
|
||
selector_config=selectors,
|
||
selector_digest=selector_digest,
|
||
)
|
||
config = assemble_writer_gate_config(
|
||
base_config=base_config,
|
||
selector_config=selectors,
|
||
selector_digest=selector_digest,
|
||
rows=rows,
|
||
)
|
||
output_dir = args.output_dir or (
|
||
PRIVATE_TMP / f"writer-gate-a-{uuid.uuid4().hex}"
|
||
)
|
||
config_path = write_temporary_config(config, output_dir)
|
||
summary = {
|
||
"status": "ready_for_writer_dry_run",
|
||
"workId": config["referenceWork"]["id"],
|
||
"sampleCount": len(config["samples"]),
|
||
"contentMode": "canonical_frozen_prose",
|
||
"configPath": str(config_path),
|
||
}
|
||
print(json.dumps(summary, ensure_ascii=False, sort_keys=True))
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
raise SystemExit(main())
|
||
except (
|
||
WriterReferenceWorkError,
|
||
RetrievalError,
|
||
OSError,
|
||
json.JSONDecodeError,
|
||
psycopg.Error,
|
||
) as error:
|
||
print(
|
||
json.dumps(
|
||
{"status": "blocked_writer_reference_adapter", "error": str(error)},
|
||
ensure_ascii=False,
|
||
)
|
||
)
|
||
raise SystemExit(2)
|
||
|
||
|
||
__all__ = [
|
||
"WriterReferenceWorkError",
|
||
"SnapshotProseRepository",
|
||
"assemble_writer_gate_config",
|
||
"index_unique_canonical_blocks",
|
||
"load_writer_reference_rows",
|
||
"milestone_reference_chapters",
|
||
"resolve_card_selectors",
|
||
"selector_sha256",
|
||
"select_recent_canonical_blocks",
|
||
"write_temporary_config",
|
||
]
|