zizi fa922f8cc5 实现: 装配正文五章真实冻结评测
将卡稳定选择器、冻结历史正文和目标 scaffold 绑定到同一只读快照;固定上下文预算并净化诊断索引,收紧盲评输入边界,防止目标事实和选卡信息泄漏。
2026-07-21 16:09:23 +08:00

571 lines
23 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""根据固定卡索引计划回读冻结历史原文。
抽取卡只提供定位与历史状态索引。任何由卡承载的历史事实都必须展开
sourceRefs 并成功读取 asOf 以内的原文;正式设定、Canonical 状态和细纲
新事实通过独立不可变引用进入事实证据。
"""
from __future__ import annotations
import copy
import hashlib
import pathlib
import sys
from typing import Any, Callable, Mapping, Protocol, Sequence
from writer_contract import PLAN_VERSION, TIE_BREAK, canonical_json, normalize_text, retrieval_identity
class RetrievalError(ValueError):
"""检索计划、授权、冻结或来源引用不能证明安全时抛出。"""
class CardIndexRepository(Protocol):
"""卡索引仓储只接收固定计划,不允许写手自行追加查询。"""
def search(self, plan: Mapping[str, Any]) -> list[dict[str, Any]]:
"""返回带稳定排序字段、历史里程碑和来源指针的卡索引。"""
class ProseRepository(Protocol):
"""原文仓储只能读取冻结线内的不可变来源引用。"""
def read_source_refs(
self,
*,
work_id: int,
as_of: int,
source_refs: Sequence[Mapping[str, Any]],
) -> list[dict[str, Any]]:
"""展开已校验的来源指针。"""
def _load_search_cards() -> Callable[..., list[dict[str, Any]]]:
"""延迟导入 search skill,保持纯数据测试不触发嵌入依赖。"""
search_scripts = pathlib.Path(__file__).resolve().parents[2] / "search" / "scripts"
sys.path.insert(0, str(search_scripts))
from search import search_cards
return search_cards
def _chapter(value: Any, path: str) -> int:
"""严格解析正整数章号,拒绝 bool 和猜测性字符串。"""
if isinstance(value, bool):
raise RetrievalError(f"{path} 必须是正整数章号")
if isinstance(value, str) and value.isdigit():
value = int(value)
if not isinstance(value, int) or value <= 0:
raise RetrievalError(f"{path} 必须是正整数章号")
return value
def _milestone_chapter(value: Mapping[str, Any]) -> int:
"""兼容实验库中英文里程碑章号字段,但不解析模糊文本。"""
for key in ("chapter", "章", "chapterNo", "order_no"):
if key in value:
return _chapter(value[key], f"milestone.{key}")
raise RetrievalError("卡里程碑缺少可证明的绝对章号")
def stable_sort_cards(cards: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
"""按合同固定卡片顺序,同分召回不会因数据库执行计划漂移。"""
copied = [copy.deepcopy(dict(card)) for card in cards]
try:
return sorted(
copied,
key=lambda item: (
-float(item["score"]),
str(item["sourceVersion"]),
str(item["sourceId"]),
int(item.get("sourceOffset") or 0),
),
)
except (KeyError, TypeError, ValueError) as error:
raise RetrievalError("卡缺少稳定排序字段") from error
def _outline_elements(fine_outline: Mapping[str, Any]) -> list[dict[str, str]]:
"""按合同顺序提取需要检索的人物、关系、物品、地点和力量体系。"""
groups = (
("entities", None),
("relations", "character_relation"),
("items", "item"),
("locations", "location"),
("powerSystems", "power_system"),
("stateNeeds", "unknown"),
)
result: list[dict[str, str]] = []
seen: set[tuple[str, str]] = set()
for field, fallback_type in groups:
values = fine_outline.get(field, [])
if not isinstance(values, list):
raise RetrievalError(f"fineOutline.{field} 必须是数组")
for index, raw in enumerate(values):
if isinstance(raw, str):
item = {"id": f"{field}:{index}", "type": fallback_type or "unknown", "name": raw}
elif isinstance(raw, Mapping):
item = {
"id": str(raw.get("id") or f"{field}:{index}"),
"type": str(raw.get("type") or fallback_type or "unknown"),
"name": str(raw.get("name") or ""),
}
else:
raise RetrievalError(f"fineOutline.{field}[{index}] 类型非法")
if not item["name"]:
raise RetrievalError(f"fineOutline.{field}[{index}] 缺少 name")
identity = (item["type"], item["name"])
if identity not in seen:
seen.add(identity)
result.append(item)
return result
def build_retrieval_plan(
*,
run_id: str,
work_id: int,
target_chapter: int,
as_of: int,
fine_outline: Mapping[str, Any],
card_index_version: str,
prose_index_version: str,
token_budget: Mapping[str, int],
top_k: int = 5,
) -> dict[str, Any]:
"""从已确认细纲确定查询集合,并在执行前冻结计划身份。"""
target = _chapter(target_chapter, "targetChapter")
freeze = _chapter(as_of, "asOf")
if freeze >= target:
raise RetrievalError("asOf 必须早于 targetChapter")
if not isinstance(work_id, int) or isinstance(work_id, bool) or work_id <= 0:
raise RetrievalError("workId 必须是正整数")
if not isinstance(top_k, int) or isinstance(top_k, bool) or top_k <= 0:
raise RetrievalError("topK 必须是正整数")
hard_constraints = fine_outline.get("hardConstraints", [])
if not isinstance(hard_constraints, list) or any(not isinstance(item, str) for item in hard_constraints):
raise RetrievalError("fineOutline.hardConstraints 必须是字符串数组")
suffix = " ".join(hard_constraints)
queries = [
{
"queryId": f"query-{index:03d}-{item['id']}",
"text": normalize_text(f"{item['name']} {suffix}".strip()),
"entityTypes": [item["type"]],
"purpose": "fact_and_prose_evidence",
"topK": top_k,
}
for index, item in enumerate(_outline_elements(fine_outline), 1)
]
plan = {
"planVersion": PLAN_VERSION,
"runId": run_id,
"asOf": freeze,
"queries": queries,
"cardIndexVersion": str(card_index_version),
"proseIndexVersion": str(prose_index_version),
"filters": {
"workId": work_id,
"asOfChapter": freeze,
"sourceStatus": "active",
"authorizationRequired": True,
},
"tieBreak": TIE_BREAK,
"tokenBudget": copy.deepcopy(dict(token_budget)),
}
plan["planId"] = retrieval_identity(plan)
return plan
def _validate_source_ref(ref: Mapping[str, Any], *, as_of: int, path: str) -> dict[str, Any]:
"""校验块级原文引用完整且不越过冻结线。"""
required = {"sourceId", "sourceVersion", "chapter", "blockId", "startCodePoint", "endCodePoint"}
if not isinstance(ref, Mapping) or not required.issubset(ref):
raise RetrievalError(f"{path} 缺少块级来源定位")
chapter = _chapter(ref["chapter"], f"{path}.chapter")
if chapter > as_of:
raise RetrievalError(f"{path} 包含目标章或未来章")
start = ref["startCodePoint"]
end = ref["endCodePoint"]
if any(isinstance(value, bool) or not isinstance(value, int) for value in (ref["blockId"], start, end)):
raise RetrievalError(f"{path} 的块或字符区间非法")
if ref["blockId"] <= 0 or start < 0 or end <= start:
raise RetrievalError(f"{path} 的块或字符区间非法")
if not str(ref["sourceId"]).strip() or not str(ref["sourceVersion"]).strip():
raise RetrievalError(f"{path} 缺少来源 ID 或版本")
return copy.deepcopy(dict(ref))
def freeze_card(card: Mapping[str, Any], *, as_of: int) -> dict[str, Any]:
"""仅用冻结线内里程碑重建卡状态,不透传终态摘要和未来字段。"""
freeze = _chapter(as_of, "asOf")
milestones = card.get("milestones")
if not isinstance(milestones, list):
raise RetrievalError(f"卡 {card.get('cardId')} 缺少历史里程碑")
kept: list[dict[str, Any]] = []
for index, milestone in enumerate(milestones):
if not isinstance(milestone, Mapping):
raise RetrievalError(f"卡里程碑[{index}] 必须是对象")
chapter = _milestone_chapter(milestone)
if chapter <= freeze:
normalized = copy.deepcopy(dict(milestone))
normalized["chapter"] = chapter
for alias in ("章", "chapterNo", "order_no"):
normalized.pop(alias, None)
kept.append(normalized)
kept.sort(key=lambda item: (item["chapter"], str(item.get("id") or "")))
if not kept:
raise RetrievalError(f"卡 {card.get('cardId')} 在冻结线内没有可证明状态")
refs = [
_validate_source_ref(ref, as_of=freeze, path=f"card.sourceRefs[{index}]")
for index, ref in enumerate(card.get("sourceRefs") or [])
]
return {
"cardId": str(card.get("cardId") or ""),
"type": str(card.get("type") or ""),
"name": str(card.get("name") or ""),
"score": float(card.get("score") or 0),
"sourceId": str(card.get("sourceId") or ""),
"sourceVersion": str(card.get("sourceVersion") or ""),
"sourceOffset": int(card.get("sourceOffset") or 0),
"sourceRefs": refs,
"stateAsOf": kept,
"sourceKind": str(card.get("sourceKind") or ""),
"productionRetrievalEligible": bool(card.get("productionRetrievalEligible")),
}
class ProductionCardIndexRepository:
"""生产卡仓储固定调用 search.py 的 work 授权面。"""
def __init__(self, *, search_function: Callable[..., list[dict[str, Any]]] | None = None):
self._search = search_function or _load_search_cards()
def search(self, plan: Mapping[str, Any]) -> list[dict[str, Any]]:
"""按计划逐查询召回,并再次失败关闭验证生产资格。"""
cards: list[dict[str, Any]] = []
for query in plan.get("queries", []):
entity_types = query.get("entityTypes") or []
card_type = entity_types[0] if len(entity_types) == 1 and entity_types[0] != "unknown" else None
cards.extend(
self._search(
intent=query["text"],
scope="work",
work_id=plan["filters"]["workId"],
ttype=card_type,
purpose="generation",
top=query["topK"],
)
)
unique: dict[str, dict[str, Any]] = {}
for card in stable_sort_cards(cards):
if (
card.get("sourceKind") != "canonical_entity"
or card.get("sourceStatus") not in {"active", "authorized"}
or card.get("bindingStatus") != "active"
or card.get("productionRetrievalEligible") is not True
):
raise RetrievalError("生产检索命中非 active Canonical entity 或无效 binding")
unique.setdefault(str(card.get("cardId")), copy.deepcopy(card))
return stable_sort_cards(list(unique.values()))
class ReplayCardIndexRepository:
"""回放卡仓储只接受预注册 upgrade_book 评测投影。"""
def __init__(
self,
cards: Sequence[Mapping[str, Any]],
*,
preregistered_card_ids: Sequence[str],
_validated_replay: bool = False,
):
if not _validated_replay:
raise RetrievalError("回放仓储必须通过 from_replay_config 完成授权与泄露门禁")
expected = {str(item) for item in preregistered_card_ids}
actual = {str(item.get("cardId")) for item in cards}
if not expected or expected != actual:
raise RetrievalError("回放卡与预注册 ID 不一致")
self._cards = []
for card in cards:
if (
card.get("sourceType") != "upgrade_book"
or card.get("evaluationStatus") != "eval_draft"
or card.get("productionRetrievalEligible") is not False
):
raise RetrievalError("回放只允许不可生产检索的 upgrade_book 评测卡")
self._cards.append(copy.deepcopy(dict(card)))
@classmethod
def from_replay_config(
cls,
config: Mapping[str, Any],
*,
cards: Sequence[Mapping[str, Any]],
preregistered_card_ids: Sequence[str],
) -> "ReplayCardIndexRepository":
"""复用 replay-eval 的授权、冻结来源和内容泄露审计后构造仓储。"""
scripts = pathlib.Path(__file__).resolve().parents[2] / "replay-eval" / "scripts"
sys.path.insert(0, str(scripts))
from audit_leakage import audit_snapshot
from check_snapshot import check_authorization, check_target_sources
if not isinstance(config, Mapping):
raise RetrievalError("回放配置必须是对象")
target = _chapter(config.get("targetChapter"), "config.targetChapter")
snapshot = config.get("snapshot")
if not isinstance(snapshot, Mapping):
raise RetrievalError("回放配置缺少冻结快照")
as_of = _chapter(snapshot.get("asOfChapter"), "config.snapshot.asOfChapter")
if target != as_of + 1:
raise RetrievalError("回放 targetChapter 必须等于 asOfChapter+1")
authorization_result = check_authorization(config.get("authorization"))
if not authorization_result.get("ok"):
raise RetrievalError("回放授权快照未通过既有门禁")
source_result = check_target_sources(target, config.get("sources", []))
if not source_result.get("ok"):
raise RetrievalError("回放来源包含目标章、未来章或不可证明区间")
leakage = config.get("leakageAudit")
target_facts = leakage.get("targetFacts") if isinstance(leakage, Mapping) else None
audit_result = audit_snapshot(snapshot, target_facts, as_of=as_of, target=target)
if not audit_result.get("ok"):
raise RetrievalError("回放冻结快照未通过既有内容泄露审计")
return cls(
cards,
preregistered_card_ids=preregistered_card_ids,
_validated_replay=True,
)
def search(self, plan: Mapping[str, Any]) -> list[dict[str, Any]]:
"""返回预注册集合;计划不能把评测卡转换为生产卡。"""
del plan
return stable_sort_cards(self._cards)
class FrozenProseRepository:
"""复用 load_reference_work 的唯一冻结原文 SQL 入口。"""
def __init__(self, *, dsn: str, tenant_id: int):
self.dsn = dsn
self.tenant_id = tenant_id
def read_source_refs(self, *, work_id: int, as_of: int, source_refs: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
"""延迟导入 replay-eval,避免复制 SQL 或建立第二套权限语义。"""
scripts = pathlib.Path(__file__).resolve().parents[2] / "replay-eval" / "scripts"
sys.path.insert(0, str(scripts))
from load_reference_work import load_frozen_prose_rows
return load_frozen_prose_rows(
dsn=self.dsn,
tenant_id=self.tenant_id,
work_id=work_id,
as_of=as_of,
source_refs=source_refs,
)
def _content_hash(text: str) -> str:
"""对规范化文本计算带算法前缀的内容哈希。"""
normalized = normalize_text(text)
return "sha256:" + hashlib.sha256(normalized.encode("utf-8")).hexdigest()
def _index_hint(card: Mapping[str, Any], *, as_of: int) -> dict[str, Any]:
"""把冻结卡投影成稳定索引提示,是否存在原文指针不影响卡进入索引。"""
state_as_of = card.get("stateAsOf")
if not isinstance(state_as_of, list) or not state_as_of:
raise RetrievalError(f"卡 {card.get('cardId')} 缺少冻结状态")
latest = state_as_of[-1]
if not isinstance(latest, Mapping):
raise RetrievalError(f"卡 {card.get('cardId')} 最新冻结状态非法")
if card.get("sourceKind") == "eval_draft":
# 诊断卡只能公开索引元数据;里程碑台阶属于被测信息,不能再透传给 B/C 臂。
content = canonical_json(
{
"name": str(card.get("name") or ""),
"type": str(card.get("type") or ""),
"sourceChapters": sorted(
{_chapter(ref.get("chapter"), "card.sourceRefs.chapter") for ref in card.get("sourceRefs", [])}
),
}
)
else:
# 生产 Canonical 卡保持原有冻结状态提示行为。
content = normalize_text(
str(latest.get("fact") or latest.get("台阶") or canonical_json(latest))
)
hint = {
"cardId": str(card.get("cardId") or ""),
"name": str(card.get("name") or ""),
"type": str(card.get("type") or ""),
"content": content,
"sourceId": str(card.get("sourceId") or ""),
"sourceVersion": str(card.get("sourceVersion") or ""),
"asOf": as_of,
}
if any(not str(hint[field]).strip() for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion")):
raise RetrievalError("冻结卡缺少生成索引提示所需字段")
return hint
def _authoritative_evidence(facts: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
"""把正式设定、Canonical 状态和细纲新事实转换为独立事实证据。"""
allowed = {"formal_setting", "canonical_state", "fine_outline_declared_new"}
result: list[dict[str, Any]] = []
for index, raw in enumerate(facts):
source_type = raw.get("sourceType")
source_ref = raw.get("sourceRef")
fact = normalize_text(str(raw.get("fact") or ""))
if source_type not in allowed or not fact or not isinstance(source_ref, Mapping):
raise RetrievalError(f"authoritativeFacts[{index}] 合同非法")
if not source_ref.get("sourceId") or not source_ref.get("sourceVersion"):
raise RetrievalError(f"authoritativeFacts[{index}] 缺少不可变来源引用")
result.append(
{
"evidenceId": str(raw.get("factId") or f"authoritative:{index}"),
"fact": fact,
"sourceType": source_type,
"sourceRef": copy.deepcopy(dict(source_ref)),
"contentSha256": _content_hash(fact),
"riskLevel": str(raw.get("riskLevel") or "medium"),
}
)
return result
def retrieve_writer_sources(
*,
plan: Mapping[str, Any],
card_repository: CardIndexRepository,
prose_repository: ProseRepository,
authoritative_facts: Sequence[Mapping[str, Any]] = (),
) -> dict[str, Any]:
"""执行固定计划,展开卡来源并形成事实/原文两条证据线。"""
as_of = _chapter(plan.get("asOf"), "plan.asOf")
work_id = plan.get("filters", {}).get("workId")
if not isinstance(work_id, int) or isinstance(work_id, bool) or work_id <= 0:
raise RetrievalError("plan.filters.workId 非法")
cards = [freeze_card(card, as_of=as_of) for card in stable_sort_cards(card_repository.search(plan))]
index_hints = [_index_hint(card_item, as_of=as_of) for card_item in cards]
source_refs: list[dict[str, Any]] = []
unverified: list[dict[str, Any]] = []
for card_item in cards:
if card_item["sourceRefs"]:
source_refs.extend(card_item["sourceRefs"])
else:
unverified.append(
{
"cardId": card_item["cardId"],
"reason": "missing_source_refs",
"classification": "unverifiedIndexHint",
}
)
deduplicated = {
(str(ref["sourceVersion"]), str(ref["sourceId"]), int(ref["startCodePoint"])): ref
for ref in source_refs
}
ordered_refs = [deduplicated[key] for key in sorted(deduplicated)]
prose_rows = prose_repository.read_source_refs(work_id=work_id, as_of=as_of, source_refs=ordered_refs) if ordered_refs else []
def ref_key(ref: Mapping[str, Any]) -> tuple[str, str, int]:
"""同一块可有多个字符区间,映射键必须包含偏移。"""
return (str(ref.get("sourceVersion")), str(ref.get("sourceId")), int(ref.get("startCodePoint") or 0))
rows_by_source = {ref_key(row.get("sourceRef", {})): row for row in prose_rows}
missing = [ref["sourceId"] for ref in ordered_refs if ref_key(ref) not in rows_by_source]
if missing:
raise RetrievalError(f"卡来源未能回读原文: {','.join(map(str, missing))}")
prose_evidence: list[dict[str, Any]] = []
for index, ref in enumerate(ordered_refs):
row = rows_by_source[ref_key(ref)]
text = normalize_text(str(row.get("text") or ""))
if not text:
raise RetrievalError(f"来源 {ref['sourceId']} 原文为空")
prose_evidence.append(
{
"evidenceId": f"prose:card:{index:04d}",
"chapter": _chapter(row.get("chapter"), "prose.chapter"),
"sourceRef": copy.deepcopy(ref),
"contentSha256": _content_hash(text),
"purpose": (
"card_chapter_proxy"
if ref.get("sourceType") == "card_chapter_proxy"
else str(row.get("purpose") or "card_source")
),
"text": text,
"isRecentBaseline": False,
}
)
fact_evidence = _authoritative_evidence(authoritative_facts)
for card_item in cards:
if not card_item["sourceRefs"]:
continue
latest = card_item["stateAsOf"][-1]
fact_text = normalize_text(str(latest.get("fact") or latest.get("台阶") or canonical_json(latest)))
fact_evidence.append(
{
"evidenceId": f"fact:card:{card_item['cardId']}",
"fact": fact_text,
"sourceType": "historical_prose",
"sourceRef": copy.deepcopy(card_item["sourceRefs"][0]),
"contentSha256": _content_hash(fact_text),
"riskLevel": "medium",
}
)
fact_evidence.sort(key=lambda item: item["evidenceId"])
sources = sorted(
[copy.deepcopy(ref) for ref in ordered_refs]
+ [copy.deepcopy(item["sourceRef"]) for item in fact_evidence if item["sourceType"] != "historical_prose"],
key=lambda item: (str(item["sourceVersion"]), str(item["sourceId"]), int(item.get("startCodePoint") or 0)),
)
manifest_payload = {
"manifestVersion": "writer-retrieval-manifest-v1",
"planId": plan["planId"],
"sources": sources,
"omittedSources": [
{
"sourceId": next(
card_item["sourceId"]
for card_item in cards
if card_item["cardId"] == item["cardId"]
),
"reason": item["reason"],
}
for item in unverified
],
}
manifest = {**manifest_payload, "manifestId": retrieval_identity(manifest_payload)}
return {
"cards": cards,
"factEvidence": fact_evidence,
"proseEvidence": prose_evidence,
"indexHints": index_hints,
"unverifiedIndexHints": unverified,
"manifest": manifest,
}
__all__ = [
"RetrievalError", "CardIndexRepository", "ProseRepository", "ProductionCardIndexRepository",
"ReplayCardIndexRepository", "FrozenProseRepository", "stable_sort_cards", "freeze_card",
"build_retrieval_plan", "retrieve_writer_sources",
]