实现: 以完整卡索引驱动冻结原文回读
This commit is contained in:
parent
de0f870b96
commit
8727eeaa46
@ -106,6 +106,26 @@ def _normalize_fact(raw: Mapping[str, Any]) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def _normalize_index_hint(raw: Mapping[str, Any], *, as_of: int) -> dict[str, Any]:
|
||||
"""规范化冻结卡索引提示;它只用于诊断,绝不升级为事实证据。"""
|
||||
|
||||
required = {"cardId", "name", "type", "content", "sourceId", "sourceVersion", "asOf"}
|
||||
if not isinstance(raw, Mapping) or set(raw) != required:
|
||||
raise AssemblyError("indexHints 每项字段必须严格匹配 WriterContext v1")
|
||||
hint_as_of = raw["asOf"]
|
||||
if isinstance(hint_as_of, bool) or not isinstance(hint_as_of, int) or hint_as_of <= 0:
|
||||
raise AssemblyError("indexHints.asOf 必须是正整数")
|
||||
if hint_as_of > as_of:
|
||||
raise AssemblyError("indexHints 包含目标章或未来章")
|
||||
normalized = {
|
||||
key: normalize_text(str(raw[key]))
|
||||
for key in required - {"asOf"}
|
||||
}
|
||||
if any(not value.strip() for value in normalized.values()):
|
||||
raise AssemblyError("indexHints 字符串字段不能为空")
|
||||
return {**normalized, "asOf": hint_as_of}
|
||||
|
||||
|
||||
def _select_recent_baseline(recent_chapters: Sequence[Mapping[str, Any]], *, as_of: int) -> list[dict[str, Any]]:
|
||||
"""选择冻结章起连续四章全文;作品不足四章时从第一章开始。"""
|
||||
|
||||
@ -230,6 +250,7 @@ def _manifest(
|
||||
prose: Sequence[Mapping[str, Any]],
|
||||
pattern_references: Sequence[Mapping[str, Any]],
|
||||
omitted: Sequence[Mapping[str, str]],
|
||||
index_hints: Sequence[Mapping[str, Any]] = (),
|
||||
) -> dict[str, Any]:
|
||||
"""由最终入包来源集合计算稳定 manifest,不含 runId 或时间戳。"""
|
||||
|
||||
@ -240,6 +261,15 @@ def _manifest(
|
||||
for raw_ref in pattern_references:
|
||||
ref = copy.deepcopy(dict(raw_ref))
|
||||
unique[_source_key(ref)] = ref
|
||||
for hint in index_hints:
|
||||
# 这里记录的是全部冻结卡索引,并不代表卡缺少原文来源。
|
||||
ref = {
|
||||
"sourceId": hint["sourceId"],
|
||||
"sourceVersion": hint["sourceVersion"],
|
||||
"chapter": hint["asOf"],
|
||||
"sourceType": "diagnostic_card_index",
|
||||
}
|
||||
unique[_source_key(ref)] = ref
|
||||
sources = [unique[key] for key in sorted(unique)]
|
||||
omitted_rows = sorted(
|
||||
[copy.deepcopy(dict(item)) for item in omitted],
|
||||
@ -311,6 +341,7 @@ def assemble_context(
|
||||
token_budget: Mapping[str, int],
|
||||
pattern_references: Sequence[Mapping[str, Any]] = (),
|
||||
generated_at: str,
|
||||
evidence_strategy: str = "production_dual_evidence",
|
||||
) -> dict[str, Any]:
|
||||
"""组装稳定 WriterContext,并返回规范 JSON 与 Markdown manifest。"""
|
||||
|
||||
@ -322,7 +353,26 @@ def assemble_context(
|
||||
if isinstance(max_chars, bool) or not isinstance(max_chars, int) or max_chars <= 0:
|
||||
raise AssemblyError("tokenBudget.maxContextChars 必须是正整数")
|
||||
|
||||
baseline = _select_recent_baseline(recent_chapters, as_of=as_of)
|
||||
strategies = {
|
||||
"production_dual_evidence",
|
||||
"historical_prose_only",
|
||||
"card_index_only",
|
||||
"card_index_plus_prose",
|
||||
}
|
||||
if evidence_strategy not in strategies:
|
||||
raise AssemblyError("evidenceStrategy 非法")
|
||||
if mode == "production" and evidence_strategy != "production_dual_evidence":
|
||||
raise AssemblyError("生产上下文必须使用 production_dual_evidence")
|
||||
if evidence_strategy != "production_dual_evidence" and not (
|
||||
mode == "diagnostic_only" and purpose in {"evaluation", "diagnostic"}
|
||||
):
|
||||
raise AssemblyError("诊断证据策略只允许用于 diagnostic_only 评测或诊断")
|
||||
|
||||
baseline = (
|
||||
[]
|
||||
if evidence_strategy == "card_index_only"
|
||||
else _select_recent_baseline(recent_chapters, as_of=as_of)
|
||||
)
|
||||
baseline_blocks = {_whole_source_key(item["sourceRef"]) for item in baseline}
|
||||
supplemental: list[dict[str, Any]] = []
|
||||
seen_fragments = {_source_key(item["sourceRef"]) for item in baseline}
|
||||
@ -342,6 +392,22 @@ def assemble_context(
|
||||
key=lambda item: ({"high": 0, "medium": 1, "low": 2}[item["riskLevel"]], item["evidenceId"]),
|
||||
)
|
||||
cards = [copy.deepcopy(dict(item)) for item in retrieval_result.get("cards", [])]
|
||||
# 新合同由检索器显式给出覆盖全部冻结命中卡的 indexHints。旧字段只在
|
||||
# 历史评测夹具尚未迁移时回退,不能再决定有 sourceRefs 的卡是否入包。
|
||||
raw_index_hints = retrieval_result.get("indexHints")
|
||||
if raw_index_hints is None:
|
||||
raw_index_hints = retrieval_result.get("unverifiedIndexHints", [])
|
||||
index_hints = sorted(
|
||||
[
|
||||
_normalize_index_hint(item, as_of=as_of)
|
||||
for item in raw_index_hints
|
||||
],
|
||||
key=lambda item: (item["sourceVersion"], item["sourceId"], item["cardId"]),
|
||||
)
|
||||
if evidence_strategy in {"card_index_only", "card_index_plus_prose"} and not index_hints:
|
||||
raise AssemblyError("卡索引诊断策略缺少 indexHints")
|
||||
if evidence_strategy in {"production_dual_evidence", "historical_prose_only"}:
|
||||
index_hints = []
|
||||
inherited_omitted = [
|
||||
{"sourceId": str(item.get("sourceId") or "unknown"), "reason": str(item.get("reason") or "not_relevant")}
|
||||
for item in retrieval_result.get("manifest", {}).get("omittedSources", [])
|
||||
@ -354,10 +420,13 @@ def assemble_context(
|
||||
selected_prose: list[dict[str, Any]] = list(baseline)
|
||||
omitted = list(inherited_omitted)
|
||||
candidates: list[tuple[str, dict[str, Any]]] = []
|
||||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "high")
|
||||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "medium")
|
||||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "low")
|
||||
candidates.extend(("prose", item) for item in supplemental)
|
||||
if evidence_strategy == "production_dual_evidence":
|
||||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "high")
|
||||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "medium")
|
||||
candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "low")
|
||||
candidates.extend(("prose", item) for item in supplemental)
|
||||
elif evidence_strategy == "card_index_plus_prose":
|
||||
candidates.extend(("prose", item) for item in supplemental)
|
||||
|
||||
def make_context() -> dict[str, Any]:
|
||||
"""用当前选择集生成完整上下文,供预算试算与最终冻结。"""
|
||||
@ -377,6 +446,7 @@ def assemble_context(
|
||||
prose=ordered_prose,
|
||||
pattern_references=pattern_references,
|
||||
omitted=omitted,
|
||||
index_hints=index_hints,
|
||||
)
|
||||
context = {
|
||||
"schemaVersion": "writer-context-v1",
|
||||
@ -402,6 +472,8 @@ def assemble_context(
|
||||
"narrativeState": copy.deepcopy(dict(narrative_state)),
|
||||
"factEvidence": ordered_facts,
|
||||
"proseEvidence": ordered_prose,
|
||||
"evidenceStrategy": evidence_strategy,
|
||||
"indexHints": copy.deepcopy(index_hints),
|
||||
"patternReferences": [copy.deepcopy(dict(item)) for item in pattern_references],
|
||||
"evidenceCoverage": _build_coverage(fine_outline, ordered_facts, ordered_prose, cards),
|
||||
"outputContract": copy.deepcopy(dict(output_contract)),
|
||||
|
||||
@ -383,6 +383,32 @@ def _content_hash(text: str) -> str:
|
||||
return "sha256:" + hashlib.sha256(normalized.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _index_hint(card: Mapping[str, Any], *, as_of: int) -> dict[str, Any]:
|
||||
"""把冻结卡投影成稳定索引提示,是否存在原文指针不影响卡进入索引。"""
|
||||
|
||||
state_as_of = card.get("stateAsOf")
|
||||
if not isinstance(state_as_of, list) or not state_as_of:
|
||||
raise RetrievalError(f"卡 {card.get('cardId')} 缺少冻结状态")
|
||||
latest = state_as_of[-1]
|
||||
if not isinstance(latest, Mapping):
|
||||
raise RetrievalError(f"卡 {card.get('cardId')} 最新冻结状态非法")
|
||||
content = normalize_text(
|
||||
str(latest.get("fact") or latest.get("台阶") or canonical_json(latest))
|
||||
)
|
||||
hint = {
|
||||
"cardId": str(card.get("cardId") or ""),
|
||||
"name": str(card.get("name") or ""),
|
||||
"type": str(card.get("type") or ""),
|
||||
"content": content,
|
||||
"sourceId": str(card.get("sourceId") or ""),
|
||||
"sourceVersion": str(card.get("sourceVersion") or ""),
|
||||
"asOf": as_of,
|
||||
}
|
||||
if any(not str(hint[field]).strip() for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion")):
|
||||
raise RetrievalError("冻结卡缺少生成索引提示所需字段")
|
||||
return hint
|
||||
|
||||
|
||||
def _authoritative_evidence(facts: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""把正式设定、Canonical 状态和细纲新事实转换为独立事实证据。"""
|
||||
|
||||
@ -423,6 +449,7 @@ def retrieve_writer_sources(
|
||||
if not isinstance(work_id, int) or isinstance(work_id, bool) or work_id <= 0:
|
||||
raise RetrievalError("plan.filters.workId 非法")
|
||||
cards = [freeze_card(card, as_of=as_of) for card in stable_sort_cards(card_repository.search(plan))]
|
||||
index_hints = [_index_hint(card_item, as_of=as_of) for card_item in cards]
|
||||
source_refs: list[dict[str, Any]] = []
|
||||
unverified: list[dict[str, Any]] = []
|
||||
for card_item in cards:
|
||||
@ -497,7 +524,14 @@ def retrieve_writer_sources(
|
||||
"planId": plan["planId"],
|
||||
"sources": sources,
|
||||
"omittedSources": [
|
||||
{"sourceId": f"card:{item['cardId']}", "reason": item["reason"]}
|
||||
{
|
||||
"sourceId": next(
|
||||
card_item["sourceId"]
|
||||
for card_item in cards
|
||||
if card_item["cardId"] == item["cardId"]
|
||||
),
|
||||
"reason": item["reason"],
|
||||
}
|
||||
for item in unverified
|
||||
],
|
||||
}
|
||||
@ -506,6 +540,7 @@ def retrieve_writer_sources(
|
||||
"cards": cards,
|
||||
"factEvidence": fact_evidence,
|
||||
"proseEvidence": prose_evidence,
|
||||
"indexHints": index_hints,
|
||||
"unverifiedIndexHints": unverified,
|
||||
"manifest": manifest,
|
||||
}
|
||||
|
||||
@ -12,6 +12,7 @@ import unittest
|
||||
sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent))
|
||||
|
||||
from assemble_writer_context import AssemblyError, assemble_context # noqa: E402
|
||||
from retrieve_writer_sources import retrieve_writer_sources # noqa: E402
|
||||
from writer_contract import retrieval_identity, validate_writer_context # noqa: E402
|
||||
|
||||
|
||||
@ -127,6 +128,7 @@ def base_kwargs() -> dict:
|
||||
"cards": [],
|
||||
"factEvidence": [],
|
||||
"proseEvidence": [],
|
||||
"indexHints": [],
|
||||
"unverifiedIndexHints": [],
|
||||
"manifest": {
|
||||
"manifestVersion": "writer-retrieval-manifest-v1",
|
||||
@ -186,6 +188,175 @@ class AssembleWriterContextTest(unittest.TestCase):
|
||||
with self.assertRaises(AssemblyError):
|
||||
assemble_context(**kwargs)
|
||||
|
||||
def test_diagnostic_arms_use_frozen_index_hints_without_bypassing_contract(self):
|
||||
"""A/B/C 均由同一组装入口产生,B 只含提示,A/C 保留近章基线。"""
|
||||
|
||||
hint = {
|
||||
"cardId": "card:甲",
|
||||
"name": "甲",
|
||||
"type": "character",
|
||||
"content": "甲在冻结点仍位于城外",
|
||||
"sourceId": "card-source:甲",
|
||||
"sourceVersion": "cards-v1",
|
||||
"asOf": 6,
|
||||
}
|
||||
contexts = {}
|
||||
for arm, strategy in (
|
||||
("A", "historical_prose_only"),
|
||||
("B", "card_index_only"),
|
||||
("C", "card_index_plus_prose"),
|
||||
):
|
||||
kwargs = base_kwargs()
|
||||
kwargs.update(
|
||||
{
|
||||
"mode": "diagnostic_only",
|
||||
"purpose": "evaluation",
|
||||
"quality_policy_version": "writer-eval-v1",
|
||||
"evidence_strategy": strategy,
|
||||
}
|
||||
)
|
||||
kwargs["retrieval_result"]["indexHints"] = [copy.deepcopy(hint)]
|
||||
contexts[arm] = assemble_context(**kwargs)["context"]
|
||||
|
||||
self.assertEqual(contexts["A"]["indexHints"], [])
|
||||
self.assertEqual(contexts["A"]["factEvidence"], [])
|
||||
self.assertEqual([item["chapter"] for item in contexts["A"]["proseEvidence"]], [3, 4, 5, 6])
|
||||
self.assertEqual(contexts["B"]["proseEvidence"], [])
|
||||
self.assertEqual(contexts["B"]["factEvidence"], [])
|
||||
self.assertEqual(contexts["B"]["indexHints"], [hint])
|
||||
self.assertEqual([item["chapter"] for item in contexts["C"]["proseEvidence"]], [3, 4, 5, 6])
|
||||
self.assertEqual(contexts["C"]["indexHints"], [hint])
|
||||
self.assertTrue(all(not context["acceptanceEligible"] for context in contexts.values()))
|
||||
|
||||
def test_index_hint_strict_fields_and_freeze_are_enforced(self):
|
||||
"""组装器不接受检索器夹带字段或越过 asOf 的提示。"""
|
||||
|
||||
base_hint = {
|
||||
"cardId": "card:甲", "name": "甲", "type": "character",
|
||||
"content": "冻结状态", "sourceId": "card-source:甲",
|
||||
"sourceVersion": "cards-v1", "asOf": 6,
|
||||
}
|
||||
for mutation in ("unknown", "future"):
|
||||
kwargs = base_kwargs()
|
||||
kwargs.update(
|
||||
{
|
||||
"mode": "diagnostic_only",
|
||||
"purpose": "diagnostic",
|
||||
"quality_policy_version": "writer-eval-v1",
|
||||
"evidence_strategy": "card_index_only",
|
||||
}
|
||||
)
|
||||
hint = copy.deepcopy(base_hint)
|
||||
if mutation == "unknown":
|
||||
hint["score"] = 1.0
|
||||
else:
|
||||
hint["asOf"] = 7
|
||||
kwargs["retrieval_result"]["indexHints"] = [hint]
|
||||
with self.subTest(mutation=mutation), self.assertRaises(AssemblyError):
|
||||
assemble_context(**kwargs)
|
||||
|
||||
def test_retrieve_to_assemble_keeps_all_cards_as_indexes_and_only_c_reads_card_sources(self):
|
||||
"""真实串联证明卡索引与原文回读独立,A/B/C 只改变约定变量。"""
|
||||
|
||||
with_ref = {
|
||||
"cardId": "card:甲",
|
||||
"type": "character",
|
||||
"name": "甲",
|
||||
"score": 0.9,
|
||||
"sourceVersion": "cards-v1",
|
||||
"sourceId": "canonical-entity:甲",
|
||||
"sourceOffset": 0,
|
||||
"sourceRefs": [{
|
||||
"sourceId": "chapter:2:block:202",
|
||||
"sourceVersion": "chapter-2-block-202-v1",
|
||||
"chapter": 2,
|
||||
"blockId": 202,
|
||||
"startCodePoint": 0,
|
||||
"endCodePoint": 6,
|
||||
}],
|
||||
"milestones": [
|
||||
{"chapter": 2, "fact": "甲在城门受伤"},
|
||||
{"chapter": 8, "fact": "未来状态"},
|
||||
],
|
||||
"sourceKind": "canonical_entity",
|
||||
"productionRetrievalEligible": True,
|
||||
}
|
||||
without_ref = {
|
||||
**copy.deepcopy(with_ref),
|
||||
"cardId": "card:乙",
|
||||
"name": "乙",
|
||||
"sourceId": "canonical-entity:乙",
|
||||
"sourceRefs": [],
|
||||
"milestones": [{"chapter": 4, "fact": "乙仍留守城内"}],
|
||||
}
|
||||
|
||||
class CardRepository:
|
||||
"""返回同时含有、缺少原文指针的冻结卡。"""
|
||||
|
||||
def search(self, retrieval_plan):
|
||||
del retrieval_plan
|
||||
return [copy.deepcopy(with_ref), copy.deepcopy(without_ref)]
|
||||
|
||||
class ProseRepository:
|
||||
"""按卡指针返回补充历史原文。"""
|
||||
|
||||
def read_source_refs(self, *, work_id, as_of, source_refs):
|
||||
self.call = (work_id, as_of, copy.deepcopy(source_refs))
|
||||
return [{
|
||||
"sourceRef": copy.deepcopy(source_refs[0]),
|
||||
"chapter": 2,
|
||||
"text": "甲在城门受伤。",
|
||||
"purpose": "card_source",
|
||||
}]
|
||||
|
||||
kwargs = base_kwargs()
|
||||
retrieval_result = retrieve_writer_sources(
|
||||
plan=kwargs["retrieval_plan"],
|
||||
card_repository=CardRepository(),
|
||||
prose_repository=ProseRepository(),
|
||||
)
|
||||
contexts = {}
|
||||
for arm, strategy in (
|
||||
("A", "historical_prose_only"),
|
||||
("B", "card_index_only"),
|
||||
("C", "card_index_plus_prose"),
|
||||
):
|
||||
arm_kwargs = copy.deepcopy(kwargs)
|
||||
arm_kwargs.update({
|
||||
"mode": "diagnostic_only",
|
||||
"purpose": "evaluation",
|
||||
"quality_policy_version": "writer-eval-v1",
|
||||
"evidence_strategy": strategy,
|
||||
"retrieval_result": copy.deepcopy(retrieval_result),
|
||||
})
|
||||
contexts[arm] = assemble_context(**arm_kwargs)["context"]
|
||||
|
||||
expected_cards = {"card:甲", "card:乙"}
|
||||
self.assertEqual(contexts["A"]["indexHints"], [])
|
||||
self.assertEqual([item["chapter"] for item in contexts["A"]["proseEvidence"]], [3, 4, 5, 6])
|
||||
self.assertEqual({item["cardId"] for item in contexts["B"]["indexHints"]}, expected_cards)
|
||||
self.assertEqual(contexts["B"]["proseEvidence"], [])
|
||||
self.assertEqual({item["cardId"] for item in contexts["C"]["indexHints"]}, expected_cards)
|
||||
card_manifest_sources = [
|
||||
item
|
||||
for item in contexts["C"]["retrievalManifest"]["sources"]
|
||||
if item.get("sourceType") == "diagnostic_card_index"
|
||||
]
|
||||
self.assertEqual(
|
||||
{item["sourceId"] for item in card_manifest_sources},
|
||||
{"canonical-entity:甲", "canonical-entity:乙"},
|
||||
)
|
||||
self.assertEqual(
|
||||
[item["chapter"] for item in contexts["C"]["proseEvidence"]],
|
||||
[3, 4, 5, 6, 2],
|
||||
)
|
||||
self.assertIn(
|
||||
{"sourceId": "canonical-entity:乙", "reason": "missing_source_refs"},
|
||||
contexts["C"]["omittedSources"],
|
||||
)
|
||||
self.assertTrue(all(not context["factEvidence"] for context in contexts.values()))
|
||||
self.assertTrue(all(not context["acceptanceEligible"] for context in contexts.values()))
|
||||
|
||||
def test_card_prose_is_deduplicated_and_stably_sorted_after_baseline(self):
|
||||
kwargs = base_kwargs()
|
||||
duplicate = prose(6, 106, "第6章完整原文。")
|
||||
|
||||
@ -273,7 +273,24 @@ class RetrieveWriterSourcesTest(unittest.TestCase):
|
||||
{item["sourceType"] for item in result["factEvidence"]},
|
||||
{"historical_prose", "formal_setting", "fine_outline_declared_new"},
|
||||
)
|
||||
self.assertEqual([item["cardId"] for item in result["indexHints"]], ["1", "2"])
|
||||
self.assertEqual(
|
||||
result["indexHints"][0],
|
||||
{
|
||||
"cardId": "1",
|
||||
"name": "角色1",
|
||||
"type": "character",
|
||||
"content": "冻结线内状态",
|
||||
"sourceId": "canonical-entity:1",
|
||||
"sourceVersion": SOURCE_VERSION,
|
||||
"asOf": 5,
|
||||
},
|
||||
)
|
||||
self.assertEqual(result["unverifiedIndexHints"][0]["cardId"], "2")
|
||||
self.assertEqual(
|
||||
result["manifest"]["omittedSources"],
|
||||
[{"sourceId": "canonical-entity:2", "reason": "missing_source_refs"}],
|
||||
)
|
||||
|
||||
def test_snapshot_transaction_is_repeatable_read_and_read_only(self):
|
||||
connection = FakeConnection()
|
||||
|
||||
@ -26,6 +26,27 @@ from writer_contract import ( # noqa: E402
|
||||
def valid_context() -> dict:
|
||||
"""构造覆盖全部必填字段的最小合法上下文。"""
|
||||
|
||||
prose_evidence = []
|
||||
for chapter in range(485, 489):
|
||||
text = f"第{chapter}章冻结原文。"
|
||||
prose_evidence.append(
|
||||
{
|
||||
"evidenceId": f"prose:{chapter}",
|
||||
"chapter": chapter,
|
||||
"sourceRef": {
|
||||
"sourceId": f"chapter:{chapter}",
|
||||
"sourceVersion": f"chapter-{chapter}-v1",
|
||||
"chapter": chapter,
|
||||
"blockId": 1,
|
||||
"startCodePoint": 0,
|
||||
"endCodePoint": len(text),
|
||||
},
|
||||
"contentSha256": "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest(),
|
||||
"purpose": "recent_full_chapter",
|
||||
"text": text,
|
||||
"isRecentBaseline": True,
|
||||
}
|
||||
)
|
||||
context = {
|
||||
"schemaVersion": "writer-context-v1",
|
||||
"runId": "writer-run-001",
|
||||
@ -89,7 +110,7 @@ def valid_context() -> dict:
|
||||
"immediateSituation": "战斗持续",
|
||||
},
|
||||
"factEvidence": [],
|
||||
"proseEvidence": [],
|
||||
"proseEvidence": prose_evidence,
|
||||
"patternReferences": [],
|
||||
"evidenceCoverage": [],
|
||||
"outputContract": {
|
||||
@ -138,6 +159,116 @@ def valid_output() -> dict:
|
||||
|
||||
|
||||
class WriterContractTest(unittest.TestCase):
|
||||
def test_index_hints_are_strict_diagnostic_only_and_frozen(self):
|
||||
"""诊断卡索引提示只能进入不可接受的诊断上下文。"""
|
||||
|
||||
hint = {
|
||||
"cardId": "card-1",
|
||||
"name": "甲",
|
||||
"type": "character",
|
||||
"content": "甲曾在城门出现",
|
||||
"sourceId": "upgrade-book:card-1",
|
||||
"sourceVersion": "upgrade-book-v1",
|
||||
"asOf": 488,
|
||||
}
|
||||
context = valid_context()
|
||||
context.update(
|
||||
{
|
||||
"mode": "diagnostic_only",
|
||||
"purpose": "evaluation",
|
||||
"qualityPolicyVersion": "writer-eval-v1",
|
||||
"acceptanceEligible": False,
|
||||
"evidenceStrategy": "card_index_only",
|
||||
"indexHints": [hint],
|
||||
}
|
||||
)
|
||||
context["proseEvidence"] = []
|
||||
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
|
||||
validate_writer_context(context)
|
||||
|
||||
for mutation in ("production", "unknown_field", "future"):
|
||||
invalid = copy.deepcopy(context)
|
||||
if mutation == "production":
|
||||
invalid.update({"mode": "production", "purpose": "production", "acceptanceEligible": True})
|
||||
elif mutation == "unknown_field":
|
||||
invalid["indexHints"][0]["reason"] = "不得透传"
|
||||
else:
|
||||
invalid["indexHints"][0]["asOf"] = 489
|
||||
invalid["contextSnapshot"]["contextSha256"] = retrieval_identity(invalid)
|
||||
with self.subTest(mutation=mutation), self.assertRaises(ContractError):
|
||||
validate_writer_context(invalid)
|
||||
|
||||
def test_card_index_only_requires_explicit_strategy_and_no_fact_or_prose(self):
|
||||
"""B 臂跳过四章基线必须留在合同字段中,不能隐式旁路。"""
|
||||
|
||||
context = valid_context()
|
||||
context.update(
|
||||
{
|
||||
"mode": "diagnostic_only",
|
||||
"purpose": "diagnostic",
|
||||
"qualityPolicyVersion": "writer-eval-v1",
|
||||
"acceptanceEligible": False,
|
||||
"evidenceStrategy": "card_index_only",
|
||||
"indexHints": [{
|
||||
"cardId": "card-1", "name": "甲", "type": "character",
|
||||
"content": "甲的冻结状态", "sourceId": "card-source:1",
|
||||
"sourceVersion": "cards-v1", "asOf": 488,
|
||||
}],
|
||||
}
|
||||
)
|
||||
context["proseEvidence"] = []
|
||||
context["factEvidence"] = [{
|
||||
"evidenceId": "fact-1", "fact": "甲在场", "sourceType": "formal_setting",
|
||||
"sourceRef": {"sourceId": "setting:1", "sourceVersion": "setting-v1"},
|
||||
"contentSha256": "sha256:" + hashlib.sha256("甲在场".encode()).hexdigest(),
|
||||
"riskLevel": "high",
|
||||
}]
|
||||
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
|
||||
with self.assertRaises(ContractError):
|
||||
validate_writer_context(context)
|
||||
|
||||
def test_diagnostic_prose_strategies_are_explicit_and_keep_baseline(self):
|
||||
"""A/C 臂保留连续历史原文,C 臂只额外加入诊断卡索引提示。"""
|
||||
|
||||
hint = {
|
||||
"cardId": "card-1", "name": "甲", "type": "character",
|
||||
"content": "甲的冻结状态", "sourceId": "card-source:1",
|
||||
"sourceVersion": "cards-v1", "asOf": 488,
|
||||
}
|
||||
for strategy, hints in (
|
||||
("historical_prose_only", []),
|
||||
("card_index_plus_prose", [hint]),
|
||||
):
|
||||
context = valid_context()
|
||||
context.update(
|
||||
{
|
||||
"mode": "diagnostic_only",
|
||||
"purpose": "evaluation",
|
||||
"qualityPolicyVersion": "writer-eval-v1",
|
||||
"acceptanceEligible": False,
|
||||
"evidenceStrategy": strategy,
|
||||
"indexHints": hints,
|
||||
}
|
||||
)
|
||||
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
|
||||
with self.subTest(strategy=strategy):
|
||||
self.assertEqual(
|
||||
[item["chapter"] for item in validate_writer_context(context)["proseEvidence"]],
|
||||
[485, 486, 487, 488],
|
||||
)
|
||||
|
||||
def test_production_cannot_skip_or_break_recent_baseline(self):
|
||||
"""生产上下文不能借诊断策略或不连续近章绕过四章基线。"""
|
||||
|
||||
for mutation in ("diagnostic_strategy", "missing_baseline"):
|
||||
context = valid_context()
|
||||
if mutation == "diagnostic_strategy":
|
||||
context["evidenceStrategy"] = "historical_prose_only"
|
||||
else:
|
||||
context["proseEvidence"].pop(1)
|
||||
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
|
||||
with self.subTest(mutation=mutation), self.assertRaises(ContractError):
|
||||
validate_writer_context(context)
|
||||
def test_run_id_does_not_change_retrieval_identity(self):
|
||||
first = {"runId": "run-a", "query": "咖啡\u0301", "nested": {"value": 1}}
|
||||
second = {"runId": "run-b", "query": "咖啡\u0301", "nested": {"value": 1}}
|
||||
|
||||
@ -341,7 +341,12 @@ def validate_writer_context(value: Any) -> dict[str, Any]:
|
||||
"omittedSources", "acceptanceEligible",
|
||||
}
|
||||
)
|
||||
context = _object(value, "$", required)
|
||||
context = _object(
|
||||
value,
|
||||
"$",
|
||||
required,
|
||||
frozenset({"evidenceStrategy", "indexHints"}),
|
||||
)
|
||||
if context["schemaVersion"] != CONTEXT_VERSION:
|
||||
raise ContractError("$.schemaVersion 版本不支持")
|
||||
run_id = _string(context["runId"], "$.runId")
|
||||
@ -352,6 +357,16 @@ def validate_writer_context(value: Any) -> dict[str, Any]:
|
||||
raise ContractError("$.mode 枚举非法")
|
||||
if context["purpose"] not in {"production", "evaluation", "diagnostic"}:
|
||||
raise ContractError("$.purpose 枚举非法")
|
||||
evidence_strategy = context.get("evidenceStrategy", "production_dual_evidence")
|
||||
if evidence_strategy not in {
|
||||
"production_dual_evidence",
|
||||
"historical_prose_only",
|
||||
"card_index_only",
|
||||
"card_index_plus_prose",
|
||||
}:
|
||||
raise ContractError("$.evidenceStrategy 枚举非法")
|
||||
if context["mode"] == "production" and evidence_strategy != "production_dual_evidence":
|
||||
raise ContractError("生产上下文必须使用 production_dual_evidence")
|
||||
_string(context["qualityPolicyVersion"], "$.qualityPolicyVersion")
|
||||
_integer(context["workId"], "$.workId", minimum=1)
|
||||
target = _integer(context["targetChapter"], "$.targetChapter", minimum=1)
|
||||
@ -435,6 +450,60 @@ def validate_writer_context(value: Any) -> dict[str, Any]:
|
||||
if item["contentSha256"] != expected:
|
||||
raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配")
|
||||
|
||||
index_hints = _array(context.get("indexHints", []), "$.indexHints")
|
||||
for index, hint in enumerate(index_hints):
|
||||
item = _object(
|
||||
hint,
|
||||
f"$.indexHints[{index}]",
|
||||
frozenset(
|
||||
{
|
||||
"cardId",
|
||||
"name",
|
||||
"type",
|
||||
"content",
|
||||
"sourceId",
|
||||
"sourceVersion",
|
||||
"asOf",
|
||||
}
|
||||
),
|
||||
)
|
||||
for field in ("cardId", "name", "type", "content", "sourceId", "sourceVersion"):
|
||||
_string(item[field], f"$.indexHints[{index}].{field}")
|
||||
hint_as_of = _integer(item["asOf"], f"$.indexHints[{index}].asOf", minimum=1)
|
||||
if hint_as_of > as_of:
|
||||
raise ContractError(f"$.indexHints[{index}] 超出冻结线")
|
||||
if index_hints and not (
|
||||
context["mode"] == "diagnostic_only"
|
||||
and context["purpose"] in {"evaluation", "diagnostic"}
|
||||
and context["acceptanceEligible"] is False
|
||||
):
|
||||
raise ContractError("indexHints 只允许用于不可接受的评测或诊断上下文")
|
||||
if context["mode"] == "production" and index_hints:
|
||||
raise ContractError("生产上下文禁止 indexHints")
|
||||
if evidence_strategy == "card_index_only":
|
||||
if context["mode"] != "diagnostic_only" or context["purpose"] not in {"evaluation", "diagnostic"}:
|
||||
raise ContractError("card_index_only 只允许用于诊断上下文")
|
||||
if context["factEvidence"] or context["proseEvidence"] or not index_hints:
|
||||
raise ContractError("card_index_only 必须仅包含 indexHints")
|
||||
elif evidence_strategy == "historical_prose_only":
|
||||
if index_hints or context["factEvidence"] or not context["proseEvidence"]:
|
||||
raise ContractError("historical_prose_only 必须仅包含历史原文")
|
||||
elif evidence_strategy == "card_index_plus_prose":
|
||||
if context["factEvidence"] or not index_hints or not context["proseEvidence"]:
|
||||
raise ContractError("card_index_plus_prose 必须包含 indexHints 与历史原文")
|
||||
|
||||
# v1 生产正文必须携带截至冻结点的连续前四章。诊断 A/C 保持同一
|
||||
# 原文基线,只有显式 card_index_only 策略获准跳过该生产要求。
|
||||
if evidence_strategy != "card_index_only":
|
||||
baseline_chapters = sorted(
|
||||
item["chapter"]
|
||||
for item in context["proseEvidence"]
|
||||
if item["isRecentBaseline"] is True
|
||||
)
|
||||
expected_baseline = list(range(max(1, as_of - 3), as_of + 1))
|
||||
if baseline_chapters != expected_baseline:
|
||||
raise ContractError("原文证据必须包含截至冻结点的连续前四章基线")
|
||||
|
||||
for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")):
|
||||
_source_ref(reference, f"$.patternReferences[{index}]")
|
||||
for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")):
|
||||
|
||||
@ -41,6 +41,14 @@ scope: agent
|
||||
- tokenBudget
|
||||
- omittedSources
|
||||
- acceptanceEligible
|
||||
WriterContext可选字段:
|
||||
evidenceStrategy:
|
||||
枚举: [production_dual_evidence, historical_prose_only, card_index_only, card_index_plus_prose]
|
||||
约束: 生产固定 production_dual_evidence;card_index_only 仅诊断可跳过连续前四章基线
|
||||
indexHints:
|
||||
每项严格字段: [cardId, name, type, content, sourceId, sourceVersion, asOf]
|
||||
来源: retrieval_result.indexHints,覆盖同批检索命中的全部冻结卡,不以 sourceRefs 是否存在为准
|
||||
约束: 仅 mode=diagnostic_only、purpose 为 evaluation/diagnostic、acceptanceEligible=false;生产硬拒绝;asOf 不得越过上下文冻结线
|
||||
WriterOutput必填字段:
|
||||
- schemaVersion
|
||||
- runId
|
||||
@ -60,6 +68,7 @@ scope: agent
|
||||
双证据:
|
||||
factEvidence来源: [historical_prose, formal_setting, canonical_state, fine_outline_declared_new]
|
||||
proseEvidence来源: 历史 Canonical 原文,必须带章号、块、字符区间与内容哈希
|
||||
indexHints用途: 只用于诊断卡索引替代效应,卡负责定位原文而不替代原文事实,不是事实证据,claimLedger不得引用
|
||||
接受边界: evaluation、diagnostic、diagnostic_only 一律 acceptanceEligible=false
|
||||
稳定排序: score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC
|
||||
实现校验器: .claude/skills/read-context/scripts/writer_contract.py
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user