265 lines
11 KiB
Python
265 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""事实证据与原文证据双线组装测试。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import copy
|
|
import hashlib
|
|
import pathlib
|
|
import sys
|
|
import unittest
|
|
|
|
sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent))
|
|
|
|
from assemble_writer_context import AssemblyError, assemble_context # noqa: E402
|
|
from writer_contract import retrieval_identity, validate_writer_context # noqa: E402
|
|
|
|
|
|
def digest(text: str) -> str:
|
|
"""生成测试证据的规范 SHA-256。"""
|
|
|
|
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def prose(chapter: int, block_id: int, text: str, *, recent: bool = False, purpose: str = "card_source") -> dict:
|
|
"""构造块级可引用原文。"""
|
|
|
|
return {
|
|
"evidenceId": f"prose:{chapter}:{block_id}",
|
|
"chapter": chapter,
|
|
"sourceRef": {
|
|
"sourceId": f"chapter:{chapter}:block:{block_id}",
|
|
"sourceVersion": f"chapter-{chapter}-block-{block_id}-v1",
|
|
"chapter": chapter,
|
|
"blockId": block_id,
|
|
"startCodePoint": 0,
|
|
"endCodePoint": len(text),
|
|
},
|
|
"contentSha256": digest(text),
|
|
"purpose": purpose,
|
|
"text": text,
|
|
"isRecentBaseline": recent,
|
|
}
|
|
|
|
|
|
def fact(fact_id: str, text: str, source_type: str = "formal_setting", risk: str = "medium") -> dict:
|
|
"""构造独立权威事实证据。"""
|
|
|
|
return {
|
|
"evidenceId": fact_id,
|
|
"fact": text,
|
|
"sourceType": source_type,
|
|
"sourceRef": {
|
|
"sourceId": f"{source_type}:{fact_id}",
|
|
"sourceVersion": f"{source_type}-v1",
|
|
},
|
|
"contentSha256": digest(text),
|
|
"riskLevel": risk,
|
|
}
|
|
|
|
|
|
def plan(run_id: str = "run-a") -> dict:
|
|
"""构造合法固定检索计划。"""
|
|
|
|
value = {
|
|
"planVersion": "writer-retrieval-plan-v1",
|
|
"runId": run_id,
|
|
"asOf": 6,
|
|
"queries": [],
|
|
"cardIndexVersion": "cards-v1",
|
|
"proseIndexVersion": "prose-v1",
|
|
"filters": {
|
|
"workId": 8,
|
|
"asOfChapter": 6,
|
|
"sourceStatus": "active",
|
|
"authorizationRequired": True,
|
|
},
|
|
"tieBreak": "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC",
|
|
"tokenBudget": {"maxContextChars": 50000},
|
|
}
|
|
value["planId"] = retrieval_identity(value)
|
|
return value
|
|
|
|
|
|
def fine_outline() -> dict:
|
|
"""构造含五类覆盖目标的细纲输入。"""
|
|
|
|
return {
|
|
"sourceRef": {"sourceId": "fine-outline:7", "sourceVersion": "outline-v3", "chapter": 7},
|
|
"hardConstraints": ["甲携剑抵达城门"],
|
|
"adjustableBeats": ["可调整过渡方式"],
|
|
"declaredNewFacts": [],
|
|
"entities": [{"id": "character:甲", "type": "character", "name": "甲"}],
|
|
"relations": [{"id": "relation:甲乙", "type": "character_relation", "name": "甲乙关系"}],
|
|
"items": [{"id": "item:剑", "type": "item", "name": "剑"}],
|
|
"locations": [{"id": "location:城门", "type": "location", "name": "城门"}],
|
|
"powerSystems": [{"id": "power:灵力", "type": "power_system", "name": "灵力"}],
|
|
}
|
|
|
|
|
|
def recent_chapters(last: int = 6) -> list[dict]:
|
|
"""构造第一章到冻结章的完整章节原文。"""
|
|
|
|
return [prose(chapter, 100 + chapter, f"第{chapter}章完整原文。") for chapter in range(1, last + 1)]
|
|
|
|
|
|
def base_kwargs() -> dict:
|
|
"""构造 assemble_context 的公共合法输入。"""
|
|
|
|
return {
|
|
"run_id": "run-a",
|
|
"attempt": 1,
|
|
"mode": "production",
|
|
"purpose": "production",
|
|
"quality_policy_version": "writer-production-v1",
|
|
"work_id": 8,
|
|
"target_chapter": 7,
|
|
"as_of": 6,
|
|
"source_version": "raw-file-v1:sha256:" + "a" * 64,
|
|
"authorization_snapshot": {
|
|
"snapshotId": "auth-1",
|
|
"allowedPurpose": "generation",
|
|
"verifiedAt": "2026-07-20T00:00:00Z",
|
|
},
|
|
"source_status": "active",
|
|
"retrieval_plan": plan(),
|
|
"retrieval_result": {
|
|
"cards": [],
|
|
"factEvidence": [],
|
|
"proseEvidence": [],
|
|
"unverifiedIndexHints": [],
|
|
"manifest": {
|
|
"manifestVersion": "writer-retrieval-manifest-v1",
|
|
"manifestId": "sha256:" + "b" * 64,
|
|
"planId": plan()["planId"],
|
|
"sources": [],
|
|
"omittedSources": [],
|
|
},
|
|
},
|
|
"fine_outline": fine_outline(),
|
|
"narrative_state": {
|
|
"time": "当日",
|
|
"location": "城门",
|
|
"characterPositions": {"甲": "城外"},
|
|
"immediateSituation": "即将交战",
|
|
},
|
|
"recent_chapters": recent_chapters(),
|
|
"output_contract": {
|
|
"targetChars": 4000,
|
|
"minChars": 3600,
|
|
"maxChars": 4400,
|
|
"frontmatterRequired": False,
|
|
"newSettingDeclarationRequired": True,
|
|
},
|
|
"token_budget": {"maxContextChars": 50000},
|
|
"generated_at": "2026-07-20T00:00:00Z",
|
|
}
|
|
|
|
|
|
class AssembleWriterContextTest(unittest.TestCase):
|
|
def test_previous_four_full_chapters_are_baseline(self):
|
|
result = assemble_context(**base_kwargs())
|
|
baseline = [item for item in result["context"]["proseEvidence"] if item["isRecentBaseline"]]
|
|
self.assertEqual([item["chapter"] for item in baseline], [3, 4, 5, 6])
|
|
self.assertTrue(all(item["purpose"] == "recent_full_chapter" for item in baseline))
|
|
validate_writer_context(result["context"])
|
|
|
|
def test_when_fewer_than_four_exist_start_at_chapter_one_without_duplicates(self):
|
|
kwargs = base_kwargs()
|
|
kwargs.update(
|
|
{
|
|
"target_chapter": 3,
|
|
"as_of": 2,
|
|
"recent_chapters": recent_chapters(2),
|
|
"retrieval_plan": {**plan(), "asOf": 2, "filters": {**plan()["filters"], "asOfChapter": 2}},
|
|
}
|
|
)
|
|
kwargs["retrieval_plan"]["planId"] = retrieval_identity({key: value for key, value in kwargs["retrieval_plan"].items() if key != "planId"})
|
|
kwargs["retrieval_result"]["manifest"]["planId"] = kwargs["retrieval_plan"]["planId"]
|
|
result = assemble_context(**kwargs)
|
|
chapters = [item["chapter"] for item in result["context"]["proseEvidence"]]
|
|
self.assertEqual(chapters, [1, 2])
|
|
|
|
def test_missing_chapter_in_required_continuous_baseline_fails_closed(self):
|
|
kwargs = base_kwargs()
|
|
kwargs["recent_chapters"] = [item for item in recent_chapters() if item["chapter"] != 4]
|
|
with self.assertRaises(AssemblyError):
|
|
assemble_context(**kwargs)
|
|
|
|
def test_card_prose_is_deduplicated_and_stably_sorted_after_baseline(self):
|
|
kwargs = base_kwargs()
|
|
duplicate = prose(6, 106, "第6章完整原文。")
|
|
supplemental_b = prose(2, 202, "补充乙。")
|
|
supplemental_a = prose(1, 201, "补充甲。")
|
|
kwargs["retrieval_result"]["proseEvidence"] = [supplemental_b, duplicate, supplemental_a]
|
|
result = assemble_context(**kwargs)
|
|
evidence = result["context"]["proseEvidence"]
|
|
self.assertEqual(sum(item["sourceRef"]["sourceId"] == duplicate["sourceRef"]["sourceId"] for item in evidence), 1)
|
|
supplemental = [item for item in evidence if not item["isRecentBaseline"]]
|
|
self.assertEqual([item["sourceRef"]["sourceId"] for item in supplemental], ["chapter:1:block:201", "chapter:2:block:202"])
|
|
|
|
def test_fact_and_prose_evidence_are_separate_and_five_types_report_coverage(self):
|
|
kwargs = base_kwargs()
|
|
kwargs["retrieval_result"]["factEvidence"] = [
|
|
fact("fact:character", "甲保持警惕", risk="high"),
|
|
fact("fact:relation", "甲乙关系稳定"),
|
|
fact("fact:item", "剑仍在甲手中"),
|
|
fact("fact:location", "城门已经关闭"),
|
|
fact("fact:power", "灵力消耗受限"),
|
|
]
|
|
result = assemble_context(**kwargs)
|
|
context = result["context"]
|
|
self.assertTrue(all("fact" in item and "text" not in item for item in context["factEvidence"]))
|
|
self.assertTrue(all("text" in item and "fact" not in item for item in context["proseEvidence"]))
|
|
self.assertEqual(
|
|
{item["elementType"] for item in context["evidenceCoverage"]},
|
|
{"character", "character_relation", "item", "location", "power_system"},
|
|
)
|
|
self.assertTrue(all(item["status"] in {"supported", "style_gap"} for item in context["evidenceCoverage"]))
|
|
|
|
def test_same_snapshot_and_inputs_have_same_manifest_and_context_hash(self):
|
|
first = assemble_context(**base_kwargs())
|
|
second_kwargs = base_kwargs()
|
|
second_kwargs["run_id"] = "run-b"
|
|
second_kwargs["retrieval_plan"]["runId"] = "run-b"
|
|
second = assemble_context(**second_kwargs)
|
|
self.assertEqual(first["context"]["retrievalManifest"]["manifestId"], second["context"]["retrievalManifest"]["manifestId"])
|
|
self.assertEqual(first["context"]["contextSnapshot"]["contextSha256"], second["context"]["contextSnapshot"]["contextSha256"])
|
|
self.assertEqual(first["manifestMarkdown"], second["manifestMarkdown"])
|
|
|
|
def test_budget_keeps_hard_outline_recent_prose_and_high_risk_fact_with_trace(self):
|
|
kwargs = base_kwargs()
|
|
kwargs["retrieval_result"]["factEvidence"] = [
|
|
fact("fact:high", "甲的高风险身份事实" + "甲" * 300, risk="high"),
|
|
fact("fact:low", "低风险背景" + "乙" * 300, risk="low"),
|
|
]
|
|
kwargs["retrieval_result"]["proseEvidence"] = [prose(1, 900, "很早的补充原文" + "旧" * 500)]
|
|
kwargs["recent_chapters"] = [
|
|
prose(chapter, 100 + chapter, f"第{chapter}章" + "近" * 350)
|
|
for chapter in range(1, 7)
|
|
]
|
|
kwargs["token_budget"] = {"maxContextChars": 7500}
|
|
result = assemble_context(**kwargs)
|
|
context = result["context"]
|
|
self.assertIn("甲携剑抵达城门", context["fineOutline"]["hardConstraints"])
|
|
self.assertIn("fact:high", {item["evidenceId"] for item in context["factEvidence"]})
|
|
baseline = [item["chapter"] for item in context["proseEvidence"] if item["isRecentBaseline"]]
|
|
self.assertEqual(baseline, [3, 4, 5, 6])
|
|
self.assertLessEqual(context["tokenBudget"]["usedContextChars"], 7500)
|
|
self.assertTrue(context["omittedSources"])
|
|
self.assertIn("token_budget", {item["reason"] for item in context["omittedSources"]})
|
|
|
|
def test_budget_that_cannot_hold_four_full_chapters_fails_closed(self):
|
|
kwargs = base_kwargs()
|
|
kwargs["recent_chapters"] = [
|
|
prose(chapter, 100 + chapter, f"第{chapter}章" + "近" * 500)
|
|
for chapter in range(1, 7)
|
|
]
|
|
kwargs["token_budget"] = {"maxContextChars": 3000}
|
|
with self.assertRaisesRegex(AssemblyError, "连续前四章全文基线"):
|
|
assemble_context(**kwargs)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|