muse-agent-example/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py
zizi e36cd57010 框架: 评测合同与提示词去支架化 + skill 三层重组(Phase 0-4)
【评测合同与提示词】
- writer/detector/judge 评测提示词去支架化:删评测身份/输出格式叮嘱/盲化反向叮嘱/机器字段
  (改由运行时 --json-schema/--tools ""/--strict-mcp-config/输入设计强制),装回各角色本业纪律
- writer 加写作纪律:戏剧化节拍禁抄细纲概述句、具体压倒抽象、每场戏三件套、篇幅用场景写够不注水不缩水
- 篇幅自纠环修订指令区分偏短(加场景)/偏长(删冗余);机械门要求清单(章末钩子/硬事件/角色/伏笔)注入写手硬约束
- 引文校验从「必须恰好1次」放宽为「0次失败关闭、≥1次绑首次」(检测+盲评同口径)
- llm 立 chat_governed 为主入口(chat 降为仅调试),clean_detect 改用 chat_governed 弃自带降级链
- rewrite/expansion/polish 输出合同对齐 writer.md(返回文本、写手不读写工作区、主会话落工作区)
- 修复 confirm/replay-eval 探针两组坏自测(夹具适配现行合同、探针测试自包含不硬编码漂移哈希);补 clean_detect 离线测试
- AGENTS.md 新增 §10「Agent 提示词与 Skill 审查标准」

【skill 三层重组:单向依赖 底座→能力→编排,断两环+修生产倒挂】
- Phase 0: 新建 runtime 底座(claude_runtime/file_cas/raw_vault),断环 C1、修 continuation 生产倒挂
- Phase 1: 评分尺+门判(writer_rubric/fine_outline_rubric/writer_gate/gate_input_builder)收进 quality-gate,断环 C2、名实相符
- Phase 2: 升格管线从 parse-book 独立成 upgrade skill(备份审计契约键名稳定、仅改路径定位)
- Phase 3: replay-eval 瘦成纯编排
- read-context 共用簇(build_snapshot/audit_leakage/check_snapshot/load_reference_work)下沉到新 snapshot skill,消除能力层向上引用
- Phase 4: run_writer_replay.py(130KB)拆成包(_common/budget/authorization/sample/blind/execute),__init__ 全量 re-export 测试零改动

【base 配置】三角色 effort 提 high;提示词更新;探针重测绑定 writer 合同;预算 writer 45 次/总 450 美元(含篇幅修订)

全量离线测试 36 个文件全绿;函数逻辑零改动(仅搬位置/改 import/改文档,capture_code_identity 仅改路径定位)。
2026-07-26 03:28:22 +08:00

446 lines
16 KiB
Python

#!/usr/bin/env python3
"""卡索引驱动的冻结原文检索测试。"""
from __future__ import annotations
import copy
import pathlib
import sys
import unittest
SCRIPT_DIR = pathlib.Path(__file__).resolve().parent
sys.path.insert(0, str(SCRIPT_DIR))
sys.path.insert(0, str(SCRIPT_DIR.parents[1] / "snapshot" / "scripts"))
sys.path.insert(0, str(SCRIPT_DIR.parents[1] / "search" / "scripts"))
from load_reference_work import begin_read_snapshot # noqa: E402
from search import search_cards # noqa: E402
from retrieve_writer_sources import ( # noqa: E402
ProductionCardIndexRepository,
ReplayCardIndexRepository,
RetrievalError,
build_retrieval_plan,
freeze_card,
retrieve_writer_sources,
stable_sort_cards,
)
SOURCE_VERSION = "raw-file-v1:sha256:" + "a" * 64
def source_ref(chapter: int, block_id: int, offset: int = 0) -> dict:
"""构造可回读的冻结原文引用。"""
return {
"sourceId": f"chapter:{chapter}:block:{block_id}",
"sourceVersion": SOURCE_VERSION,
"chapter": chapter,
"blockId": block_id,
"startCodePoint": offset,
"endCodePoint": offset + 4,
}
def card(card_id: str, score: float, chapter: int = 3, offset: int = 0) -> dict:
"""构造含历史里程碑和原文指针的抽取卡索引。"""
return {
"cardId": card_id,
"type": "character",
"name": f"角色{card_id}",
"score": score,
"sourceVersion": SOURCE_VERSION,
"sourceId": f"canonical-entity:{card_id}",
"sourceOffset": offset,
"sourceRefs": [source_ref(chapter, int(card_id), offset)],
"milestones": [
{"chapter": 1, "fact": "初始状态"},
{"chapter": chapter, "fact": "冻结线内状态"},
{"chapter": 9, "fact": "未来终态"},
],
"sourceKind": "canonical_entity",
"sourceStatus": "active",
"bindingStatus": "active",
"productionRetrievalEligible": True,
}
class FakeCardRepository:
"""只返回预置卡片并记录固定计划查询。"""
def __init__(self, cards):
self.cards = cards
self.plans = []
def search(self, plan):
self.plans.append(copy.deepcopy(plan))
return copy.deepcopy(self.cards)
class FakeProseRepository:
"""按来源引用返回冻结片段,不访问数据库。"""
def __init__(self):
self.calls = []
def read_source_refs(self, *, work_id, as_of, source_refs):
self.calls.append((work_id, as_of, copy.deepcopy(source_refs)))
rows = []
for ref in source_refs:
rows.append(
{
"sourceRef": copy.deepcopy(ref),
"chapter": ref["chapter"],
"text": "历史原文",
"purpose": "card_source",
}
)
return rows
class FakeConnection:
"""记录事务声明,证明读取器先锁定只读可重复读快照。"""
def __init__(self):
self.statements = []
def execute(self, statement, params=None):
self.statements.append((" ".join(statement.split()), params))
return self
class FakeQueryResult:
"""模拟 psycopg 查询结果。"""
def __init__(self, rows):
self.rows = rows
def fetchall(self):
return self.rows
class FakeSearchConnection:
"""为 search_cards 提供字段策略与单条 Canonical 卡结果。"""
def __init__(self):
self.statements = []
def __enter__(self):
return self
def __exit__(self, exc_type, exc, traceback):
return False
def execute(self, statement, params=None):
normalized = " ".join(statement.split())
self.statements.append((normalized, params))
if "muse_meta_schema" in normalized:
return FakeQueryResult([("character", {"秘密": False})])
return FakeQueryResult(
[
(
"entity",
11,
{
"型": "character",
"名称": "甲",
"一句话摘要": "历史角色",
"字段": {
"秘密": "不可见",
"sourceRefs": [source_ref(3, 11)],
"milestones": [{"chapter": 3, "fact": "历史事实"}],
},
},
"active",
0.9,
2,
{"sourceRefs": [source_ref(3, 11)]},
"active",
"active",
)
]
)
class RetrieveWriterSourcesTest(unittest.TestCase):
def setUp(self):
self.plan = build_retrieval_plan(
run_id="run-a",
work_id=8,
target_chapter=6,
as_of=5,
fine_outline={
"entities": [
{"id": "character:甲", "type": "character", "name": "甲"},
{"id": "item:剑", "type": "item", "name": "剑"},
],
"relations": [],
"locations": [],
"powerSystems": [],
"hardConstraints": ["甲使用剑"],
},
card_index_version="cards-v1",
prose_index_version="prose-v1",
token_budget={"maxContextChars": 20000},
)
def test_cards_use_documented_stable_order_and_ties_repeat(self):
cards = [card("3", 0.8, offset=2), card("2", 0.8, offset=1), card("1", 0.9)]
expected = ["1", "2", "3"]
for _ in range(3):
self.assertEqual([item["cardId"] for item in stable_sort_cards(cards)], expected)
def test_plan_identity_excludes_run_id(self):
other = copy.deepcopy(self.plan)
other["runId"] = "run-b"
rebuilt = build_retrieval_plan(
run_id="run-b",
work_id=8,
target_chapter=6,
as_of=5,
fine_outline={
"entities": [
{"id": "character:甲", "type": "character", "name": "甲"},
{"id": "item:剑", "type": "item", "name": "剑"},
],
"relations": [],
"locations": [],
"powerSystems": [],
"hardConstraints": ["甲使用剑"],
},
card_index_version="cards-v1",
prose_index_version="prose-v1",
token_budget={"maxContextChars": 20000},
)
self.assertEqual(self.plan["planId"], rebuilt["planId"])
def test_state_as_of_uses_only_milestones_at_or_before_freeze(self):
frozen = freeze_card(card("1", 1.0), as_of=5)
self.assertEqual([item["chapter"] for item in frozen["stateAsOf"]], [1, 3])
self.assertNotIn("未来终态", str(frozen))
def test_target_future_and_unprovable_sources_fail_closed(self):
for bad_ref in (
source_ref(6, 1),
source_ref(7, 1),
{"sourceId": "unknown", "sourceVersion": SOURCE_VERSION},
):
unsafe = card("1", 1.0)
unsafe["sourceRefs"] = [bad_ref]
with self.subTest(bad_ref=bad_ref), self.assertRaises(RetrievalError):
retrieve_writer_sources(
plan=self.plan,
card_repository=FakeCardRepository([unsafe]),
prose_repository=FakeProseRepository(),
)
def test_cards_expand_to_prose_and_authoritative_facts_keep_independent_refs(self):
prose = FakeProseRepository()
no_ref_card = card("2", 0.7)
no_ref_card["sourceRefs"] = []
result = retrieve_writer_sources(
plan=self.plan,
card_repository=FakeCardRepository([card("1", 0.9), no_ref_card]),
prose_repository=prose,
authoritative_facts=[
{
"factId": "setting:1",
"fact": "正式设定事实",
"sourceType": "formal_setting",
"sourceRef": {
"sourceId": "setting:8",
"sourceVersion": "setting-v2",
},
"riskLevel": "high",
},
{
"factId": "outline:new:1",
"fact": "细纲声明的新事实",
"sourceType": "fine_outline_declared_new",
"sourceRef": {
"sourceId": "fine-outline:6",
"sourceVersion": "outline-v3",
"chapter": 6,
},
"riskLevel": "medium",
},
],
)
self.assertEqual(len(prose.calls[0][2]), 1)
self.assertEqual(result["proseEvidence"][0]["text"], "历史原文")
self.assertEqual(
{item["sourceType"] for item in result["factEvidence"]},
{"historical_prose", "formal_setting", "fine_outline_declared_new"},
)
self.assertEqual([item["cardId"] for item in result["indexHints"]], ["1", "2"])
self.assertEqual(
result["indexHints"][0],
{
"cardId": "1",
"name": "角色1",
"type": "character",
"content": "冻结线内状态",
"sourceId": "canonical-entity:1",
"sourceVersion": SOURCE_VERSION,
"asOf": 5,
},
)
self.assertEqual(result["unverifiedIndexHints"][0]["cardId"], "2")
self.assertEqual(
result["manifest"]["omittedSources"],
[{"sourceId": "canonical-entity:2", "reason": "missing_source_refs"}],
)
def test_eval_draft_hint_only_contains_index_metadata_and_proxy_is_explicit(self):
"""评测卡不透传里程碑台阶,整章代理必须保留降级来源和用途。"""
replay_card = card("1", 0.9)
replay_card["sourceKind"] = "eval_draft"
replay_card["sourceRefs"][0]["sourceType"] = "card_chapter_proxy"
result = retrieve_writer_sources(
plan=self.plan,
card_repository=FakeCardRepository([replay_card]),
prose_repository=FakeProseRepository(),
)
hint_content = result["indexHints"][0]["content"]
self.assertEqual(
hint_content,
'{"name":"角色1","sourceChapters":[3],"type":"character"}',
)
self.assertNotIn("冻结线内状态", hint_content)
self.assertEqual(
result["proseEvidence"][0]["sourceRef"]["sourceType"],
"card_chapter_proxy",
)
self.assertEqual(result["proseEvidence"][0]["purpose"], "card_chapter_proxy")
def test_snapshot_transaction_is_repeatable_read_and_read_only(self):
connection = FakeConnection()
begin_read_snapshot(connection)
self.assertEqual(
connection.statements[0][0],
"SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY",
)
def test_production_repository_only_accepts_active_canonical_binding(self):
calls = []
def search_function(**kwargs):
calls.append(kwargs)
return [card("1", 0.9)]
repository = ProductionCardIndexRepository(search_function=search_function)
result = repository.search(self.plan)
self.assertEqual(result[0]["cardId"], "1")
self.assertEqual(calls[0]["scope"], "work")
for field, value in (
("sourceKind", "eval_draft"),
("sourceStatus", "draft"),
("bindingStatus", "inactive"),
("productionRetrievalEligible", False),
):
unsafe = card("1", 0.9)
unsafe[field] = value
repository = ProductionCardIndexRepository(search_function=lambda **_: [unsafe])
with self.subTest(field=field), self.assertRaises(RetrievalError):
repository.search(self.plan)
def test_search_cards_reuses_active_entity_and_binding_sql(self):
connection = FakeSearchConnection()
result = search_cards(
"甲的历史状态",
scope="work",
work_id=8,
ttype="character",
purpose="generation",
top=5,
connection_factory=lambda _: connection,
embedder=lambda _: [0.1, 0.2],
)
sql = connection.statements[1][0]
self.assertIn("en.status='active'", sql)
self.assertIn("b.binding_status='active'", sql)
self.assertIn("en.source_action_policy='allowed'", sql)
self.assertEqual(result[0]["sourceKind"], "canonical_entity")
self.assertTrue(result[0]["productionRetrievalEligible"])
self.assertNotIn("秘密", result[0]["visibleFields"])
def test_replay_repository_is_preregistered_upgrade_book_and_never_production_eligible(self):
replay_card = card("11", 0.9)
replay_card.update(
{
"sourceKind": "eval_draft",
"evaluationStatus": "eval_draft",
"sourceType": "upgrade_book",
"productionRetrievalEligible": False,
}
)
replay_config = {
"targetChapter": 6,
"snapshot": {"asOfChapter": 5, "data": {"chapters": [{"chapter": 5, "text": "安全历史"}]}},
"sources": [
{"sourceId": "chapter:5", "sourceVersion": "chapter-v1", "chapter": 5}
],
"authorization": {
"sourceStatus": "active",
"copyrightStatus": "research_only",
"sourceHash": "sha256:" + "a" * 64,
"sourceVersion": SOURCE_VERSION,
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution"],
"authorizationSnapshot": {
"id": "auth-1",
"version": "v1",
"immutable": True,
"sourceHash": "sha256:" + "a" * 64,
"sourceVersion": SOURCE_VERSION,
"sourceStatus": "active",
"copyrightStatus": "research_only",
"authorizationBasis": "user_authorization",
"allowedPurpose": ["offline_evaluation"],
"forbiddenPurpose": ["external_distribution"],
"checkedAt": "2026-07-20T00:00:00Z",
"revalidationAt": "2099-07-21T00:00:00Z",
},
},
"leakageAudit": {
"targetFacts": {"targetChapter": 6, "forbiddenFacts": []}
},
}
repository = ReplayCardIndexRepository.from_replay_config(
replay_config,
cards=[replay_card],
preregistered_card_ids=["11"],
)
self.assertFalse(repository.search(self.plan)[0]["productionRetrievalEligible"])
denied = copy.deepcopy(replay_config)
denied["authorization"] = {}
with self.assertRaises(RetrievalError):
ReplayCardIndexRepository.from_replay_config(
denied,
cards=[replay_card],
preregistered_card_ids=["11"],
)
wrong = copy.deepcopy(replay_card)
wrong["sourceType"] = "extract_chapter"
with self.assertRaises(RetrievalError):
ReplayCardIndexRepository.from_replay_config(
replay_config,
cards=[wrong],
preregistered_card_ids=["11"],
)
with self.assertRaises(RetrievalError):
ReplayCardIndexRepository.from_replay_config(
replay_config,
cards=[replay_card],
preregistered_card_ids=["12"],
)
if __name__ == "__main__":
unittest.main()