范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
197 lines
7.9 KiB
Python
197 lines
7.9 KiB
Python
#!/usr/bin/env python3
|
|
"""build_snapshot.py 的无网络离线测试。"""
|
|
|
|
import pathlib
|
|
import sys
|
|
import unittest
|
|
|
|
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
|
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "freeze-context" / "scripts"
|
|
sys.path.insert(0, str(SCRIPT_DIR))
|
|
from build_snapshot import ( # noqa: E402
|
|
SnapshotError,
|
|
build_snapshot,
|
|
build_snapshot_manifest,
|
|
filter_milestones,
|
|
filter_outline_windows,
|
|
normalize_chapter,
|
|
remove_terminal_fields,
|
|
)
|
|
|
|
|
|
META = {
|
|
"targetChapter": 489,
|
|
"referenceWork": {"id": "deep-space", "version": "v1"},
|
|
"evaluationSetVersion": "set-v1",
|
|
"strategyVersion": "strategy-v1",
|
|
"authorizationSnapshot": {
|
|
"id": "auth-1",
|
|
"version": "v1",
|
|
"immutable": True,
|
|
"sourceVersion": "work-v1",
|
|
"sourceStatus": "active",
|
|
"allowedPurpose": ["offline_evaluation"],
|
|
"checkedAt": "2026-07-19T00:00:00Z",
|
|
"revalidationAt": "2026-07-20T00:00:00Z",
|
|
},
|
|
"runPermissions": {"purpose": "offline_evaluation", "mode": "dry_run"},
|
|
"armConfig": {"arms": ["outline_only", "outline_plus_cards", "outline_plus_placebo_cards"]},
|
|
}
|
|
|
|
|
|
class BuildSnapshotTest(unittest.TestCase):
|
|
def test_normalize_chapter_rejects_guessing(self):
|
|
self.assertEqual(normalize_chapter(488), 488)
|
|
self.assertEqual(normalize_chapter(" 488 "), 488)
|
|
self.assertIsNone(normalize_chapter(True))
|
|
self.assertIsNone(normalize_chapter(488.0))
|
|
self.assertIsNone(normalize_chapter("第488章"))
|
|
|
|
def test_milestones_are_frozen_by_absolute_chapter(self):
|
|
kept, omitted = filter_milestones(
|
|
[
|
|
{"章": 487, "台阶": "已知"},
|
|
{"章": "488", "台阶": "已知字符串"},
|
|
{"章": "489", "台阶": "未来"},
|
|
{"章": "487-490", "台阶": "跨冻结点"},
|
|
{"台阶": "没有章号"},
|
|
],
|
|
488,
|
|
)
|
|
self.assertEqual([item["台阶"] for item in kept], ["已知", "已知字符串"])
|
|
self.assertEqual(
|
|
[item["reason"] for item in omitted],
|
|
["future_or_crosses_as_of", "future_or_crosses_as_of", "missing_or_unbounded_chapter"],
|
|
)
|
|
self.assertTrue(all("台阶" not in item for item in omitted))
|
|
|
|
def test_outline_windows_require_complete_to_order(self):
|
|
kept, omitted = filter_outline_windows(
|
|
[
|
|
{"window_no": 2, "from_order": 401, "to_order": 430},
|
|
{"window_no": 1, "from_order": 431, "to_order": 490},
|
|
{"window_no": 3, "from_order": "bad", "to_order": 500},
|
|
],
|
|
488,
|
|
)
|
|
self.assertEqual([item["from_order"] for item in kept], [401])
|
|
self.assertEqual([item["reason"] for item in omitted], ["future_or_crosses_as_of", "missing_or_invalid_window_bounds"])
|
|
|
|
def test_terminal_fields_are_removed_recursively(self):
|
|
value = {
|
|
"name": "苏铭",
|
|
"current_state": "终局状态",
|
|
"nested": {"future_arc": "未来计划", "safe": "保留"},
|
|
"items": [{"终态摘要": "不应出现"}, {"safe": "保留"}],
|
|
}
|
|
self.assertEqual(remove_terminal_fields(value), {"name": "苏铭", "nested": {"safe": "保留"}, "items": [{}, {"safe": "保留"}]})
|
|
|
|
def test_nested_card_history_is_frozen(self):
|
|
result = build_snapshot(
|
|
{
|
|
"cards": [
|
|
{
|
|
"name": "生物机甲",
|
|
"milestones": [
|
|
{"chapter": 488, "step": "可见"},
|
|
{"chapter": 489, "step": "未来"},
|
|
],
|
|
"current_state": "终态字段不得透传",
|
|
}
|
|
]
|
|
},
|
|
488,
|
|
"v0",
|
|
target_chapter=489,
|
|
manifest_metadata=META,
|
|
)
|
|
card = result["snapshot"]["cards"][0]
|
|
self.assertEqual(card["milestones"], [{"chapter": 488, "step": "可见"}])
|
|
self.assertNotIn("current_state", card)
|
|
self.assertEqual(result["manifest"]["omittedSources"][0]["reason"], "future_or_crosses_as_of")
|
|
|
|
def test_manifest_is_stable_and_does_not_store_payload(self):
|
|
kwargs = {
|
|
"as_of": 488,
|
|
"target_chapter": 489,
|
|
"snapshot_version": "v0",
|
|
"reference_work": META["referenceWork"],
|
|
"evaluation_set_version": META["evaluationSetVersion"],
|
|
"strategy_version": META["strategyVersion"],
|
|
"authorization_snapshot": META["authorizationSnapshot"],
|
|
"run_permissions": META["runPermissions"],
|
|
"arm_config": META["armConfig"],
|
|
"sections": {"l0": {"sourceId": "task", "sourceVersion": "1", "payload": {"target": 489}}},
|
|
"final_report": {"status": "ready", "hashes": {"sourceHash": "abc"}},
|
|
}
|
|
first = build_snapshot_manifest(**kwargs)
|
|
second = build_snapshot_manifest(**kwargs)
|
|
self.assertEqual(first, second)
|
|
self.assertNotIn("payload", first["sections"]["l0"])
|
|
self.assertEqual(len(first["manifestSha256"]), 64)
|
|
self.assertNotIn("原书", str(first["finalReport"]))
|
|
|
|
def test_final_report_rejects_original_text_fields(self):
|
|
with self.assertRaises(SnapshotError):
|
|
build_snapshot_manifest(
|
|
as_of=488,
|
|
target_chapter=489,
|
|
snapshot_version="v0",
|
|
reference_work=META["referenceWork"],
|
|
evaluation_set_version=META["evaluationSetVersion"],
|
|
strategy_version=META["strategyVersion"],
|
|
authorization_snapshot=META["authorizationSnapshot"],
|
|
run_permissions=META["runPermissions"],
|
|
arm_config=META["armConfig"],
|
|
sections={},
|
|
final_report={"raw_text": "原书全文"},
|
|
)
|
|
|
|
def test_unknown_chapter_collection_is_filtered_and_terminal_fields_are_audited(self):
|
|
result = build_snapshot(
|
|
{
|
|
"chapters": [{"chapter": 489, "fact": "未来"}, {"chapter": 488, "fact": "已知"}],
|
|
"cards": [{"name": "事件", "event_result": "未来结果", "relationshipPlan": "未来计划"}],
|
|
},
|
|
488,
|
|
"v0",
|
|
target_chapter=489,
|
|
manifest_metadata=META,
|
|
)
|
|
self.assertEqual(result["snapshot"]["chapters"], [{"chapter": 488, "fact": "已知"}])
|
|
self.assertNotIn("event_result", result["snapshot"]["cards"][0])
|
|
self.assertNotIn("relationshipPlan", result["snapshot"]["cards"][0])
|
|
reasons = [item["reason"] for item in result["manifest"]["omittedSources"]]
|
|
self.assertIn("future_or_crosses_as_of", reasons)
|
|
self.assertTrue(any(item["reason"] == "terminal_field_removed" for item in result["manifest"]["omittedFields"]))
|
|
|
|
def test_unknown_top_level_section_fails_closed(self):
|
|
with self.assertRaises(SnapshotError):
|
|
build_snapshot(
|
|
{"chapters": [], "futurePayload": "不得透传"},
|
|
488,
|
|
"v0",
|
|
target_chapter=489,
|
|
manifest_metadata=META,
|
|
)
|
|
|
|
def test_final_report_is_closed_set(self):
|
|
with self.assertRaises(SnapshotError):
|
|
build_snapshot_manifest(
|
|
target_chapter=489,
|
|
as_of=488,
|
|
snapshot_version="v0",
|
|
reference_work=META["referenceWork"],
|
|
evaluation_set_version=META["evaluationSetVersion"],
|
|
strategy_version=META["strategyVersion"],
|
|
authorization_snapshot=META["authorizationSnapshot"],
|
|
run_permissions=META["runPermissions"],
|
|
arm_config=META["armConfig"],
|
|
sections={},
|
|
final_report={"payload": "原书全文"},
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|