范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
113 lines
4.2 KiB
Python
113 lines
4.2 KiB
Python
#!/usr/bin/env python3
|
|
"""Writer 全评测集预注册顺序与盲化分配测试。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import pathlib
|
|
import sys
|
|
import unittest
|
|
|
|
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
|
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "evaluate-frozen-replay" / "scripts"
|
|
if str(SCRIPT_DIR) not in sys.path:
|
|
sys.path.insert(0, str(SCRIPT_DIR))
|
|
|
|
from writer_eval_preregister import ( # noqa: E402
|
|
PreregistrationError,
|
|
build_balanced_preregistration,
|
|
validate_balanced_position_counts,
|
|
)
|
|
|
|
|
|
class WriterEvalPreregisterTest(unittest.TestCase):
|
|
"""覆盖评测集级平衡、确定性和独立盲化命名空间。"""
|
|
|
|
def setUp(self):
|
|
"""固定五个样本,复现 Gate A 最小预注册集合。"""
|
|
|
|
self.evaluation_set_version = "writer-gate-a-deep-space-v1"
|
|
self.sample_ids = [
|
|
"deep-space-489-battle",
|
|
"deep-space-321-character-dialogue",
|
|
"deep-space-544-turning-point",
|
|
"deep-space-199-information-reveal",
|
|
"deep-space-523-returning-character",
|
|
]
|
|
|
|
def test_five_samples_use_hash_sort_and_balanced_abc_rotations(self):
|
|
"""五样本先按规定哈希排序,再整体循环 ABC/BCA/CAB。"""
|
|
|
|
registration = build_balanced_preregistration(
|
|
evaluation_set_version=self.evaluation_set_version,
|
|
sample_ids=self.sample_ids,
|
|
)
|
|
expected_order = sorted(
|
|
self.sample_ids,
|
|
key=lambda sample_id: hashlib.sha256(
|
|
f"{self.evaluation_set_version}{sample_id}".encode("utf-8")
|
|
).hexdigest(),
|
|
)
|
|
self.assertEqual(
|
|
[item["sampleId"] for item in registration["armOrderTable"]],
|
|
expected_order,
|
|
)
|
|
self.assertEqual(
|
|
[item["armOrder"] for item in registration["armOrderTable"]],
|
|
[
|
|
["A", "B", "C"],
|
|
["B", "C", "A"],
|
|
["C", "A", "B"],
|
|
["A", "B", "C"],
|
|
["B", "C", "A"],
|
|
],
|
|
)
|
|
validate_balanced_position_counts(registration["armOrderTable"], order_field="armOrder")
|
|
validate_balanced_position_counts(
|
|
registration["blindAssignmentTable"], order_field="armOrder"
|
|
)
|
|
for table_name in ("armOrderTable", "blindAssignmentTable"):
|
|
counts = registration["positionCounts"][table_name]
|
|
for position in counts:
|
|
self.assertLessEqual(max(position.values()) - min(position.values()), 1)
|
|
|
|
def test_blind_assignment_is_deterministic_but_uses_independent_namespace(self):
|
|
"""盲化分配可复现,但不能复用运行顺序表的样本排序。"""
|
|
|
|
first = build_balanced_preregistration(
|
|
evaluation_set_version=self.evaluation_set_version,
|
|
sample_ids=self.sample_ids,
|
|
)
|
|
second = build_balanced_preregistration(
|
|
evaluation_set_version=self.evaluation_set_version,
|
|
sample_ids=list(reversed(self.sample_ids)),
|
|
)
|
|
self.assertEqual(first, second)
|
|
self.assertNotEqual(
|
|
[item["sampleId"] for item in first["armOrderTable"]],
|
|
[item["sampleId"] for item in first["blindAssignmentTable"]],
|
|
)
|
|
self.assertEqual(first["blindNamespace"], "writer-blind-assignment-v1")
|
|
|
|
def test_duplicate_samples_and_unbalanced_tables_fail_closed(self):
|
|
"""重复样本或逐样本随机构造出的失衡表都不能通过预注册。"""
|
|
|
|
with self.assertRaises(PreregistrationError):
|
|
build_balanced_preregistration(
|
|
evaluation_set_version=self.evaluation_set_version,
|
|
sample_ids=["same", "same"],
|
|
)
|
|
with self.assertRaises(PreregistrationError):
|
|
validate_balanced_position_counts(
|
|
[
|
|
{"sampleId": "one", "armOrder": ["A", "B", "C"]},
|
|
{"sampleId": "two", "armOrder": ["A", "B", "C"]},
|
|
{"sampleId": "three", "armOrder": ["A", "B", "C"]},
|
|
],
|
|
order_field="armOrder",
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|