范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
128 lines
4.4 KiB
Python
128 lines
4.4 KiB
Python
#!/usr/bin/env python3
|
|
"""细纲 detector 机器合同的离线测试。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pathlib
|
|
import sys
|
|
import unittest
|
|
|
|
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
|
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "evaluate-frozen-replay" / "scripts"
|
|
if str(SCRIPT_DIR) not in sys.path:
|
|
sys.path.insert(0, str(SCRIPT_DIR))
|
|
|
|
from fine_outline_detector import validate_detector_report # noqa: E402
|
|
|
|
|
|
class FineOutlineDetectorTest(unittest.TestCase):
|
|
def test_report_must_not_reveal_arm_or_judge_target_role_coverage(self):
|
|
arm_leak = {
|
|
"protocol": "fine_outline_detector_v0",
|
|
"candidateId": "blind-1",
|
|
"arm": "outline_only",
|
|
"findings": [],
|
|
"coverageFindings": [],
|
|
}
|
|
self.assertFalse(validate_detector_report(arm_leak, "blind-1")["ok"])
|
|
|
|
target_role_judgment = {
|
|
"protocol": "fine_outline_detector_v0",
|
|
"candidateId": "blind-1",
|
|
"findings": [],
|
|
"coverageFindings": [
|
|
{"category": "missing_target_role_card", "summary": "越权判断"}
|
|
],
|
|
}
|
|
self.assertFalse(validate_detector_report(target_role_judgment, "blind-1")["ok"])
|
|
|
|
def test_semantic_alias_and_unknown_categories_fail_closed(self):
|
|
for section, category in (
|
|
("findings", "target_new_character_card_missing"),
|
|
("coverageFindings", "new_role_without_card"),
|
|
("findings", "invented_detector_category"),
|
|
("coverageFindings", ["frozen_context_gap"]),
|
|
):
|
|
report = {
|
|
"protocol": "fine_outline_detector_v0",
|
|
"candidateId": "blind-1",
|
|
"findings": [],
|
|
"coverageFindings": [],
|
|
}
|
|
finding = {
|
|
"category": category,
|
|
"severity": "low",
|
|
"location": "candidate",
|
|
"evidenceSummary": "测试",
|
|
}
|
|
report[section] = [finding]
|
|
self.assertFalse(validate_detector_report(report, "blind-1")["ok"])
|
|
|
|
def test_registered_categories_are_accepted_and_detect_skill_assigns_target_coverage_to_eval(self):
|
|
report = {
|
|
"protocol": "fine_outline_detector_v0",
|
|
"candidateId": "blind-1",
|
|
"findings": [
|
|
{
|
|
"category": "entity_state",
|
|
"severity": "medium",
|
|
"location": "entities[0]",
|
|
"evidenceSummary": "与冻结状态不一致",
|
|
}
|
|
],
|
|
"coverageFindings": [
|
|
{
|
|
"category": "frozen_context_gap",
|
|
"summary": "冻结资料缺少已知实体字段",
|
|
}
|
|
],
|
|
}
|
|
self.assertTrue(validate_detector_report(report, "blind-1")["ok"])
|
|
|
|
skill_path = (
|
|
PROJECT_ROOT
|
|
/ ".claude"
|
|
/ "skills"
|
|
/ "check-content-consistency"
|
|
/ "SKILL.md"
|
|
)
|
|
skill = skill_path.read_text(encoding="utf-8")
|
|
self.assertNotIn("要单列为资料覆盖发现", skill)
|
|
self.assertIn("归 judge/eval", skill)
|
|
|
|
def test_nested_findings_reject_arm_and_card_manifest_leakage(self):
|
|
base = {
|
|
"protocol": "fine_outline_detector_v0",
|
|
"candidateId": "blind-1",
|
|
"findings": [
|
|
{
|
|
"category": "entity_state",
|
|
"severity": "low",
|
|
"location": "entities[0]",
|
|
"evidenceSummary": "冻结状态核对",
|
|
}
|
|
],
|
|
"coverageFindings": [
|
|
{
|
|
"category": "frozen_context_gap",
|
|
"summary": "公共冻结事实缺字段",
|
|
}
|
|
],
|
|
}
|
|
leaking_finding = {
|
|
**base,
|
|
"findings": [{**base["findings"][0], "arm": "outline_plus_cards"}],
|
|
}
|
|
leaking_coverage = {
|
|
**base,
|
|
"coverageFindings": [
|
|
{**base["coverageFindings"][0], "cardManifest": {"count": 1}}
|
|
],
|
|
}
|
|
self.assertFalse(validate_detector_report(leaking_finding, "blind-1")["ok"])
|
|
self.assertFalse(validate_detector_report(leaking_coverage, "blind-1")["ok"])
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|