zizi c9f69d9d6d 治理: Skill 测试治理第一阶段——harness 控制平面 + 实现测试迁出运行时目录
范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):

1. 新增 harness/ 控制平面
   - skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
   - run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
     空跑与 skip-only 失败关闭、AST 测试形状门
   - manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
   - manifests/test-inventory.json:81 个测试资产登记
   - specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责

2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
   - 71 个测试文件迁移并修复项目根与临时目录运行导入
   - 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
   - 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
   - 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变

3. 运行时文档清理
   - 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
     只保留运行时合同;业务运行合同、额度、授权与离线模式均保留

4. SoT 同步
   - AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
   - 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
   - humanization 覆盖矩阵:活动测试路径同步迁移

验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。

已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
2026-08-19 01:50:20 +08:00

147 lines
5.1 KiB
Python

#!/usr/bin/env python3
"""细纲 rubric 的离线回归测试。"""
import pathlib
import inspect
import sys
import unittest
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "score-content-quality" / "scripts"
if str(SCRIPT_DIR) not in sys.path:
sys.path.insert(0, str(SCRIPT_DIR))
from fine_outline_rubric import ( # noqa: E402
DIMENSIONS,
RUBRIC_PROFILE,
stability_warning,
validate_report,
validate_scores,
)
def valid_scores():
return {
dimension: {"score": 4, "evidence": f"证据-{dimension}"}
for dimension in DIMENSIONS
}
class FineOutlineRubricTest(unittest.TestCase):
def test_every_dimension_requires_evidence(self):
self.assertEqual(validate_scores(valid_scores()), [])
missing_evidence = valid_scores()
missing_evidence[DIMENSIONS[0]] = {"score": 4}
self.assertTrue(any("缺少证据" in error for error in validate_scores(missing_evidence)))
def test_prose_dimensions_are_rejected(self):
scores = valid_scores()
scores["style_fit"] = {"score": 5, "evidence": "不应出现"}
self.assertTrue(any("禁止正文质量维度" in error for error in validate_scores(scores)))
def test_profile_and_score_range_are_checked(self):
report = {"profile": RUBRIC_PROFILE, "scores": valid_scores()}
self.assertEqual(validate_report(report), [])
bad = {"profile": "quality_gate", "scores": valid_scores()}
bad["scores"][DIMENSIONS[1]] = {"score": 6, "evidence": "超范围"}
self.assertEqual(len(validate_report(bad)), 2)
def test_large_reviewer_gap_warns(self):
first = {dimension: 4 for dimension in DIMENSIONS}
second = {dimension: 4 for dimension in DIMENSIONS}
second[DIMENSIONS[2]] = 5
result = stability_warning(first, second)
self.assertFalse(result["stable"])
self.assertEqual(result["gaps"][DIMENSIONS[2]], 1.0)
def test_missing_dimension_fails_stability_closed(self):
first = {dimension: 4 for dimension in DIMENSIONS}
second = {dimension: 4 for dimension in DIMENSIONS[:-1]}
result = stability_warning(first, second)
self.assertFalse(result["stable"])
self.assertIn(DIMENSIONS[-1], result["missingDimensions"])
def test_batch_report_requires_exact_candidates_and_distinct_judge_identity(self):
self.assertIn("expected_judge_id", inspect.signature(validate_report).parameters)
report = {
"profile": RUBRIC_PROFILE,
"judgeId": "judge-primary",
"evaluations": [
{"candidateId": "blind-1", "scores": valid_scores(), "summary": "摘要一"},
{"candidateId": "blind-2", "scores": valid_scores(), "summary": "摘要二"},
],
}
self.assertEqual(
validate_report(
report,
expected_judge_id="judge-primary",
expected_candidate_ids=("blind-1", "blind-2"),
),
[],
)
duplicate = {**report, "evaluations": [report["evaluations"][0], report["evaluations"][0]]}
self.assertTrue(
validate_report(
duplicate,
expected_judge_id="judge-primary",
expected_candidate_ids=("blind-1", "blind-2"),
)
)
invalid_id = {
**report,
"evaluations": [{**report["evaluations"][0], "candidateId": ["blind-1"]}],
}
self.assertTrue(validate_report(invalid_id, expected_judge_id="judge-primary"))
arm_leak = {**report, "arm": "outline_only"}
self.assertTrue(
validate_report(
arm_leak,
expected_judge_id="judge-primary",
expected_candidate_ids=("blind-1", "blind-2"),
)
)
def test_nested_evaluation_and_score_reject_arm_card_manifest_leakage(self):
evaluation = {
"candidateId": "blind-1",
"scores": valid_scores(),
"summary": "短安全摘要",
}
report = {
"profile": RUBRIC_PROFILE,
"judgeId": "judge-primary",
"evaluations": [evaluation],
}
leaking_evaluation = {
**report,
"evaluations": [{**evaluation, "arm": "outline_only"}],
}
leaking_scores = valid_scores()
leaking_scores[DIMENSIONS[0]] = {
**leaking_scores[DIMENSIONS[0]],
"cardManifest": {"count": 1},
}
leaking_score = {
**report,
"evaluations": [{**evaluation, "scores": leaking_scores}],
}
self.assertTrue(
validate_report(
leaking_evaluation,
expected_judge_id="judge-primary",
expected_candidate_ids=("blind-1",),
)
)
self.assertTrue(
validate_report(
leaking_score,
expected_judge_id="judge-primary",
expected_candidate_ids=("blind-1",),
)
)
if __name__ == "__main__":
unittest.main()