范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
107 lines
4.8 KiB
Python
107 lines
4.8 KiB
Python
#!/usr/bin/env python3
|
|
"""CandidateEnvelope v2 机械硬门的离线测试。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import copy
|
|
import hashlib
|
|
import pathlib
|
|
import sys
|
|
import unittest
|
|
|
|
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
|
SKILLS_DIR = PROJECT_ROOT / ".claude" / "skills"
|
|
SCRIPT_DIR = SKILLS_DIR / "check-content-consistency" / "scripts"
|
|
CONTINUATION_DIR = SKILLS_DIR / "write-next-chapter" / "scripts"
|
|
READ_CONTEXT_DIR = SKILLS_DIR / "assemble-context" / "scripts"
|
|
WRITER_TEST_DIR = PROJECT_ROOT / "tests" / "skills" / "write-next-chapter"
|
|
for path in (SCRIPT_DIR, CONTINUATION_DIR, READ_CONTEXT_DIR, WRITER_TEST_DIR):
|
|
if str(path) not in sys.path:
|
|
sys.path.insert(0, str(path))
|
|
|
|
from test_run_writer import _bound_context # noqa: E402
|
|
from check_writer_candidate import check_writer_candidate # noqa: E402
|
|
from writer_contract import build_candidate_envelope, han_count # noqa: E402
|
|
|
|
|
|
def _candidate_body() -> str:
|
|
prefix = "林澈完成围攻突围,又把旧徽章压回掌心。"
|
|
hook = "城门忽然打开"
|
|
return prefix + "文" * (4000 - han_count(prefix) - han_count(hook)) + hook
|
|
|
|
|
|
def _requirements() -> dict:
|
|
return {
|
|
"requiredEvents": [{"requirementId": "event-1", "anchors": ["完成围攻突围"]}],
|
|
"requiredCharacters": ["林澈"],
|
|
"foreshadowingActions": [{"requirementId": "foreshadow-1", "anchors": ["旧徽章"]}],
|
|
"chapterEndHook": {"requirementId": "hook-1", "anchors": ["城门忽然打开"], "maxDistanceFromEnd": 20},
|
|
}
|
|
|
|
|
|
def _rehash(output: dict) -> None:
|
|
output["candidateSha256"] = "sha256:" + hashlib.sha256(output["candidateBody"].encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _valid_pair() -> tuple[dict, dict]:
|
|
context = _bound_context()
|
|
return context, build_candidate_envelope(context, {"candidateBody": _candidate_body()})
|
|
|
|
|
|
class CheckWriterCandidateTest(unittest.TestCase):
|
|
def test_missing_outline_anchors_block(self) -> None:
|
|
context, output = _valid_pair()
|
|
mutations = {
|
|
"HARD_EVENT_MISSING": ("完成围攻突围", "完成撤离"),
|
|
"REQUIRED_CHARACTER_MISSING": ("林澈", "无名人"),
|
|
"FORESHADOWING_ACTION_MISSING": ("旧徽章", "旧物件"),
|
|
"CHAPTER_END_HOOK_MISSING": ("城门忽然打开", "风声渐渐平息"),
|
|
}
|
|
for expected_code, (old, new) in mutations.items():
|
|
candidate = copy.deepcopy(output)
|
|
candidate["candidateBody"] = candidate["candidateBody"].replace(old, new)
|
|
_rehash(candidate)
|
|
report = check_writer_candidate(context, candidate, _requirements())
|
|
with self.subTest(expected_code=expected_code):
|
|
self.assertFalse(report["passed"])
|
|
self.assertIn(expected_code, {item["code"] for item in report["blockingFailures"]})
|
|
|
|
def test_hash_context_and_length_are_mechanical(self) -> None:
|
|
context, output = _valid_pair()
|
|
bad_hash = copy.deepcopy(output)
|
|
bad_hash["candidateSha256"] = "sha256:" + "f" * 64
|
|
self.assertIn("CANDIDATE_HASH_MISMATCH", {item["code"] for item in check_writer_candidate(context, bad_hash, _requirements())["blockingFailures"]})
|
|
|
|
bad_context = copy.deepcopy(output)
|
|
bad_context["contextSnapshotSha256"] = "sha256:" + "e" * 64
|
|
self.assertIn("CONTEXT_BINDING_MISMATCH", {item["code"] for item in check_writer_candidate(context, bad_context, _requirements())["blockingFailures"]})
|
|
|
|
short = copy.deepcopy(output)
|
|
short["candidateBody"] = "文" * 1000
|
|
_rehash(short)
|
|
self.assertIn("CANDIDATE_LENGTH_OUT_OF_RANGE", {item["code"] for item in check_writer_candidate(context, short, _requirements())["blockingFailures"]})
|
|
|
|
def test_writer_owned_semantic_fields_are_rejected_not_consumed(self) -> None:
|
|
context, output = _valid_pair()
|
|
for field, value in (
|
|
("claimLedger", []), ("evidenceRequests", []), ("newSettingDeclarations", []),
|
|
):
|
|
candidate = copy.deepcopy(output)
|
|
candidate[field] = value
|
|
report = check_writer_candidate(context, candidate, _requirements())
|
|
with self.subTest(field=field):
|
|
self.assertFalse(report["passed"])
|
|
self.assertIn("OUTPUT_CONTRACT_INVALID", {item["code"] for item in report["blockingFailures"]})
|
|
|
|
def test_valid_candidate_passes_and_requests_independent_semantic_detection(self) -> None:
|
|
context, output = _valid_pair()
|
|
report = check_writer_candidate(context, output, _requirements())
|
|
self.assertTrue(report["passed"])
|
|
self.assertEqual(report["candidateSha256"], output["candidateSha256"])
|
|
self.assertEqual(report["semanticReview"]["status"], "pending")
|
|
self.assertEqual(report["semanticReview"]["inputContract"], "WriterContext v1 + CandidateEnvelope v2 -> SemanticDetection v3")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|