zizi c9f69d9d6d 治理: Skill 测试治理第一阶段——harness 控制平面 + 实现测试迁出运行时目录
范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):

1. 新增 harness/ 控制平面
   - skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
   - run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
     空跑与 skip-only 失败关闭、AST 测试形状门
   - manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
   - manifests/test-inventory.json:81 个测试资产登记
   - specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责

2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
   - 71 个测试文件迁移并修复项目根与临时目录运行导入
   - 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
   - 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
   - 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变

3. 运行时文档清理
   - 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
     只保留运行时合同;业务运行合同、额度、授权与离线模式均保留

4. SoT 同步
   - AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
   - 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
   - humanization 覆盖矩阵:活动测试路径同步迁移

验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。

已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
2026-08-19 01:50:20 +08:00

343 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""无工具正文写手 adapter 的隔离、合同与失败关闭测试。"""
from __future__ import annotations
import copy
import hashlib
import json
import pathlib
import subprocess
import sys
import unittest
from decimal import Decimal
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "write-next-chapter" / "scripts"
READ_CONTEXT_DIR = PROJECT_ROOT / ".claude" / "skills" / "assemble-context" / "scripts"
TEST_CONTEXT_DIR = PROJECT_ROOT / "tests" / "skills" / "assemble-context"
sys.path.insert(0, str(SCRIPT_DIR))
sys.path.insert(0, str(READ_CONTEXT_DIR))
sys.path.insert(0, str(TEST_CONTEXT_DIR))
from test_writer_contract import valid_context, valid_draft # noqa: E402
from writer_contract import build_writer_creative_input, canonical_json, retrieval_identity # noqa: E402
from run_writer import ( # noqa: E402
WriterAdapterError,
build_writer_execution_profile,
calculate_dynamic_output_contract,
run_writer,
)
def _bound_context() -> dict:
"""构造带一条事实证据且可供 adapter 绑定检查的合法上下文。"""
context = valid_context()
context["factEvidence"] = [
{
"evidenceId": "fact-1",
"fact": "林澈仍在圣蒂曼城内",
"sourceType": "canonical_state",
"sourceRef": {
"sourceId": "state:8",
"sourceVersion": "state-v1",
"sourceType": "canonical_state",
},
"contentSha256": "sha256:" + "a" * 64,
"riskLevel": "high",
}
]
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
return context
def _writer_draft(_context: dict, *, body: str | None = None) -> dict:
"""构造 writer 模型允许返回的单字段草稿。"""
return valid_draft(body=body if body is not None else "文" * 4000)
class FakeRunner:
"""记录参数并返回预设结果,确保测试不会调用真实模型。"""
def __init__(self, completed: subprocess.CompletedProcess[str] | BaseException):
self.completed = completed
self.calls: list[tuple[list[str], dict]] = []
def __call__(self, command: list[str], **kwargs):
self.calls.append((command, kwargs))
if isinstance(self.completed, BaseException):
raise self.completed
return self.completed
FULL_MODEL_ID = "claude-opus-4-1-20250805"
def _profile(*, timeout_seconds: float = 30):
"""构造绑定完整模型和 WriterDraft v2 schema 的测试 profile。"""
executable = pathlib.Path("/usr/bin/true")
return build_writer_execution_profile(
claude_executable_path=str(executable),
claude_executable_sha256=hashlib.sha256(executable.read_bytes()).hexdigest(),
claude_cli_version="2.1.211",
resolved_model_id=FULL_MODEL_ID,
effort="high",
max_budget_usd_per_call=Decimal("1.500000"),
timeout_seconds=timeout_seconds,
max_context_chars=200000,
system_prompt="只返回 WriterDraft v2:一个只含 candidateBody 的对象。",
)
def _success_runner(context: dict, output: dict | None = None) -> FakeRunner:
"""返回模拟 Claude structured_output 成功 envelope 的 runner。"""
payload = output if output is not None else _writer_draft(context)
stdout = json.dumps(
{
"type": "result",
"subtype": "success",
"is_error": False,
# CLI 2.1.211 正常结束的 terminal_reason 是 "completed"(subtype 才是 "success")
"terminal_reason": "completed",
"stop_reason": "end_turn",
"api_error_status": None,
"total_cost_usd": "0.120000",
"usage": {"input_tokens": 100, "output_tokens": 20},
"modelUsage": {
FULL_MODEL_ID: {
"inputTokens": 100,
"outputTokens": 20,
"costUSD": "0.120000",
}
},
"structured_output": payload,
"result": "禁止作为业务 fallback",
},
ensure_ascii=False,
)
return FakeRunner(subprocess.CompletedProcess(["claude"], 0, stdout=stdout, stderr=""))
class RunWriterTest(unittest.TestCase):
def test_adapter_uses_verified_isolation_flags_and_stdin_only(self):
context = _bound_context()
runner = _success_runner(context)
result = run_writer(
context,
profile=_profile(),
runner=runner,
binding_verifier=lambda _profile: None,
)
self.assertEqual(result["candidateVersion"], 1)
self.assertEqual(result["schemaVersion"], "candidate-envelope-v2")
command, kwargs = runner.calls[0]
required_pairs = {
"--output-format": "json",
"--model": FULL_MODEL_ID,
"--effort": "high",
"--max-budget-usd": "1.500000",
"--tools": "",
}
self.assertIn("--print", command)
self.assertIn("--bare", command)
self.assertIn("--no-session-persistence", command)
self.assertIn("--disable-slash-commands", command)
self.assertIn("--strict-mcp-config", command)
self.assertIn("--json-schema", command)
self.assertIn("--system-prompt", command)
for flag, value in required_pairs.items():
self.assertEqual(command[command.index(flag) + 1], value)
self.assertEqual(command[0], "/usr/bin/sandbox-exec")
self.assertIn("/usr/bin/true", command)
creative_input = build_writer_creative_input(context)
self.assertEqual(kwargs["input"], canonical_json(creative_input))
self.assertTrue(kwargs["text"])
self.assertTrue(kwargs["capture_output"])
self.assertNotIn("shell", kwargs)
self.assertNotIn(canonical_json(creative_input), command)
self.assertNotIn("PATH", kwargs["env"])
serialized_input = kwargs["input"]
for forbidden in (
"runId", "authorizationSnapshot", "retrievalManifest", "contextSnapshot",
"candidateVersion", "acceptanceEligible", "contentSha256",
):
self.assertNotIn(forbidden, serialized_input)
def test_timeout_nonzero_non_json_and_schema_error_fail_closed(self):
context = _bound_context()
invalid_schema = _writer_draft(context)
invalid_schema["candidateSha256"] = "sha256:" + "a" * 64
cases = {
"WRITER_TIMEOUT": FakeRunner(subprocess.TimeoutExpired(["claude"], 1)),
"WRITER_NONZERO_EXIT": FakeRunner(
subprocess.CompletedProcess(["claude"], 7, stdout="", stderr="拒绝执行")
),
"WRITER_RECEIPT_INVALID": FakeRunner(
subprocess.CompletedProcess(["claude"], 0, stdout="not-json", stderr="")
),
"WRITER_SCHEMA_INVALID": _success_runner(context, invalid_schema),
}
for expected_code, runner in cases.items():
with self.subTest(expected_code=expected_code), self.assertRaises(WriterAdapterError) as raised:
run_writer(
context,
profile=_profile(timeout_seconds=1),
runner=runner,
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.code, expected_code)
self.assertFalse(raised.exception.acceptance_eligible)
self.assertNotIn("拒绝执行", str(raised.exception.details))
def test_candidate_length_must_fit_dynamic_context_range(self):
context = _bound_context()
too_short = _writer_draft(context, body="短" * 3599)
with self.assertRaises(WriterAdapterError) as raised:
run_writer(
context,
profile=_profile(),
runner=_success_runner(context, too_short),
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.code, "candidate_length_out_of_range")
self.assertEqual(raised.exception.details["actualHanChars"], 3599)
def test_model_cannot_supply_candidate_hash_or_other_envelope_fields(self):
context = _bound_context()
output = _writer_draft(context)
output["candidateSha256"] = "sha256:" + "a" * 64
with self.assertRaises(WriterAdapterError) as raised:
run_writer(
context,
profile=_profile(),
runner=_success_runner(context, output),
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.code, "WRITER_SCHEMA_INVALID")
def test_dynamic_output_contract_uses_outline_density_and_frozen_history(self):
contract = calculate_dynamic_output_contract(
fine_outline={
"hardConstraints": ["事件一", "事件二", "事件三"],
"foreshadowingActions": [],
"requiredScenes": [],
},
recent_chapter_bodies=["文" * 2501, "文" * 2502, "文" * 2503, "文" * 2504],
)
self.assertEqual(contract["targetChars"], 2100)
self.assertEqual(contract["minChars"], 2000)
self.assertEqual(contract["maxChars"], 2730)
self.assertFalse(contract["frontmatterRequired"])
self.assertNotIn("newSettingDeclarationRequired", contract)
def test_candidate_envelope_metadata_is_bound_by_adapter(self):
context = _bound_context()
result = run_writer(
copy.deepcopy(context),
profile=_profile(),
runner=_success_runner(context),
binding_verifier=lambda _profile: None,
)
self.assertEqual(result["runId"], context["runId"])
self.assertEqual(result["attempt"], context["attempt"])
self.assertEqual(result["contextSnapshotId"], context["contextSnapshot"]["manifestId"])
self.assertEqual(result["contextSnapshotSha256"], context["contextSnapshot"]["contextSha256"])
self.assertEqual(result["acceptanceEligible"], context["acceptanceEligible"])
self.assertEqual(
result["candidateSha256"],
"sha256:" + hashlib.sha256(result["candidateBody"].encode("utf-8")).hexdigest(),
)
def test_writer_profile_binds_v2_schema_and_prompt_ids(self):
profile = _profile()
self.assertEqual(profile.profile_version, "claude-offline-writer-v2")
self.assertEqual(profile.json_schema_id, "writer-draft-v2")
self.assertEqual(profile.system_prompt_id, "writer-system-prompt-v2")
self.assertEqual(profile.json_schema["required"], ["candidateBody"])
self.assertFalse(profile.json_schema["additionalProperties"])
def test_real_default_requires_explicit_frozen_profile(self):
"""未显式传入冻结 profile 时,真实默认入口必须失败关闭。"""
with self.assertRaises(WriterAdapterError) as raised:
run_writer(_bound_context())
self.assertEqual(raised.exception.code, "WRITER_PROFILE_REQUIRED")
def test_run_identity_and_persistence_forward_to_runtime(self):
"""生产记账:运行号默认随上下文,调用方与存证入口显式转发给 runtime。"""
context = _bound_context()
runner = _success_runner(context)
events = []
result = run_writer(
context,
profile=_profile(),
runner=runner,
binding_verifier=lambda _profile: None,
persist_call=events.append,
)
self.assertIn("candidateBody", result)
self.assertEqual(len(events), 1)
event = events[0]
self.assertEqual(event["run_id"], context["runId"])
self.assertEqual(event["caller"], "writer")
self.assertEqual(event["purpose"], "production_generation")
self.assertIn("businessInput", event["prompt"])
self.assertIn("candidateBody", event["response"])
def test_result_field_is_never_business_fallback(self):
"""structured_output 缺失时,即使 result 含合法 WriterDraft 也必须拒绝。"""
context = _bound_context()
envelope = {
"type": "result",
"subtype": "success",
"is_error": False,
# 与 CLI 2.1.211 真实 envelope 一致;若失配 RECEIPT_INVALID 会抢占主码
"terminal_reason": "completed",
"stop_reason": "end_turn",
"api_error_status": None,
"total_cost_usd": "0.120000",
"usage": {"input_tokens": 100, "output_tokens": 20},
"modelUsage": {
FULL_MODEL_ID: {
"inputTokens": 100,
"outputTokens": 20,
"costUSD": "0.120000",
}
},
"result": json.dumps(_writer_draft(context), ensure_ascii=False),
}
runner = FakeRunner(
subprocess.CompletedProcess(["claude"], 0, stdout=json.dumps(envelope), stderr="")
)
with self.assertRaises(WriterAdapterError) as raised:
run_writer(
context,
profile=_profile(),
runner=runner,
binding_verifier=lambda _profile: None,
)
self.assertEqual(raised.exception.code, "WRITER_SCHEMA_INVALID")
if __name__ == "__main__":
unittest.main()