muse-agent-example/harness/manifests/test-inventory.json
zizi c9f69d9d6d 治理: Skill 测试治理第一阶段——harness 控制平面 + 实现测试迁出运行时目录
范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):

1. 新增 harness/ 控制平面
   - skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
   - run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
     空跑与 skip-only 失败关闭、AST 测试形状门
   - manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
   - manifests/test-inventory.json:81 个测试资产登记
   - specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责

2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
   - 71 个测试文件迁移并修复项目根与临时目录运行导入
   - 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
   - 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
   - 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变

3. 运行时文档清理
   - 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
     只保留运行时合同;业务运行合同、额度、授权与离线模式均保留

4. SoT 同步
   - AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
   - 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
   - humanization 覆盖矩阵:活动测试路径同步迁移

验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。

已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
2026-08-19 01:50:20 +08:00

1003 lines
44 KiB
JSON

{
"schema_version": 1,
"generated_scope": "Current agent-example source test assets: .claude/skills/**/test_*.py and *_test.py, tests/skills/** source files, humanization/tests/** source files, harness/**/test_*.py, dashboard/test_server_display.py, and other obvious source test files; excludes .git, .venv, __pycache__ compiled artifacts, deleted working-tree files, and harness specification documents.",
"entries": [
{
"path": "tests/skills/access-database/test_authorization_snapshot_ddl.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "access-database",
"kind": "tool_contract",
"evidence_level": "static_structure",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver."
},
{
"path": "tests/skills/access-database/test_db_params.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "access-database",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked."
},
{
"path": "tests/skills/access-database/test_skill_catalog.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "access-database",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files."
},
{
"path": "tests/skills/assemble-context/test_assemble_writer_context.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts."
},
{
"path": "tests/skills/assemble-context/test_fine_outline_reader.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing."
},
{
"path": "tests/skills/assemble-context/test_fine_outline_unification.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler."
},
{
"path": "tests/skills/assemble-context/test_pattern_binding_reader.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior."
},
{
"path": "tests/skills/assemble-context/test_retrieve_writer_sources.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections."
},
{
"path": "tests/skills/assemble-context/test_style_loader.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection."
},
{
"path": "tests/skills/assemble-context/test_writer_contract.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "assemble-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory."
},
{
"path": "tests/skills/call-content-model/test_call_persistence.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "call-content-model",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter."
},
{
"path": "tests/skills/call-content-model/test_quota.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "call-content-model",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls."
},
{
"path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "capture-ai-flavor-cases",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked."
},
{
"path": "tests/skills/check-content-consistency/test_build_semantic_input.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "check-content-consistency",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection."
},
{
"path": "tests/skills/check-content-consistency/test_check_writer_candidate.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "check-content-consistency",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields."
},
{
"path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "check-content-consistency",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called."
},
{
"path": "tests/skills/clean-book-text/test_clean_detect_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "clean-book-text",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked."
},
{
"path": "tests/skills/decide-candidate/test_confirm_knowledge_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "decide-candidate",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs."
},
{
"path": "tests/skills/decide-candidate/test_fact_delta.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "decide-candidate",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions."
},
{
"path": "tests/skills/decide-candidate/test_fact_delta_db.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "decide-candidate",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["postgresql"],
"side_effects": ["postgresql"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows."
},
{
"path": "tests/skills/decide-candidate/test_projection_db.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "decide-candidate",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["postgresql"],
"side_effects": ["postgresql"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup."
},
{
"path": "tests/skills/decide-candidate/test_write_canonical_db.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "decide-candidate",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["postgresql"],
"side_effects": ["postgresql"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards."
},
{
"path": "tests/skills/decide-candidate/test_writer_acceptance.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "decide-candidate",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store."
},
{
"path": "tests/skills/deconstruct-book/test_parse_llm_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "deconstruct-book",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files."
},
{
"path": "tests/skills/deconstruct-book/test_parse_outline_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "deconstruct-book",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function."
},
{
"path": ".claude/skills/design-story-foundation/scripts/test_serial_merge.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "design-story-foundation",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests packet construction and raw-output parsing with temporary Markdown files; no model or service driver."
},
{
"path": ".claude/skills/design-story-foundation/scripts/test_validate_candidates.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "design-story-foundation",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates candidate tree/heading/placeholder/root contracts using temporary candidate files."
},
{
"path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "diagnose-ai-flavor",
"kind": "domain_eval",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge."
},
{
"path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "embed-knowledge",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows."
},
{
"path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "establish-voice-baseline",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "tool_contract",
"evidence_level": "static_structure",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_gate_input_builder.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_load_writer_reference_work.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_pattern_reference_injection.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_refresh_runtime_probe.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "runtime_probe",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Exercises runtime-probe refresh and authorization gates with fake invocation results and temp output files."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_run_replay.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem", "subprocess"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs the replay orchestrator with a generated fake-agent executable, temp output, and synthetic planner/detector/judge responses."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_run_writer_replay.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem", "raw_vault"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_writer_eval_preregister.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration."
},
{
"path": "tests/skills/evaluate-frozen-replay/test_writer_gate.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "evaluate-frozen-replay",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output."
},
{
"path": "tests/skills/execute-claude-task/test_claude_runtime.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "execute-claude-task",
"kind": "runtime_probe",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests runtime profile/receipt/sandbox/environment handling through a mocked subprocess runner and temp isolation directories."
},
{
"path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "extract-chapter-knowledge",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions."
},
{
"path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "extract-work-knowledge",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry."
},
{
"path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "extract-work-knowledge",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made."
},
{
"path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "extract-work-knowledge",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["postgresql"],
"side_effects": ["postgresql"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1."
},
{
"path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "extract-work-knowledge",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics."
},
{
"path": "tests/skills/freeze-context/test_audit_leakage.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "freeze-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records."
},
{
"path": "tests/skills/freeze-context/test_build_snapshot.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "freeze-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory."
},
{
"path": "tests/skills/freeze-context/test_check_snapshot.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "freeze-context",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution."
},
{
"path": "tests/skills/freeze-context/test_load_reference_work.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "freeze-context",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection."
},
{
"path": "tests/skills/maintain-work-extraction/test_backup_upgrade_work_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "maintain-work-extraction",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched."
},
{
"path": "tests/skills/maintain-work-extraction/test_reset_upgrade_work_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "maintain-work-extraction",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries."
},
{
"path": "tests/skills/plan-chapter/test_contract.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "plan-chapter",
"kind": "tool_contract",
"evidence_level": "static_structure",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts."
},
{
"path": "tests/skills/plan-story/test_field_coverage.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "plan-story",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write."
},
{
"path": "tests/skills/plan-story/test_record_planning_execution.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "plan-story",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions."
},
{
"path": "tests/skills/plan-story/test_repair_deterministic_receipt.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "plan-story",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries."
},
{
"path": "tests/skills/plan-story/test_select_patterns_offline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "plan-story",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results."
},
{
"path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "prevent-ai-flavor",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched."
},
{
"path": "tests/skills/record-run-evidence/test_file_cas.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "record-run-evidence",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions."
},
{
"path": "tests/skills/record-run-evidence/test_persist_raw.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "record-run-evidence",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests only the raw secret-pattern validator with direct strings."
},
{
"path": "tests/skills/record-run-evidence/test_raw_vault.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "record-run-evidence",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["offline", "filesystem", "raw_vault"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery."
},
{
"path": "tests/skills/record-run-evidence/test_record_failed_run.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "record-run-evidence",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence."
},
{
"path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "record-run-evidence",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks receipt eligibility predicates using in-memory values only."
},
{
"path": "tests/skills/record-run-evidence/test_run_registry.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "record-run-evidence",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests run ID formatting and terminal-state rejection before database access."
},
{
"path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "revise-ai-flavor",
"kind": "domain_eval",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked."
},
{
"path": "tests/skills/score-content-quality/test_lesson_registry_db.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "score-content-quality",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["postgresql"],
"side_effects": ["postgresql"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them."
},
{
"path": "tests/skills/score-content-quality/test_rubric.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "score-content-quality",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory."
},
{
"path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "score-content-quality",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used."
},
{
"path": "tests/skills/score-content-quality/test_writer_rubric.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "score-content-quality",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts."
},
{
"path": "tests/skills/search-knowledge/test_search.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "search-knowledge",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding."
},
{
"path": "tests/skills/write-next-chapter/test_candidate_cas.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "write-next-chapter",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops."
},
{
"path": "tests/skills/write-next-chapter/test_candidate_cas_db.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "write-next-chapter",
"kind": "integration",
"evidence_level": "real_dependency_integration",
"requires": ["postgresql"],
"side_effects": ["postgresql"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows."
},
{
"path": "tests/skills/write-next-chapter/test_persist_writer_run.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "write-next-chapter",
"kind": "tool_unit",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call."
},
{
"path": "tests/skills/write-next-chapter/test_run_writer.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "write-next-chapter",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Runs the writer adapter against a FakeRunner/CompletedProcess and temporary profile inputs; no real Claude process or model."
},
{
"path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "write-next-chapter",
"kind": "fake_pipeline",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors."
},
{
"path": "tests/skills/write-next-chapter/test_semantic_verdict.py",
"scope": "runtime_skill",
"owner_skill_or_domain": "write-next-chapter",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence."
},
{
"path": "dashboard/test_server_display.py",
"scope": "other",
"owner_skill_or_domain": "dashboard",
"kind": "tool_contract",
"evidence_level": "deterministic_offline",
"requires": ["offline"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Tests dashboard display/encoding helpers and synthetic AI-flavor views; the file declares no database connection."
},
{
"path": "harness/test_run_selected.py",
"scope": "harness",
"owner_skill_or_domain": "harness",
"kind": "harness_self_test",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem", "subprocess"],
"side_effects": ["filesystem", "subprocess"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Uses temporary manifests and a fake Python child process to test selector, dependency, timeout, nonzero and output-summary handling."
},
{
"path": "harness/test_skill_harness.py",
"scope": "harness",
"owner_skill_or_domain": "harness",
"kind": "harness_self_test",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Creates temporary SKILL.md/manifest fixtures and tests harness static-audit reports and CLI exit codes."
},
{
"path": "humanization/tests/test_contracts.py",
"scope": "domain",
"owner_skill_or_domain": "humanization",
"kind": "domain_eval",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Checks humanization asset contracts and executes the synthetic U0 patch/review replay; no external Agent/model driver."
},
{
"path": "humanization/tests/test_framework_coverage.py",
"scope": "domain",
"owner_skill_or_domain": "humanization",
"kind": "tool_contract",
"evidence_level": "static_structure",
"requires": ["offline", "filesystem"],
"side_effects": ["none"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Reads the research coverage YAML and checks capability owners, implementation paths, and status values."
},
{
"path": "humanization/tests/test_humanization_v2.py",
"scope": "domain",
"owner_skill_or_domain": "humanization",
"kind": "domain_eval",
"evidence_level": "deterministic_offline",
"requires": ["offline", "filesystem"],
"side_effects": ["filesystem"],
"skill_behavior_eval": false,
"classification_confidence": "high",
"classification_basis": "Evaluates synthetic voice/rule/carrier gates and lifecycle fixtures with temporary files; no external Agent/model reads SKILL.md."
}
],
"summary": {
"entry_count": 81,
"by_scope": {
"runtime_skill": 75,
"domain": 3,
"harness": 2,
"other": 1
},
"by_kind": {
"tool_unit": 13,
"tool_contract": 30,
"integration": 8,
"fake_pipeline": 22,
"runtime_probe": 2,
"skill_behavior_eval": 0,
"domain_eval": 4,
"harness_self_test": 2
},
"by_evidence_level": {
"static_structure": 4,
"deterministic_offline": 69,
"real_dependency_integration": 8
}
}
}