1831 lines
58 KiB
JSON
1831 lines
58 KiB
JSON
{
|
||
"schema_version": 1,
|
||
"generated_scope": "Current agent-example source test assets: .agent/skills/**/test_*.py and *_test.py, tests/skills/** source files, humanization/tests/** source files, harness/**/test_*.py, dashboard/test_server_display.py, and other obvious source test files; excludes .git, .venv, __pycache__ compiled artifacts, deleted working-tree files, and harness specification documents.",
|
||
"entries": [
|
||
{
|
||
"path": "dashboard/test_server_display.py",
|
||
"scope": "other",
|
||
"owner_skill_or_domain": "dashboard",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests dashboard display/encoding helpers and synthetic AI-flavor views; the file declares no database connection."
|
||
},
|
||
{
|
||
"path": "harness/evals/skills/diagnose-ai-flavor/run_eval.py",
|
||
"kind": "skill_behavior_eval",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "diagnose-ai-flavor",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"model",
|
||
"credentials"
|
||
],
|
||
"side_effects": [
|
||
"model"
|
||
],
|
||
"skill_behavior_eval": true,
|
||
"classification_basis": "Skill 行为评测入口:默认真实模型适配器,未授权时以稳定码失败关闭;fake 适配器只验证评测管道",
|
||
"classification_confidence": "high"
|
||
},
|
||
{
|
||
"path": "harness/evals/test_skill_eval.py",
|
||
"kind": "harness_self_test",
|
||
"scope": "harness",
|
||
"owner_skill_or_domain": "harness",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [],
|
||
"skill_behavior_eval": false,
|
||
"classification_basis": "行为评测引擎的确定性离线自测:裁决逻辑、失败关闭和报告合同;不构成 Skill 行为证据",
|
||
"classification_confidence": "high"
|
||
},
|
||
{
|
||
"path": "harness/test_run_selected.py",
|
||
"scope": "harness",
|
||
"owner_skill_or_domain": "harness",
|
||
"kind": "harness_self_test",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem",
|
||
"subprocess"
|
||
],
|
||
"side_effects": [
|
||
"filesystem",
|
||
"subprocess"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses temporary manifests and a fake Python child process to test selector, dependency, timeout, nonzero and output-summary handling."
|
||
},
|
||
{
|
||
"path": "harness/test_skill_harness.py",
|
||
"scope": "harness",
|
||
"owner_skill_or_domain": "harness",
|
||
"kind": "harness_self_test",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Creates temporary SKILL.md/manifest fixtures and tests harness static-audit reports and CLI exit codes."
|
||
},
|
||
{
|
||
"path": "humanization/tests/test_contracts.py",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "humanization",
|
||
"kind": "domain_eval",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks humanization asset contracts and executes the synthetic U0 patch/review replay; no external Agent/model driver."
|
||
},
|
||
{
|
||
"path": "humanization/tests/test_framework_coverage.py",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "humanization",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "static_structure",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Reads the research coverage YAML and checks capability owners, implementation paths, and status values."
|
||
},
|
||
{
|
||
"path": "humanization/tests/test_humanization_v2.py",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "humanization",
|
||
"kind": "domain_eval",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Evaluates synthetic voice/rule/carrier gates and lifecycle fixtures with temporary files; no external Agent/model reads SKILL.md."
|
||
},
|
||
{
|
||
"path": "humanization/tests/test_load_db.py",
|
||
"kind": "tool_contract",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "humanization",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [],
|
||
"skill_behavior_eval": false,
|
||
"classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同",
|
||
"classification_confidence": "high"
|
||
},
|
||
{
|
||
"path": "humanization/tests/test_load_db_pg_smoke.py",
|
||
"kind": "integration",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "humanization",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_basis": "真实 PostgreSQL 冒烟:数据库规则库与文件种子指纹一致性,需显式环境变量授权",
|
||
"classification_confidence": "high"
|
||
},
|
||
{
|
||
"path": "humanization/tests/test_seed_rules_db.py",
|
||
"kind": "tool_contract",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "humanization",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [],
|
||
"skill_behavior_eval": false,
|
||
"classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同",
|
||
"classification_confidence": "high"
|
||
},
|
||
{
|
||
"path": "tests/architecture/test_import_boundaries.py",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "architecture",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "static_structure",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Scans Skill and dashboard Python files for forbidden sys.path injections into shared runtime implementations (humanization/src, access-database/scripts, call-content-model/scripts, embed-knowledge/scripts, execute-role-task/scripts, establish-voice-baseline/scripts) and Skill imports from the dashboard."
|
||
},
|
||
{
|
||
"path": "tests/architecture/test_skills_index.py",
|
||
"scope": "domain",
|
||
"owner_skill_or_domain": "architecture",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "static_structure",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "校验 Skill 发现总索引 .agent/skills/_index.md 与磁盘 skill、SKILL.md frontmatter 和 harness/manifests/skills.json 三方一致,拒绝增删改名漏同步或手改造成的漂移。"
|
||
},
|
||
{
|
||
"path": "tests/skills/access-database/test_authorization_snapshot_ddl.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "access-database",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "static_structure",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver."
|
||
},
|
||
{
|
||
"path": "tests/skills/access-database/test_db_params.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "access-database",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked."
|
||
},
|
||
{
|
||
"path": "tests/skills/access-database/test_skill_catalog.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "access-database",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files."
|
||
},
|
||
{
|
||
"path": "tests/skills/adjudicate-quality-gate/test_gate_input_builder.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "adjudicate-quality-gate",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls."
|
||
},
|
||
{
|
||
"path": "tests/skills/adjudicate-quality-gate/test_writer_gate.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "adjudicate-quality-gate",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_assemble_writer_context.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_fine_outline_reader.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_fine_outline_unification.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_freeze_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for context freeze win lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_pattern_binding_reader.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_retrieve_writer_sources.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_style_loader.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection."
|
||
},
|
||
{
|
||
"path": "tests/skills/assemble-context/test_writer_contract.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "assemble-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory."
|
||
},
|
||
{
|
||
"path": "tests/skills/backup-work-extraction/test_backup_upgrade_work_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "backup-work-extraction",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched."
|
||
},
|
||
{
|
||
"path": "tests/skills/call-content-model/test_call_persistence.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "call-content-model",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter."
|
||
},
|
||
{
|
||
"path": "tests/skills/call-content-model/test_quota.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "call-content-model",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls."
|
||
},
|
||
{
|
||
"path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "capture-ai-flavor-cases",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked."
|
||
},
|
||
{
|
||
"path": "tests/skills/capture-ai-flavor-cases/test_capture_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "capture-ai-flavor-cases",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for capture batch persist lessons; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/check-content-consistency/test_build_semantic_input.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "check-content-consistency",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection."
|
||
},
|
||
{
|
||
"path": "tests/skills/check-content-consistency/test_check_writer_candidate.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "check-content-consistency",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields."
|
||
},
|
||
{
|
||
"path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "check-content-consistency",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called."
|
||
},
|
||
{
|
||
"path": "tests/skills/clean-book-text/test_clean_detect_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "clean-book-text",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked."
|
||
},
|
||
{
|
||
"path": "tests/skills/confirm-knowledge-draft/test_confirm_knowledge_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "confirm-knowledge-draft",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_decision_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup; checks accept=win and discard=lesson payloads without database."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_fact_delta.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_fact_delta_db.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_next_steps_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Asserts accept/discard next_steps are suggestions with auto=false and human_authorize where content would change."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_projection_db.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_write_canonical_db.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards."
|
||
},
|
||
{
|
||
"path": "tests/skills/decide-candidate/test_writer_acceptance.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "decide-candidate",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store."
|
||
},
|
||
{
|
||
"path": "tests/skills/deconstruct-book/test_deconstruct_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "deconstruct-book",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for deconstruct outline window lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/deconstruct-book/test_parse_llm_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "deconstruct-book",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files."
|
||
},
|
||
{
|
||
"path": "tests/skills/deconstruct-book/test_parse_outline_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "deconstruct-book",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function."
|
||
},
|
||
{
|
||
"path": "tests/skills/design-story-foundation/test_assert_selection_handoff.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "design-story-foundation",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks selection handoff JSON required fields and fail-closed paths; no database or model."
|
||
},
|
||
{
|
||
"path": "tests/skills/design-story-foundation/test_validate_candidates.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "design-story-foundation",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates candidate tree/heading/placeholder/root contracts using temporary candidate files."
|
||
},
|
||
{
|
||
"path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "diagnose-ai-flavor",
|
||
"kind": "domain_eval",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge."
|
||
},
|
||
{
|
||
"path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "embed-knowledge",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows."
|
||
},
|
||
{
|
||
"path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "establish-voice-baseline",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked."
|
||
},
|
||
{
|
||
"path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "evaluate-frozen-replay",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "static_structure",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution."
|
||
},
|
||
{
|
||
"path": "tests/skills/evaluate-frozen-replay/test_run_replay.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "evaluate-frozen-replay",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs the replay orchestrator with an in-memory governed-chat fake, temp output, and synthetic planner/detector/judge responses."
|
||
},
|
||
{
|
||
"path": "tests/skills/execute-role-task/test_muse_role.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "execute-role-task",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates provider-neutral role profiles, prompt injection, model-policy binding, schema validation, budgets, timeouts and receipts through an in-memory governed-chat fake."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-chapter-knowledge/test_chapter_extract_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-chapter-knowledge",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for chapter extraction win lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-chapter-knowledge",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-work-knowledge",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-work-knowledge",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-work-knowledge",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-work-knowledge/test_upgrade_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-work-knowledge",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for upgrade window batch lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "extract-work-knowledge",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics."
|
||
},
|
||
{
|
||
"path": "tests/skills/freeze-context/test_audit_leakage.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "freeze-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records."
|
||
},
|
||
{
|
||
"path": "tests/skills/freeze-context/test_build_snapshot.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "freeze-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory."
|
||
},
|
||
{
|
||
"path": "tests/skills/freeze-context/test_check_snapshot.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "freeze-context",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution."
|
||
},
|
||
{
|
||
"path": "tests/skills/freeze-context/test_load_reference_work.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "freeze-context",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection."
|
||
},
|
||
{
|
||
"path": "tests/skills/load-replay-reference-work/test_load_writer_reference_work.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "load-replay-reference-work",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files."
|
||
},
|
||
{
|
||
"path": "tests/skills/load-replay-reference-work/test_pattern_reference_injection.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "load-replay-reference-work",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection."
|
||
},
|
||
{
|
||
"path": "tests/skills/merge-story-candidates/test_serial_merge.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "merge-story-candidates",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests packet construction and raw-output parsing with temporary Markdown files; no model or service driver."
|
||
},
|
||
{
|
||
"path": "tests/skills/plan-chapter/test_contract.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "plan-chapter",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "static_structure",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts."
|
||
},
|
||
{
|
||
"path": "tests/skills/plan-story/test_field_coverage.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "plan-story",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write."
|
||
},
|
||
{
|
||
"path": "tests/skills/plan-story/test_planning_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "plan-story",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for planning section persistence lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/plan-story/test_record_planning_execution.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "plan-story",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions."
|
||
},
|
||
{
|
||
"path": "tests/skills/plan-story/test_repair_deterministic_receipt.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "plan-story",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries."
|
||
},
|
||
{
|
||
"path": "tests/skills/plan-story/test_select_patterns_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "plan-story",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results."
|
||
},
|
||
{
|
||
"path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "prevent-ai-flavor",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched."
|
||
},
|
||
{
|
||
"path": "tests/skills/prevent-ai-flavor/test_prevention_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "prevent-ai-flavor",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for prevention contract win lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/promote-ai-flavor-rule/test_propose_rule.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "promote-ai-flavor-rule",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests rule candidate induction gates and CLI ownership after split from capture-ai-flavor-cases."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_file_cas.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_lesson_registry_db.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_persist_raw.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests only the raw secret-pattern validator with direct strings."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_raw_vault.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem",
|
||
"raw_vault"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_record_failed_run.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks receipt eligibility predicates using in-memory values only."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_run_registry.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests run ID formatting and terminal-state rejection before database access."
|
||
},
|
||
{
|
||
"path": "tests/skills/refresh-runtime-probe/test_refresh_runtime_probe.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "refresh-runtime-probe",
|
||
"kind": "runtime_probe",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises provider-neutral runtime probe refresh, dry-run non-acceptance and authorization gates with fake role results and temp output files."
|
||
},
|
||
{
|
||
"path": "tests/skills/replay-writer-gate/test_run_writer_replay.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "replay-writer-gate",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem",
|
||
"raw_vault"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model."
|
||
},
|
||
{
|
||
"path": "tests/skills/replay-writer-gate/test_writer_eval_preregister.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "replay-writer-gate",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration."
|
||
},
|
||
{
|
||
"path": "tests/skills/reset-work-extraction/test_reset_upgrade_work_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "reset-work-extraction",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries."
|
||
},
|
||
{
|
||
"path": "tests/skills/review-knowledge-cards/test_calibrate_stamp.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "review-knowledge-cards",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises calibrate stamp fingerprint, MAE gate, and fail-closed missing/stale/failed stamps using a temp file; no database or model call."
|
||
},
|
||
{
|
||
"path": "tests/skills/review-knowledge-cards/test_review_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "review-knowledge-cards",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for knowledge card review writeback lesson; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "revise-ai-flavor",
|
||
"kind": "domain_eval",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked."
|
||
},
|
||
{
|
||
"path": "tests/skills/revise-ai-flavor/test_revision_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "revise-ai-flavor",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for revision conclusion lesson kinds; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/rewrite-selection/test_assert_expected_revision.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "rewrite-selection",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks expectedRevision match/mismatch as a pure function; database access is not invoked."
|
||
},
|
||
{
|
||
"path": "tests/skills/score-content-quality/test_rubric.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "score-content-quality",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory."
|
||
},
|
||
{
|
||
"path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "score-content-quality",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used."
|
||
},
|
||
{
|
||
"path": "tests/skills/score-content-quality/test_score_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "score-content-quality",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for stable blind-judge score lessons; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/score-content-quality/test_writer_rubric.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "score-content-quality",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts."
|
||
},
|
||
{
|
||
"path": "tests/skills/search-knowledge/test_search.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "search-knowledge",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_candidate_cas.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_candidate_cas_db.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "integration",
|
||
"evidence_level": "real_dependency_integration",
|
||
"requires": [
|
||
"postgresql"
|
||
],
|
||
"side_effects": [
|
||
"postgresql"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_gate_anchor_projection.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "medium",
|
||
"classification_basis": "加载 .agent/skills/write-next-chapter/scripts/produce_next_chapter.py 并断言机械验收子串(门锚)对写手可见;只测纯函数投影,不产生系统事实。"
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_persist_writer_run.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_production_evidence_reassemble.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "离线单测 production_evidence_reassemble:缺口识别生成新 attempt、新快照与创作输入变化,全程 mock,不连库不调模型。"
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_run_writer.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Runs the writer adapter against an in-memory governed-chat fake and frozen role profiles; no real model call."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "fake_pipeline",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_semantic_verdict.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "tool_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence."
|
||
},
|
||
{
|
||
"path": "tests/skills/write-next-chapter/test_writer_lesson_offline.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "write-next-chapter",
|
||
"kind": "tool_unit",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Mocks propose_lesson_dedup for writer mechanical gate lessons; no database."
|
||
},
|
||
{
|
||
"path": "tests/skills/dispatch-agent-task/test_dispatch_agent_task.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "dispatch-agent-task",
|
||
"kind": "runtime_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises portable task spec contract, pi adapter argv/stream normalization and run_dispatch fail-closed chain with fake pi streams and injected DB connections; no network, no real framework binary."
|
||
},
|
||
{
|
||
"path": "tests/skills/dispatch-agent-task/test_read_tools.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "dispatch-agent-task",
|
||
"kind": "runtime_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline",
|
||
"filesystem"
|
||
],
|
||
"side_effects": [
|
||
"filesystem"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Exercises read-only exploration tool registry contract with injected fake DB connections plus CLI subprocess list/rejection; no network, no real DB, no real framework binary."
|
||
},
|
||
{
|
||
"path": "tests/skills/record-run-evidence/test_agent_trace.py",
|
||
"scope": "runtime_skill",
|
||
"owner_skill_or_domain": "record-run-evidence",
|
||
"kind": "runtime_contract",
|
||
"evidence_level": "deterministic_offline",
|
||
"requires": [
|
||
"offline"
|
||
],
|
||
"side_effects": [
|
||
"none"
|
||
],
|
||
"skill_behavior_eval": false,
|
||
"classification_confidence": "high",
|
||
"classification_basis": "Fixes agent event ledger writer shapes and atomic framework evidence persistence with fake connections; no DB, no network."
|
||
}
|
||
],
|
||
"summary": {
|
||
"entry_count": 110,
|
||
"by_scope": {
|
||
"other": 1,
|
||
"runtime_skill": 98,
|
||
"harness": 3,
|
||
"domain": 8
|
||
},
|
||
"by_kind": {
|
||
"tool_contract": 35,
|
||
"skill_behavior_eval": 1,
|
||
"harness_self_test": 3,
|
||
"domain_eval": 4,
|
||
"integration": 9,
|
||
"tool_unit": 32,
|
||
"fake_pipeline": 22,
|
||
"runtime_probe": 1,
|
||
"runtime_contract": 3
|
||
},
|
||
"by_evidence_level": {
|
||
"deterministic_offline": 94,
|
||
"real_dependency_integration": 10,
|
||
"static_structure": 6
|
||
},
|
||
"total": 110
|
||
}
|
||
} |