{ "schema_version": 1, "generated_scope": "Current agent-example source test assets: .agent/skills/**/test_*.py and *_test.py, tests/skills/** source files, muse/lifecycle/quality/humanization/tests/** source files, muse/lifecycle/quality/harness/**/test_*.py, muse/authority/studio/read/test_server_display.py, and other obvious source test files; excludes .git, .venv, __pycache__ compiled artifacts, deleted working-tree files, and harness specification documents.", "entries": [ { "path": "muse/authority/studio/read/test_server_display.py", "scope": "other", "owner_skill_or_domain": "dashboard", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests dashboard display/encoding helpers and synthetic AI-flavor views; the file declares no database connection." }, { "path": "muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/run_eval.py", "kind": "skill_behavior_eval", "scope": "runtime_skill", "owner_skill_or_domain": "diagnose-ai-flavor", "evidence_level": "real_dependency_integration", "requires": [ "model", "credentials" ], "side_effects": [ "model" ], "skill_behavior_eval": true, "classification_basis": "Skill 行为评测入口:默认真实模型适配器,未授权时以稳定码失败关闭;fake 适配器只验证评测管道", "classification_confidence": "high" }, { "path": "muse/lifecycle/quality/harness/evals/test_skill_eval.py", "kind": "harness_self_test", "scope": "harness", "owner_skill_or_domain": "harness", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [], "skill_behavior_eval": false, "classification_basis": "行为评测引擎的确定性离线自测:裁决逻辑、失败关闭和报告合同;不构成 Skill 行为证据", "classification_confidence": "high" }, { "path": "muse/lifecycle/quality/harness/test_run_selected.py", "scope": "harness", "owner_skill_or_domain": "harness", "kind": "harness_self_test", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem", "subprocess" ], "side_effects": [ "filesystem", "subprocess" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses temporary manifests and a fake Python child process to test selector, dependency, timeout, nonzero and output-summary handling." }, { "path": "muse/lifecycle/quality/harness/test_skill_harness.py", "scope": "harness", "owner_skill_or_domain": "harness", "kind": "harness_self_test", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Creates temporary SKILL.md/manifest fixtures and tests harness static-audit reports and CLI exit codes." }, { "path": "muse/lifecycle/quality/humanization/tests/test_contracts.py", "scope": "domain", "owner_skill_or_domain": "humanization", "kind": "domain_eval", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks humanization asset contracts and executes the synthetic U0 patch/review replay; no external Agent/model driver." }, { "path": "muse/lifecycle/quality/humanization/tests/test_framework_coverage.py", "scope": "domain", "owner_skill_or_domain": "humanization", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Reads the research coverage YAML and checks capability owners, implementation paths, and status values." }, { "path": "muse/lifecycle/quality/humanization/tests/test_humanization_v2.py", "scope": "domain", "owner_skill_or_domain": "humanization", "kind": "domain_eval", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Evaluates synthetic voice/rule/carrier gates and lifecycle fixtures with temporary files; no external Agent/model reads SKILL.md." }, { "path": "muse/lifecycle/quality/humanization/tests/test_load_db.py", "kind": "tool_contract", "scope": "domain", "owner_skill_or_domain": "humanization", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [], "skill_behavior_eval": false, "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", "classification_confidence": "high" }, { "path": "muse/lifecycle/quality/humanization/tests/test_load_db_pg_smoke.py", "kind": "integration", "scope": "domain", "owner_skill_or_domain": "humanization", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_basis": "真实 PostgreSQL 冒烟:数据库规则库与文件种子指纹一致性,需显式环境变量授权", "classification_confidence": "high" }, { "path": "muse/lifecycle/quality/humanization/tests/test_seed_rules_db.py", "kind": "tool_contract", "scope": "domain", "owner_skill_or_domain": "humanization", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [], "skill_behavior_eval": false, "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", "classification_confidence": "high" }, { "path": "tests/architecture/test_import_boundaries.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Scans Skill and dashboard Python files for forbidden sys.path injections into shared runtime implementations (humanization/src, access-database/scripts, call-content-model/scripts, embed-knowledge/scripts, execute-role-task/scripts, establish-voice-baseline/scripts) and Skill imports from the dashboard." }, { "path": "tests/architecture/test_skills_index.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "校验 Skill 发现总索引 .agent/skills/_index.md 与磁盘 skill、SKILL.md frontmatter 和 muse/lifecycle/quality/harness/manifests/skills.json 三方一致,拒绝增删改名漏同步或手改造成的漂移。" }, { "path": "tests/skills/access-database/test_authorization_snapshot_ddl.py", "scope": "runtime_skill", "owner_skill_or_domain": "access-database", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver." }, { "path": "tests/skills/access-database/test_db_params.py", "scope": "runtime_skill", "owner_skill_or_domain": "access-database", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked." }, { "path": "tests/skills/access-database/test_skill_catalog.py", "scope": "runtime_skill", "owner_skill_or_domain": "access-database", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files." }, { "path": "tests/skills/adjudicate-quality-gate/test_gate_input_builder.py", "scope": "runtime_skill", "owner_skill_or_domain": "adjudicate-quality-gate", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls." }, { "path": "tests/skills/adjudicate-quality-gate/test_writer_gate.py", "scope": "runtime_skill", "owner_skill_or_domain": "adjudicate-quality-gate", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output." }, { "path": "tests/skills/assemble-context/test_assemble_writer_context.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts." }, { "path": "tests/skills/assemble-context/test_fine_outline_reader.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing." }, { "path": "tests/skills/assemble-context/test_fine_outline_unification.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler." }, { "path": "tests/skills/assemble-context/test_freeze_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for context freeze win lesson; no database." }, { "path": "tests/skills/assemble-context/test_pattern_binding_reader.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior." }, { "path": "tests/skills/assemble-context/test_retrieve_writer_sources.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections." }, { "path": "tests/skills/assemble-context/test_style_loader.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection." }, { "path": "tests/skills/assemble-context/test_writer_contract.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory." }, { "path": "tests/skills/backup-work-extraction/test_backup_upgrade_work_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "backup-work-extraction", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched." }, { "path": "tests/skills/call-content-model/test_call_persistence.py", "scope": "runtime_skill", "owner_skill_or_domain": "call-content-model", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter." }, { "path": "tests/skills/call-content-model/test_quota.py", "scope": "runtime_skill", "owner_skill_or_domain": "call-content-model", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls." }, { "path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py", "scope": "runtime_skill", "owner_skill_or_domain": "capture-ai-flavor-cases", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked." }, { "path": "tests/skills/capture-ai-flavor-cases/test_capture_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "capture-ai-flavor-cases", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for capture batch persist lessons; no database." }, { "path": "tests/skills/check-content-consistency/test_build_semantic_input.py", "scope": "runtime_skill", "owner_skill_or_domain": "check-content-consistency", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection." }, { "path": "tests/skills/check-content-consistency/test_check_writer_candidate.py", "scope": "runtime_skill", "owner_skill_or_domain": "check-content-consistency", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields." }, { "path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py", "scope": "runtime_skill", "owner_skill_or_domain": "check-content-consistency", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called." }, { "path": "tests/skills/clean-book-text/test_clean_detect_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "clean-book-text", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked." }, { "path": "tests/skills/confirm-knowledge-draft/test_confirm_knowledge_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "confirm-knowledge-draft", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs." }, { "path": "tests/skills/decide-candidate/test_decision_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup; checks accept=win and discard=lesson payloads without database." }, { "path": "tests/skills/decide-candidate/test_fact_delta.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions." }, { "path": "tests/skills/decide-candidate/test_fact_delta_db.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows." }, { "path": "tests/skills/decide-candidate/test_next_steps_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Asserts accept/discard next_steps are suggestions with auto=false and human_authorize where content would change." }, { "path": "tests/skills/decide-candidate/test_projection_db.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup." }, { "path": "tests/skills/decide-candidate/test_write_canonical_db.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards." }, { "path": "tests/skills/decide-candidate/test_writer_acceptance.py", "scope": "runtime_skill", "owner_skill_or_domain": "decide-candidate", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store." }, { "path": "tests/skills/deconstruct-book/test_deconstruct_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "deconstruct-book", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for deconstruct outline window lesson; no database." }, { "path": "tests/skills/deconstruct-book/test_parse_llm_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "deconstruct-book", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files." }, { "path": "tests/skills/deconstruct-book/test_parse_outline_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "deconstruct-book", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function." }, { "path": "tests/skills/design-story-foundation/test_assert_selection_handoff.py", "scope": "runtime_skill", "owner_skill_or_domain": "design-story-foundation", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks selection handoff JSON required fields and fail-closed paths; no database or model." }, { "path": "tests/skills/design-story-foundation/test_validate_candidates.py", "scope": "runtime_skill", "owner_skill_or_domain": "design-story-foundation", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates candidate tree/heading/placeholder/root contracts using temporary candidate files." }, { "path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py", "scope": "runtime_skill", "owner_skill_or_domain": "diagnose-ai-flavor", "kind": "domain_eval", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge." }, { "path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "embed-knowledge", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows." }, { "path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py", "scope": "runtime_skill", "owner_skill_or_domain": "establish-voice-baseline", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked." }, { "path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py", "scope": "runtime_skill", "owner_skill_or_domain": "evaluate-frozen-replay", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution." }, { "path": "tests/skills/evaluate-frozen-replay/test_run_replay.py", "scope": "runtime_skill", "owner_skill_or_domain": "evaluate-frozen-replay", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs the replay orchestrator with an in-memory governed-chat fake, temp output, and synthetic planner/detector/judge responses." }, { "path": "tests/skills/execute-role-task/test_muse_role.py", "scope": "runtime_skill", "owner_skill_or_domain": "execute-role-task", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates provider-neutral role profiles, prompt injection, model-policy binding, schema validation, budgets, timeouts and receipts through an in-memory governed-chat fake." }, { "path": "tests/skills/extract-chapter-knowledge/test_chapter_extract_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-chapter-knowledge", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for chapter extraction win lesson; no database." }, { "path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-chapter-knowledge", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions." }, { "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-work-knowledge", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry." }, { "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-work-knowledge", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1." }, { "path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-work-knowledge", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made." }, { "path": "tests/skills/extract-work-knowledge/test_upgrade_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-work-knowledge", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for upgrade window batch lesson; no database." }, { "path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-work-knowledge", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics." }, { "path": "tests/skills/freeze-context/test_audit_leakage.py", "scope": "runtime_skill", "owner_skill_or_domain": "freeze-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records." }, { "path": "tests/skills/freeze-context/test_build_snapshot.py", "scope": "runtime_skill", "owner_skill_or_domain": "freeze-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory." }, { "path": "tests/skills/freeze-context/test_check_snapshot.py", "scope": "runtime_skill", "owner_skill_or_domain": "freeze-context", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution." }, { "path": "tests/skills/freeze-context/test_load_reference_work.py", "scope": "runtime_skill", "owner_skill_or_domain": "freeze-context", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection." }, { "path": "tests/skills/load-replay-reference-work/test_load_writer_reference_work.py", "scope": "runtime_skill", "owner_skill_or_domain": "load-replay-reference-work", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files." }, { "path": "tests/skills/load-replay-reference-work/test_pattern_reference_injection.py", "scope": "runtime_skill", "owner_skill_or_domain": "load-replay-reference-work", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection." }, { "path": "tests/skills/merge-story-candidates/test_serial_merge.py", "scope": "runtime_skill", "owner_skill_or_domain": "merge-story-candidates", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests packet construction and raw-output parsing with temporary Markdown files; no model or service driver." }, { "path": "tests/skills/plan-chapter/test_contract.py", "scope": "runtime_skill", "owner_skill_or_domain": "plan-chapter", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts." }, { "path": "tests/skills/plan-story/test_field_coverage.py", "scope": "runtime_skill", "owner_skill_or_domain": "plan-story", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write." }, { "path": "tests/skills/plan-story/test_planning_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "plan-story", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for planning section persistence lesson; no database." }, { "path": "tests/skills/plan-story/test_record_planning_execution.py", "scope": "runtime_skill", "owner_skill_or_domain": "plan-story", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions." }, { "path": "tests/skills/plan-story/test_repair_deterministic_receipt.py", "scope": "runtime_skill", "owner_skill_or_domain": "plan-story", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries." }, { "path": "tests/skills/plan-story/test_select_patterns_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "plan-story", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results." }, { "path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py", "scope": "runtime_skill", "owner_skill_or_domain": "prevent-ai-flavor", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched." }, { "path": "tests/skills/prevent-ai-flavor/test_prevention_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "prevent-ai-flavor", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for prevention contract win lesson; no database." }, { "path": "tests/skills/promote-ai-flavor-rule/test_propose_rule.py", "scope": "runtime_skill", "owner_skill_or_domain": "promote-ai-flavor-rule", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests rule candidate induction gates and CLI ownership after split from capture-ai-flavor-cases." }, { "path": "tests/skills/record-run-evidence/test_file_cas.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions." }, { "path": "tests/skills/record-run-evidence/test_lesson_registry_db.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them." }, { "path": "tests/skills/record-run-evidence/test_persist_raw.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests only the raw secret-pattern validator with direct strings." }, { "path": "tests/skills/record-run-evidence/test_raw_vault.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "offline", "filesystem", "raw_vault" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery." }, { "path": "tests/skills/record-run-evidence/test_record_failed_run.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence." }, { "path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks receipt eligibility predicates using in-memory values only." }, { "path": "tests/skills/record-run-evidence/test_run_registry.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests run ID formatting and terminal-state rejection before database access." }, { "path": "tests/skills/refresh-runtime-probe/test_refresh_runtime_probe.py", "scope": "runtime_skill", "owner_skill_or_domain": "refresh-runtime-probe", "kind": "runtime_probe", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises provider-neutral runtime probe refresh, dry-run non-acceptance and authorization gates with fake role results and temp output files." }, { "path": "tests/skills/replay-writer-gate/test_run_writer_replay.py", "scope": "runtime_skill", "owner_skill_or_domain": "replay-writer-gate", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem", "raw_vault" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model." }, { "path": "tests/skills/replay-writer-gate/test_writer_eval_preregister.py", "scope": "runtime_skill", "owner_skill_or_domain": "replay-writer-gate", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration." }, { "path": "tests/skills/reset-work-extraction/test_reset_upgrade_work_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "reset-work-extraction", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries." }, { "path": "tests/skills/review-knowledge-cards/test_calibrate_stamp.py", "scope": "runtime_skill", "owner_skill_or_domain": "review-knowledge-cards", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises calibrate stamp fingerprint, MAE gate, and fail-closed missing/stale/failed stamps using a temp file; no database or model call." }, { "path": "tests/skills/review-knowledge-cards/test_review_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "review-knowledge-cards", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for knowledge card review writeback lesson; no database." }, { "path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py", "scope": "runtime_skill", "owner_skill_or_domain": "revise-ai-flavor", "kind": "domain_eval", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked." }, { "path": "tests/skills/revise-ai-flavor/test_revision_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "revise-ai-flavor", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for revision conclusion lesson kinds; no database." }, { "path": "tests/skills/rewrite-selection/test_assert_expected_revision.py", "scope": "runtime_skill", "owner_skill_or_domain": "rewrite-selection", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks expectedRevision match/mismatch as a pure function; database access is not invoked." }, { "path": "tests/skills/score-content-quality/test_rubric.py", "scope": "runtime_skill", "owner_skill_or_domain": "score-content-quality", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory." }, { "path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py", "scope": "runtime_skill", "owner_skill_or_domain": "score-content-quality", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used." }, { "path": "tests/skills/score-content-quality/test_score_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "score-content-quality", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for stable blind-judge score lessons; no database." }, { "path": "tests/skills/score-content-quality/test_writer_rubric.py", "scope": "runtime_skill", "owner_skill_or_domain": "score-content-quality", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts." }, { "path": "tests/skills/search-knowledge/test_search.py", "scope": "runtime_skill", "owner_skill_or_domain": "search-knowledge", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding." }, { "path": "tests/skills/write-next-chapter/test_candidate_cas.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops." }, { "path": "tests/skills/write-next-chapter/test_candidate_cas_db.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows." }, { "path": "tests/skills/write-next-chapter/test_gate_anchor_projection.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "medium", "classification_basis": "加载 muse/content/work/skills/generate/write-next-chapter/scripts/produce_next_chapter.py 并断言机械验收子串(门锚)对写手可见;只测纯函数投影,不产生系统事实。" }, { "path": "tests/skills/write-next-chapter/test_persist_writer_run.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call." }, { "path": "tests/skills/write-next-chapter/test_production_evidence_reassemble.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "离线单测 production_evidence_reassemble:缺口识别生成新 attempt、新快照与创作输入变化,全程 mock,不连库不调模型。" }, { "path": "tests/skills/write-next-chapter/test_run_writer.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Runs the writer adapter against an in-memory governed-chat fake and frozen role profiles; no real model call." }, { "path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "fake_pipeline", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors." }, { "path": "tests/skills/write-next-chapter/test_dispatch_writer_bridge.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises writer dispatch bridge contract with fake pi streams and injected DB connections: spec freeze, per-chapter session identity, envelope binding, evidence-reference fail-closed; no network, no real framework binary." }, { "path": "tests/skills/write-next-chapter/test_semantic_verdict.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence." }, { "path": "tests/skills/write-next-chapter/test_writer_lesson_offline.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "tool_unit", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Mocks propose_lesson_dedup for writer mechanical gate lessons; no database." }, { "path": "tests/skills/dispatch-agent-task/test_dispatch_agent_task.py", "scope": "runtime_skill", "owner_skill_or_domain": "dispatch-agent-task", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises portable task spec contract, pi adapter argv/stream normalization and run_dispatch fail-closed chain with fake pi streams and injected DB connections; no network, no real framework binary." }, { "path": "tests/skills/dispatch-agent-task/test_read_tools.py", "scope": "runtime_skill", "owner_skill_or_domain": "dispatch-agent-task", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Exercises read-only exploration tool registry contract with injected fake DB connections plus CLI subprocess list/rejection; no network, no real DB, no real framework binary." }, { "path": "tests/skills/record-run-evidence/test_agent_trace.py", "scope": "runtime_skill", "owner_skill_or_domain": "record-run-evidence", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Fixes agent event ledger writer shapes and atomic framework evidence persistence with fake connections; no DB, no network." }, { "path": "tests/skills/assemble-context/test_persist_freeze_db.py", "scope": "runtime_skill", "owner_skill_or_domain": "assemble-context", "kind": "integration", "evidence_level": "real_dependency_integration", "requires": [ "postgresql" ], "side_effects": [ "postgresql" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "真实库验证冻结幂等键 (manifest_sha256, context_sha256) 对:按既有一行原值重放不新增行;append-only 触发器禁止清理,故零污染设计。" }, { "path": "tests/skills/write-next-chapter/test_two_phase_writer.py", "scope": "runtime_skill", "owner_skill_or_domain": "write-next-chapter", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "两阶段写手离线契约:探索清单严格校验、确定性回放失败关闭、生成输入只含回放材料与合同、两阶段编排与探索摘要落盘;假派发器与假工具,不连库不真调。" }, { "path": "tests/skills/extract-chapter-knowledge/test_dispatch_extraction_bridge.py", "scope": "runtime_skill", "owner_skill_or_domain": "extract-chapter-knowledge", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "filesystem" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "抽取派发桥离线契约:全量正文注入任务输入、证据逐字绑定机械校验、修复重派语义、缺证失败关闭;假派发器,不连库不真调。" }, { "path": "tests/skills/score-content-quality/test_dispatch_judge_bridge.py", "scope": "runtime_skill", "owner_skill_or_domain": "score-content-quality", "kind": "runtime_contract", "evidence_level": "deterministic_offline", "requires": [ "offline" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "评委派发桥离线契约:圈定授权工具白名单、盲评输入防泄漏哈希绑定、任务包与盲评输入全等(桥零附加身份)、ModelRunner 协议形状;不连库不真调。" }, { "path": "tests/architecture/test_dsh_adapter.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "用 DSH session JSONL 夹具验证 headless patch、模型/工具边界、未知事件保留、flush 工件归一和非零/压缩/续接失败关闭;不启动真实模型。" }, { "path": "tests/architecture/test_framework_port_purity.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "扫描 framework 目录,阻断框架对 Muse 角色合同、证据账本、数据库和模型策略的反向 import。" }, { "path": "tests/architecture/test_framework_protocol.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "验证 FrameworkExecutionRequest 与 Draft 2020-12 Schema、JSONL 工件 flush/replay、序列缺口和重复发布失败关闭。" }, { "path": "tests/architecture/test_markdown_links.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "扫描入口、SoT 和 framework/muse 活文档,解析仓内 Markdown 链接并失败关闭死链。" }, { "path": "tests/architecture/test_dsh_pi_route.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "验证 Pi models.json 的 provider、协议、地址和模型目录投影为不含密钥的 DSH pi-ai patch,并阻断 policy 与路由身份不一致。" }, { "path": "tests/architecture/test_studio_path_contract.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "验证迁移后的只读看板、候选决策和经验确认入口从独立仓根加载业务脚本,并覆盖写通道页面地标与键盘可达滚动区。" }, { "path": "tests/architecture/test_runtime_purity.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "static_structure", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "扫描 runtime 目录,阻断对 Muse 业务模块的反向 import,并要求 dispatch_agent_task.py 保持薄 CLI。" }, { "path": "tests/protocol/test_muse_store.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "验证 muse.db WAL、五表、向量 blob 余弦排序,以及 revise 必须写入 revisions diff。" }, { "path": "tests/e2e/test_sqlite_write_path.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "经 muse.flow.dispatch 假 launcher 写入 run+events,revise 落 revisions,同 input 重跑 output_text 可 diff,且不产生未忽略运行文件。" }, { "path": "tests/adapters/test_host_adapter_consistency.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Pi/DSH/Claude 三适配器用同一 FrameworkExecutionRequest 构建 argv,且适配器源码不 import muse 业务模块,守护宿主适配层统一消费与纯净度合同。" }, { "path": "tests/e2e/test_compounding.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "lesson 空证据被 Python 与 SQLite 约束双层拒绝、升格 merge 保留 source_run_ids/source_review_ids、回放复用生产 run_dispatch 入口、web 写面闭集 reviews/revisions/adopt;全程临时库离线执行。" }, { "path": "tests/e2e/test_ranking_attribution.py", "scope": "domain", "owner_skill_or_domain": "architecture", "kind": "tool_contract", "evidence_level": "deterministic_offline", "requires": [ "offline", "filesystem" ], "side_effects": [ "none" ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "起点/番茄 CSV 导入计数正确,归因报告非空落盘并含 run 与 skills 哈希回链;全程临时库离线执行。" } ], "summary": { "entry_count": 127, "by_scope": { "domain": 20, "harness": 3, "other": 1, "runtime_skill": 103 }, "by_kind": { "domain_eval": 4, "fake_pipeline": 22, "harness_self_test": 3, "integration": 10, "runtime_contract": 7, "runtime_probe": 1, "skill_behavior_eval": 1, "tool_contract": 47, "tool_unit": 32 }, "by_evidence_level": { "deterministic_offline": 106, "real_dependency_integration": 11, "static_structure": 10 }, "total": 127 } }