diff --git a/.agent/skills/dispatch-agent-task/SKILL.md b/.agent/skills/dispatch-agent-task/SKILL.md index 02f721c..45b202f 100644 --- a/.agent/skills/dispatch-agent-task/SKILL.md +++ b/.agent/skills/dispatch-agent-task/SKILL.md @@ -13,20 +13,25 @@ disable-model-invocation: true ``` .venv/bin/python .agent/skills/dispatch-agent-task/scripts/dispatch_agent_task.py \ --spec task.json --provider P --model M [--thinking low] \ - [--repo-root .] [--run-id ID] [--run-dir DIR] [--trigger-source user] + [--repo-root .] [--run-id ID] [--run-dir DIR] [--trigger-source user] \ + [--session-id ID --session-dir DIR] [--enable-read-tools] ``` | 模块 | 职责 | |---|---| | `scripts/agent_task.py` | 可移植任务包合同:spec 加载校验、角色身份+中央角色合同+Schema 装配、结构化输出校验。 | -| `scripts/pi_runner.py` | pi 框架适配器:argv 构造(`--system-prompt` 注入、`--tools` 白名单、`--no-context-files/--no-skills/--no-extensions` 隔离)、JSON 事件流消费、看门狗超时。 | -| `scripts/dispatch_agent_task.py` | CLI:运行登记 -> 事件入账 -> 框架执行 -> 校验 -> 证据落库 -> 回执。 | +| `scripts/pi_runner.py` | pi 框架适配器:argv 构造(`--system-prompt` 注入、`--tools` 白名单、`--no-context-files/--no-skills/--no-extensions` 隔离、会话复用 `--session-id/--session-dir`)、JSON 事件流消费、看门狗超时。 | +| `scripts/read_tools.py` | 探索工具 server 只读实现:五个登记工具(细纲/文风/范式绑定/章节正文/实体检索),只读连接、未知工具拒绝、结果有界;`--list` 输出登记表。 | +| `scripts/muse_read_tools_extension.ts` | 工具 server 的框架扩展:装载时从 `read_tools.py --list` 动态注册工具,每次执行转给只读实现;不经派发器注入环境变量时不注册任何工具。 | +| `scripts/dispatch_agent_task.py` | CLI:运行登记 -> 事件入账 -> 框架执行 -> 校验 -> 证据落库 -> 依赖清单 -> 回执。 | ## 输入与输出 - 输入:`AgentTaskSpec` JSON(specVersion=agent-task-v1;role 限五个角色,角色合同来自 `.agent/docs/architecture/角色合同.md`;outputSchema 必须是合法 Draft 2020-12;inputSha256 可选校验)。`provider`、`model`、`thinking` 属于执行策略,其中 `provider` 和 `model` 必须由调用方显式传入并如实记账。 - 输出:回执 JSON(runId、框架、请求/实际模型、逐回合用量与成本、哈希链、证据 ID)与退出码;结构化输出经 `muse_llm.extract_json` + 完整 Draft 2020-12 校验,失败关闭(退出码 4)。 - 事件协议(九类闭集,逐条追加 `example_agent_event`):run.started / agent.started / model.completed / tool.started / tool.completed / agent.completed / agent.failed / run.completed / run.failed。 +- 会话复用:`--session-id` + `--session-dir` 继续/创建指定框架会话(同一章补证/修改复用);缺省为一次性会话。会话文件由框架写入指定目录,随运行目录同权限保护。 +- 探索工具:`--enable-read-tools` 启用只读工具 server(扩展经 `-e` 显式加载,自动发现仍被 `--no-extensions` 禁用);任务包含登记表工具名但未启用时失败关闭(READ_TOOLS_NOT_ENABLED)。工具调用参数进事件账本,成功后产 `dependencies.json`(依赖清单:本次运行实际读了什么),回执含 `dependencies` 摘要。 - 退出码:0 成功;2 spec 非法;3 框架失败(超时/非零退出/流不可解析/无模型回合);4 输出不合 Schema;5 证据落库失败。 ## 红线 @@ -35,6 +40,7 @@ disable-model-invocation: true - 只有 `pi_runner.py` 可以直接调用框架二进制;其余任何位置 shell 调模型 CLI 都被架构门禁阻断。 - 不读取本机模型客户端配置文件;框架凭据走框架自身环境变量,本 Skill 不经手。 - 角色文件只提供身份提示;中央角色合同是稳定边界唯一事实源。系统提示词由适配器按固定顺序装配,不裁剪角色合同;工具白名单外的能力不开放(空名单 = `--no-tools`)。 +- 工具 server 只提供登记过的只读工具:不提供裸查询、不写库;登记表(`read_tools.py` TOOL_REGISTRY)是唯一事实源,扩展不自带工具清单。 - 失败一律关闭:框架异常、Schema 不符、证据落库失败都终止运行并记 run.failed,不部分成功。 ## 数据边界 @@ -42,7 +48,7 @@ disable-model-invocation: true - `example_agent_event`(DDL-113,append-only):归一事件账本,只存身份、用量、成本与安全摘要。 - `example_llm_call`:每个模型回合一条投影(无额度窗时 `window_key=NULL`,以 `run_id` 归属本次派发,raw 指针指向全量转录)。 - raw 表:system prompt(prompt)、最终输出(response)、框架全量转录(supplier)经 `record-run-evidence/agent_trace.persist_agent_evidence` 单事务原子落库,写前密钥拦截。 -- `example_run`:start_run/finish_run 登记终态;运行目录(/tmp/muse-agent-runs/)保留 task-spec、system-prompt、user-message、transcript、output、receipt 审计件。 +- `example_run`:start_run/finish_run 登记终态;运行目录(/tmp/muse-agent-runs/)保留 task-spec、system-prompt、user-message、transcript、output、dependencies(如有)、receipt 审计件。 - 留痕是旁路义务:派发路径不提供「不留痕」选项,业务调用方不能决定是否记录。 ## 复利合同 diff --git a/.agent/skills/dispatch-agent-task/scripts/dispatch_agent_task.py b/.agent/skills/dispatch-agent-task/scripts/dispatch_agent_task.py index 513eec7..605c4d1 100644 --- a/.agent/skills/dispatch-agent-task/scripts/dispatch_agent_task.py +++ b/.agent/skills/dispatch-agent-task/scripts/dispatch_agent_task.py @@ -39,6 +39,7 @@ from agent_task import ( # noqa: E402 ) from agent_trace import AgentTraceWriter, persist_agent_evidence # noqa: E402 from persist_raw import _check_no_secrets # noqa: E402 +import read_tools from pi_runner import ( # noqa: E402 DEFAULT_PI_BIN, AgentStreamOutcome, @@ -129,6 +130,11 @@ def _write_private_json(path: Path, value: Mapping[str, Any]) -> None: _write_private_text(path, json.dumps(value, ensure_ascii=False, indent=2)) +# 工具 server 扩展路径与内建只读文件工具(探索白名单的两类合法成员)。 +READ_TOOL_EXTENSION = Path(__file__).resolve().parent / "muse_read_tools_extension.ts" +BUILTIN_TOOL_NAMES = frozenset({"read", "grep", "find", "ls"}) + + def run_dispatch( spec_path: str | Path, *, @@ -140,6 +146,9 @@ def run_dispatch( launcher: Callable[..., Iterable] | None = None, trigger_source: str = "user", trigger_detail: Mapping[str, Any] | None = None, + session_id: str | None = None, + session_dir: str | Path | None = None, + enable_read_tools: bool = False, ) -> tuple[dict[str, Any], int]: """执行一次完整派发;所有可控失败都返回稳定回执与退出码。""" @@ -170,7 +179,34 @@ def run_dispatch( EXIT_SPEC_INVALID, ) - effective_policy = replace(policy, cwd=policy.cwd or str(repo_root_path)) + disabled_read_tools = sorted( + { + tool + for tool in package.spec.tool_allowlist + if tool in read_tools.TOOL_REGISTRY and not enable_read_tools + } + ) + if disabled_read_tools: + return ( + { + "status": "failed", + "errorCode": "READ_TOOLS_NOT_ENABLED", + "error": f"任务包含探索工具但未启用工具 server: {disabled_read_tools}", + }, + EXIT_SPEC_INVALID, + ) + policy_kwargs: dict[str, Any] = {"cwd": policy.cwd or str(repo_root_path)} + if session_id is not None: + policy_kwargs["session_id"] = session_id + policy_kwargs["session_dir"] = str(session_dir) if session_dir is not None else None + if enable_read_tools: + # 工具 server 扩展经环境变量定位只读实现;框架进程继承该环境。 + os.environ["MUSE_READ_TOOLS_PYTHON"] = sys.executable + os.environ["MUSE_READ_TOOLS_SCRIPT"] = str( + Path(__file__).resolve().parent / "read_tools.py" + ) + policy_kwargs["extension_path"] = str(READ_TOOL_EXTENSION) + effective_policy = replace(policy, **policy_kwargs) if ( package.role_contract.model_policy == "fixed-opus" and "opus" not in effective_policy.model.lower() @@ -546,6 +582,29 @@ def run_dispatch( evidence=evidence, ) + # 依赖清单:本次运行实际读取了什么(工具调用序列与参数),链路透视的读侧材料。 + dependency_count = len(outcome.tool_calls) + if dependency_count: + dependencies = [ + { + "seq": index + 1, + "tool": call.name, + "args": dict(call.args) if call.args else {}, + "isError": call.is_error, + } + for index, call in enumerate(outcome.tool_calls) + ] + try: + _write_private_json(run_dir_path / "dependencies.json", dependencies) + except OSError as exc: + return _failed( + "AUDIT_WRITE_FAILED", + f"依赖清单写入失败: {type(exc).__name__}", + EXIT_EVIDENCE_FAILED, + session_id=outcome.session_id, + evidence=evidence, + ) + usage = _usage_totals(outcome) total_cost, cost_complete = _cost_totals(outcome) try: @@ -586,6 +645,11 @@ def run_dispatch( durationMs=outcome.duration_ms, turns=outcome.turns, toolCallCount=len(outcome.tool_calls), + dependencies=( + {"count": dependency_count, "file": "dependencies.json"} + if dependency_count + else None + ), modelCallCount=len(outcome.model_calls), actualModelIds=[call.actual_model_id for call in outcome.model_calls], usage=usage, @@ -621,6 +685,9 @@ def main(argv: list[str] | None = None) -> int: parser.add_argument("--run-id", default=None, help="指定 run_id(默认自动生成)") parser.add_argument("--run-dir", default=None, help="运行目录(默认 /tmp/muse-agent-runs/)") parser.add_argument("--trigger-source", default="user", choices=["user", "replay_eval", "diagnostic"]) + parser.add_argument("--session-id", default=None, help="框架会话 ID(复用=继续原会话,缺省=一次性会话)") + parser.add_argument("--session-dir", default=None, help="框架会话目录(与 --session-id 同给)") + parser.add_argument("--enable-read-tools", action="store_true", help="启用只读探索工具 server") args = parser.parse_args(argv) policy = ExecutionPolicy( @@ -633,6 +700,9 @@ def main(argv: list[str] | None = None) -> int: run_id=args.run_id, run_dir=args.run_dir, trigger_source=args.trigger_source, + session_id=args.session_id, + session_dir=args.session_dir, + enable_read_tools=args.enable_read_tools, ) print(json.dumps(receipt, ensure_ascii=False, indent=2)) return code diff --git a/.agent/skills/dispatch-agent-task/scripts/muse_read_tools_extension.ts b/.agent/skills/dispatch-agent-task/scripts/muse_read_tools_extension.ts new file mode 100644 index 0000000..ddcf9c2 --- /dev/null +++ b/.agent/skills/dispatch-agent-task/scripts/muse_read_tools_extension.ts @@ -0,0 +1,91 @@ +// 探索工具 server 的框架扩展(阶段 D)。 +// +// 边界合同中的工具层:给创作智能体暴露只读探索工具。本扩展不自带任何查询逻辑, +// 工具登记表与实现都在 read_tools.py(Python,只读连接、未知工具拒绝、结果有界)。 +// 加载时执行 `read_tools.py --list` 取登记表,按登记表动态注册工具,保证单一事实源; +// 每次工具执行把参数原样转给 `read_tools.py <工具名> ''`,stdout 即结果。 +// +// 派发器经环境变量定位实现:MUSE_READ_TOOLS_PYTHON(解释器)、MUSE_READ_TOOLS_SCRIPT(脚本)。 +// 未经派发器注入这两个变量时不注册任何工具(静默跳过),隔离会话不受影响。 +import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; +import { Type } from "typebox"; + +interface ToolSpec { + description: string; + tables: string[]; + args: Record; +} + +function runReadTools(python: string, script: string, argv: string[]): string { + const bun = (globalThis as { Bun?: any }).Bun; + if (bun?.spawnSync) { + const res = bun.spawnSync([python, script, ...argv], { + stdout: "pipe", + stderr: "pipe", + }); + const stdout = new TextDecoder().decode(res.stdout); + if (res.exitCode !== 0) { + const stderr = new TextDecoder().decode(res.stderr).trim(); + throw new Error(stderr || `read_tools 退出码 ${res.exitCode}`); + } + return stdout; + } + // Node 运行时的回退路径。 + // eslint-disable-next-line @typescript-eslint/no-var-requires + const { execFileSync } = require("node:child_process"); + return execFileSync(python, [script, ...argv], { + encoding: "utf-8", + stdio: ["ignore", "pipe", "pipe"], + }); +} + +export default function (pi: ExtensionAPI) { + const python = process.env.MUSE_READ_TOOLS_PYTHON; + const script = process.env.MUSE_READ_TOOLS_SCRIPT; + if (!python || !script) { + return; // 非工具 server 派发:不注册任何工具。 + } + + let registry: Record = {}; + try { + registry = JSON.parse(runReadTools(python, script, ["--list"])); + } catch { + return; // 登记表不可读时保持无工具,不放行裸查询。 + } + + for (const [name, spec] of Object.entries(registry)) { + const props: Record = {}; + for (const [rawName, argType] of Object.entries(spec.args ?? {})) { + const optional = rawName.endsWith("?"); + const argName = optional ? rawName.slice(0, -1) : rawName; + const base = + argType === "int" + ? Type.Integer({ description: argName }) + : Type.String({ description: argName }); + props[argName] = optional ? Type.Optional(base) : base; + } + pi.registerTool({ + name, + label: name, + description: `${spec.description}(只读;读取 ${spec.tables.join("、")})`, + promptGuidelines: [ + `Use ${name} only for read-only exploration; it never modifies data.`, + ], + parameters: Type.Object(props), + async execute(_toolCallId, params) { + try { + const stdout = runReadTools(python, script, [ + name, + JSON.stringify(params ?? {}), + ]); + return { content: [{ type: "text", text: stdout }] }; + } catch (error) { + // 抛出错误使框架把该工具调用标记为失败,模型可据此调整探索。 + throw new Error( + error instanceof Error ? error.message : String(error) + ); + } + }, + }); + } +} diff --git a/.agent/skills/dispatch-agent-task/scripts/pi_runner.py b/.agent/skills/dispatch-agent-task/scripts/pi_runner.py index bfea6d0..253a5e0 100644 --- a/.agent/skills/dispatch-agent-task/scripts/pi_runner.py +++ b/.agent/skills/dispatch-agent-task/scripts/pi_runner.py @@ -53,6 +53,11 @@ class ExecutionPolicy: thinking: str | None = None pi_bin: str = DEFAULT_PI_BIN cwd: str | None = None + # 会话复用:同一章补证/修改继续原会话;两者同给或同缺。 + session_id: str | None = None + session_dir: str | None = None + # 显式加载的框架扩展(工具 server);--no-extensions 只禁自动发现,-e 照常加载。 + extension_path: str | None = None def __post_init__(self) -> None: if not isinstance(self.provider, str) or not self.provider.strip(): @@ -63,6 +68,8 @@ class ExecutionPolicy: "off", "minimal", "low", "medium", "high", "xhigh", "max" }: raise ValueError("thinking 不受支持") + if bool(self.session_id) != bool(self.session_dir): + raise ValueError("session_id 与 session_dir 必须同时给出") @property def framework(self) -> str: @@ -91,6 +98,7 @@ class ToolCallRecord: tool_call_id: str name: str is_error: bool = False + args: Mapping[str, Any] | None = None @dataclass @@ -111,12 +119,19 @@ class AgentStreamOutcome: def build_pi_argv(package: TaskPackage, policy: ExecutionPolicy) -> list[str]: """构造 pi 子代理 argv:system prompt 注入、工具白名单、上下文隔离。""" - argv = [policy.pi_bin, "--print", "--mode", "json", "--no-session"] + argv = [policy.pi_bin, "--print", "--mode", "json"] + if policy.session_id: + argv += ["--session-id", policy.session_id, "--session-dir", policy.session_dir or ""] + else: + argv.append("--no-session") argv += ["--provider", policy.provider, "--model", policy.model] if policy.thinking: argv += ["--thinking", policy.thinking] # 上下文隔离:不加载项目 AGENTS.md/skills/extensions,角色合同全部来自任务包。 + # --no-extensions 只禁自动发现,-e 显式传入的工具 server 扩展不受影响。 argv += ["--no-context-files", "--no-skills", "--no-extensions", "--no-approve"] + if policy.extension_path: + argv += ["-e", policy.extension_path] allowlist = package.spec.tool_allowlist if allowlist: argv += ["--tools", ",".join(allowlist)] @@ -386,12 +401,22 @@ class PiAgentRunner: raise FrameworkError("TOOL_EVENT_INVALID", "工具开始事件缺 toolCallId") if name not in allowed_tools: raise FrameworkError("TOOL_NOT_ALLOWED", f"框架执行了未授权工具: {name or ''}") - outcome.tool_calls.append(ToolCallRecord(tool_call_id=tool_call_id, name=name)) + raw_args = event.get("args") + args = raw_args if isinstance(raw_args, Mapping) else None + outcome.tool_calls.append( + ToolCallRecord(tool_call_id=tool_call_id, name=name, args=args) + ) + # 依赖清单材料:工具读取参数进事件账本(截断,避免撑爆事件行)。 + args_summary: str | None = None + if args is not None: + args_summary = json.dumps(dict(args), ensure_ascii=False, sort_keys=True) + if len(args_summary) > 512: + args_summary = args_summary[:512] + "…" sink.emit( "tool.started", status="ok", tool_name=name, - details={"toolCallId": tool_call_id}, + details={"toolCallId": tool_call_id, "args": args_summary}, ) elif kind == "tool_execution_end": is_error = bool(event.get("isError")) diff --git a/.agent/skills/dispatch-agent-task/scripts/read_tools.py b/.agent/skills/dispatch-agent-task/scripts/read_tools.py new file mode 100644 index 0000000..556e080 --- /dev/null +++ b/.agent/skills/dispatch-agent-task/scripts/read_tools.py @@ -0,0 +1,280 @@ +#!/usr/bin/env python3 +"""探索工具 server 的只读取数实现(阶段 D)。 + +边界合同中的工具 server 组件:给创作智能体提供只读取数能力的统一入口。 +每个工具声明读什么表、什么版本语义、返回什么结构;不提供裸查询,不写库。 +连接固定 `muse_db.connect(readonly=True)`(会话级只读,写语句被库直接拒)。 + +查询语义与 assemble-context 的三个一等取数端保持一致(按边界合同工具 server +自持实现,不跨 Skill 导入):已确认细纲、已确认文风投影、已确认范式绑定; +另加正文章节读取与实体检索两个探索工具。 + +CLI:`read_tools.py <工具名> ''`,JSON 结果输出到 stdout; +未知工具或参数非法退出码 2(合同拒绝),运行时错误退出码 1。 +`--list` 输出工具登记表(名称、描述、读取表),供扩展与测试对账。 +""" +from __future__ import annotations + +import json +import sys +from typing import Any, Callable, Mapping + +import muse_db + +# 返回规模上限:防御性裁剪,避免单次工具返回撑爆模型上下文。 +MAX_TEXT_CHARS = 30000 +MAX_ROWS = 50 +MAX_DESC_CHARS = 500 + + +def _payload(value: Any) -> Any: + """规划行 payload 统一解成对象(库里可能是 jsonb 或字符串)。""" + + if isinstance(value, str): + return json.loads(value) + return value + + +def _style_constraints(style_payload: Any) -> list[str]: + """把文风投影成约束字符串列表(与 assemble-context 口径一致)。""" + + if isinstance(style_payload, str): + text = style_payload.strip() + return [text] if text else [] + if isinstance(style_payload, dict): + rules = [] + for aspect, value in style_payload.items(): + text = str(value).strip() + if text: + rules.append(f"{aspect}:{text}") + return rules + return [] + + +def _require_int(args: Mapping[str, Any], name: str) -> int: + value = args.get(name) + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"参数 {name} 必须是整数") + return value + + +def read_fine_outline(conn, args: Mapping[str, Any]) -> dict[str, Any]: + """读指定章最新一条已确认细纲(state=confirmed,version 倒序)。""" + + work_id = _require_int(args, "work_id") + target_chapter = _require_int(args, "target_chapter") + row = conn.execute( + "SELECT payload, version FROM example_planning_section WHERE work_id=%s AND " + "section_type='fine_outline' AND target_chapter=%s AND state='confirmed' AND " + "deleted=false ORDER BY version DESC LIMIT 1", + (work_id, target_chapter), + ).fetchone() + if not row: + return {"found": False, "reason": f"无第{target_chapter}章已确认细纲"} + return {"found": True, "version": row[1], "payload": _payload(row[0])} + + +def read_style_constraints(conn, args: Mapping[str, Any]) -> dict[str, Any]: + """读已确认文风并投影为约束;优先独立 style 行,回退设定行 style 字段。""" + + work_id = _require_int(args, "work_id") + row = conn.execute( + "SELECT payload FROM example_planning_section WHERE work_id=%s AND " + "schema_type='style' AND state='confirmed' AND deleted=false " + "ORDER BY version DESC LIMIT 1", + (work_id,), + ).fetchone() + if row: + return {"constraints": _style_constraints(_payload(row[0]))} + setting = conn.execute( + "SELECT payload FROM example_planning_section WHERE work_id=%s AND " + "section_type='setting' AND state='confirmed' AND deleted=false " + "ORDER BY version DESC LIMIT 1", + (work_id,), + ).fetchone() + if setting: + payload = _payload(setting[0]) + if isinstance(payload, dict): + return {"constraints": _style_constraints(payload.get("style"))} + return {"constraints": []} + + +def read_pattern_bindings(conn, args: Mapping[str, Any]) -> dict[str, Any]: + """读最新一条已确认 assembly 规划行的 patternReferences;无绑定诚实返空。""" + + work_id = _require_int(args, "work_id") + row = conn.execute( + "SELECT payload FROM example_planning_section WHERE work_id=%s AND " + "section_type='assembly' AND state='confirmed' AND deleted=false " + "ORDER BY version DESC LIMIT 1", + (work_id,), + ).fetchone() + if not row: + return {"bindings": []} + payload = _payload(row[0]) + if not isinstance(payload, dict): + return {"bindings": []} + references = payload.get("patternReferences", []) + return {"bindings": [dict(item) for item in references if isinstance(item, dict)]} + + +def read_chapter_text(conn, args: Mapping[str, Any]) -> dict[str, Any]: + """读指定章正式正文(章节行 + 正文块按序拼接);超长截断并标记。""" + + work_id = _require_int(args, "work_id") + chapter_order = _require_int(args, "chapter_order") + chapter = conn.execute( + "SELECT c.id, c.order_no, c.title FROM muse_content_chapter c " + "WHERE c.work_id=%s AND c.order_no=%s AND c.deleted=false LIMIT 1", + (work_id, chapter_order), + ).fetchone() + if not chapter: + return {"found": False, "reason": f"无第{chapter_order}章(work={work_id})"} + blocks = conn.execute( + "SELECT b.content_text FROM muse_content_block b " + "WHERE b.chapter_id=%s AND b.deleted=false ORDER BY b.order_no", + (chapter[0],), + ).fetchall() + text = "\n".join(str(block[0] or "") for block in blocks) + truncated = len(text) > MAX_TEXT_CHARS + return { + "found": True, + "chapterOrder": chapter[1], + "title": chapter[2], + "text": text[:MAX_TEXT_CHARS], + "truncated": truncated, + "totalChars": len(text), + } + + +def search_entities(conn, args: Mapping[str, Any]) -> dict[str, Any]: + """按名称模糊检索作品实体(人物、势力、地点、力量体系等);结果有界。""" + + work_id = _require_int(args, "work_id") + keyword = str(args.get("keyword") or "").strip() + if not keyword: + raise ValueError("参数 keyword 不能为空") + entity_type = args.get("entity_type") + if entity_type is not None and not isinstance(entity_type, str): + raise ValueError("参数 entity_type 必须是字符串") + query = ( + "SELECT entity_type, normalized_name, description, status " + "FROM muse_knowledge_entity WHERE work_id=%s AND deleted=false " + "AND normalized_name ILIKE %s" + ) + params: list[Any] = [work_id, f"%{keyword}%"] + if entity_type: + query += " AND entity_type=%s" + params.append(entity_type) + query += " ORDER BY normalized_name LIMIT %s" + params.append(MAX_ROWS) + rows = conn.execute(query, tuple(params)).fetchall() + return { + "entities": [ + { + "entityType": row[0], + "name": row[1], + "description": str(row[2] or "")[:MAX_DESC_CHARS], + "status": row[3], + } + for row in rows + ] + } + + +# 工具登记表:名称 -> (描述、读取表、参数、实现)。扩展与测试以此对账,保证单一事实源。 +# args 是参数说明:{参数名: 类型},类型取 int / str;可选参数名后加 ?。 +TOOL_REGISTRY: dict[str, dict[str, Any]] = { + "read_fine_outline": { + "description": "读指定作品指定章的最新一条已确认细纲(结构骨架与硬约束)。", + "tables": ("example_planning_section",), + "args": {"work_id": "int", "target_chapter": "int"}, + "impl": read_fine_outline, + }, + "read_style_constraints": { + "description": "读作品已确认文风并投影为约束列表。", + "tables": ("example_planning_section",), + "args": {"work_id": "int"}, + "impl": read_style_constraints, + }, + "read_pattern_bindings": { + "description": "读作品已确认的范式绑定(规划期选定,写作期只消费)。", + "tables": ("example_planning_section",), + "args": {"work_id": "int"}, + "impl": read_pattern_bindings, + }, + "read_chapter_text": { + "description": "读指定章的正式正文全文(超长截断并标记)。", + "tables": ("muse_content_chapter", "muse_content_block"), + "args": {"work_id": "int", "chapter_order": "int"}, + "impl": read_chapter_text, + }, + "search_entities": { + "description": "按名称模糊检索作品实体(人物、势力、地点、力量体系等)。", + "tables": ("muse_knowledge_entity",), + "args": {"work_id": "int", "keyword": "str", "entity_type?": "str"}, + "impl": search_entities, + }, +} + + +def execute_tool(name: str, args: Mapping[str, Any], connect: Callable[..., Any] | None = None) -> dict[str, Any]: + """执行一个登记工具;只读连接,未知工具拒绝。""" + + entry = TOOL_REGISTRY.get(name) + if entry is None: + raise KeyError(f"未知探索工具: {name}") + factory = connect or (lambda: muse_db.connect(readonly=True)) + conn = factory() + try: + return entry["impl"](conn, args) + finally: + closer = getattr(conn, "close", None) + if callable(closer): + closer() + + +def main(argv: list[str]) -> int: + if argv and argv[0] == "--list": + listing = { + name: { + "description": entry["description"], + "tables": list(entry["tables"]), + "args": entry.get("args", {}), + } + for name, entry in TOOL_REGISTRY.items() + } + print(json.dumps(listing, ensure_ascii=False, indent=1)) + return 0 + if len(argv) != 2: + print("用法: read_tools.py <工具名> '' | --list", file=sys.stderr) + return 2 + name, raw_args = argv + if name not in TOOL_REGISTRY: + print(f"未知探索工具: {name}", file=sys.stderr) + return 2 + try: + args = json.loads(raw_args) + except json.JSONDecodeError: + print("参数不是合法 JSON", file=sys.stderr) + return 2 + if not isinstance(args, dict): + print("参数必须是 JSON 对象", file=sys.stderr) + return 2 + try: + result = execute_tool(name, args) + except ValueError as exc: + print(f"参数合同拒绝: {exc}", file=sys.stderr) + return 2 + except Exception as exc: # 运行时失败如实上报,不静默 + print(f"工具执行失败: {type(exc).__name__}: {exc}", file=sys.stderr) + return 1 + print(json.dumps(result, ensure_ascii=False)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) + + +__all__ = ["TOOL_REGISTRY", "execute_tool", "main"] diff --git a/docs/plans/2026-08-22-agent-example整体收敛总plan.md b/docs/plans/2026-08-22-agent-example整体收敛总plan.md index e39fdb5..1f4ae63 100644 --- a/docs/plans/2026-08-22-agent-example整体收敛总plan.md +++ b/docs/plans/2026-08-22-agent-example整体收敛总plan.md @@ -147,7 +147,7 @@ A、B 可并行;C 依赖 A;E 依赖 D;F 依赖 E;G 依赖 D(链路透 ### 阶段 B:链路盘点与清理 -- 意图:盘点全部模型调用路径、59 个技能、一次性脚本与散落设计文档,每项定终态归属,删除无消费者的遗留。 +- 意图:盘点全部模型调用路径、58 个技能、一次性脚本与散落设计文档,每项定终态归属,删除无消费者的遗留。 - 边界:4 条模型调用路径定终态--框架派发链为生产与评测主链;治理直调链为对照实验与内容模型任务;角色运行时链随直调链处置;批处理执行链定保留或退役。删除以零引用检索为依据。 - 验证:被删入口全仓零引用;全量离线清单无新增失败;单一提交可一次对比删除面。 diff --git a/docs/plans/2026-08-22-阶段D-能力建设.md b/docs/plans/2026-08-22-阶段D-能力建设.md new file mode 100644 index 0000000..5e7db20 --- /dev/null +++ b/docs/plans/2026-08-22-阶段D-能力建设.md @@ -0,0 +1,58 @@ +# 阶段 D:能力建设(只读工具 server / 会话复用 / 依赖清单) + +日期:2026-08-22 +总 plan:[2026-08-22-agent-example整体收敛总plan.md](2026-08-22-agent-example整体收敛总plan.md) +状态:离线部分完成;真实派发验证待人工授权(一次带探索工具的显式模型调用)。 + +## 1. 意图 + +给框架派发链补齐三块底座:只读工具 server、会话复用、依赖清单。 + +## 2. 实现落点 + +### 2.1 只读工具 server(边界合同的工具层组件) + +- `read_tools.py`:五个登记工具,单一事实源 `TOOL_REGISTRY`(名称、描述、读取表、参数说明、实现): + - `read_fine_outline`(已确认细纲)、`read_style_constraints`(已确认文风投影)、`read_pattern_bindings`(已确认范式绑定)——查询语义与 assemble-context 三个一等取数端一致(工具 server 按边界合同自持实现,不跨 Skill 导入)。 + - `read_chapter_text`(章节正式正文,30000 字截断标记)、`search_entities`(实体模糊检索,50 行上限)。 + - 只读连接 `muse_db.connect(readonly=True)`;未知工具/非法参数合同拒绝(退出码 2);`--list` 输出登记表。 +- `muse_read_tools_extension.ts`:框架扩展,装载时从 `--list` 动态注册工具(不硬编码工具清单,防双源漂移);每次执行转给只读实现;派发器未注入环境变量时不注册任何工具。 +- 派发器经 `-e` 显式加载该扩展;`--no-extensions` 只禁自动发现(框架资源加载器合同已核实),隔离与工具加载共存。 +- 派发门禁:任务白名单含登记表工具名但未 `--enable-read-tools` 时,运行前失败关闭(READ_TOOLS_NOT_ENABLED)。 + +### 2.2 会话复用 + +- `ExecutionPolicy` 增加 `session_id` / `session_dir`(同给或同缺,否则构造拒绝);`build_pi_argv` 映射为框架原生 `--session-id` + `--session-dir`(不自造会话存储)。 +- 派发语义:每次派发仍是一次独立运行记录;会话连续通过共享会话目录实现——同一章补证/修改传前次运行的会话 ID 与会话目录继续原会话;完全重写新建。 +- CLI 新增 `--session-id` / `--session-dir` / `--enable-read-tools`。 + +### 2.3 依赖清单 + +- 框架工具事件携带参数(`tool_execution_start.args`);适配器捕获进 `ToolCallRecord.args`,截断后随 `tool.started` 事件入账本。 +- 成功运行把工具调用序列写为运行目录 `dependencies.json`(seq/tool/args/isError);回执含 `dependencies` 摘要(count/file)。无工具调用不产文件。 +- 不加新表:事件账本 + 运行目录审计件承载;创作台链路透视(阶段 G)读这两处。 + +## 3. 文件台账 + +| 处置 | 文件 | 原因 | +|---|---|---| +| 新增 | `scripts/read_tools.py` | 填补边界合同工具层空白:探索读取的唯一受控入口;登记表为单一事实源;入口为扩展装载与测试 | +| 新增 | `scripts/muse_read_tools_extension.ts` | 工具 server 的框架适配:只从登记表动态注册,无第二套工具清单;经派发器 `-e` 加载 | +| 修改 | `scripts/pi_runner.py` | 会话复用字段与 argv、扩展加载、工具参数捕获(依赖清单材料);职责仍是薄适配,无业务决策 | +| 修改 | `scripts/dispatch_agent_task.py` | 三个 CLI 开关、工具门禁、环境变量注入、依赖清单写入、回执字段 | +| 修改 | `SKILL.md`(dispatch-agent-task) | 登记新模块、命令参数、事件与依赖清单合同、工具 server 红线 | +| 新增 | `tests/skills/dispatch-agent-task/test_read_tools.py` | 工具合同离线测试:五工具行为、拒绝路径、登记表 CLI、扩展动态装载对账 | +| 修改 | `tests/skills/dispatch-agent-task/test_dispatch_agent_task.py` | 新增 5 项:会话成对校验与 argv、扩展标志、工具参数捕获、工具门禁失败关闭、依赖清单产出 | +| 修改 | `harness/manifests/test-inventory.json` | 登记新测试条目,摘要同步重算(110 条) | +| 保留 | `execute-role-task`、直调链 | 阶段 F 裁决对象,本阶段不动 | +| 遗留 | 真实派发验证(一次带工具的显式模型调用,查库核对事件/依赖清单) | 需人工授权;授权后执行并回填本文件验证节 | + +## 4. 验证 + +- 工具离线测试:10 项通过(五工具行为、拒绝路径、登记表、扩展对账)。 +- 派发测试:25 项通过(原 20 项 + 会话/工具/依赖清单 5 项)。 +- 选择器定向:新增两条目通过。 +- 全量离线清单:102 通过、8 项外部依赖阻断、0 失败(较阶段 C 基线 +1 条目)。 +- 技能严格审计:58 个 Skill,阻断 0,质量发现 0;索引一致性通过。 +- 架构门禁(导入边界 + 索引):通过。`git diff --check`:通过。 +- 真实派发验证:待授权(见 §3 遗留)。 diff --git a/harness/manifests/test-inventory.json b/harness/manifests/test-inventory.json index 71768a7..1a8fc4a 100644 --- a/harness/manifests/test-inventory.json +++ b/harness/manifests/test-inventory.json @@ -1,1814 +1,1831 @@ { - "schema_version": 1, - "generated_scope": "Current agent-example source test assets: .agent/skills/**/test_*.py and *_test.py, tests/skills/** source files, humanization/tests/** source files, harness/**/test_*.py, dashboard/test_server_display.py, and other obvious source test files; excludes .git, .venv, __pycache__ compiled artifacts, deleted working-tree files, and harness specification documents.", - "entries": [ - { - "path": "dashboard/test_server_display.py", - "scope": "other", - "owner_skill_or_domain": "dashboard", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests dashboard display/encoding helpers and synthetic AI-flavor views; the file declares no database connection." - }, - { - "path": "harness/evals/skills/diagnose-ai-flavor/run_eval.py", - "kind": "skill_behavior_eval", - "scope": "runtime_skill", - "owner_skill_or_domain": "diagnose-ai-flavor", - "evidence_level": "real_dependency_integration", - "requires": [ - "model", - "credentials" - ], - "side_effects": [ - "model" - ], - "skill_behavior_eval": true, - "classification_basis": "Skill 行为评测入口:默认真实模型适配器,未授权时以稳定码失败关闭;fake 适配器只验证评测管道", - "classification_confidence": "high" - }, - { - "path": "harness/evals/test_skill_eval.py", - "kind": "harness_self_test", - "scope": "harness", - "owner_skill_or_domain": "harness", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [], - "skill_behavior_eval": false, - "classification_basis": "行为评测引擎的确定性离线自测:裁决逻辑、失败关闭和报告合同;不构成 Skill 行为证据", - "classification_confidence": "high" - }, - { - "path": "harness/test_run_selected.py", - "scope": "harness", - "owner_skill_or_domain": "harness", - "kind": "harness_self_test", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem", - "subprocess" - ], - "side_effects": [ - "filesystem", - "subprocess" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses temporary manifests and a fake Python child process to test selector, dependency, timeout, nonzero and output-summary handling." - }, - { - "path": "harness/test_skill_harness.py", - "scope": "harness", - "owner_skill_or_domain": "harness", - "kind": "harness_self_test", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Creates temporary SKILL.md/manifest fixtures and tests harness static-audit reports and CLI exit codes." - }, - { - "path": "humanization/tests/test_contracts.py", - "scope": "domain", - "owner_skill_or_domain": "humanization", - "kind": "domain_eval", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks humanization asset contracts and executes the synthetic U0 patch/review replay; no external Agent/model driver." - }, - { - "path": "humanization/tests/test_framework_coverage.py", - "scope": "domain", - "owner_skill_or_domain": "humanization", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Reads the research coverage YAML and checks capability owners, implementation paths, and status values." - }, - { - "path": "humanization/tests/test_humanization_v2.py", - "scope": "domain", - "owner_skill_or_domain": "humanization", - "kind": "domain_eval", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Evaluates synthetic voice/rule/carrier gates and lifecycle fixtures with temporary files; no external Agent/model reads SKILL.md." - }, - { - "path": "humanization/tests/test_load_db.py", - "kind": "tool_contract", - "scope": "domain", - "owner_skill_or_domain": "humanization", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [], - "skill_behavior_eval": false, - "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", - "classification_confidence": "high" - }, - { - "path": "humanization/tests/test_load_db_pg_smoke.py", - "kind": "integration", - "scope": "domain", - "owner_skill_or_domain": "humanization", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_basis": "真实 PostgreSQL 冒烟:数据库规则库与文件种子指纹一致性,需显式环境变量授权", - "classification_confidence": "high" - }, - { - "path": "humanization/tests/test_seed_rules_db.py", - "kind": "tool_contract", - "scope": "domain", - "owner_skill_or_domain": "humanization", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [], - "skill_behavior_eval": false, - "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", - "classification_confidence": "high" - }, - { - "path": "tests/architecture/test_import_boundaries.py", - "scope": "domain", - "owner_skill_or_domain": "architecture", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Scans Skill and dashboard Python files for forbidden sys.path injections into shared runtime implementations (humanization/src, access-database/scripts, call-content-model/scripts, embed-knowledge/scripts, execute-role-task/scripts, establish-voice-baseline/scripts) and Skill imports from the dashboard." - }, - { - "path": "tests/architecture/test_skills_index.py", - "scope": "domain", - "owner_skill_or_domain": "architecture", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "校验 Skill 发现总索引 .agent/skills/_index.md 与磁盘 skill、SKILL.md frontmatter 和 harness/manifests/skills.json 三方一致,拒绝增删改名漏同步或手改造成的漂移。" - }, - { - "path": "tests/skills/access-database/test_authorization_snapshot_ddl.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "access-database", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver." - }, - { - "path": "tests/skills/access-database/test_db_params.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "access-database", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked." - }, - { - "path": "tests/skills/access-database/test_skill_catalog.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "access-database", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files." - }, - { - "path": "tests/skills/adjudicate-quality-gate/test_gate_input_builder.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "adjudicate-quality-gate", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls." - }, - { - "path": "tests/skills/adjudicate-quality-gate/test_writer_gate.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "adjudicate-quality-gate", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output." - }, - { - "path": "tests/skills/assemble-context/test_assemble_writer_context.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts." - }, - { - "path": "tests/skills/assemble-context/test_fine_outline_reader.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing." - }, - { - "path": "tests/skills/assemble-context/test_fine_outline_unification.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler." - }, - { - "path": "tests/skills/assemble-context/test_freeze_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for context freeze win lesson; no database." - }, - { - "path": "tests/skills/assemble-context/test_pattern_binding_reader.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior." - }, - { - "path": "tests/skills/assemble-context/test_retrieve_writer_sources.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections." - }, - { - "path": "tests/skills/assemble-context/test_style_loader.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection." - }, - { - "path": "tests/skills/assemble-context/test_writer_contract.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory." - }, - { - "path": "tests/skills/backup-work-extraction/test_backup_upgrade_work_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "backup-work-extraction", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched." - }, - { - "path": "tests/skills/call-content-model/test_call_persistence.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "call-content-model", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter." - }, - { - "path": "tests/skills/call-content-model/test_quota.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "call-content-model", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls." - }, - { - "path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "capture-ai-flavor-cases", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked." - }, - { - "path": "tests/skills/capture-ai-flavor-cases/test_capture_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "capture-ai-flavor-cases", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for capture batch persist lessons; no database." - }, - { - "path": "tests/skills/check-content-consistency/test_build_semantic_input.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "check-content-consistency", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection." - }, - { - "path": "tests/skills/check-content-consistency/test_check_writer_candidate.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "check-content-consistency", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields." - }, - { - "path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "check-content-consistency", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called." - }, - { - "path": "tests/skills/clean-book-text/test_clean_detect_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "clean-book-text", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked." - }, - { - "path": "tests/skills/confirm-knowledge-draft/test_confirm_knowledge_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "confirm-knowledge-draft", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs." - }, - { - "path": "tests/skills/decide-candidate/test_decision_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup; checks accept=win and discard=lesson payloads without database." - }, - { - "path": "tests/skills/decide-candidate/test_fact_delta.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions." - }, - { - "path": "tests/skills/decide-candidate/test_fact_delta_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows." - }, - { - "path": "tests/skills/decide-candidate/test_next_steps_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Asserts accept/discard next_steps are suggestions with auto=false and human_authorize where content would change." - }, - { - "path": "tests/skills/decide-candidate/test_projection_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup." - }, - { - "path": "tests/skills/decide-candidate/test_write_canonical_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards." - }, - { - "path": "tests/skills/decide-candidate/test_writer_acceptance.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store." - }, - { - "path": "tests/skills/deconstruct-book/test_deconstruct_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "deconstruct-book", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for deconstruct outline window lesson; no database." - }, - { - "path": "tests/skills/deconstruct-book/test_parse_llm_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "deconstruct-book", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files." - }, - { - "path": "tests/skills/deconstruct-book/test_parse_outline_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "deconstruct-book", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function." - }, - { - "path": "tests/skills/design-story-foundation/test_assert_selection_handoff.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "design-story-foundation", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks selection handoff JSON required fields and fail-closed paths; no database or model." - }, - { - "path": "tests/skills/design-story-foundation/test_validate_candidates.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "design-story-foundation", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates candidate tree/heading/placeholder/root contracts using temporary candidate files." - }, - { - "path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "diagnose-ai-flavor", - "kind": "domain_eval", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge." - }, - { - "path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "embed-knowledge", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows." - }, - { - "path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "establish-voice-baseline", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_run_replay.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the replay orchestrator with an in-memory governed-chat fake, temp output, and synthetic planner/detector/judge responses." - }, - { - "path": "tests/skills/execute-role-task/test_muse_role.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "execute-role-task", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates provider-neutral role profiles, prompt injection, model-policy binding, schema validation, budgets, timeouts and receipts through an in-memory governed-chat fake." - }, - { - "path": "tests/skills/extract-chapter-knowledge/test_chapter_extract_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-chapter-knowledge", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for chapter extraction win lesson; no database." - }, - { - "path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-chapter-knowledge", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions." - }, - { - "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry." - }, - { - "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1." - }, - { - "path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made." - }, - { - "path": "tests/skills/extract-work-knowledge/test_upgrade_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for upgrade window batch lesson; no database." - }, - { - "path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics." - }, - { - "path": "tests/skills/freeze-context/test_audit_leakage.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records." - }, - { - "path": "tests/skills/freeze-context/test_build_snapshot.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory." - }, - { - "path": "tests/skills/freeze-context/test_check_snapshot.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution." - }, - { - "path": "tests/skills/freeze-context/test_load_reference_work.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection." - }, - { - "path": "tests/skills/load-replay-reference-work/test_load_writer_reference_work.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "load-replay-reference-work", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files." - }, - { - "path": "tests/skills/load-replay-reference-work/test_pattern_reference_injection.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "load-replay-reference-work", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection." - }, - { - "path": "tests/skills/merge-story-candidates/test_serial_merge.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "merge-story-candidates", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests packet construction and raw-output parsing with temporary Markdown files; no model or service driver." - }, - { - "path": "tests/skills/plan-chapter/test_contract.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-chapter", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts." - }, - { - "path": "tests/skills/plan-story/test_field_coverage.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write." - }, - { - "path": "tests/skills/plan-story/test_planning_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for planning section persistence lesson; no database." - }, - { - "path": "tests/skills/plan-story/test_record_planning_execution.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions." - }, - { - "path": "tests/skills/plan-story/test_repair_deterministic_receipt.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries." - }, - { - "path": "tests/skills/plan-story/test_select_patterns_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results." - }, - { - "path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "prevent-ai-flavor", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched." - }, - { - "path": "tests/skills/prevent-ai-flavor/test_prevention_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "prevent-ai-flavor", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for prevention contract win lesson; no database." - }, - { - "path": "tests/skills/promote-ai-flavor-rule/test_propose_rule.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "promote-ai-flavor-rule", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests rule candidate induction gates and CLI ownership after split from capture-ai-flavor-cases." - }, - { - "path": "tests/skills/record-run-evidence/test_file_cas.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions." - }, - { - "path": "tests/skills/record-run-evidence/test_lesson_registry_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them." - }, - { - "path": "tests/skills/record-run-evidence/test_persist_raw.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests only the raw secret-pattern validator with direct strings." - }, - { - "path": "tests/skills/record-run-evidence/test_raw_vault.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "offline", - "filesystem", - "raw_vault" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery." - }, - { - "path": "tests/skills/record-run-evidence/test_record_failed_run.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence." - }, - { - "path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks receipt eligibility predicates using in-memory values only." - }, - { - "path": "tests/skills/record-run-evidence/test_run_registry.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests run ID formatting and terminal-state rejection before database access." - }, - { - "path": "tests/skills/refresh-runtime-probe/test_refresh_runtime_probe.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "refresh-runtime-probe", - "kind": "runtime_probe", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises provider-neutral runtime probe refresh, dry-run non-acceptance and authorization gates with fake role results and temp output files." - }, - { - "path": "tests/skills/replay-writer-gate/test_run_writer_replay.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "replay-writer-gate", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem", - "raw_vault" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model." - }, - { - "path": "tests/skills/replay-writer-gate/test_writer_eval_preregister.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "replay-writer-gate", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration." - }, - { - "path": "tests/skills/reset-work-extraction/test_reset_upgrade_work_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "reset-work-extraction", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries." - }, - { - "path": "tests/skills/review-knowledge-cards/test_calibrate_stamp.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "review-knowledge-cards", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises calibrate stamp fingerprint, MAE gate, and fail-closed missing/stale/failed stamps using a temp file; no database or model call." - }, - { - "path": "tests/skills/review-knowledge-cards/test_review_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "review-knowledge-cards", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for knowledge card review writeback lesson; no database." - }, - { - "path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "revise-ai-flavor", - "kind": "domain_eval", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked." - }, - { - "path": "tests/skills/revise-ai-flavor/test_revision_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "revise-ai-flavor", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for revision conclusion lesson kinds; no database." - }, - { - "path": "tests/skills/rewrite-selection/test_assert_expected_revision.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "rewrite-selection", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks expectedRevision match/mismatch as a pure function; database access is not invoked." - }, - { - "path": "tests/skills/score-content-quality/test_rubric.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory." - }, - { - "path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used." - }, - { - "path": "tests/skills/score-content-quality/test_score_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for stable blind-judge score lessons; no database." - }, - { - "path": "tests/skills/score-content-quality/test_writer_rubric.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts." - }, - { - "path": "tests/skills/search-knowledge/test_search.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "search-knowledge", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding." - }, - { - "path": "tests/skills/write-next-chapter/test_candidate_cas.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops." - }, - { - "path": "tests/skills/write-next-chapter/test_candidate_cas_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": [ - "postgresql" - ], - "side_effects": [ - "postgresql" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows." - }, - { - "path": "tests/skills/write-next-chapter/test_gate_anchor_projection.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "medium", - "classification_basis": "加载 .agent/skills/write-next-chapter/scripts/produce_next_chapter.py 并断言机械验收子串(门锚)对写手可见;只测纯函数投影,不产生系统事实。" - }, - { - "path": "tests/skills/write-next-chapter/test_persist_writer_run.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call." - }, - { - "path": "tests/skills/write-next-chapter/test_production_evidence_reassemble.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "离线单测 production_evidence_reassemble:缺口识别生成新 attempt、新快照与创作输入变化,全程 mock,不连库不调模型。" - }, - { - "path": "tests/skills/write-next-chapter/test_run_writer.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the writer adapter against an in-memory governed-chat fake and frozen role profiles; no real model call." - }, - { - "path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors." - }, - { - "path": "tests/skills/write-next-chapter/test_semantic_verdict.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence." - }, - { - "path": "tests/skills/write-next-chapter/test_writer_lesson_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks propose_lesson_dedup for writer mechanical gate lessons; no database." - }, - { - "path": "tests/skills/dispatch-agent-task/test_dispatch_agent_task.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "dispatch-agent-task", - "kind": "runtime_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline", - "filesystem" - ], - "side_effects": [ - "filesystem" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises portable task spec contract, pi adapter argv/stream normalization and run_dispatch fail-closed chain with fake pi streams and injected DB connections; no network, no real framework binary." - }, - { - "path": "tests/skills/record-run-evidence/test_agent_trace.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "runtime_contract", - "evidence_level": "deterministic_offline", - "requires": [ - "offline" - ], - "side_effects": [ - "none" - ], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Fixes agent event ledger writer shapes and atomic framework evidence persistence with fake connections; no DB, no network." - } - ], - "summary": { - "entry_count": 109, - "by_scope": { - "other": 1, - "runtime_skill": 97, - "harness": 3, - "domain": 8 - }, - "by_kind": { - "tool_contract": 35, - "skill_behavior_eval": 1, - "harness_self_test": 3, - "domain_eval": 4, - "integration": 9, - "tool_unit": 32, - "fake_pipeline": 22, - "runtime_probe": 1, - "runtime_contract": 2 - }, - "by_evidence_level": { - "deterministic_offline": 93, - "real_dependency_integration": 10, - "static_structure": 6 - }, - "total": 109 + "schema_version": 1, + "generated_scope": "Current agent-example source test assets: .agent/skills/**/test_*.py and *_test.py, tests/skills/** source files, humanization/tests/** source files, harness/**/test_*.py, dashboard/test_server_display.py, and other obvious source test files; excludes .git, .venv, __pycache__ compiled artifacts, deleted working-tree files, and harness specification documents.", + "entries": [ + { + "path": "dashboard/test_server_display.py", + "scope": "other", + "owner_skill_or_domain": "dashboard", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests dashboard display/encoding helpers and synthetic AI-flavor views; the file declares no database connection." + }, + { + "path": "harness/evals/skills/diagnose-ai-flavor/run_eval.py", + "kind": "skill_behavior_eval", + "scope": "runtime_skill", + "owner_skill_or_domain": "diagnose-ai-flavor", + "evidence_level": "real_dependency_integration", + "requires": [ + "model", + "credentials" + ], + "side_effects": [ + "model" + ], + "skill_behavior_eval": true, + "classification_basis": "Skill 行为评测入口:默认真实模型适配器,未授权时以稳定码失败关闭;fake 适配器只验证评测管道", + "classification_confidence": "high" + }, + { + "path": "harness/evals/test_skill_eval.py", + "kind": "harness_self_test", + "scope": "harness", + "owner_skill_or_domain": "harness", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [], + "skill_behavior_eval": false, + "classification_basis": "行为评测引擎的确定性离线自测:裁决逻辑、失败关闭和报告合同;不构成 Skill 行为证据", + "classification_confidence": "high" + }, + { + "path": "harness/test_run_selected.py", + "scope": "harness", + "owner_skill_or_domain": "harness", + "kind": "harness_self_test", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem", + "subprocess" + ], + "side_effects": [ + "filesystem", + "subprocess" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses temporary manifests and a fake Python child process to test selector, dependency, timeout, nonzero and output-summary handling." + }, + { + "path": "harness/test_skill_harness.py", + "scope": "harness", + "owner_skill_or_domain": "harness", + "kind": "harness_self_test", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Creates temporary SKILL.md/manifest fixtures and tests harness static-audit reports and CLI exit codes." + }, + { + "path": "humanization/tests/test_contracts.py", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "kind": "domain_eval", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks humanization asset contracts and executes the synthetic U0 patch/review replay; no external Agent/model driver." + }, + { + "path": "humanization/tests/test_framework_coverage.py", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Reads the research coverage YAML and checks capability owners, implementation paths, and status values." + }, + { + "path": "humanization/tests/test_humanization_v2.py", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "kind": "domain_eval", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Evaluates synthetic voice/rule/carrier gates and lifecycle fixtures with temporary files; no external Agent/model reads SKILL.md." + }, + { + "path": "humanization/tests/test_load_db.py", + "kind": "tool_contract", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [], + "skill_behavior_eval": false, + "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", + "classification_confidence": "high" + }, + { + "path": "humanization/tests/test_load_db_pg_smoke.py", + "kind": "integration", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_basis": "真实 PostgreSQL 冒烟:数据库规则库与文件种子指纹一致性,需显式环境变量授权", + "classification_confidence": "high" + }, + { + "path": "humanization/tests/test_seed_rules_db.py", + "kind": "tool_contract", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [], + "skill_behavior_eval": false, + "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", + "classification_confidence": "high" + }, + { + "path": "tests/architecture/test_import_boundaries.py", + "scope": "domain", + "owner_skill_or_domain": "architecture", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Scans Skill and dashboard Python files for forbidden sys.path injections into shared runtime implementations (humanization/src, access-database/scripts, call-content-model/scripts, embed-knowledge/scripts, execute-role-task/scripts, establish-voice-baseline/scripts) and Skill imports from the dashboard." + }, + { + "path": "tests/architecture/test_skills_index.py", + "scope": "domain", + "owner_skill_or_domain": "architecture", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "校验 Skill 发现总索引 .agent/skills/_index.md 与磁盘 skill、SKILL.md frontmatter 和 harness/manifests/skills.json 三方一致,拒绝增删改名漏同步或手改造成的漂移。" + }, + { + "path": "tests/skills/access-database/test_authorization_snapshot_ddl.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "access-database", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver." + }, + { + "path": "tests/skills/access-database/test_db_params.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "access-database", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked." + }, + { + "path": "tests/skills/access-database/test_skill_catalog.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "access-database", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files." + }, + { + "path": "tests/skills/adjudicate-quality-gate/test_gate_input_builder.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "adjudicate-quality-gate", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls." + }, + { + "path": "tests/skills/adjudicate-quality-gate/test_writer_gate.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "adjudicate-quality-gate", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output." + }, + { + "path": "tests/skills/assemble-context/test_assemble_writer_context.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts." + }, + { + "path": "tests/skills/assemble-context/test_fine_outline_reader.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing." + }, + { + "path": "tests/skills/assemble-context/test_fine_outline_unification.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler." + }, + { + "path": "tests/skills/assemble-context/test_freeze_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for context freeze win lesson; no database." + }, + { + "path": "tests/skills/assemble-context/test_pattern_binding_reader.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior." + }, + { + "path": "tests/skills/assemble-context/test_retrieve_writer_sources.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections." + }, + { + "path": "tests/skills/assemble-context/test_style_loader.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection." + }, + { + "path": "tests/skills/assemble-context/test_writer_contract.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory." + }, + { + "path": "tests/skills/backup-work-extraction/test_backup_upgrade_work_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "backup-work-extraction", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched." + }, + { + "path": "tests/skills/call-content-model/test_call_persistence.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "call-content-model", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter." + }, + { + "path": "tests/skills/call-content-model/test_quota.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "call-content-model", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls." + }, + { + "path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "capture-ai-flavor-cases", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked." + }, + { + "path": "tests/skills/capture-ai-flavor-cases/test_capture_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "capture-ai-flavor-cases", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for capture batch persist lessons; no database." + }, + { + "path": "tests/skills/check-content-consistency/test_build_semantic_input.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "check-content-consistency", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection." + }, + { + "path": "tests/skills/check-content-consistency/test_check_writer_candidate.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "check-content-consistency", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields." + }, + { + "path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "check-content-consistency", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called." + }, + { + "path": "tests/skills/clean-book-text/test_clean_detect_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "clean-book-text", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked." + }, + { + "path": "tests/skills/confirm-knowledge-draft/test_confirm_knowledge_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "confirm-knowledge-draft", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs." + }, + { + "path": "tests/skills/decide-candidate/test_decision_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup; checks accept=win and discard=lesson payloads without database." + }, + { + "path": "tests/skills/decide-candidate/test_fact_delta.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions." + }, + { + "path": "tests/skills/decide-candidate/test_fact_delta_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows." + }, + { + "path": "tests/skills/decide-candidate/test_next_steps_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Asserts accept/discard next_steps are suggestions with auto=false and human_authorize where content would change." + }, + { + "path": "tests/skills/decide-candidate/test_projection_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup." + }, + { + "path": "tests/skills/decide-candidate/test_write_canonical_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards." + }, + { + "path": "tests/skills/decide-candidate/test_writer_acceptance.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store." + }, + { + "path": "tests/skills/deconstruct-book/test_deconstruct_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "deconstruct-book", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for deconstruct outline window lesson; no database." + }, + { + "path": "tests/skills/deconstruct-book/test_parse_llm_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "deconstruct-book", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files." + }, + { + "path": "tests/skills/deconstruct-book/test_parse_outline_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "deconstruct-book", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function." + }, + { + "path": "tests/skills/design-story-foundation/test_assert_selection_handoff.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "design-story-foundation", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks selection handoff JSON required fields and fail-closed paths; no database or model." + }, + { + "path": "tests/skills/design-story-foundation/test_validate_candidates.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "design-story-foundation", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates candidate tree/heading/placeholder/root contracts using temporary candidate files." + }, + { + "path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "diagnose-ai-flavor", + "kind": "domain_eval", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge." + }, + { + "path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "embed-knowledge", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows." + }, + { + "path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "establish-voice-baseline", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_run_replay.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the replay orchestrator with an in-memory governed-chat fake, temp output, and synthetic planner/detector/judge responses." + }, + { + "path": "tests/skills/execute-role-task/test_muse_role.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "execute-role-task", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates provider-neutral role profiles, prompt injection, model-policy binding, schema validation, budgets, timeouts and receipts through an in-memory governed-chat fake." + }, + { + "path": "tests/skills/extract-chapter-knowledge/test_chapter_extract_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-chapter-knowledge", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for chapter extraction win lesson; no database." + }, + { + "path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-chapter-knowledge", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions." + }, + { + "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry." + }, + { + "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1." + }, + { + "path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made." + }, + { + "path": "tests/skills/extract-work-knowledge/test_upgrade_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for upgrade window batch lesson; no database." + }, + { + "path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics." + }, + { + "path": "tests/skills/freeze-context/test_audit_leakage.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records." + }, + { + "path": "tests/skills/freeze-context/test_build_snapshot.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory." + }, + { + "path": "tests/skills/freeze-context/test_check_snapshot.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution." + }, + { + "path": "tests/skills/freeze-context/test_load_reference_work.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection." + }, + { + "path": "tests/skills/load-replay-reference-work/test_load_writer_reference_work.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "load-replay-reference-work", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files." + }, + { + "path": "tests/skills/load-replay-reference-work/test_pattern_reference_injection.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "load-replay-reference-work", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection." + }, + { + "path": "tests/skills/merge-story-candidates/test_serial_merge.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "merge-story-candidates", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests packet construction and raw-output parsing with temporary Markdown files; no model or service driver." + }, + { + "path": "tests/skills/plan-chapter/test_contract.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-chapter", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts." + }, + { + "path": "tests/skills/plan-story/test_field_coverage.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write." + }, + { + "path": "tests/skills/plan-story/test_planning_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for planning section persistence lesson; no database." + }, + { + "path": "tests/skills/plan-story/test_record_planning_execution.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions." + }, + { + "path": "tests/skills/plan-story/test_repair_deterministic_receipt.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries." + }, + { + "path": "tests/skills/plan-story/test_select_patterns_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results." + }, + { + "path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "prevent-ai-flavor", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched." + }, + { + "path": "tests/skills/prevent-ai-flavor/test_prevention_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "prevent-ai-flavor", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for prevention contract win lesson; no database." + }, + { + "path": "tests/skills/promote-ai-flavor-rule/test_propose_rule.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "promote-ai-flavor-rule", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests rule candidate induction gates and CLI ownership after split from capture-ai-flavor-cases." + }, + { + "path": "tests/skills/record-run-evidence/test_file_cas.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions." + }, + { + "path": "tests/skills/record-run-evidence/test_lesson_registry_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them." + }, + { + "path": "tests/skills/record-run-evidence/test_persist_raw.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests only the raw secret-pattern validator with direct strings." + }, + { + "path": "tests/skills/record-run-evidence/test_raw_vault.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "offline", + "filesystem", + "raw_vault" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery." + }, + { + "path": "tests/skills/record-run-evidence/test_record_failed_run.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence." + }, + { + "path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks receipt eligibility predicates using in-memory values only." + }, + { + "path": "tests/skills/record-run-evidence/test_run_registry.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests run ID formatting and terminal-state rejection before database access." + }, + { + "path": "tests/skills/refresh-runtime-probe/test_refresh_runtime_probe.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "refresh-runtime-probe", + "kind": "runtime_probe", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises provider-neutral runtime probe refresh, dry-run non-acceptance and authorization gates with fake role results and temp output files." + }, + { + "path": "tests/skills/replay-writer-gate/test_run_writer_replay.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "replay-writer-gate", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem", + "raw_vault" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model." + }, + { + "path": "tests/skills/replay-writer-gate/test_writer_eval_preregister.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "replay-writer-gate", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration." + }, + { + "path": "tests/skills/reset-work-extraction/test_reset_upgrade_work_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "reset-work-extraction", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries." + }, + { + "path": "tests/skills/review-knowledge-cards/test_calibrate_stamp.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "review-knowledge-cards", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises calibrate stamp fingerprint, MAE gate, and fail-closed missing/stale/failed stamps using a temp file; no database or model call." + }, + { + "path": "tests/skills/review-knowledge-cards/test_review_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "review-knowledge-cards", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for knowledge card review writeback lesson; no database." + }, + { + "path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "revise-ai-flavor", + "kind": "domain_eval", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked." + }, + { + "path": "tests/skills/revise-ai-flavor/test_revision_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "revise-ai-flavor", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for revision conclusion lesson kinds; no database." + }, + { + "path": "tests/skills/rewrite-selection/test_assert_expected_revision.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "rewrite-selection", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks expectedRevision match/mismatch as a pure function; database access is not invoked." + }, + { + "path": "tests/skills/score-content-quality/test_rubric.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory." + }, + { + "path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used." + }, + { + "path": "tests/skills/score-content-quality/test_score_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for stable blind-judge score lessons; no database." + }, + { + "path": "tests/skills/score-content-quality/test_writer_rubric.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts." + }, + { + "path": "tests/skills/search-knowledge/test_search.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "search-knowledge", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding." + }, + { + "path": "tests/skills/write-next-chapter/test_candidate_cas.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops." + }, + { + "path": "tests/skills/write-next-chapter/test_candidate_cas_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows." + }, + { + "path": "tests/skills/write-next-chapter/test_gate_anchor_projection.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "medium", + "classification_basis": "加载 .agent/skills/write-next-chapter/scripts/produce_next_chapter.py 并断言机械验收子串(门锚)对写手可见;只测纯函数投影,不产生系统事实。" + }, + { + "path": "tests/skills/write-next-chapter/test_persist_writer_run.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call." + }, + { + "path": "tests/skills/write-next-chapter/test_production_evidence_reassemble.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "离线单测 production_evidence_reassemble:缺口识别生成新 attempt、新快照与创作输入变化,全程 mock,不连库不调模型。" + }, + { + "path": "tests/skills/write-next-chapter/test_run_writer.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the writer adapter against an in-memory governed-chat fake and frozen role profiles; no real model call." + }, + { + "path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors." + }, + { + "path": "tests/skills/write-next-chapter/test_semantic_verdict.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence." + }, + { + "path": "tests/skills/write-next-chapter/test_writer_lesson_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks propose_lesson_dedup for writer mechanical gate lessons; no database." + }, + { + "path": "tests/skills/dispatch-agent-task/test_dispatch_agent_task.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "dispatch-agent-task", + "kind": "runtime_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises portable task spec contract, pi adapter argv/stream normalization and run_dispatch fail-closed chain with fake pi streams and injected DB connections; no network, no real framework binary." + }, + { + "path": "tests/skills/dispatch-agent-task/test_read_tools.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "dispatch-agent-task", + "kind": "runtime_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises read-only exploration tool registry contract with injected fake DB connections plus CLI subprocess list/rejection; no network, no real DB, no real framework binary." + }, + { + "path": "tests/skills/record-run-evidence/test_agent_trace.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "runtime_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Fixes agent event ledger writer shapes and atomic framework evidence persistence with fake connections; no DB, no network." } -} + ], + "summary": { + "entry_count": 110, + "by_scope": { + "other": 1, + "runtime_skill": 98, + "harness": 3, + "domain": 8 + }, + "by_kind": { + "tool_contract": 35, + "skill_behavior_eval": 1, + "harness_self_test": 3, + "domain_eval": 4, + "integration": 9, + "tool_unit": 32, + "fake_pipeline": 22, + "runtime_probe": 1, + "runtime_contract": 3 + }, + "by_evidence_level": { + "deterministic_offline": 94, + "real_dependency_integration": 10, + "static_structure": 6 + }, + "total": 110 + } +} \ No newline at end of file diff --git a/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py b/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py index 907dca2..338ba91 100644 --- a/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py +++ b/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py @@ -35,7 +35,7 @@ from pi_runner import ( # noqa: E402 PiAgentRunner, build_pi_argv, ) -from dispatch_agent_task import EXIT_OUTPUT_INVALID, run_dispatch # noqa: E402 +from dispatch_agent_task import EXIT_OUTPUT_INVALID, EXIT_SPEC_INVALID, run_dispatch # noqa: E402 from agent_trace import AgentTraceWriter # noqa: E402 REPO_ROOT = PROJECT_ROOT @@ -548,5 +548,129 @@ class ValidateOutputTest(unittest.TestCase): validate_structured_output('{"title":"t"}', spec) + + +class SessionAndReadToolsTest(unittest.TestCase): + """阶段 D:会话复用、工具 server 开关与依赖清单。""" + + def setUp(self): + self.tmp = pathlib.Path(tempfile.mkdtemp()) + + def test_session_requires_dir_and_maps_to_argv(self): + with self.assertRaises(ValueError): + ExecutionPolicy(provider="p", model="m", session_id="s1") + package = build_task_package(load_spec(make_spec(self.tmp)), REPO_ROOT) + argv = build_pi_argv( + package, + ExecutionPolicy(provider="p", model="m", session_id="s1", session_dir="/tmp/sd"), + ) + self.assertIn("--session-id", argv) + self.assertEqual(argv[argv.index("--session-id") + 1], "s1") + self.assertEqual(argv[argv.index("--session-dir") + 1], "/tmp/sd") + self.assertNotIn("--no-session", argv) + + def test_extension_path_maps_to_e_flag_with_isolation(self): + package = build_task_package(load_spec(make_spec(self.tmp)), REPO_ROOT) + argv = build_pi_argv( + package, + ExecutionPolicy(provider="p", model="m", extension_path="/x/ext.ts"), + ) + self.assertIn("-e", argv) + self.assertEqual(argv[argv.index("-e") + 1], "/x/ext.ts") + # 自动发现仍被禁:只加载显式扩展。 + self.assertIn("--no-extensions", argv) + + def test_tool_args_captured_for_dependency_manifest(self): + sink_holder = {} + + class RecordingSink(AgentTraceWriter): + def __init__(self): + super().__init__( + run_id="t", framework="pi", agent_role="planner", + connect=lambda: (_ for _ in ()).throw(AssertionError("离线测试不应触库")), + ) + self.rows = [] + + def emit(self, event_type, **kwargs): + self.rows.append((event_type, kwargs)) + return len(self.rows) + + sink = RecordingSink() + sink_holder["sink"] = sink + lines = pi_stream_lines('{"title":"t","beats":["b"]}', with_tool=True) + # 给工具事件注入真实参数形状(依赖清单材料)。 + patched = [] + for raw in lines: + event = json.loads(raw) + if event.get("type") == "tool_execution_start": + event["args"] = {"work_id": 12, "target_chapter": 4} + patched.append(json.dumps(event, ensure_ascii=False).encode("utf-8") + b"\n") + package = build_task_package( + load_spec(make_spec(self.tmp, toolAllowlist=["read"])), REPO_ROOT + ) + runner = PiAgentRunner(launcher=fake_launcher(patched)) + outcome = runner.run( + package, + ExecutionPolicy(provider="p", model="m"), + sink, + timeout_seconds=30, + ) + self.assertEqual(outcome.tool_calls[0].args, {"work_id": 12, "target_chapter": 4}) + started = [kwargs for kind, kwargs in sink.rows if kind == "tool.started"] + self.assertIn("work_id", started[0]["details"]["args"]) + + def test_read_tools_disabled_fails_closed(self): + spec_path = make_spec(self.tmp, toolAllowlist=["read_fine_outline"]) + receipt, code = run_dispatch( + spec_path, + repo_root=REPO_ROOT, + policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + run_id="unittest-agent-dispatch-rt", + run_dir=self.tmp / "run", + connect_factory=RecordingConnect(), + launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), + trigger_source="diagnostic", + ) + self.assertEqual(code, EXIT_SPEC_INVALID) + self.assertEqual(receipt["errorCode"], "READ_TOOLS_NOT_ENABLED") + + def test_dependency_manifest_written_on_tool_use(self): + spec_path = make_spec(self.tmp, toolAllowlist=["read"]) + run_dir = self.tmp / "run" + receipt, code = run_dispatch( + spec_path, + repo_root=REPO_ROOT, + policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + run_id="unittest-agent-dispatch-dep", + run_dir=run_dir, + connect_factory=RecordingConnect(), + launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}', with_tool=True)), + trigger_source="diagnostic", + enable_read_tools=True, + ) + self.assertEqual(code, 0, receipt) + dependencies = json.loads((run_dir / "dependencies.json").read_text(encoding="utf-8")) + self.assertEqual(len(dependencies), 1) + self.assertEqual(dependencies[0]["tool"], "read") + self.assertEqual(dependencies[0]["seq"], 1) + self.assertEqual(receipt["dependencies"], {"count": 1, "file": "dependencies.json"}) + # 无工具调用的运行不产依赖清单文件。 + tmp2 = pathlib.Path(tempfile.mkdtemp()) + run_dir2 = tmp2 / "run" + receipt2, code2 = run_dispatch( + make_spec(tmp2), + repo_root=REPO_ROOT, + policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + run_id="unittest-agent-dispatch-nodep", + run_dir=run_dir2, + connect_factory=RecordingConnect(), + launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), + trigger_source="diagnostic", + ) + self.assertEqual(code2, 0, receipt2) + self.assertFalse((run_dir2 / "dependencies.json").exists()) + self.assertIsNone(receipt2["dependencies"]) + + if __name__ == "__main__": unittest.main(verbosity=2) diff --git a/tests/skills/dispatch-agent-task/test_read_tools.py b/tests/skills/dispatch-agent-task/test_read_tools.py new file mode 100644 index 0000000..2e15075 --- /dev/null +++ b/tests/skills/dispatch-agent-task/test_read_tools.py @@ -0,0 +1,201 @@ +#!/usr/bin/env python3 +"""探索工具 server(read_tools)的离线测试:不连真实库。 + +固定工具合同:五个只读工具按登记表执行、未知工具与非法参数拒绝、 +结果有界截断、登记表 CLI 输出结构稳定;并校验派发器的工具开关门禁 +(探索工具未启用工具 server 时失败关闭)与依赖清单产出。 +""" +from __future__ import annotations + +import json +import pathlib +import subprocess +import sys +import tempfile +import unittest + +PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3] +SKILL_DIR = PROJECT_ROOT / ".agent" / "skills" / "dispatch-agent-task" / "scripts" +if str(SKILL_DIR) not in sys.path: + sys.path.insert(0, str(SKILL_DIR)) + +import read_tools # noqa: E402 +from read_tools import TOOL_REGISTRY, execute_tool # noqa: E402 + +READ_TOOLS_PY = SKILL_DIR / "read_tools.py" +PYTHON = sys.executable + + +class FakeCursor: + def __init__(self, rows): + self._rows = rows + + def fetchone(self): + return self._rows[0] if self._rows else None + + def fetchall(self): + return self._rows + + +class FakeConn: + """记录每条 SQL 与参数、按构造行返回的假连接。""" + + def __init__(self, rows): + self.rows = rows + self.queries = [] + + def execute(self, sql, params=None): + self.queries.append((sql, params)) + return FakeCursor(self.rows) + + def close(self): + pass + + +class ToolBehaviourTest(unittest.TestCase): + def test_read_fine_outline_hit_and_miss(self): + conn = FakeConn([({"chapterGoal": "g"}, 3)]) + result = execute_tool( + "read_fine_outline", {"work_id": 12, "target_chapter": 4}, connect=lambda: conn + ) + self.assertTrue(result["found"]) + self.assertEqual(result["version"], 3) + self.assertEqual(result["payload"], {"chapterGoal": "g"}) + sql, params = conn.queries[0] + self.assertIn("state='confirmed'", sql) + self.assertEqual(params, (12, 4)) + + conn = FakeConn([]) + result = execute_tool( + "read_fine_outline", {"work_id": 12, "target_chapter": 9}, connect=lambda: conn + ) + self.assertFalse(result["found"]) + + def test_read_style_constraints_dict_projection(self): + conn = FakeConn([({"句式": "短句", "用词质感": "冷"},)]) + result = execute_tool("read_style_constraints", {"work_id": 12}, connect=lambda: conn) + self.assertEqual(result["constraints"], ["句式:短句", "用词质感:冷"]) + + def test_read_style_falls_back_to_setting_row(self): + class TwoStepConn: + def __init__(self): + self._step = 0 + + def execute(self, sql, params=None): + self._step += 1 + rows = [] if self._step == 1 else [({"style": "一句话文风"},)] + return FakeCursor(rows) + + def close(self): + pass + + result = execute_tool("read_style_constraints", {"work_id": 12}, connect=TwoStepConn) + self.assertEqual(result["constraints"], ["一句话文风"]) + + def test_read_pattern_bindings(self): + conn = FakeConn([({"patternReferences": [{"id": 1}, {"x": 2}]},)]) + result = execute_tool("read_pattern_bindings", {"work_id": 12}, connect=lambda: conn) + self.assertEqual(result["bindings"], [{"id": 1}, {"x": 2}]) + conn = FakeConn([]) + result = execute_tool("read_pattern_bindings", {"work_id": 12}, connect=lambda: conn) + self.assertEqual(result["bindings"], []) + + def test_read_chapter_text_truncates(self): + class ChapterConn: + def __init__(self): + self._step = 0 + + def execute(self, sql, params=None): + self._step += 1 + rows = [(1, 3, "第三章")] if self._step == 1 else [("字" * 40000,)] + return FakeCursor(rows) + + def close(self): + pass + + result = execute_tool( + "read_chapter_text", {"work_id": 12, "chapter_order": 3}, connect=ChapterConn + ) + self.assertTrue(result["found"]) + self.assertTrue(result["truncated"]) + self.assertEqual(len(result["text"]), 30000) + self.assertEqual(result["totalChars"], 40000) + + def test_search_entities_like_and_bounds(self): + conn = FakeConn([("character", "林深", "主角", "active")]) + result = execute_tool( + "search_entities", {"work_id": 12, "keyword": "林"}, connect=lambda: conn + ) + self.assertEqual(result["entities"][0]["name"], "林深") + sql, params = conn.queries[0] + self.assertIn("ILIKE", sql) + self.assertEqual(params, (12, "%林%", 50)) + + def test_unknown_tool_and_bad_args_rejected(self): + with self.assertRaises(KeyError): + execute_tool("drop_table", {}, connect=lambda: FakeConn([])) + with self.assertRaises(ValueError): + execute_tool( + "search_entities", {"work_id": 12, "keyword": ""}, connect=lambda: FakeConn([]) + ) + with self.assertRaises(ValueError): + execute_tool( + "read_fine_outline", {"work_id": "x", "target_chapter": 1}, + connect=lambda: FakeConn([]), + ) + + +class RegistryCliTest(unittest.TestCase): + def test_registry_shape_and_cli_list(self): + self.assertEqual( + set(TOOL_REGISTRY), + { + "read_fine_outline", + "read_style_constraints", + "read_pattern_bindings", + "read_chapter_text", + "search_entities", + }, + ) + for entry in TOOL_REGISTRY.values(): + self.assertTrue(entry["description"]) + self.assertTrue(entry["tables"]) + self.assertTrue(entry["args"]) + proc = subprocess.run( + [PYTHON, str(READ_TOOLS_PY), "--list"], + capture_output=True, + text=True, + timeout=30, + ) + self.assertEqual(proc.returncode, 0, proc.stderr) + listing = json.loads(proc.stdout) + self.assertEqual(set(listing), set(TOOL_REGISTRY)) + for name, item in listing.items(): + self.assertEqual(item["args"], TOOL_REGISTRY[name]["args"]) + + def test_cli_rejects_unknown_tool_with_contract_exit_code(self): + proc = subprocess.run( + [PYTHON, str(READ_TOOLS_PY), "drop_table", "{}"], + capture_output=True, + text=True, + timeout=30, + ) + self.assertEqual(proc.returncode, 2) + + +class ExtensionSyncTest(unittest.TestCase): + """扩展不自带工具清单:它必须从登记表 --list 动态装载,防止双源漂移。""" + + def test_extension_loads_registry_dynamically(self): + extension = (SKILL_DIR / "muse_read_tools_extension.ts").read_text(encoding="utf-8") + self.assertIn("--list", extension) + self.assertIn("registerTool", extension) + self.assertIn("MUSE_READ_TOOLS_PYTHON", extension) + self.assertIn("MUSE_READ_TOOLS_SCRIPT", extension) + # 扩展不得硬编码登记表之外的工具名(只能出现动态注册语句)。 + for name in ("read_fine_outline", "search_entities"): + self.assertNotIn(f'name: "{name}"', extension) + + +if __name__ == "__main__": + unittest.main()