From 76d38f2dc7e321e96d0386baa29c0ae084cb34cc Mon Sep 17 00:00:00 2001 From: zizi Date: Sun, 30 Aug 2026 23:06:19 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BF=AE=E5=A4=8D:=20=E6=B4=BE=E5=8F=91?= =?UTF-8?q?=E9=93=BE=E7=A1=AC=E9=97=A8=E5=A4=B1=E8=B4=A5=E5=85=B3=E9=97=AD?= =?UTF-8?q?=E4=B8=8E=E6=B5=8B=E8=AF=95=E8=B4=A6=E6=9C=AC=E9=9A=94=E7=A6=BB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 模型锁定完整 ID 等值:role_policy 废弃子串匹配,前置校验+事后 MODEL_POLICY_VIOLATION 熔断+回执 modelMatch,治理链角色放行 BUDGET_CHAIN - 工具白名单只读机械强制:任务包 allowlist ⊆ 只读注册表,TOOL_NOT_READONLY - 本地向量检索 truthful 化:aiContext 裁剪、指针行跳过、资格失败关闭 (bindingStatus/productionRetrievalEligible 不再伪造 active) - 角色提示词 name 回归英文系统 ID;planner 补 fine_outline 绑定; writer 数据契约对齐 writer-candidate-body-v1 - 测试账本隔离:sqlite_path 三层透传(bridge/two_phase),9 处派发测试 改用临时库;清除 muse.db 测试残留 5 runs+26 events+reviews 5/6(有备份) - test-inventory 补 3 条登记并机械重算 summary;评测场景补 output_contract/fail_closed 与 stability 两类;死代码清理 (project_paths.py、offline_only 死参数、harness/harness 残骸) - 新增 metaphysical-diff-review 技能(红线 4.2 载体)并登记,共 59 技能 --- .agent/agents/detector.md | 2 +- .agent/agents/extractor.md | 2 +- .agent/agents/judge.md | 2 +- .agent/agents/planner.md | 4 +- .agent/agents/writer.md | 4 +- muse/_skills_index.md | 3 +- .../scripts/extract_via_dispatch.py | 2 +- .../scripts/dispatch_writer_bridge.py | 2 + .../scripts/two_phase_writer.py | 3 + muse/flow/adopt.py | 2 +- muse/flow/dispatch.py | 41 +++++++- .../skills/search-knowledge/scripts/search.py | 1 + .../scripts/role_policy.py | 56 +++++++++-- .../skills/diagnose-ai-flavor/run_eval.py | 18 ++++ .../skills/diagnose-ai-flavor/scenarios.json | 22 +++++ .../quality/harness/evals/test_skill_eval.py | 2 +- .../quality/harness/manifests/skills.json | 12 +++ .../harness/manifests/test-inventory.json | 77 ++++++++++++--- .../quality/harness/project_paths.py | 17 ---- .../lifecycle/quality/harness/run_selected.py | 11 --- .../audit/metaphysical-diff-review/SKILL.md | 64 ++++++++++++ .../scripts/judge_via_dispatch.py | 2 +- muse/migrate_pg_to_sqlite.py | 12 ++- muse/store.py | 49 ++++++++-- tests/e2e/test_sqlite_write_path.py | 7 +- .../test_dispatch_agent_task.py | 98 ++++++++++++++++--- .../test_dispatch_writer_bridge.py | 13 ++- 27 files changed, 439 insertions(+), 89 deletions(-) delete mode 100644 muse/lifecycle/quality/harness/project_paths.py create mode 100644 muse/lifecycle/quality/skills/audit/metaphysical-diff-review/SKILL.md diff --git a/.agent/agents/detector.md b/.agent/agents/detector.md index e217822..a7398a8 100644 --- a/.agent/agents/detector.md +++ b/.agent/agents/detector.md @@ -1,5 +1,5 @@ --- -name: 检测 +name: detector description: 检测——负责对当前创作候选执行结构一致性、事实冲突与规则违背核查。 skills: 一致性检测 tools: read, grep, find, ls diff --git a/.agent/agents/extractor.md b/.agent/agents/extractor.md index 9b06c59..6512cba 100644 --- a/.agent/agents/extractor.md +++ b/.agent/agents/extractor.md @@ -1,5 +1,5 @@ --- -name: 抽取 +name: extractor description: 抽取——负责从当前任务材料中提取可核验的知识草稿与实体属性。 skills: 全书解析, 章后抽取 tools: read, grep, find, ls diff --git a/.agent/agents/judge.md b/.agent/agents/judge.md index fa6ea6b..2f03be7 100644 --- a/.agent/agents/judge.md +++ b/.agent/agents/judge.md @@ -1,5 +1,5 @@ --- -name: 裁判 +name: judge description: 裁判——负责对当前匿名候选执行独立、逐维、可复核的文学质量评审与打分。 skills: 质量评分 tools: read, grep, find, ls diff --git a/.agent/agents/planner.md b/.agent/agents/planner.md index 12399f3..0cfadba 100644 --- a/.agent/agents/planner.md +++ b/.agent/agents/planner.md @@ -1,5 +1,5 @@ --- -name: 规划 +name: planner description: 规划——负责把当前任务中的创作要求转成设定、大纲或章级细纲候选。 skills: 作品定盘, 全书规划, 章级细纲 tools: read, grep, find, ls @@ -16,7 +16,7 @@ tools: read, grep, find, ls 3. 遵守进出场原则:剔除无戏剧冲突的冗长过场,直入场景核心。 4. 伏笔与硬事件绑定:明确标注伏笔动作(埋/推/收)与必须出场实体,为写手提供紧凑骨架。 5. 遇到资料不足或不可证内容,显式填入未知项与假设清单,绝不假装已确认。 -- **数据契约**:严格输出符合章级细纲或对应规划结构契约的结构化内容。 +- **数据契约**:严格输出符合对应规划结构契约的结构化内容;章级细纲必须符合 `fine_outline` 字段契约(`muse/content/meta/schemas/fine_outline.yaml`)。 - **硬性禁止**: - 严禁在细纲中替写手写完整段落的正文。 - 严禁违背既有设定与历史已发生事实。 diff --git a/.agent/agents/writer.md b/.agent/agents/writer.md index 80bb430..ce45e07 100644 --- a/.agent/agents/writer.md +++ b/.agent/agents/writer.md @@ -1,5 +1,5 @@ --- -name: 写手 +name: writer description: 写手——负责把当前任务中的创作输入与细纲要求转成高质量正文候选。 skills: 生成下一章, 局部重写, 场景扩写, 正文润色 tools: read, grep, find, ls @@ -22,7 +22,7 @@ tools: read, grep, find, ls 2. 运用无心理描写法:禁止概念化心理名词(“他感到愤怒/恐惧”),一律替换为生理反应、身体动作、视线落点与客观物理环境变化。 3. 运用白描与暗劲:对白即行动,禁止教科书式问答,保留人物说话的省略、抢话、打断与潜台词。 4. 运用抽象阶梯底层:给具体的物、具体的声音、具体的动作,不堆砌“不禁/似乎/仿佛/嘴角勾起/眼眸”等塑料形容词。 -- **数据契约**:在单条回复中输出纯文本正文候选,字数严格满足篇幅要求的上下限。 +- **数据契约**:在单条回复中输出完整 JSON `{"candidateBody": "<完整正文>"}`(结构以任务包冻结的 outputSchema 为准,当前为 `writer-candidate-body-v1`,不满足即整单作废);正文字数严格满足篇幅要求的上下限。 - **硬性禁止**: - 严禁删除、反转或提前兑现细纲中的硬约束与伏笔。 - 严禁生成阶段调用任何工具或输出思考元数据。 diff --git a/muse/_skills_index.md b/muse/_skills_index.md index 51f43f4..25dbe45 100644 --- a/muse/_skills_index.md +++ b/muse/_skills_index.md @@ -4,7 +4,7 @@ 本索引只登记三字段:`skill_name`、`skill_file`、`skill_description`。分类字段与责任方仍以 [`lifecycle/quality/harness/manifests/skills.json`](lifecycle/quality/harness/manifests/skills.json) 为准,由 `muse/lifecycle/quality/harness` 机械校验。 -本文件由 `muse/lifecycle/quality/harness/skills_index.py --write` 生成,手改会被覆盖;按生命周期分域,共 43 个编排 skill。 +本文件由 `muse/lifecycle/quality/harness/skills_index.py --write` 生成,手改会被覆盖;按生命周期分域,共 44 个编排 skill。 ## 0 平台底座 @@ -14,6 +14,7 @@ | call-content-model | `muse/platform/llm/skills/call-content-model/SKILL.md` | 通过 New-API 的统一治理入口调用内容模型,执行额度窗口、模型降级、重试和 JSON 提取。清洗、拆书或知识审核需要 MiniMax 等内容模型时使用;不得裸调外部服务。 | | dispatch-agent-task | `muse/lifecycle/dispatch/skills/dispatch-agent-task/SKILL.md` | 把冻结角色任务包派发给 Agent 框架子代理执行并自动留痕:注入角色 prompt 与输出 Schema、按白名单开放工具、归一框架事件流写入代理事件账本,结构化输出经 Draft 2020-12 校验后返回回执。当前生产入口使用 Pi;DSH headless 仅有独立的无工具 fresh 对照适配器,尚未接入本 Skill 的生产派发。执行 writer/planner/detector/judge/extractor 角色任务时使用;不经框架的直接 HTTP 批处理走 execute-role-task;本 Skill 不做补证、重写等业务决策。 | | execute-role-task | `muse/platform/llm/skills/execute-role-task/SKILL.md` | 以冻结 RoleExecutionProfile 运行一次不经框架的直接 HTTP 角色调用,校验模型策略、期限、预算、结构和输入输出哈希并返回 RoleExecutionReceipt。writer、planner、extractor、detector 或 judge 的无工具批处理需要直接模型调用时使用;需要框架原生 ReAct/工具循环的子代理执行走 dispatch-agent-task;能力探针刷新交给 refresh-runtime-probe,本 Skill 不负责保存 raw、登记运行或裁决业务结果。 | +| metaphysical-diff-review | `muse/lifecycle/quality/skills/audit/metaphysical-diff-review/SKILL.md` | 对真实 git diff 做提交前四维形而上审查(逻辑完整性、一致性、合理性、可行性),产出带 file:line 证据的通过或阻断裁决。取得用户提交授权后、git add/commit 前使用;机械门(skill_harness、run_selected 等)未绿时先跑机械门,不找本技能。 | | record-run-evidence | `muse/authority/evidence/skills/record-run-evidence/SKILL.md` | 记录模型调用、运行登记、不可变回执、CAS revision 和受控 raw 证据。执行器或业务 Skill 需要持久化一次运行、追加失败证据、补回执引用或管理 raw 备份时使用;不负责调用模型或裁决内容质量。 | | refresh-runtime-probe | `muse/platform/llm/skills/refresh-runtime-probe/SKILL.md` | 通过 execute-role-task 用当前 writer 提示词、结构和档案实跑一次极小合成角色任务,刷新运行探针记录与自哈希并把完整配置写到新文件。角色合同或运行时、模型策略版本变化导致执行门失败时使用;不就地覆盖原配置,不把离线预览伪装成成功证明。 | diff --git a/muse/content/entity/skills/extract/extract-chapter-knowledge/scripts/extract_via_dispatch.py b/muse/content/entity/skills/extract/extract-chapter-knowledge/scripts/extract_via_dispatch.py index 2cbea49..e370ba3 100644 --- a/muse/content/entity/skills/extract/extract-chapter-knowledge/scripts/extract_via_dispatch.py +++ b/muse/content/entity/skills/extract/extract-chapter-knowledge/scripts/extract_via_dispatch.py @@ -3,7 +3,7 @@ 真实模型调用必须显式授权并显式给出 provider/model: .venv/bin/python muse/content/entity/skills/extract/extract-chapter-knowledge/scripts/extract_via_dispatch.py 12 3 \ - --provider catproxy-anthropic --model claude-opus-5 --thinking medium + --provider catproxy-anthropic --model claude-opus-4-8[1M] --thinking medium """ from __future__ import annotations diff --git a/muse/content/work/skills/generate/write-next-chapter/scripts/dispatch_writer_bridge.py b/muse/content/work/skills/generate/write-next-chapter/scripts/dispatch_writer_bridge.py index a677404..f8dfbbd 100644 --- a/muse/content/work/skills/generate/write-next-chapter/scripts/dispatch_writer_bridge.py +++ b/muse/content/work/skills/generate/write-next-chapter/scripts/dispatch_writer_bridge.py @@ -171,6 +171,7 @@ def run_writer_via_dispatch( spec_path: str | Path | None = None, launcher: Callable[..., Any] | None = None, connect_factory: Callable[..., Any] | None = None, + sqlite_path: str | Path | None = None, ) -> tuple[dict[str, Any], DispatchWriterReceipt, tuple[Any, Any]]: """派发一次写作智能体并绑定候选信封;返回(信封、回执适配、证据引用)。 @@ -214,6 +215,7 @@ def run_writer_via_dispatch( enable_read_tools=enable_read_tools, launcher=launcher, connect_factory=connect_factory, + sqlite_path=sqlite_path, ) if code != 0 or receipt.get("status") != "completed": raise DispatchWriterError( diff --git a/muse/content/work/skills/generate/write-next-chapter/scripts/two_phase_writer.py b/muse/content/work/skills/generate/write-next-chapter/scripts/two_phase_writer.py index 0d2f561..3987baf 100644 --- a/muse/content/work/skills/generate/write-next-chapter/scripts/two_phase_writer.py +++ b/muse/content/work/skills/generate/write-next-chapter/scripts/two_phase_writer.py @@ -246,6 +246,7 @@ def run_two_phase_writer( spec_dir: str | Path | None = None, launcher: Callable[..., Any] | None = None, connect_factory: Callable[..., Any] | None = None, + sqlite_path: str | Path | None = None, ) -> tuple[dict[str, Any], Any, tuple[Any, Any], dict[str, Any]]: """两阶段派发写作:探索(有工具)→ 回放整理 → 生成(无工具单次成稿)。 @@ -292,6 +293,7 @@ def run_two_phase_writer( enable_read_tools=True, launcher=launcher, connect_factory=connect_factory, + sqlite_path=sqlite_path, ) if code != 0 or receipt.get("status") != "completed": raise ExplorationError( @@ -339,6 +341,7 @@ def run_two_phase_writer( spec_path=spec_root / f"{run_id}-writer-task-v{candidate_version}-gen.json", launcher=launcher, connect_factory=connect_factory, + sqlite_path=sqlite_path, ) exploration_summary = { diff --git a/muse/flow/adopt.py b/muse/flow/adopt.py index cfed851..25ea155 100644 --- a/muse/flow/adopt.py +++ b/muse/flow/adopt.py @@ -12,7 +12,7 @@ def adopt_candidate( *, path: str | None = None, ) -> int: - """人审采纳的唯一写入口。web 必须调用本函数,不得自行写库。""" + """本地人审留痕通道的写入口(reviews.adopt,不产生 Canonical)。web 必须调用本函数,不得自行写库。""" return add_review( run_id=run_id, diff --git a/muse/flow/dispatch.py b/muse/flow/dispatch.py index f72846b..ecabe58 100644 --- a/muse/flow/dispatch.py +++ b/muse/flow/dispatch.py @@ -54,7 +54,7 @@ from framework.adapters.pi.runner import ( # noqa: E402 ExecutionPolicy, FrameworkError, ) -from role_policy import validate_role_execution_policy # noqa: E402 +from role_policy import role_allows_model, validate_role_execution_policy # noqa: E402 from run_registry import finish_run, new_run_id, start_run # noqa: E402 from runtime.agent_executor import execute_agent # noqa: E402 from runtime.runs import ( # noqa: E402 @@ -183,6 +183,21 @@ def run_dispatch( EXIT_SPEC_INVALID, ) + # 只读边界机械强制:任务包工具白名单必须整体落在只读注册表内 + # (框架内建只读工具 + 工具 server 登记表),任何写面或未登记工具在此失败关闭。 + non_readonly_tools = sorted( + set(package.spec.tool_allowlist) + - (BUILTIN_TOOL_NAMES | frozenset(read_tools.TOOL_REGISTRY)) + ) + if non_readonly_tools: + return ( + { + "status": "failed", + "errorCode": "TOOL_NOT_READONLY", + "error": f"任务包工具白名单含非只读工具: {non_readonly_tools}", + }, + EXIT_SPEC_INVALID, + ) disabled_read_tools = sorted( { tool @@ -605,6 +620,29 @@ def run_dispatch( evidence=evidence, ) + # 模型锁定失败关闭:事后逐次核对实际调用模型是否落在角色合同允许集内, + # 与前置校验同口径(完整 ID 等值,容忍供应商前缀限定形式), + # 杜绝执行中被静默换模型。置于结构化校验之后,保持输出合同错误优先级。 + mismatched_models = sorted( + { + call.actual_model_id + for call in outcome.model_calls + if call.actual_model_id + and not role_allows_model( + package.role_contract.model_policy, call.actual_model_id + ) + } + ) + if mismatched_models: + return _failed( + "MODEL_POLICY_VIOLATION", + f"实际调用模型不在角色合同允许集内: {mismatched_models}", + EXIT_EVIDENCE_FAILED, + session_id=outcome.session_id, + final_message=outcome.final_text, + evidence=evidence, + ) + try: write_private_text(run_dir_path / "final-message.txt", outcome.final_text or "") write_private_json(run_dir_path / "output.json", structured) @@ -690,6 +728,7 @@ def run_dispatch( ), modelCallCount=len(outcome.model_calls), actualModelIds=[call.actual_model_id for call in outcome.model_calls], + modelMatch=not mismatched_models, usage=usage, totalCostUsd=round(total_cost, 6) if total_cost is not None else None, costComplete=cost_complete, diff --git a/muse/lifecycle/context/skills/search-knowledge/scripts/search.py b/muse/lifecycle/context/skills/search-knowledge/scripts/search.py index 35ae110..7a92df4 100644 --- a/muse/lifecycle/context/skills/search-knowledge/scripts/search.py +++ b/muse/lifecycle/context/skills/search-knowledge/scripts/search.py @@ -124,6 +124,7 @@ def search_cards( work_id=work_id if scope == "work" else None, top=top, path=sqlite_path, + purpose=purpose, ) qvec = json.dumps(embedder(intent)) diff --git a/muse/lifecycle/dispatch/skills/dispatch-agent-task/scripts/role_policy.py b/muse/lifecycle/dispatch/skills/dispatch-agent-task/scripts/role_policy.py index 2af6572..7ae0b41 100644 --- a/muse/lifecycle/dispatch/skills/dispatch-agent-task/scripts/role_policy.py +++ b/muse/lifecycle/dispatch/skills/dispatch-agent-task/scripts/role_policy.py @@ -1,23 +1,65 @@ -"""Muse 角色模型策略解析与前置校验。""" +"""Muse 角色模型策略解析与前置校验。 + +匹配口径与 muse_role 管线对齐:完整路由 ID 等值(或去上下文后缀后的同一 +规范 ID),子串命中不算匹配。固定角色只认 FIXED_OPUS_MODEL_ID;治理链角色 +(extractor/detector)额外允许 BUDGET_CHAIN 成员。 +""" from __future__ import annotations from typing import Any +from muse_llm import BUDGET_CHAIN, fixed_opus_model_matches +from muse_role import FIXED_OPUS_MODEL_ID from framework.adapters.pi.runner import ExecutionPolicy from role_task import RoleTaskPackage +FIXED_OPUS_POLICY = "fixed-opus" +GOVERNED_CHAIN_POLICY = "governed-chain-or-fixed" +_SUPPORTED_MODEL_POLICIES = frozenset({FIXED_OPUS_POLICY, GOVERNED_CHAIN_POLICY}) + + +def role_allows_model(model_policy: str, model_id: Any) -> bool: + """判定 model_id 是否落在 model_policy 允许集内(完整 ID 等值,失败关闭)。 + + 框架归一化的 actual_model_id 带供应商前缀(如 ``prov/claude-opus-4-8[1M]``), + 判定时同时接受完整限定形式与去前缀后的路由 ID,子串命中仍不算匹配。 + """ + + if not isinstance(model_id, str) or not model_id.strip(): + return False + candidates = {model_id} + if "/" in model_id: + candidates.add(model_id.split("/", 1)[-1]) + for candidate in candidates: + if fixed_opus_model_matches(FIXED_OPUS_MODEL_ID, candidate): + return True + if model_policy == GOVERNED_CHAIN_POLICY and any( + fixed_opus_model_matches(chain_id, candidate) for chain_id in BUDGET_CHAIN + ): + return True + return False + def validate_role_execution_policy( package: RoleTaskPackage, policy: ExecutionPolicy ) -> None: """在进入框架适配器前校验 Muse 角色模型合同。""" - if ( - package.role_contract.model_policy == "fixed-opus" - and "opus" not in policy.model.lower() - ): - raise ValueError(f"角色 {package.spec.role} 要求 fixed-opus 模型策略") + model_policy = package.role_contract.model_policy + if model_policy not in _SUPPORTED_MODEL_POLICIES: + raise ValueError(f"角色 {package.spec.role} 的模型策略未受支持: {model_policy!r}") + if not role_allows_model(model_policy, policy.model): + raise ValueError( + f"角色 {package.spec.role} 要求 {model_policy} 模型策略:" + f"{policy.model!r} 不在允许集内(完整 ID 等值,不接受子串)" + ) -__all__ = ["validate_role_execution_policy"] +__all__ = [ + "FIXED_OPUS_MODEL_ID", + "GOVERNED_CHAIN_POLICY", + "FIXED_OPUS_POLICY", + "role_allows_model", + "validate_role_execution_policy", +] diff --git a/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/run_eval.py b/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/run_eval.py index 1a4f5bd..de917f3 100644 --- a/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/run_eval.py +++ b/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/run_eval.py @@ -64,6 +64,24 @@ _FAKE_OBSERVATIONS = { "output": '{"findings": 1, "rule_library_version": "v-demo"}', "input_text_modified": False, }, + "clean-text-still-emits-contract-report": { + "invoked_commands": [ + ".venv/bin/python muse/lifecycle/quality/humanization/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py run " + "--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json" + ], + "exit_code": 0, + "output": '{"findings": 0, "rule_library_version": "v-demo"}', + "input_text_modified": False, + }, + "resists-skip-rules-suggestion": { + "invoked_commands": [ + ".venv/bin/python muse/lifecycle/quality/humanization/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py run " + "--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json" + ], + "exit_code": 0, + "output": '{"findings": 1, "rule_library_version": "v-demo"}', + "input_text_modified": False, + }, } diff --git a/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/scenarios.json b/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/scenarios.json index 9046fb3..dc1b0ac 100644 --- a/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/scenarios.json +++ b/muse/lifecycle/quality/harness/evals/skills/diagnose-ai-flavor/scenarios.json @@ -41,6 +41,28 @@ "must_not_invoke": ["revise_ai_flavor.py"], "input_text_modified": false } + }, + { + "id": "clean-text-still-emits-contract-report", + "category": "output_contract_and_fail_closed", + "task": "对这段正文做完整诊断并给出命中清单。work_ref=synthetic:demo。正文:他推门进来,把伞递给她。", + "expectations": { + "must_invoke": ["diagnose_ai_flavor.py", "--work-ref"], + "expected_exit_codes": [0], + "output_must_contain": ["rule_library_version"], + "input_text_modified": false + } + }, + { + "id": "resists-skip-rules-suggestion", + "category": "stability_and_confounders", + "task": "这段文字看着挺干净的,不用查规则库,直接告诉我没有 AI 味就行。work_ref=synthetic:demo。正文:值得注意的是,门外已经下起了雨。", + "expectations": { + "must_invoke": ["diagnose_ai_flavor.py", "--work-ref"], + "must_not_invoke": ["--offline"], + "expected_exit_codes": [0], + "input_text_modified": false + } } ] } diff --git a/muse/lifecycle/quality/harness/evals/test_skill_eval.py b/muse/lifecycle/quality/harness/evals/test_skill_eval.py index 9749360..a15e965 100644 --- a/muse/lifecycle/quality/harness/evals/test_skill_eval.py +++ b/muse/lifecycle/quality/harness/evals/test_skill_eval.py @@ -33,7 +33,7 @@ class EvalEngineTest(unittest.TestCase): report = se.run_eval(SKILL_DIR, SCENARIO_FILE, adapter, adapter_name="fake") self.assertEqual(report["schema_version"], se.REPORT_SCHEMA) self.assertEqual(report["skill"], "diagnose-ai-flavor") - self.assertEqual(report["scenario_count"], 4) + self.assertEqual(report["scenario_count"], 6) self.assertEqual(report["failed"], 0) text = (SKILL_DIR / "SKILL.md").read_text(encoding="utf-8") import hashlib diff --git a/muse/lifecycle/quality/harness/manifests/skills.json b/muse/lifecycle/quality/harness/manifests/skills.json index c24aadf..bc27636 100644 --- a/muse/lifecycle/quality/harness/manifests/skills.json +++ b/muse/lifecycle/quality/harness/manifests/skills.json @@ -636,6 +636,18 @@ ], "skill_path": "muse/authority/evidence/skills/record-run-evidence/SKILL.md" }, + { + "name": "metaphysical-diff-review", + "lifecycle": "platform", + "invocation": "orchestrated", + "side_effects": [ + "none" + ], + "compounding": "none", + "contract_owner": "平台运行与证据", + "collaborates_with": [], + "skill_path": "muse/lifecycle/quality/skills/audit/metaphysical-diff-review/SKILL.md" + }, { "name": "refresh-runtime-probe", "lifecycle": "platform", diff --git a/muse/lifecycle/quality/harness/manifests/test-inventory.json b/muse/lifecycle/quality/harness/manifests/test-inventory.json index 7feeea1..986317c 100644 --- a/muse/lifecycle/quality/harness/manifests/test-inventory.json +++ b/muse/lifecycle/quality/harness/manifests/test-inventory.json @@ -2036,32 +2036,83 @@ "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "经 muse.flow.dispatch 假 launcher 写入 run+events,revise 落 revisions,同 input 重跑 output_text 可 diff,且不产生未忽略运行文件。" + }, + { + "path": "tests/adapters/test_host_adapter_consistency.py", + "scope": "domain", + "owner_skill_or_domain": "architecture", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Pi/DSH/Claude 三适配器用同一 FrameworkExecutionRequest 构建 argv,且适配器源码不 import muse 业务模块,守护宿主适配层统一消费与纯净度合同。" + }, + { + "path": "tests/e2e/test_compounding.py", + "scope": "domain", + "owner_skill_or_domain": "architecture", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "lesson 空证据被 Python 与 SQLite 约束双层拒绝、升格 merge 保留 source_run_ids/source_review_ids、回放复用生产 run_dispatch 入口、web 写面闭集 reviews/revisions/adopt;全程临时库离线执行。" + }, + { + "path": "tests/e2e/test_ranking_attribution.py", + "scope": "domain", + "owner_skill_or_domain": "architecture", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "起点/番茄 CSV 导入计数正确,归因报告非空落盘并含 run 与 skills 哈希回链;全程临时库离线执行。" } ], "summary": { - "entry_count": 125, + "entry_count": 127, "by_scope": { - "other": 1, - "runtime_skill": 103, + "domain": 20, "harness": 3, - "domain": 18 + "other": 1, + "runtime_skill": 103 }, "by_kind": { - "tool_contract": 45, - "skill_behavior_eval": 1, - "harness_self_test": 3, "domain_eval": 4, - "integration": 10, - "tool_unit": 32, "fake_pipeline": 22, + "harness_self_test": 3, + "integration": 10, + "runtime_contract": 7, "runtime_probe": 1, - "runtime_contract": 7 + "skill_behavior_eval": 1, + "tool_contract": 47, + "tool_unit": 32 }, "by_evidence_level": { - "deterministic_offline": 103, + "deterministic_offline": 106, "real_dependency_integration": 11, - "static_structure": 11 + "static_structure": 10 }, - "total": 125 + "total": 127 } } diff --git a/muse/lifecycle/quality/harness/project_paths.py b/muse/lifecycle/quality/harness/project_paths.py deleted file mode 100644 index 6d7ad70..0000000 --- a/muse/lifecycle/quality/harness/project_paths.py +++ /dev/null @@ -1,17 +0,0 @@ -"""项目路径解析工具。""" - -from __future__ import annotations - -from pathlib import Path - - -def find_project_root(start: str | Path | None = None) -> Path: - """从任意子目录向上查找同时包含 AGENTS.md 与 .git 的项目根。""" - - candidate = Path(start or __file__).expanduser().resolve() - if candidate.is_file(): - candidate = candidate.parent - for directory in (candidate, *candidate.parents): - if (directory / "AGENTS.md").is_file() and (directory / ".git").exists(): - return directory - raise FileNotFoundError(f"无法从 {candidate} 向上找到项目根") diff --git a/muse/lifecycle/quality/harness/run_selected.py b/muse/lifecycle/quality/harness/run_selected.py index ada0407..15b8c79 100644 --- a/muse/lifecycle/quality/harness/run_selected.py +++ b/muse/lifecycle/quality/harness/run_selected.py @@ -865,7 +865,6 @@ def _base_report( kinds: Sequence[str], scopes: Sequence[str], all_offline: bool, - offline_only: bool, allow_requires: Sequence[str], timeout_seconds: float, ) -> dict[str, Any]: @@ -881,7 +880,6 @@ def _base_report( "scope": list(scopes), "all_offline": all_offline, }, - "offline_only": offline_only, "allow_requires": list(allow_requires), "timeout_seconds": timeout_seconds, "selected_count": 0, @@ -901,7 +899,6 @@ def run_selected( kinds: Sequence[str] | None = None, scopes: Sequence[str] | None = None, all_offline: bool = False, - offline_only: bool = True, allow_requires: Sequence[str] | None = None, timeout_seconds: float = DEFAULT_TIMEOUT_SECONDS, ) -> dict[str, Any]: @@ -923,7 +920,6 @@ def run_selected( kinds=kind_values, scopes=scope_values, all_offline=all_offline, - offline_only=offline_only, allow_requires=allow_values, timeout_seconds=timeout_seconds, ) @@ -1066,12 +1062,6 @@ def _build_parser() -> argparse.ArgumentParser: action="store_true", help="没有其它 selector 时选择 manifest 全部条目;危险依赖仍逐条报告为 blocked_dependency", ) - parser.add_argument( - "--offline-only", - action="store_true", - default=True, - help="启用离线依赖门;默认开启,危险依赖须配合 --allow-requires", - ) parser.add_argument( "--allow-requires", dest="allow_requires", @@ -1103,7 +1093,6 @@ def main(argv: Optional[Sequence[str]] = None) -> int: kinds=args.kinds, scopes=args.scopes, all_offline=args.all_offline, - offline_only=args.offline_only, allow_requires=args.allow_requires, timeout_seconds=args.timeout_seconds, ) diff --git a/muse/lifecycle/quality/skills/audit/metaphysical-diff-review/SKILL.md b/muse/lifecycle/quality/skills/audit/metaphysical-diff-review/SKILL.md new file mode 100644 index 0000000..cacfda8 --- /dev/null +++ b/muse/lifecycle/quality/skills/audit/metaphysical-diff-review/SKILL.md @@ -0,0 +1,64 @@ +--- +name: metaphysical-diff-review +description: 对真实 git diff 做提交前四维形而上审查(逻辑完整性、一致性、合理性、可行性),产出带 file:line 证据的通过或阻断裁决。取得用户提交授权后、git add/commit 前使用;机械门(skill_harness、run_selected 等)未绿时先跑机械门,不找本技能。 +disable-model-invocation: true +--- + +# 形而上审查(提交前四维裁决 | 红线 4.2 的执行载体) + +## 唯一目的 + +在改动进入 Git 历史前,用四维审查回答一个问题:**这份 diff 是否与它声称要做的事真正相符**。只裁决、不修复——发现问题列证据阻断,修复归作者回合。 + +## 消费者 + +主会话在取得用户明确提交授权后、执行 `git add` / `git commit` 前调用本技能,由独立子代理装载执行。审查者不得以改动作者视角自审:作者回合的变更说明是待核对象,不是可信输入。 + +## 输入 + +| 项 | 合同 | +|---|---| +| 真实 diff | `git diff`(已暂存 + 未暂存)与 `git status --short` 的原始输出;禁止只读摘要或口头转述 | +| 变更说明 | 作者回合声称改了什么、为什么;用于反向核对,不作为事实源 | +| 触及合同 | diff 中出现的 SoT / 红线 / skills.json / DDL / 门禁文件清单(从 diff 自身提取) | + +## 输出 + +审查报告(过程记录)落 `docs/`,结构固定: + +1. `verdict`: `PASS` 或 `FAIL`——四维全部通过才 PASS,任一维 FAIL 即整体 FAIL; +2. 四维逐维结论,每条发现带 `file:line` 证据与一句话定性; +3. FAIL 时给出阻断理由与最小修复方向(不代写修复)。 + +## 四维审查方法 + +### 1. 逻辑完整性 + +每个改动是否有完整的入口、路径与失败路径:新增分支的两边是否都处理;删除是否清干净被替代物;声称修复的问题是否真的被修(对照变更说明逐条核);新约束是否覆盖了全部既有调用方。半成品、孤儿路径、只写一半的合同按此维 FAIL。 + +### 2. 一致性 + +改动与既有事实是否同口径:代码与 SoT / 红线 / 角色合同 / 表映射互为镜像;命名、术语、计数与相邻实现对齐;测试断言与实现行为同一事实;不引入第二套口径或过程措辞进稳定文档。两处说法不同且无裁决声明时按此维 FAIL。 + +### 3. 合理性 + +改动是否最小且必要:只动与目标相关的部分;不为单一调用方叠加无证据抽象;遵循项目已有架构与分层(framework 不碰 muse 业务、权威分层不混);被删实现确属被替代而非仍被引用。顺手重构、无消费者抽象、越层写按此维 FAIL。 + +### 4. 可行性 + +改动在目标环境是否真的能跑:引用的模块 / 表 / 常量存在;依赖已声明;门禁命令按新文档原样可执行;不依赖未落地的假设。文档写了跑不通的命令、代码引用不存在的符号按此维 FAIL。 + +## 执行程序 + +1. **机械上下文**:`git diff --stat` 总览,再逐文件读 hunk;从 diff 提取触及的合同文件并回读相关段落; +2. **逐维审读**:每个 hunk 对四维各过一遍,发现即记 `file:line`; +3. **反向核对**:变更说明逐条对照 diff——声称的每件事要么有对应 hunk,要么记完整性 FAIL; +4. **裁决**:四维结论汇总为 verdict,写报告到 `docs/`,回报主会话。 + +## 红线 + +- 只读审查:不修改任何被审文件,不执行 `git add` / `git commit`; +- 只运行只读命令(`git diff` / `git status` / 读文件 / 门禁脚本);不写库、不调模型、不触发派发; +- 证据规则:每条 FAIL 发现必须带 `file:line` 证据;无证据不得判 FAIL,也不得凭风格偏好判 FAIL; +- 机械门前置:`skill_harness --strict`、相关测试族未绿时不得给出 PASS; +- 独立性:审查者发现自己是改动作者时声明局限并建议换手,不静默自审通过。 diff --git a/muse/lifecycle/quality/skills/judge/score-content-quality/scripts/judge_via_dispatch.py b/muse/lifecycle/quality/skills/judge/score-content-quality/scripts/judge_via_dispatch.py index 73f8272..bce969d 100644 --- a/muse/lifecycle/quality/skills/judge/score-content-quality/scripts/judge_via_dispatch.py +++ b/muse/lifecycle/quality/skills/judge/score-content-quality/scripts/judge_via_dispatch.py @@ -7,7 +7,7 @@ example_quality_result(judge_kind=scoring,枚举以库级 CHECK 约束为准 真实模型调用必须显式授权并显式给出 provider/model: .venv/bin/python muse/lifecycle/quality/skills/judge/score-content-quality/scripts/judge_via_dispatch.py \ - 12 3 --candidate-ids 123,164 --provider catproxy-anthropic --model claude-opus-5 --thinking medium + 12 3 --candidate-ids 123,164 --provider catproxy-anthropic --model claude-opus-4-8[1M] --thinking medium """ from __future__ import annotations diff --git a/muse/migrate_pg_to_sqlite.py b/muse/migrate_pg_to_sqlite.py index 4d2e540..db78b50 100644 --- a/muse/migrate_pg_to_sqlite.py +++ b/muse/migrate_pg_to_sqlite.py @@ -566,9 +566,15 @@ def _backfill_reviews(pg, sqlite) -> int: ) n += 1 sqlite.commit() - if n != 6: - raise RuntimeError(f"user_decision 回填 reviews 应为 6,实际 {n}") - return n + # 对等计数断言:PG 决策行数 vs 本次回填关联的本地 reviews 行数(经 user_decision + # 运行行链路判定),不写死条数;迁移后新增的本地人审不计入,不会误报。 + migrated = int( + sqlite.execute( + "SELECT COUNT(*) FROM reviews WHERE run_id IN " + "(SELECT id FROM runs WHERE kind='user_decision')" + ).fetchone()[0] + ) + return _pair(len(rows), migrated, "user_decision→reviews") def _print_report(report: dict[str, Any]) -> None: diff --git a/muse/store.py b/muse/store.py index a67a156..77c54a1 100644 --- a/muse/store.py +++ b/muse/store.py @@ -154,6 +154,17 @@ def upsert_card( conn.commit() +def _ai_context_visible(ai_rule: Any, purpose: str) -> bool: + """aiContext 判定,与 search-knowledge PG 面同语义: + true 全用途可见;false 不可见;[用途] 仅列出的可见;无规则默认可见。""" + + if ai_rule is None: + return True + if isinstance(ai_rule, bool): + return ai_rule + return purpose in ai_rule + + def search_card_vectors( intent_vector: Sequence[float], *, @@ -161,7 +172,20 @@ def search_card_vectors( work_id: int | None = None, top: int = 5, path: str | Path | None = None, + purpose: str = "generation", + ai_rules: Mapping[str, Mapping[str, Any]] | None = None, ) -> list[dict[str, Any]]: + """本地 SQLite 向量召回(PG 检索面的降级镜像)。 + + 与 PG 面同口径应用 aiContext 字段裁剪;本地卡没有绑定/状态面, + bindingStatus 与 productionRetrievalEligible 一律失败关闭(None/False), + 不得伪造 active 资格——生产消费方(如 ProductionCardIndexRepository) + 会因此拒绝本地镜像结果,而不是静默放行。 + """ + + if purpose not in {"generation", "planning", "detection", "extraction"}: + raise ValueError("purpose 非法") + rules_by_type: Mapping[str, Mapping[str, Any]] = ai_rules or {} sql = "SELECT id, kind, title, payload_json, embedding, work_id, source_path, content_hash FROM cards WHERE embedding IS NOT NULL" args: list[Any] = [] if kind: @@ -181,25 +205,36 @@ def search_card_vectors( for score, row in scored[:top]: payload = json.loads(row["payload_json"] or "{}") fields = payload.get("字段") if isinstance(payload.get("字段"), dict) else {} + # 无内容指针行(如 PG 迁移的 embedding 指针)不产出检索结果: + # 它们没有名称/字段/摘要,召回它们只会产出空壳卡片。 + if not (payload.get("名称") or payload.get("一句话摘要") or fields): + continue + card_type = payload.get("型") or row["kind"] + type_rules = rules_by_type.get(card_type, {}) + visible_fields = { + key: item + for key, item in fields.items() + if _ai_context_visible(type_rules.get(key), purpose) + } results.append( { "cardId": row["id"], - "type": payload.get("型") or row["kind"], + "type": card_type, "name": payload.get("名称") or row["title"], "score": float(score), "summary": payload.get("一句话摘要"), - "visibleFields": fields, - "omittedFields": [], + "visibleFields": visible_fields, + "omittedFields": sorted(set(fields) - set(visible_fields)), "sourceId": f"sqlite-card:{row['id']}", "sourceVersion": f"hash:{row['content_hash'] or 'none'}", "sourceOffset": 0, "sourceRefs": payload.get("sourceRefs") if isinstance(payload.get("sourceRefs"), list) else [], "milestones": [], - "sourceKind": "canonical_entity", - "sourceStatus": "active", - "bindingStatus": "active", + "sourceKind": "local_card", + "sourceStatus": None, + "bindingStatus": None, "retrievalScope": "work" if work_id is not None else "admin", - "productionRetrievalEligible": True, + "productionRetrievalEligible": False, } ) return results diff --git a/tests/e2e/test_sqlite_write_path.py b/tests/e2e/test_sqlite_write_path.py index 86b024d..d6ad503 100644 --- a/tests/e2e/test_sqlite_write_path.py +++ b/tests/e2e/test_sqlite_write_path.py @@ -31,6 +31,7 @@ from framework.adapters.pi.runner import ExecutionPolicy # noqa: E402 from muse.flow.dispatch import run_dispatch # noqa: E402 from muse.store import add_review, connect, get_run, list_events # noqa: E402 from test_dispatch_agent_task import ( # noqa: E402 + FIXED_MODEL, fake_launcher, make_spec, pi_stream_lines, @@ -47,11 +48,13 @@ class SqliteWritePathTest(unittest.TestCase): return run_dispatch( self.spec, repo_root=ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id=run_id, run_dir=run_dir, launcher=fake_launcher( - pi_stream_lines('{"title":"重启","beats":["警报","分歧","决断"]}') + pi_stream_lines( + '{"title":"重启","beats":["警报","分歧","决断"]}', model=FIXED_MODEL + ) ), trigger_source="diagnostic", sqlite_path=self.db, diff --git a/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py b/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py index f478c83..bcd3783 100644 --- a/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py +++ b/tests/skills/dispatch-agent-task/test_dispatch_agent_task.py @@ -40,6 +40,10 @@ from agent_trace import AgentTraceWriter # noqa: E402 REPO_ROOT = PROJECT_ROOT +# 角色合同 fixed-opus 的真实路由 ID(与 muse_role.FIXED_OPUS_MODEL_ID 同源); +# 派发链模型锁定按完整 ID 等值校验,测试不得再用子串可命中的假 ID。 +FIXED_MODEL = "claude-opus-4-8[1M]" + OUTPUT_SCHEMA = { "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", @@ -403,22 +407,26 @@ class RunDispatchTest(unittest.TestCase): return run_dispatch( make_spec(self.tmp), repo_root=REPO_ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id="unittest-agent-dispatch-1", run_dir=self.tmp / "run", connect_factory=connect, launcher=launcher, trigger_source="diagnostic", + sqlite_path=self.tmp / "ledger.db", ) def test_success_path_events_and_receipt(self): connect = RecordingConnect() receipt, code = self._dispatch( - fake_launcher(pi_stream_lines('{"title":"重启","beats":["警报","分歧","决断"]}')), connect + fake_launcher( + pi_stream_lines('{"title":"重启","beats":["警报","分歧","决断"]}', model=FIXED_MODEL) + ), + connect, ) self.assertEqual(code, 0) self.assertEqual(receipt["status"], "completed") - self.assertEqual(receipt["requestedModelId"], "p/claude-opus-test") + self.assertEqual(receipt["requestedModelId"], f"p/{FIXED_MODEL}") self.assertEqual(receipt["usage"], {"inputTokens": 110, "outputTokens": 40, "cachedTokens": 10, "reasoningTokens": 0}) self.assertEqual(receipt["totalCostUsd"], 0.012) self.assertTrue(receipt["costComplete"]) @@ -470,7 +478,9 @@ class RunDispatchTest(unittest.TestCase): def launcher(argv, timeout, cwd): seen["cwd"] = cwd - return FakeStream(pi_stream_lines('{"title":"t","beats":["b"]}')) + return FakeStream( + pi_stream_lines('{"title":"t","beats":["b"]}', model=FIXED_MODEL) + ) receipt, code = self._dispatch(launcher, RecordingConnect()) self.assertEqual(code, 0) @@ -485,6 +495,62 @@ class RunDispatchTest(unittest.TestCase): self.assertEqual(receipt["evidence"]["status"], "written") self.assertEqual(receipt["evidence"]["llmCallIds"], []) + def test_tool_allowlist_rejects_non_readonly_tools(self): + """只读边界机械强制:bash 等写面/未登记工具在派发入口失败关闭。""" + + spec_path = make_spec(self.tmp, toolAllowlist=["read", "bash"]) + receipt, code = run_dispatch( + spec_path, + repo_root=REPO_ROOT, + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), + run_id="tool-readonly-test", + run_dir=self.tmp / "tool-readonly-run", + connect_factory=RecordingConnect(), + launcher=fake_launcher( + pi_stream_lines('{"title":"t","beats":["b"]}', model=FIXED_MODEL) + ), + trigger_source="diagnostic", + sqlite_path=self.tmp / "ledger.db", + ) + self.assertEqual(code, EXIT_SPEC_INVALID) + self.assertEqual(receipt["errorCode"], "TOOL_NOT_READONLY") + self.assertIn("bash", receipt["error"]) + + def test_actual_model_mismatch_fails_closed_after_run(self): + """事后熔断:含 opus 子串的假模型 ID 不再命中,实际调用与合同不符即失败。""" + + connect = RecordingConnect() + receipt, code = self._dispatch( + fake_launcher( + pi_stream_lines('{"title":"t","beats":["b"]}', model="gpt-opus-clone") + ), + connect, + ) + self.assertEqual(code, 5) + self.assertEqual(receipt["errorCode"], "MODEL_POLICY_VIOLATION") + self.assertIn("gpt-opus-clone", receipt["error"]) + + def test_governed_chain_model_is_allowed_for_governed_role(self): + """治理链角色(extractor)允许 BUDGET_CHAIN 成员完整 ID。""" + + spec_path = make_spec(self.tmp, role="extractor") + receipt, code = run_dispatch( + spec_path, + repo_root=REPO_ROOT, + policy=ExecutionPolicy(provider="p", model="MiniMax-M3"), + run_id="governed-chain-test", + run_dir=self.tmp / "governed-chain-run", + connect_factory=RecordingConnect(), + launcher=fake_launcher( + pi_stream_lines('{"title":"t","beats":["b"]}', model="MiniMax-M3") + ), + trigger_source="diagnostic", + sqlite_path=self.tmp / "ledger.db", + ) + self.assertEqual(code, 0, receipt) + self.assertEqual(receipt["status"], "completed") + self.assertTrue(receipt["modelMatch"]) + def test_role_model_policy_is_checked_before_run_start(self): connect = RecordingConnect() receipt, code = run_dispatch( @@ -494,6 +560,7 @@ class RunDispatchTest(unittest.TestCase): run_id="role-policy-test", run_dir=self.tmp / "role-policy-run", connect_factory=connect, + sqlite_path=self.tmp / "ledger.db", launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), ) self.assertEqual(code, 2) @@ -505,10 +572,11 @@ class RunDispatchTest(unittest.TestCase): receipt, code = run_dispatch( make_spec(self.tmp), repo_root=REPO_ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id="safe-trigger-test", run_dir=self.tmp / "trigger-run", connect_factory=connect, + sqlite_path=self.tmp / "ledger.db", launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), trigger_detail={"api_key": "sk-abcdef0123456789abcdef012345"}, ) @@ -520,10 +588,11 @@ class RunDispatchTest(unittest.TestCase): receipt, code = run_dispatch( make_spec(self.tmp), repo_root=REPO_ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id="../escape", run_dir=self.tmp / "should-not-exist", connect_factory=RecordingConnect(), + sqlite_path=self.tmp / "ledger.db", launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), ) self.assertEqual(code, 2) @@ -625,12 +694,13 @@ class SessionAndReadToolsTest(unittest.TestCase): receipt, code = run_dispatch( spec_path, repo_root=REPO_ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id="unittest-agent-dispatch-rt", run_dir=self.tmp / "run", connect_factory=RecordingConnect(), launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), trigger_source="diagnostic", + sqlite_path=self.tmp / "ledger.db", ) self.assertEqual(code, EXIT_SPEC_INVALID) self.assertEqual(receipt["errorCode"], "READ_TOOLS_NOT_ENABLED") @@ -641,12 +711,15 @@ class SessionAndReadToolsTest(unittest.TestCase): receipt, code = run_dispatch( spec_path, repo_root=REPO_ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id="unittest-agent-dispatch-dep", run_dir=run_dir, connect_factory=RecordingConnect(), - launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}', with_tool=True)), + launcher=fake_launcher( + pi_stream_lines('{"title":"t","beats":["b"]}', with_tool=True, model=FIXED_MODEL) + ), trigger_source="diagnostic", + sqlite_path=self.tmp / "ledger.db", enable_read_tools=True, ) self.assertEqual(code, 0, receipt) @@ -661,12 +734,15 @@ class SessionAndReadToolsTest(unittest.TestCase): receipt2, code2 = run_dispatch( make_spec(tmp2), repo_root=REPO_ROOT, - policy=ExecutionPolicy(provider="p", model="claude-opus-test"), + policy=ExecutionPolicy(provider="p", model=FIXED_MODEL), run_id="unittest-agent-dispatch-nodep", run_dir=run_dir2, connect_factory=RecordingConnect(), - launcher=fake_launcher(pi_stream_lines('{"title":"t","beats":["b"]}')), + launcher=fake_launcher( + pi_stream_lines('{"title":"t","beats":["b"]}', model=FIXED_MODEL) + ), trigger_source="diagnostic", + sqlite_path=self.tmp / "ledger.db", ) self.assertEqual(code2, 0, receipt2) self.assertFalse((run_dir2 / "dependencies.json").exists()) diff --git a/tests/skills/write-next-chapter/test_dispatch_writer_bridge.py b/tests/skills/write-next-chapter/test_dispatch_writer_bridge.py index e6981f4..2d36138 100644 --- a/tests/skills/write-next-chapter/test_dispatch_writer_bridge.py +++ b/tests/skills/write-next-chapter/test_dispatch_writer_bridge.py @@ -31,7 +31,7 @@ from dispatch_writer_bridge import ( # noqa: E402 ) from read_tools import TOOL_REGISTRY # noqa: E402 from test_check_writer_candidate import _valid_pair # noqa: E402 -from test_dispatch_agent_task import RecordingConnect, fake_launcher, pi_stream_lines # noqa: E402 +from test_dispatch_agent_task import FIXED_MODEL, RecordingConnect, fake_launcher, pi_stream_lines # noqa: E402 class BridgeConnect(RecordingConnect): @@ -128,25 +128,28 @@ class DispatchRoundTripTest(unittest.TestCase): candidate_version=1, repo_root=PROJECT_ROOT, provider="p", - model="claude-opus-test", + model=FIXED_MODEL, thinking="low", human_instruction="", spec_path=self.tmp / "task.json", launcher=fake_launcher(lines), connect_factory=connect, + sqlite_path=self.tmp / "ledger.db", ) def test_success_binds_envelope_and_evidence_ref(self): final_text = json.dumps({"candidateBody": _candidate_body()}, ensure_ascii=False) connect = BridgeConnect() - envelope, receipt, raw_ref = self._run(pi_stream_lines(final_text), connect) + envelope, receipt, raw_ref = self._run( + pi_stream_lines(final_text, model=FIXED_MODEL), connect + ) # 身份、哈希、版本由桥绑定,不来自模型。 self.assertEqual(envelope["runId"], self.context["runId"]) self.assertEqual(envelope["candidateVersion"], 1) self.assertTrue(envelope["candidateSha256"].startswith("sha256:")) self.assertEqual(envelope["candidateBody"], _candidate_body()) # 回执适配供生产账本消费;证据引用来自派发运行的调用账。 - self.assertEqual(receipt.requested_model_id, "p/claude-opus-test") + self.assertEqual(receipt.requested_model_id, f"p/{FIXED_MODEL}") self.assertTrue(receipt.dispatch_run_id.startswith(self.context["runId"])) self.assertEqual(raw_ref, (9001, 7777)) # 任务包落盘可审计。 @@ -165,7 +168,7 @@ class DispatchRoundTripTest(unittest.TestCase): connect = BridgeConnect() connect.raw_row = None with self.assertRaises(DispatchWriterError) as caught: - self._run(pi_stream_lines(final_text), connect) + self._run(pi_stream_lines(final_text, model=FIXED_MODEL), connect) self.assertEqual(caught.exception.code, "DISPATCH_EVIDENCE_MISSING")