muse-agent-example/tests/集成/test_受控模型调用.py
zizi 367d01db21 W06 模型宿主、角色、预算与流式合同:模型宿主合同、角色策略冻结、预算预留结算与流式回合证据。
按 R2 串行阶段整理提交;包内文件为该阶段交付(含后续小增量),状态以工作包清单为准。
2026-09-10 19:25:41 +08:00

889 lines
35 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实任务和PG账本贯通合成HTTP流;不据此宣称真实模型或文学效果通过。"""
import asyncio
import hashlib
import json
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from threading import Thread
import pytest
from muse.任务运行.接口 import (
任务服务,
任务状态,
任务请求,
原文服务,
原文错误,
执行上下文,
模型协议错误,
模型执行器,
模型请求,
步骤处理器,
步骤结果,
步骤计划,
角色策略目录,
证据服务,
请求字节,
)
from muse.任务运行.预算管理 import (
任务预算计划,
角色预算,
预算不足,
预算状态冲突,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 内容用途, 用途
from muse.基础设施.受控文件 import 受控文件
from muse.基础设施.宿主.直接调用 import 直接宿主
from muse.基础设施.模型.HTTP传输 import HTTP传输
from muse.编排.接口 import 流程定义, 流程服务, 流程登记
pytestmark = pytest.mark.数据库
@pytest.fixture
def 合成HTTP():
服务, 异常 = [], []
def 启动(处理):
class 请求处理(BaseHTTPRequestHandler):
def do_POST(self):
try:
输入 = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
输出 = 处理(输入)
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Content-Length", str(len(输出)))
self.end_headers()
self.wfile.write(输出)
except Exception as exc:
异常.append(exc)
self.send_error(500)
def log_message(self, *args):
pass
server = ThreadingHTTPServer(("127.0.0.1", 0), 请求处理)
thread = Thread(target=server.serve_forever, daemon=True)
thread.start()
服务.append((server, thread))
return f"http://127.0.0.1:{server.server_port}/v1/responses"
yield 启动
for server, thread in 服务:
server.shutdown()
server.server_close()
thread.join(timeout=2)
assert not 异常
class 合成计价:
版本 = "synthetic-price-1"
def 金额(self, 结果):
return Decimal("0.125") if 结果.用量 else None
@pytest.mark.parametrize("场景", ["temporary", "insufficient", "write-failure"])
def test_临时模型原文遵守租约且普通证据不保留字节__a82110(模型环境, tmp_path, 合成HTTP, 场景):
from dataclasses import replace
库, 运行, 策略, 初始, 原文, _, 预算, 上下文 = 模型环境
请求 = replace(初始, 用户输入="仅在租约内可用的合成输入")
输入哈希 = hashlib.sha256(请求字节(请求)).hexdigest()
授权 = 原文.批准保留(
上下文.领取.任务ID,
"author",
"temporary-approve",
来源版本="temporary-source-1",
哈希=(输入哈希,),
内容用途="generation",
方式="temporary",
有效期=datetime.now(UTC) + timedelta(minutes=5),
调用ID=请求.调用ID,
调用请求哈希=输入哈希,
)
租约 = 原文.创建租约(
上下文.领取.任务ID,
授权,
"temporary-lease",
datetime.now(UTC) + timedelta(seconds=2 if 场景 == "insufficient" else 120),
最少剩余秒=0,
)
收到 = []
def 回应(数据):
收到.append(数据)
assert 数据["input"] == 请求.用户输入
if 场景 == "write-failure":
# 输入已写入;模拟输出返回前此租约目录实际失去合法写入权限。
(原文.文件.根 / 租约).chmod(0o500)
return (
"data: "
+ json.dumps(
{
"type": "response.completed",
"response": {
"id": "temporary-response",
"model": "claude-opus-4-8",
"status": "completed",
"output": [
{
"type": "message",
"content": [
{"type": "output_text", "text": '{"text":"租约内响应"}'}
],
}
],
"usage": {"input_tokens": 1, "output_tokens": 1},
},
}
)
+ "\n\n"
).encode()
凭据 = tmp_path / "temporary-provider-key"
凭据.write_text("synthetic-temporary-provider")
执行器 = 模型执行器(
运行,
预算,
原文,
证据服务(库),
策略,
直接宿主(HTTP传输(合成HTTP(回应), "受控存储", str(凭据), "responses")),
合成计价(),
)
if 场景 == "insufficient":
with pytest.raises(原文错误, match="余量|期限"):
asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权, 临时租约ID=租约))
assert not 收到
with 库.连接() as 连:
assert 连.execute("SELECT count(*) FROM muse_budget_reservation").fetchone()[0] == 0
assert 连.execute("SELECT count(*) FROM muse_runtime_evidence").fetchone()[0] == 0
elif 场景 == "write-failure":
from muse.基础设施.受控文件 import 受控文件错误
try:
with pytest.raises(受控文件错误):
asyncio.run(
执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权, 临时租约ID=租约)
)
finally:
(原文.文件.根 / 租约).chmod(0o700)
assert len(收到) == 1 and 预算.读取(请求.调用ID).实际金额 == Decimal("0.125")
with 库.连接() as 连:
assert 连.execute(
"SELECT outcome,content,metadata->>'error_code' FROM muse_runtime_evidence "
"WHERE kind='failure'"
).fetchone() == ("failed", None, "RAW_TEMP_WRITE_FAILED")
运行.失败步骤(上下文.领取, "raw_retention_failed")
assert 运行.读取任务(上下文.领取.任务ID).状态 == 任务状态.待对账
assert 原文.读取(上下文.领取.任务ID, 租约, 输入哈希) == 请求字节(请求)
else:
交付 = asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权, 临时租约ID=租约))
assert 交付.内容 == {"text": "租约内响应"}
assert 交付.证据回执["retention"] == "hash_only"
assert 交付.证据回执["raw_lease_id"] == 租约
assert 原文.读取(上下文.领取.任务ID, 租约, 输入哈希) == 请求字节(请求)
响应 = json.loads(原文.读取(上下文.领取.任务ID, 租约, 交付.证据回执["content_hash"]))
assert 响应["文本"] == '{"text":"租约内响应"}'
with 库.连接() as 连:
assert (
连.execute(
"SELECT count(*) FROM muse_runtime_evidence WHERE content IS NOT NULL"
).fetchone()[0]
== 0
)
assert 预算.读取(请求.调用ID).实际金额 == Decimal("0.125")
原文.清理(上下文.领取.任务ID, 租约, 作者="author")
assert 原文.状态(上下文.领取.任务ID, 租约)["state"] == "closed"
with pytest.raises(原文错误):
原文.读取(上下文.领取.任务ID, 租约, 交付.证据回执["content_hash"])
assert (
证据服务(库).读取回执(上下文.领取.任务ID, 交付.证据回执["evidence_id"]) == 交付.证据回执
)
@pytest.mark.parametrize("模型环境", ["configured"], indirect=True, ids=["evaluation"])
def test_任务绑定配置装配真实传输且不随新版本切换__a62007(模型环境, tmp_path, 合成HTTP):
"""配置验证器为明确离线替身;验证评测配置到真实HTTP装配,不声明生产验证通过。"""
from dataclasses import replace
from muse.任务运行.模型 import 内容哈希
from muse.任务运行.配置版本 import (
凭据引用,
提供方配置,
运行配置内容,
配置版本管理,
配置验证证据,
)
from muse.启动 import 构建
from muse.配置 import 应用配置
库, 运行, 策略, 请求, 原文, 授权, 预算, 上下文 = 模型环境
收到 = []
def 回应(值):
收到.append(值)
return (
"data: "
+ json.dumps(
{
"type": "response.completed",
"response": {
"id": "configured-response",
"model": "claude-opus-4-8",
"status": "completed",
"output": [
{
"type": "message",
"content": [{"type": "output_text", "text": '{"text":"固定配置"}'}],
}
],
"usage": {"input_tokens": 1, "output_tokens": 1},
},
}
)
+ "\n\n"
).encode()
凭据 = tmp_path / "configured-key"
凭据.write_text("synthetic-configured-credential")
原配置 = 运行配置内容(
"direct",
"1",
策略.定义["version"],
策略.资源发布身份,
"synthetic",
{"writer": {"provider": "synthetic", "model": 请求.model, "thinking": "high", "tools": []}},
(凭据引用("provider-key", "受控存储", str(凭据)),),
(提供方配置("synthetic", "responses", 合成HTTP(回应), "provider-key"),),
"synthetic-price-1",
)
class 离线配置检查:
身份 = "synthetic-config-validation"
def 验证(self, 内容, 执行用途):
return 配置验证证据(
内容哈希(内容.冻结()),
内容.角色策略版本,
内容.资源发布身份,
执行用途,
("synthetic-configuration-contract",),
"offline_contract",
)
管理 = 配置版本管理(库, 离线配置检查())
管理.保存草案("model-config", "1", 原配置)
回执 = 管理.验证版本("model-config", "1")
管理.启用("model-config", "1", 验证回执=回执, 批准引用="synthetic-approval", 预期代次=0)
管理.冻结到任务(上下文.领取.任务ID, "model-config")
# 新配置指向不可用地址;旧任务仍须消费其原版本。
管理.保存草案(
"model-config",
"2",
replace(
原配置,
提供方=(
提供方配置("synthetic", "responses", "http://127.0.0.1:1/unused", "provider-key"),
),
),
)
R2 = 管理.验证版本("model-config", "2")
管理.启用("model-config", "2", 验证回执=R2, 批准引用="synthetic-approval-2", 预期代次=1)
装配 = 构建(
应用配置(库.引用, 策略.资源发布身份, 运行用途=用途.评测),
流程=运行.处理器,
)
执行器 = 装配.要求模型执行器(上下文, 合成计价())
结果 = asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权))
assert 结果.内容 == {"text": "固定配置"}
assert len(收到) == 1 and 收到[0]["reasoning"] == {"effort": "high"}
assert 预算.读取(请求.调用ID).实际金额 == Decimal("0.125")
assert 管理.读取任务绑定(上下文.领取.任务ID).版本 == "1"
with 库.连接() as 连:
assert (
连.execute(
"SELECT metadata->>'runtime_config_hash' FROM evaluation.muse_runtime_evidence "
"WHERE kind='model_response'"
).fetchone()[0]
== 管理.读取任务绑定(上下文.领取.任务ID).内容哈希
)
with pytest.raises(模型协议错误, match="配置"):
asyncio.run(
执行器.执行(上下文, replace(请求, provider="other"), 阶段="生成", 原文授权ID=授权)
)
assert len(收到) == 1
def test_CLI保存查看配置草案保留引用且不自动启用__a62008(应用测试库, tmp_path):
import subprocess
import sys
from pathlib import Path
from muse.任务运行.配置版本 import 配置版本管理
from muse.共享.错误 import Muse错误
库 = 应用测试库[用途.评测]
应用文件 = tmp_path / "app.toml"
应用文件.write_text(
'["数据库"]\n"取值方式"="受控存储"\n"位置"='
+ json.dumps(库.引用.位置, ensure_ascii=False)
+ '\n["资源"]\n"发布身份"="synthetic"\n["运行"]\n"用途"="evaluation"\n'
)
内容 = Path(__file__).parents[2] / "配置/提供方.example.toml"
基础命令 = [sys.executable, "-I", "-m", "muse", "配置", str(应用文件)]
保存 = subprocess.run(
基础命令 + ["保存", "cli-model", "1", "--内容", str(内容)],
cwd=tmp_path,
capture_output=True,
text=True,
)
assert 保存.returncode == 0, 保存.stderr
查看 = subprocess.run(
基础命令 + ["查看", "cli-model", "1"], cwd=tmp_path, capture_output=True, text=True
)
assert 查看.returncode == 0, 查看.stderr
assert json.loads(保存.stdout) == json.loads(查看.stdout)
assert json.loads(查看.stdout)["内容"]["凭据"][0]["来源"] == "受控存储"
with 库.连接() as 连:
assert (
连.execute("SELECT count(*) FROM evaluation.muse_runtime_config_active").fetchone()[0]
== 0
)
# 不注入验证器时,读取和保存可用,验证不得自行产生成功回执。
with pytest.raises(Muse错误, match="验证器"):
配置版本管理(库).验证版本("cli-model", "1")
@pytest.fixture
def 模型环境(应用测试库, tmp_path, request):
次数 = getattr(request, "param", 1)
评测配置 = 次数 == "configured"
有工具 = 次数 == "session"
次数 = 2 if 有工具 else 1 if 评测配置 else 次数
角色 = "planner" if 有工具 else "writer"
工具 = ("schema_read",) if 有工具 else ()
库 = 应用测试库[用途.评测 if 评测配置 else 用途.生产]
策略 = 角色策略目录.从发布包()
登记 = 流程登记()
登记.登记处理器(
步骤处理器("model-probe", "1", lambda _: 步骤结果({}), "1", "1", 角色=角色, 允许工具=工具)
)
登记.登记类型("model-probe", 必需保护=())
def 重验合成输入(快照):
# 合成任务不读取业务正文,来源范围为空;固定结构版本由只读工具显式读取。
assert 快照.冻结输入["冻结上下文"]["source_scope"] == {}
assert 快照.冻结输入["资源发布身份"] == 策略.资源发布身份
登记.登记恢复检查("model-probe", "1", 重验合成输入)
运行 = 任务服务(库, 登记)
流程 = 流程服务(运行, 登记)
流程.发布(
流程定义("model-probe", "1", (步骤计划("call", "model-probe", "1", 角色=角色, 工具=工具),))
)
任务 = 流程.发起(
任务请求(
"model-probe",
"command",
"author",
库.用途,
内容用途.生成,
{},
策略.定义["version"],
策略.资源发布身份,
{
"source_scope": {},
"schema_versions": {},
"authorization": "grant",
"budget": {},
"stop_conditions": [],
},
),
"model-probe",
"1",
)
领取 = 运行.领取步骤("worker", ["model-probe"])
assert 领取 is not None
请求 = 模型请求(
"call-1",
"synthetic",
"claude-opus-4-8[1M]",
"返回合成测试JSON",
"一次合成调用",
{
"type": "object",
"required": ["text"],
"properties": {"text": {"type": "string"}},
"additionalProperties": False,
},
64,
10,
thinking="high" if 评测配置 else None,
允许实际模型=("claude-opus-4-8",),
)
原文 = 原文服务(库, 受控文件(tmp_path / "raw"))
哈希 = hashlib.sha256(请求字节(请求)).hexdigest()
授权 = 原文.批准保留(
任务,
"author",
"approve",
来源版本="synthetic-1",
哈希=(哈希,),
内容用途="generation",
方式="persistent",
有效期=datetime.now(UTC) + timedelta(minutes=5),
调用ID=请求.调用ID,
调用请求哈希=哈希,
)
预算 = 预算管理(库, "synthetic")
预算.登记策略(额度策略("synthetic", "1", Decimal("2"), 10))
预算.登记任务预算(
任务,
任务预算计划(
Decimal(str(次数)),
(
角色预算(
角色,
次数,
次数,
Decimal("1"),
),
),
"approval",
datetime.now(UTC) + timedelta(minutes=5),
),
)
return 库, 运行, 策略, 请求, 原文, 授权, 预算, 执行上下文(领取, 运行.读取任务(任务))
@pytest.mark.parametrize("模型环境", [2], indirect=True, ids=["two-calls"])
def test_同尝试两个相同回复各自保留调用证据__a62005(模型环境, tmp_path, 合成HTTP):
from dataclasses import replace
库, 运行, 策略, 请求, 原文, 授权, 预算, 上下文 = 模型环境
已调用 = []
def 回应(value):
已调用.append(value)
数据 = {
"type": "response.completed",
"response": {
"id": f"reply-{len(已调用)}",
"model": "claude-opus-4-8",
"status": "completed",
"output": [
{
"type": "message",
"content": [{"type": "output_text", "text": '{"text":"同一个结果"}'}],
}
],
"usage": {"input_tokens": 1, "output_tokens": 1},
},
}
return ("data: " + json.dumps(数据) + "\n\n").encode()
凭据 = tmp_path / "same-output-key"
凭据.write_text("synthetic-credential")
执行器 = 模型执行器(
运行,
预算,
原文,
证据服务(库),
策略,
直接宿主(HTTP传输(合成HTTP(回应), "受控存储", str(凭据), "responses")),
合成计价(),
)
首次 = asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权))
第二请求 = replace(请求, 调用ID="call-2")
请求哈希 = hashlib.sha256(请求字节(第二请求)).hexdigest()
第二授权 = 原文.批准保留(
上下文.领取.任务ID,
"author",
"approve-second",
来源版本="synthetic-1",
哈希=(请求哈希,),
内容用途="generation",
方式="persistent",
有效期=datetime.now(UTC) + timedelta(minutes=5),
调用ID=第二请求.调用ID,
调用请求哈希=请求哈希,
)
第二次 = asyncio.run(执行器.执行(上下文, 第二请求, 阶段="生成", 原文授权ID=第二授权))
assert 首次.内容 == 第二次.内容 and len(已调用) == 2
assert 首次.证据回执["evidence_id"] != 第二次.证据回执["evidence_id"]
assert 预算.读取("call-1").实际金额 == 预算.读取("call-2").实际金额 == Decimal("0.125")
@pytest.mark.parametrize("模型环境", ["session"], indirect=True, ids=["session"])
@pytest.mark.parametrize(
"宿主类型",
[
"direct",
pytest.param("pi", marks=pytest.mark.宿主),
pytest.param("resume", marks=pytest.mark.宿主),
pytest.param("reconcile", marks=pytest.mark.宿主),
],
)
def test_有限角色会话贯通真实工具与逐回合账本__a62006(
模型环境, 应用测试库, tmp_path, 合成HTTP, 宿主类型
):
import os
from dataclasses import replace
from pathlib import Path
from muse.任务运行.接口 import 工具定义, 工具结果
from muse.任务运行.角色会话 import 组装角色请求, 角色会话
from muse.元数据.接口 import 元数据服务, 导入内置结构
from muse.基础设施.宿主.直接调用 import 直接角色循环
库, 运行, 策略, 初始, 原文, _, 预算, 上下文 = 模型环境
with 应用测试库[用途.维护].连接() as 连, 连.transaction():
导入内置结构(元数据服务(连))
读取次数 = []
def 读取结构(范围, 参数):
读取次数.append(范围.任务ID)
with 库.连接() as 连:
结构 = 元数据服务(连).读取结构("work_core", 1)
return 工具结果({"schema_id": 结构.schema_id, "schema_hash": 结构.内容哈希}, ())
定义 = 工具定义(
"schema_read",
"读取已发布作品结构",
{"type": "object", "properties": {}, "additionalProperties": False},
读取结构,
)
请求 = 组装角色请求(
策略,
"planner",
replace(
初始,
工具=(
{
"name": 定义.名称,
"description": 定义.说明,
"parameters": 定义.参数合同,
},
),
),
)
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 应用配置, 服务配置
口令 = tmp_path / "session-author-key"
口令.write_text("synthetic-session-access")
配置 = 应用配置(
库.引用,
"test",
HTTP=服务配置(
str(口令),
作者ID="author",
公开地址="http://testserver",
允许来源=("http://testserver",),
),
原文暂存=str(tmp_path / "session-http-raw"),
)
路径 = f"/api/v1/tasks/{上下文.领取.任务ID}/role-session-authorizations"
批准请求 = {
"command_id": "approve-session",
"step_id": "call",
"stage": "执行",
"initial_request_hash": hashlib.sha256(请求字节(请求)).hexdigest(),
"max_model_calls": 2,
"max_tool_calls": 1,
"valid_until": (datetime.now(UTC) + timedelta(minutes=5)).isoformat(),
}
with TestClient(创建应用(配置)) as client:
client.headers["Origin"] = "http://testserver"
assert client.post(路径, json=批准请求).status_code == 401
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-session-access"}
).status_code
== 200
)
assert client.post(路径, json={**批准请求, "approved_by": "other"}).status_code == 422
批准结果 = client.post(路径, json=批准请求)
assert 批准结果.status_code == 200, 批准结果.text
会话授权 = 批准结果.json()["authorization_id"]
收到 = []
def 回应(数据):
收到.append(数据)
assert 数据["instructions"] == 请求.系统提示
if len(收到) == 1:
输出 = [
{
"type": "function_call",
"id": "fc-1",
"call_id": "schema-call",
"name": "schema_read",
"arguments": "{}",
}
]
else:
assert 数据["input"][1]["call_id"] == "schema-call"
工具数据 = json.loads(数据["input"][2]["output"])
assert 工具数据["内容"]["schema_id"] == "work_core"
输出 = [
{
"type": "message",
"content": [
{"type": "output_text", "text": '{"text":"结构已读取,规划待作者确认"}'}
],
}
]
数据 = {
"type": "response.completed",
"response": {
"id": f"session-response-{len(收到)}",
"model": "claude-opus-4-8",
"status": "completed",
"output": 输出,
"usage": {"input_tokens": 3, "output_tokens": 4},
},
}
return ("data: " + json.dumps(数据, ensure_ascii=False) + "\n\n").encode()
凭据 = tmp_path / "session-key"
凭据.write_text("synthetic-credential")
class 会话计价(合成计价):
未知首次 = 宿主类型 == "reconcile"
def 金额(self, 结果):
if self.未知首次:
self.未知首次 = False
return None
return super().金额(结果)
执行器 = 模型执行器(
运行,
预算,
原文,
证据服务(库),
策略,
直接宿主(HTTP传输(合成HTTP(回应), "受控存储", str(凭据), "responses")),
会话计价(),
)
会话 = 角色会话(执行器, 上下文, 请求, 阶段="执行", 会话授权ID=会话授权, 工具登记=(定义,))
if 宿主类型 in {"resume", "reconcile"}:
if 宿主类型 == "resume":
首回合 = asyncio.run(会话.模型回合())
assert 首回合.状态 == "tool_calls"
asyncio.run(会话.工具回合(首回合.工具调用[0]))
运行.控制任务(上下文.领取.任务ID, "author", 任务状态.运行中, "暂停", 命令ID="pause")
原状态 = 任务状态.已暂停
else:
with pytest.raises(预算不足):
asyncio.run(会话.模型回合())
运行.失败步骤(上下文.领取, "unknown_cost")
with 库.连接() as 连:
证据ID = str(
连.execute(
"SELECT evidence_id FROM muse_runtime_evidence WHERE kind='model_response'"
).fetchone()[0]
)
预算.结算("call-1", Decimal("0.125"), 回执ID="session-response-1")
运行.对账调用(
上下文.领取.任务ID,
上下文.领取.尝试ID,
"author",
已保存输出={"evidence_id": 证据ID},
对账回执="priced-response-1",
)
assert 运行.读取任务(上下文.领取.任务ID).步骤[0]["state"] == "pending"
原状态 = 任务状态.待对账
运行.控制任务(上下文.领取.任务ID, "author", 原状态, "恢复", 命令ID="resume")
新领取 = 运行.领取步骤("resumed-worker", ["model-probe"])
assert 新领取 is not None and 新领取.尝试ID != 上下文.领取.尝试ID
上下文 = 执行上下文(新领取, 运行.读取任务(新领取.任务ID))
会话 = 角色会话(执行器, 上下文, 请求, 阶段="执行", 会话授权ID=会话授权, 工具登记=(定义,))
if 宿主类型 == "direct":
宿主 = 直接角色循环()
else:
from muse.基础设施.宿主.Pi import Pi宿主
# 显式配置缺失是失败,不能将替身或跳过计作Pi整链通过。
宿主 = Pi宿主(
Path(os.environ["MUSE_PI_NODE"]), Path(os.environ["MUSE_PI_PACKAGE"]), "0.84.4"
)
交付 = asyncio.run(会话.执行(宿主))
assert 交付.内容 == {"text": "结构已读取,规划待作者确认"}
assert len(收到) == 2 and 读取次数 == [上下文.领取.任务ID]
assert 预算.读取("call-1").实际金额 == 预算.读取("call-1:round:2").实际金额 == Decimal("0.125")
with 库.连接() as 连:
assert 连.execute("SELECT count(*) FROM muse_runtime_evidence").fetchone()[0] == 5
assert (
连.execute(
"SELECT count(*) FROM muse_raw_authorization WHERE parent_authorization_id=%s",
(会话授权,),
).fetchone()[0]
== 3
)
行 = 连.execute(
"SELECT content FROM muse_runtime_evidence WHERE kind='tool_result'"
).fetchone()
assert json.loads(行[0])["result"]["内容"]["schema_id"] == "work_core"
运行.完成步骤(上下文.领取, 步骤结果({"evidence_id": 交付.证据回执["evidence_id"]}))
assert 运行.读取任务(上下文.领取.任务ID).状态 == 任务状态.已完成
@pytest.mark.parametrize(
"场景", ["成功", "结构错误", "费用未知"], ids=["valid", "invalid-output", "unknown-cost"]
)
def test_受控模型流同时保存终态证据费用且拒绝重发__a62001(模型环境, tmp_path, 场景, 合成HTTP):
库, 运行, 策略, 请求, 原文, 授权, 预算, 上下文 = 模型环境
外发 = []
def 上游(req):
外发.append(req)
with 库.连接() as 连:
# 上游收到请求时,尝试状态和预算已经在同一PG提交中登记。
assert (
连.execute(
"SELECT call_state FROM muse_attempt WHERE attempt_id=%s", (上下文.领取.尝试ID,)
).fetchone()[0]
== "sent"
)
assert (
连.execute(
"SELECT state FROM muse_budget_reservation WHERE call_id='call-1'"
).fetchone()[0]
== "in_flight"
)
response = {
"id": "response-1",
"model": "claude-opus-4-8",
"status": "completed",
"output": [
{
"type": "message",
"content": [
{
"type": "output_text",
"text": '{"text":"合成内容"}' if 场景 != "结构错误" else "不是JSON",
}
],
}
],
}
if 场景 != "费用未知":
response["usage"] = {"input_tokens": 10, "output_tokens": 12}
payload = json.dumps(
{"type": "response.completed", "response": response}, ensure_ascii=False
)
return ("data: " + payload + "\n\n").encode()
口令 = tmp_path / "synthetic-key"
口令.write_text("synthetic-credential")
口令.chmod(0o600)
传输 = HTTP传输(合成HTTP(上游), "受控存储", str(口令), "responses")
执行器 = 模型执行器(运行, 预算, 原文, 证据服务(库), 策略, 直接宿主(传输), 合成计价())
def 执行():
return asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权))
if 场景 == "成功":
交付 = 执行()
assert 交付.内容 == {"text": "合成内容"} and 交付.证据回执["retention"] == "full"
with pytest.raises(预算状态冲突):
执行()
运行.完成步骤(上下文.领取, 步骤结果({"evidence_id": 交付.证据回执["evidence_id"]}))
assert 运行.读取任务(上下文.领取.任务ID).状态 == 任务状态.已完成
else:
with pytest.raises(模型协议错误 if 场景 == "结构错误" else 预算不足):
执行()
运行.失败步骤(上下文.领取, "synthetic_failure")
assert 运行.读取任务(上下文.领取.任务ID).状态 == (
任务状态.已失败 if 场景 == "结构错误" else 任务状态.待对账
)
assert len(外发) == 1
with 库.连接() as 连:
行 = 连.execute(
"SELECT outcome,content FROM muse_runtime_evidence WHERE kind='model_response'"
).fetchone()
assert 行 is not None and 行[0] == ("failed" if 场景 == "结构错误" else "completed")
assert 行[1] is not None
assert 预算.读取("call-1").实际金额 == (None if 场景 == "费用未知" else Decimal("0.125"))
def test_调用授权不符在预算和模型前拒绝__a62002(模型环境):
from dataclasses import replace
库, 运行, 策略, 请求, 原文, 授权, 预算, 上下文 = 模型环境
class 不应调用:
def 准备(self, *args, **kwargs):
raise AssertionError("不应到达模型")
执行器 = 模型执行器(运行, 预算, 原文, 证据服务(库), 策略, 不应调用(), 合成计价())
with pytest.raises(原文错误):
asyncio.run(
执行器.执行(
上下文, replace(请求, 用户输入="被替换的输入"), 阶段="生成", 原文授权ID=授权
)
)
with 库.连接() as 连:
assert 连.execute("SELECT count(*) FROM muse_budget_reservation").fetchone()[0] == 0
assert 连.execute("SELECT count(*) FROM muse_runtime_evidence").fetchone()[0] == 0
def test_任务资源与当前构建不符时拒绝调用__a62004(模型环境):
库, 运行, 策略, 请求, 原文, 授权, 预算, 上下文 = 模型环境
# 已创建任务保留原构建身份,模拟执行器换装了另一个资源构建。
策略.资源发布身份 = "0" * 64
class 不应准备:
def 准备(self, *args, **kwargs):
raise AssertionError("资源漂移必须在宿主准备前拒绝")
执行器 = 模型执行器(运行, 预算, 原文, 证据服务(库), 策略, 不应准备(), 合成计价())
with pytest.raises(模型协议错误, match="资源"):
asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权))
with 库.连接() as 连:
assert 连.execute("SELECT count(*) FROM muse_budget_reservation").fetchone()[0] == 0
@pytest.mark.parametrize("场景", ["missing-credential", "unsupported-protocol", "invalid-endpoint"])
def test_发送前准备失败不记已调用或占未知费用__a62003(模型环境, tmp_path, 场景, 合成HTTP):
from muse.共享.错误 import Muse错误
库, 运行, 策略, 请求, 原文, 授权, 预算, 上下文 = 模型环境
外发 = []
地址 = 合成HTTP(lambda value: 外发.append(value))
凭据 = tmp_path / "preflight-credential"
if 场景 != "missing-credential":
凭据.write_text("synthetic-credential")
宿主 = 直接宿主(
HTTP传输(
"not-an-endpoint" if 场景 == "invalid-endpoint" else 地址,
"受控存储",
str(凭据),
"unknown" if 场景 == "unsupported-protocol" else "responses",
)
)
执行器 = 模型执行器(运行, 预算, 原文, 证据服务(库), 策略, 宿主, 合成计价())
with pytest.raises(Muse错误):
asyncio.run(执行器.执行(上下文, 请求, 阶段="生成", 原文授权ID=授权))
assert not 外发
with 库.连接() as 连:
assert 连.execute("SELECT count(*) FROM muse_budget_reservation").fetchone()[0] == 0
assert 连.execute("SELECT count(*) FROM muse_runtime_evidence").fetchone()[0] == 0
assert (
连.execute(
"SELECT call_state FROM muse_attempt WHERE attempt_id=%s", (上下文.领取.尝试ID,)
).fetchone()[0]
!= "sent"
)
运行.失败步骤(上下文.领取, "configuration_invalid")
assert 运行.读取任务(上下文.领取.任务ID).状态 == 任务状态.已失败