muse-agent-example/src/muse/接入/cli/角色行为命令.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

83 lines
3.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""准备具体角色请求、点名执行及实际证据读回;连接由显式双配置装配。"""
import json
from datetime import datetime
from decimal import Decimal
from pathlib import Path
from pydantic import BaseModel, ConfigDict, Field, StrictInt, ValidationError
from muse.任务运行.接口 import 任务预算计划, 角色预算
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
from muse.共享.错误 import 配置错误
from muse.启动 import 构建
from muse.接入.主会话 import 参数合同错误
from muse.效果评测.接口 import 行为场景, 评测错误, 载入行为场景
from muse.编排.行为评测 import 行为角色, 行为评测编排
from muse.配置 import 读取配置
class 角色行为请求(BaseModel):
model_config = ConfigDict(extra="forbid")
scenario_id: str
command_id: str = Field(min_length=1, max_length=200)
target: dict
config_id: str
config_version: str
max_model_calls: StrictInt = Field(ge=1, le=12)
max_tool_calls: StrictInt = Field(default=8, ge=1, le=32)
max_output_tokens: StrictInt = Field(default=1200, ge=1, le=8192)
max_cost_usd: Decimal = Field(gt=0, allow_inf_nan=False)
per_call_usd: Decimal = Field(gt=0, allow_inf_nan=False)
approval_ref: str = Field(min_length=1)
valid_until: datetime
def 运行角色行为命令(场景路径: str, 评测配置: str, 业务配置: str, 动作: str, 目标: str) -> dict:
a, b = 读取配置(评测配置), 读取配置(业务配置)
if (
a.运行用途 is not 用途.评测
or b.运行用途 is not 用途.生产
or a.HTTP is None
or b.HTTP is None
or a.HTTP.作者ID != b.HTTP.作者ID
):
raise 配置错误("需要同一作者的evaluation任务配置和显式合成资料服务配置")
# 资料服务配置必须由操作者明确给出;不派生生产连接、更不从默认库回退。
app, business = 构建(a), 构建(b)
with app.生命周期(), business.生命周期():
actor = 调用身份(b.HTTP.作者ID, None, 用途.生产, 内容用途.检测)
flow = 行为评测编排(app, business, actor)
if 动作 in {"执行", "报告"}:
return flow.执行(目标) if 动作 == "执行" else flow.读取(目标)
if 动作 != "准备":
raise 评测错误("未支持的角色行为动作")
try:
scenes: tuple[行为场景, ...] = 载入行为场景(Path(场景路径).read_text(encoding="utf-8"))
req = 角色行为请求.model_validate_json(Path(目标).read_text(encoding="utf-8"))
except ValidationError as exc:
raise 参数合同错误(exc.errors(), 前缀=("body",)) from None
except (OSError, UnicodeError):
raise 评测错误("角色行为场景或请求文件不可读取") from None
scene = next((s for s in scenes if s.scenario_id == req.scenario_id), None)
if scene is None:
raise 评测错误("请求场景不属于这份数据集")
budget = 任务预算计划(
req.max_cost_usd,
(角色预算(行为角色, req.max_model_calls, req.max_model_calls, req.per_call_usd),),
req.approval_ref,
req.valid_until,
)
result = flow.准备(
scene,
req.target,
req.command_id,
req.config_id,
req.config_version,
budget,
max_tool_calls=req.max_tool_calls,
max_output_tokens=req.max_output_tokens,
)
# 保持JSON命令出口稳定;不复制连接位置或凭据。
return json.loads(json.dumps(result, ensure_ascii=False, default=str))