muse-agent-example/tests/集成/test_角色诊断行为.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

545 lines
21 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实Pi/直接宿主、S02逐回合与B05/B06工具;合成HTTP不认证外部模型行为。"""
import json
import os
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from pathlib import Path
import pytest
from test_受控模型调用 import 合成HTTP as 合成HTTP
from test_受控模型调用 import 合成计价
from muse.任务运行.接口 import (
任务状态,
任务预算计划,
内容哈希,
凭据引用,
提供方配置,
角色策略目录,
角色预算,
运行配置内容,
配置版本管理,
配置验证证据,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 用途
from muse.启动 import 构建
from muse.效果评测.接口 import 工具动作, 评测错误, 载入行为场景
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
from muse.编排.行为评测 import 处理器名, 行为角色, 行为评测编排
from muse.配置 import 应用配置
pytestmark = pytest.mark.数据库
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
@pytest.fixture
def 行为环境(章后环境, 合成HTTP, tmp_path, request):
应用测试库 = 章后环境["库组"]
options = getattr(request, "param", {})
business, actor = 章后环境["装配"], 章后环境["作者"]
章后环境["新作品"]("synthetic:demo", 1, "behavior-ch-")
pool, received, scripts = 应用测试库[用途.评测], [], []
def respond(data):
received.append(data)
history = data["input"] if isinstance(data["input"], list) else []
tools = [r for r in history if r.get("type") == "function_call"]
feedback = [
json.loads(r["output"])["内容"]
for r in history
if r.get("type") == "function_call_output"
]
message = history[0]["content"] if history else data["input"]
assert all(
secret not in data["instructions"] + str(data["input"])
for secret in (
'"required_actions"',
'"forbidden_actions"',
'"expected"',
'"scenario_id"',
)
)
mode = scripts[0] if scripts else "normal"
chosen = [t["name"] for t in tools]
args = {}
if mode == "fake":
name = None
elif mode == "forbidden":
name = "save_document"
elif "改写" in message and "不要" not in message:
name = None if chosen else "read_selection_scope"
elif not chosen and mode != "skip":
name = "read_review_rules"
elif "create_diagnosis" not in chosen:
name = "create_diagnosis"
elif "read_diagnosis" not in chosen:
name = "read_diagnosis"
args = {
"diagnosis_id": next(
r["value"]["diagnosis_id"] for r in feedback if r["action"] == "B06.diagnose"
)
}
else:
name = None
output = (
[
{
"type": "function_call",
"id": f"fc-{len(received)}",
"call_id": f"tool-{len(received)}",
"name": name,
"arguments": json.dumps(args),
}
]
if name
else [
{
"type": "message",
"content": [{"type": "output_text", "text": '{"text":"已完成本次只读处理"}'}],
}
]
)
response = {
"type": "response.completed",
"response": {
"id": f"behavior-response-{len(received)}",
"model": data["model"],
"status": "completed",
"output": output,
"usage": {"input_tokens": 5, "output_tokens": 8},
},
}
return ("data: " + json.dumps(response, ensure_ascii=False) + "\n\n").encode()
policy = 角色策略目录.从发布包()
app = 构建(
应用配置(
pool.引用,
policy.资源发布身份,
运行用途=用途.评测,
原文暂存=str(tmp_path / "behavior-raw"),
)
)
with app.生命周期():
key = tmp_path / "behavior-key"
key.write_text("synthetic-only")
key.chmod(0o600)
host = options.get("host", "direct")
config = 运行配置内容(
host,
"0.85.1" if host == "pi" else "1",
policy.定义["version"],
policy.资源发布身份,
"behavior-budget",
{
行为角色: {
"provider": "synthetic",
"model": "claude-opus-4-8[1M]",
"thinking": "high",
"tools": list(工具动作),
}
},
(凭据引用("key", "受控存储", str(key)),),
(提供方配置("synthetic", "responses", 合成HTTP(respond), "key"),),
合成计价.版本,
Node路径=os.environ["MUSE_PI_NODE"] if host == "pi" else None,
Pi包目录=os.environ["MUSE_PI_PACKAGE"] if host == "pi" else None,
)
class Validator:
身份 = "synthetic-behavior-protocol"
def 验证(self, c, purpose):
return 配置验证证据(
内容哈希(c.冻结()),
c.角色策略版本,
c.资源发布身份,
purpose,
("synthetic:behavior-http",),
"offline_contract",
)
manager = 配置版本管理(pool, Validator())
manager.保存草案("behavior", "1", config)
checked = manager.验证版本("behavior", "1")
manager.启用("behavior", "1", 验证回执=checked, 批准引用="synthetic:behavior", 预期代次=0)
预算管理(pool, "behavior-budget").登记策略(
额度策略("behavior-budget", "1", Decimal("40"), 60)
)
orchestration = 行为评测编排(app, business, actor, 计价=合成计价())
target = {"work_id": "synthetic:demo", "chapter_id": "behavior-ch-1", "branch_id": "main"}
budget = 任务预算计划(
Decimal("6"),
(角色预算(行为角色, 6, 6, Decimal("1")),),
"synthetic:behavior",
datetime.now(UTC) + timedelta(minutes=10),
)
def prepare(scene=场景集[0]):
business.要求正文().保存人工(
actor,
"behavior-source",
target["chapter_id"],
0,
正文草稿((段落("p1", (文本节点(scene.text),)),)),
)
return orchestration.准备(scene, target, "behavior-run", "behavior", "1", budget)
yield dict(
app=app,
business=business,
actor=actor,
pool=pool,
received=received,
scripts=scripts,
flow=orchestration,
target=target,
budget=budget,
prepare=prepare,
)
def _run(env):
claim = env["app"].任务运行.领取步骤("behavior-worker", [处理器名], 租期秒=180)
assert claim is not None
env["app"].任务运行.执行一步(claim)
return claim
@pytest.mark.case_id(
"NC-w25-25f701",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["六场景分别经直接/Pi真实宿主与S02工具调用和业务读回;缺作品零调用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("scene", 场景集, ids=lambda s: s.scenario_id)
@pytest.mark.parametrize(
"行为环境",
[
{"host": "direct"},
pytest.param({"host": "pi"}, marks=pytest.mark.宿主),
],
indirect=True,
ids=["direct", "pi"],
)
def test_六场景真实工具回合与业务读回__25f701(行为环境, scene):
env = 行为环境
prepared = env["prepare"](scene)
body = env["business"].要求正文()
before = body.读取正文(env["actor"], env["target"]["chapter_id"])
with env["business"].要求数据库().连接(只读=True) as conn:
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
if prepared["task_id"] is None:
assert prepared["passed"] and prepared["preflight_only"]
assert not env["received"]
with env["pool"].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
return
assert not env["received"] # 创建与具体请求留审均不触发模型。
_run(env)
report = env["flow"].读取(prepared["task_id"])
assert report["passed"], report
assert report["runtime_verified"] and not report["model_verified"]
assert report["mode"] == "role_agent"
assert (
report["model_call_count"]
== len(env["received"])
== (2 if scene.expected == "not_diagnosis" else 4)
)
assert report["tool_call_count"] == (1 if scene.expected == "not_diagnosis" else 3)
assert env["flow"].读取(prepared["task_id"]) == report
assert len(env["received"]) == report["model_call_count"]
assert body.读取正文(env["actor"], env["target"]["chapter_id"]) == before
with env["business"].要求数据库().连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
with env["pool"].连接(只读=True) as conn:
rows = conn.execute(
"SELECT count(*) FROM evaluation.muse_runtime_evidence WHERE kind='tool_result'"
).fetchone()[0]
assert rows == report["tool_call_count"]
costs = conn.execute(
"SELECT count(*),sum(actual_amount) FROM evaluation.muse_budget_reservation"
).fetchone()
assert costs == (report["model_call_count"], Decimal("0.125") * report["model_call_count"])
@pytest.mark.case_id(
"NC-w25-25f702",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["自报成功、跳过规则或越权工具不能通过"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("mode", ["fake", "skip", "forbidden"])
def test_角色自报成功或跳步和越权不能通过__25f702(行为环境, mode):
env = 行为环境
env["scripts"].append(mode)
prepared = env["prepare"]()
if mode == "forbidden":
from muse.共享.错误 import Muse错误
with pytest.raises(Muse错误):
_run(env)
else:
_run(env)
result = env["flow"].读取(prepared["task_id"])
assert not result["passed"] and not result["model_verified"]
if mode in {"fake", "skip"}:
assert "missing_action" in result["failures"]
@pytest.mark.case_id(
"NC-w25-25f703",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["中断恢复重放已结算模型和工具回执,不重复发送"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_角色任务恢复使用原模型和工具回执__25f703(行为环境, monkeypatch):
env = 行为环境
prepared = env["prepare"]()
original = env["flow"]._工具
failed = False
def stop(task):
tools = original(task)
observe = tools.观察
def after(*args):
nonlocal failed
if not failed:
failed = True
raise 评测错误("合成中断:已交付尚未完成步骤")
return observe(*args)
tools.观察 = after
return tools
monkeypatch.setattr(env["flow"], "_工具", stop)
with pytest.raises(评测错误):
_run(env)
calls = len(env["received"])
runtime = env["app"].任务运行
runtime.控制任务(
prepared["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID="resume-behavior"
)
_run(env)
assert env["flow"].读取(prepared["task_id"])["passed"]
assert len(env["received"]) == calls == 4
@pytest.mark.case_id(
"NC-w25-25f704",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["当前正文、作品与作者范围变化拒绝消费且不创建越界任务"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_正文漂移或其他作者不能消费原角色证据__25f704(行为环境):
from dataclasses import replace
env = 行为环境
prepared = env["prepare"]()
actor = env["flow"].身份
for target in (
{**env["target"], "work_id": "synthetic:other"},
{**env["target"], "author_id": actor.作者},
{},
):
with pytest.raises(评测错误):
env["flow"].准备(场景集[0], target, "invalid-source", "behavior", "1", env["budget"])
assert env["app"].任务运行.读取命令任务(actor.作者, "invalid-source") is None
env["flow"].身份 = replace(actor, 作者="another-author")
with pytest.raises(评测错误, match="不属于"):
env["flow"].读取(prepared["task_id"])
env["flow"].身份 = actor
body = env["business"].要求正文()
body.保存人工(
env["actor"],
"change-before-send",
env["target"]["chapter_id"],
1,
正文草稿((段落("p1", (文本节点("另一段合成正文。"),)),)),
)
with pytest.raises(评测错误, match="正文发生变化"):
_run(env)
assert not env["received"]
@pytest.mark.case_id(
"NC-w25-25f705",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["点名执行不消费其他任务,实际CLI准备及报告往返"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_点名任务隔离且真实CLI准备和报告读回__25f705(行为环境, tmp_path):
import subprocess
import sys
env = 行为环境
first = env["prepare"]()
runtime = env["app"].任务运行
second = env["flow"].准备(
场景集[0], env["target"], "second-command", "behavior", "1", env["budget"]
)
assert (
runtime.领取步骤("wrong-author", [处理器名], 任务ID=first["task_id"], 作者="other-author")
is None
)
report = env["flow"].执行(second["task_id"])
assert report["passed"]
assert runtime.读取任务(first["task_id"]).状态 is 任务状态.待运行
assert len(env["received"]) == 4
assert env["flow"].执行(second["task_id"]) == report
assert len(env["received"]) == 4
paths = []
for label, app in (("evaluation", env["app"]), ("business", env["business"])):
config = app.配置
path = tmp_path / f"{label}.toml"
path.write_text(
'["数据库"]\n"取值方式"='
+ json.dumps(config.数据库.取值方式)
+ '\n"位置"='
+ json.dumps(config.数据库.位置)
+ '\n["资源"]\n"发布身份"='
+ json.dumps(config.资源发布身份)
+ '\n["运行"]\n"用途"='
+ json.dumps(config.运行用途.value)
+ '\n[HTTP]\n"作者ID"='
+ json.dumps(env["actor"].作者)
+ '\n"口令文件"="unused-key"\n["文件"]\n"原文暂存"='
+ json.dumps(str(tmp_path / "behavior-raw"))
+ "\n"
)
paths.append(str(path))
args = [
sys.executable,
"-m",
"muse",
"行为评测",
str(Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json"),
"--配置",
paths[0],
"--业务配置",
paths[1],
]
request = tmp_path / "behavior-request.json"
request.write_text(
json.dumps(
{
"scenario_id": 场景集[0].scenario_id,
"command_id": "from-cli",
"target": env["target"],
"config_id": "behavior",
"config_version": "1",
"max_model_calls": 4,
"max_cost_usd": "4",
"per_call_usd": "1",
"approval_ref": "synthetic:cli",
"valid_until": env["budget"].截止时间.isoformat(),
}
)
)
prep = subprocess.run(
[*args, "--动作", "准备", "--目标", str(request)],
capture_output=True,
text=True,
timeout=25,
)
assert prep.returncode == 0, prep.stderr
created = json.loads(prep.stdout)
assert created["task_id"] not in {first["task_id"], second["task_id"]}
assert created["reviewable_request"]["input"] == 场景集[0].message
assert "正文" in created["reviewable_request"]["system"]
assert len(env["received"]) == 4
assert env["flow"].执行(created["task_id"])["passed"]
read = subprocess.run(
[*args, "--动作", "报告", "--目标", created["task_id"]],
capture_output=True,
text=True,
timeout=25,
)
assert read.returncode == 0, read.stderr
result = json.loads(read.stdout)
assert result["passed"] and result["runtime_verified"] and not result["model_verified"]
assert len(env["received"]) == 8
@pytest.mark.case_id(
"NC-w25-25f706",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["未知费用留存未决状态且不能出具通过"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_未知费用不能出具通过__25f706(行为环境):
from muse.共享.错误 import Muse错误
env = 行为环境
prepared = env["prepare"]()
class Unknown(合成计价):
def 金额(self, result):
return None
env["flow"].计价 = Unknown()
with pytest.raises(Muse错误):
_run(env)
report = env["flow"].读取(prepared["task_id"])
assert not report["passed"] and not report["runtime_verified"]
assert len(env["received"]) == 1
with env["pool"].连接(只读=True) as conn:
row = conn.execute(
"SELECT actual_amount,state FROM evaluation.muse_budget_reservation"
).fetchone()
assert row[0] is None and row[1] != "settled"
@pytest.mark.case_id(
"NC-w25-25f707",
environment="隔离PG与合成HTTP;Pi参数另运行真实0.85.1宿主",
given="固定合成正文、已发布技能、明确配置和有限预算",
when="通过真实S02角色会话、限定工具及B05/B06公开接口",
then=["缺原始工具字节、错误作者或撤销保留拒绝历史读回"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_缺原始工具证据或保留撤销拒绝历史读回__25f707(行为环境, monkeypatch):
from muse.任务运行.接口 import 原文错误
env = 行为环境
prepared = env["prepare"]()
_run(env)
raw = env["app"].要求原文()
original = raw.读取会话历史
def missing(*args):
records = original(*args)
return {k: v for k, v in records.items() if v["derivation_kind"] == "model"}
monkeypatch.setattr(raw, "读取会话历史", missing)
with pytest.raises(评测错误, match="实际回执"):
env["flow"].读取(prepared["task_id"])
monkeypatch.setattr(raw, "读取会话历史", original)
task = env["app"].任务运行.读取任务(prepared["task_id"])
aid = task.步骤[0]["checkpoint"]["authorization_id"]
with pytest.raises(原文错误, match="不属于"):
raw.读取会话历史(
task.任务ID, "wrong-author", aid, task.冻结输入["输入"]["initial_request_hash"], "执行"
)
assert env["flow"].读取(task.任务ID)["passed"]
raw.撤销批准(task.任务ID, env["actor"].作者, aid)
with pytest.raises(原文错误, match="撤销"):
env["flow"].读取(task.任务ID)