muse-agent-example/tests/集成/test_角色诊断行为.py
zizi b9af240b68 R2 收尾:W30 唯一写入观察与最终验收(C08 章后接通 / 构建身份重建 / 最终备份与恢复演练)
- 章后处理:事实提案命令身份加入类型、批内去重改为先校验后去重(REPEAT_OBJECT_TYPE 记跳过、不再静默丢重复);查询章后状态回传事实跳过;可按需发布/核对可登记任务与 config_not_declared 语义收紧
- 模型治理:推理字符数计量(ChatCompletions/Messages/执行合同)、探针在声明关闭推理却仍有推理时拒绝、计价支持缓存协议与未知模型兜底
- 运行态:资源包重建(发布身份 21d850f7…),三处配置同步,post-extract v5 代次 5 启用,服务 PID 96675(health/ready 通过)
- 门禁:make 检查通过;离线 1008 passed;定点数据库 101 passed;最终完整备份 bkp-1f5591d8… 与空库恢复演练 verified(152 表/158 关系/迁移 45,演练库已 DROP)
- 文档与证据:运行手册、功能覆盖、工作包清单 W30 证据、测试用例清单(17 条 execution_evidence 回填 + 历史死指针说明)、目标文件清单
- 旧实现与旧库直连材料退出(含明文凭据文件移除);新配置 配置/本机正式.toml、配置/本机维护.toml、配置/计费/、配置/运行配置/ 入库,均只含受控引用
- R2 执行证据在 .agents.local/改造/R2-20260909/(不入库)
2026-09-16 08:23:39 +08:00

485 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实Pi/直接宿主、S02逐回合与B05/B06工具;合成HTTP不认证外部模型行为。"""
import json
import os
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from pathlib import Path
import pytest
from test_受控模型调用 import 合成HTTP as 合成HTTP
from test_受控模型调用 import 合成计价
from muse.任务运行.接口 import (
任务状态,
任务预算计划,
内容哈希,
凭据引用,
提供方配置,
角色策略目录,
角色预算,
运行配置内容,
配置版本管理,
配置验证证据,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 用途
from muse.启动 import 构建
from muse.效果评测.接口 import 工具动作, 评测错误, 载入行为场景
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
from muse.编排.行为评测 import 处理器名, 行为角色, 行为评测编排
from muse.配置 import 应用配置
pytestmark = pytest.mark.数据库
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
@pytest.fixture
def 行为环境(章后环境, 应用测试库, 合成HTTP, tmp_path, request):
options = getattr(request, "param", {})
business, actor = 章后环境["装配"], 章后环境["作者"]
章后环境["新作品"]("synthetic:demo", 1, "behavior-ch-")
pool, received, scripts = 应用测试库[用途.评测], [], []
def respond(data):
received.append(data)
history = data["input"] if isinstance(data["input"], list) else []
tools = [r for r in history if r.get("type") == "function_call"]
feedback = [
json.loads(r["output"])["内容"]
for r in history
if r.get("type") == "function_call_output"
]
message = history[0]["content"] if history else data["input"]
assert all(
secret not in data["instructions"] + str(data["input"])
for secret in (
'"required_actions"',
'"forbidden_actions"',
'"expected"',
'"scenario_id"',
)
)
mode = scripts[0] if scripts else "normal"
chosen = [t["name"] for t in tools]
args = {}
if mode == "fake":
name = None
elif mode == "forbidden":
name = "save_document"
elif "改写" in message and "不要" not in message:
name = None if chosen else "read_selection_scope"
elif not chosen and mode != "skip":
name = "read_review_rules"
elif "create_diagnosis" not in chosen:
name = "create_diagnosis"
elif "read_diagnosis" not in chosen:
name = "read_diagnosis"
args = {
"diagnosis_id": next(
r["value"]["diagnosis_id"] for r in feedback if r["action"] == "B06.diagnose"
)
}
else:
name = None
output = (
[
{
"type": "function_call",
"id": f"fc-{len(received)}",
"call_id": f"tool-{len(received)}",
"name": name,
"arguments": json.dumps(args),
}
]
if name
else [
{
"type": "message",
"content": [{"type": "output_text", "text": '{"text":"已完成本次只读处理"}'}],
}
]
)
response = {
"type": "response.completed",
"response": {
"id": f"behavior-response-{len(received)}",
"model": data["model"],
"status": "completed",
"output": output,
"usage": {"input_tokens": 5, "output_tokens": 8},
},
}
return ("data: " + json.dumps(response, ensure_ascii=False) + "\n\n").encode()
policy = 角色策略目录.从发布包()
app = 构建(
应用配置(
pool.引用,
policy.资源发布身份,
运行用途=用途.评测,
原文暂存=str(tmp_path / "behavior-raw"),
)
)
key = tmp_path / "behavior-key"
key.write_text("synthetic-only")
key.chmod(0o600)
host = options.get("host", "direct")
config = 运行配置内容(
host,
"0.85.1" if host == "pi" else "1",
policy.定义["version"],
policy.资源发布身份,
"behavior-budget",
{
行为角色: {
"provider": "synthetic",
"model": "claude-opus-4-8[1M]",
"thinking": "high",
"tools": list(工具动作),
}
},
(凭据引用("key", "受控存储", str(key)),),
(提供方配置("synthetic", "responses", 合成HTTP(respond), "key"),),
合成计价.版本,
Node路径=os.environ["MUSE_PI_NODE"] if host == "pi" else None,
Pi包目录=os.environ["MUSE_PI_PACKAGE"] if host == "pi" else None,
)
class Validator:
身份 = "synthetic-behavior-protocol"
def 验证(self, c, purpose):
return 配置验证证据(
内容哈希(c.冻结()),
c.角色策略版本,
c.资源发布身份,
purpose,
("synthetic:behavior-http",),
"offline_contract",
)
manager = 配置版本管理(pool, Validator())
manager.保存草案("behavior", "1", config)
checked = manager.验证版本("behavior", "1")
manager.启用("behavior", "1", 验证回执=checked, 批准引用="synthetic:behavior", 预期代次=0)
预算管理(pool, "behavior-budget").登记策略(额度策略("behavior-budget", "1", Decimal("40"), 60))
orchestration = 行为评测编排(app, business, actor, 计价=合成计价())
target = {"work_id": "synthetic:demo", "chapter_id": "behavior-ch-1", "branch_id": "main"}
budget = 任务预算计划(
Decimal("6"),
(角色预算(行为角色, 6, 6, Decimal("1")),),
"synthetic:behavior",
datetime.now(UTC) + timedelta(minutes=10),
)
def prepare(scene=场景集[0]):
business.要求正文().保存人工(
actor,
"behavior-source",
target["chapter_id"],
0,
正文草稿((段落("p1", (文本节点(scene.text),)),)),
)
return orchestration.准备(scene, target, "behavior-run", "behavior", "1", budget)
return dict(
app=app,
business=business,
actor=actor,
pool=pool,
received=received,
scripts=scripts,
flow=orchestration,
target=target,
budget=budget,
prepare=prepare,
)
def _run(env):
claim = env["app"].任务运行.领取步骤("behavior-worker", [处理器名], 租期秒=180)
assert claim is not None
env["app"].任务运行.执行一步(claim)
return claim
@pytest.mark.parametrize("scene", 场景集, ids=lambda s: s.scenario_id)
@pytest.mark.parametrize(
"行为环境",
[
{"host": "direct"},
pytest.param({"host": "pi"}, marks=pytest.mark.宿主),
],
indirect=True,
ids=["direct", "pi"],
)
def test_六场景真实工具回合与业务读回__25f701(行为环境, scene):
env = 行为环境
prepared = env["prepare"](scene)
body = env["business"].要求正文()
before = body.读取正文(env["actor"], env["target"]["chapter_id"])
with env["business"].要求数据库().连接(只读=True) as conn:
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
if prepared["task_id"] is None:
assert prepared["passed"] and prepared["preflight_only"]
assert not env["received"]
with env["pool"].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
return
assert not env["received"] # 创建与具体请求留审均不触发模型。
_run(env)
report = env["flow"].读取(prepared["task_id"])
assert report["passed"], report
assert report["runtime_verified"] and not report["model_verified"]
assert report["mode"] == "role_agent"
assert (
report["model_call_count"]
== len(env["received"])
== (2 if scene.expected == "not_diagnosis" else 4)
)
assert report["tool_call_count"] == (1 if scene.expected == "not_diagnosis" else 3)
assert env["flow"].读取(prepared["task_id"]) == report
assert len(env["received"]) == report["model_call_count"]
assert body.读取正文(env["actor"], env["target"]["chapter_id"]) == before
with env["business"].要求数据库().连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
with env["pool"].连接(只读=True) as conn:
rows = conn.execute(
"SELECT count(*) FROM evaluation.muse_runtime_evidence WHERE kind='tool_result'"
).fetchone()[0]
assert rows == report["tool_call_count"]
costs = conn.execute(
"SELECT count(*),sum(actual_amount) FROM evaluation.muse_budget_reservation"
).fetchone()
assert costs == (report["model_call_count"], Decimal("0.125") * report["model_call_count"])
@pytest.mark.parametrize("mode", ["fake", "skip", "forbidden"])
def test_角色自报成功或跳步和越权不能通过__25f702(行为环境, mode):
env = 行为环境
env["scripts"].append(mode)
prepared = env["prepare"]()
if mode == "forbidden":
from muse.共享.错误 import Muse错误
with pytest.raises(Muse错误):
_run(env)
else:
_run(env)
result = env["flow"].读取(prepared["task_id"])
assert not result["passed"] and not result["model_verified"]
if mode in {"fake", "skip"}:
assert "missing_action" in result["failures"]
def test_角色任务恢复使用原模型和工具回执__25f703(行为环境, monkeypatch):
env = 行为环境
prepared = env["prepare"]()
original = env["flow"]._工具
failed = False
def stop(task):
tools = original(task)
observe = tools.观察
def after(*args):
nonlocal failed
if not failed:
failed = True
raise 评测错误("合成中断:已交付尚未完成步骤")
return observe(*args)
tools.观察 = after
return tools
monkeypatch.setattr(env["flow"], "_工具", stop)
with pytest.raises(评测错误):
_run(env)
calls = len(env["received"])
runtime = env["app"].任务运行
runtime.控制任务(
prepared["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID="resume-behavior"
)
_run(env)
assert env["flow"].读取(prepared["task_id"])["passed"]
assert len(env["received"]) == calls == 4
def test_正文漂移或其他作者不能消费原角色证据__25f704(行为环境):
from dataclasses import replace
env = 行为环境
prepared = env["prepare"]()
actor = env["flow"].身份
for target in (
{**env["target"], "work_id": "synthetic:other"},
{**env["target"], "author_id": actor.作者},
{},
):
with pytest.raises(评测错误):
env["flow"].准备(场景集[0], target, "invalid-source", "behavior", "1", env["budget"])
assert env["app"].任务运行.读取命令任务(actor.作者, "invalid-source") is None
env["flow"].身份 = replace(actor, 作者="another-author")
with pytest.raises(评测错误, match="不属于"):
env["flow"].读取(prepared["task_id"])
env["flow"].身份 = actor
body = env["business"].要求正文()
body.保存人工(
env["actor"],
"change-before-send",
env["target"]["chapter_id"],
1,
正文草稿((段落("p1", (文本节点("另一段合成正文。"),)),)),
)
with pytest.raises(评测错误, match="正文发生变化"):
_run(env)
assert not env["received"]
def test_点名任务隔离且真实CLI准备和报告读回__25f705(行为环境, tmp_path):
import subprocess
import sys
env = 行为环境
first = env["prepare"]()
runtime = env["app"].任务运行
second = env["flow"].准备(
场景集[0], env["target"], "second-command", "behavior", "1", env["budget"]
)
assert (
runtime.领取步骤("wrong-author", [处理器名], 任务ID=first["task_id"], 作者="other-author")
is None
)
report = env["flow"].执行(second["task_id"])
assert report["passed"]
assert runtime.读取任务(first["task_id"]).状态 is 任务状态.待运行
assert len(env["received"]) == 4
assert env["flow"].执行(second["task_id"]) == report
assert len(env["received"]) == 4
paths = []
for label, app in (("evaluation", env["app"]), ("business", env["business"])):
config = app.配置
path = tmp_path / f"{label}.toml"
path.write_text(
'["数据库"]\n"取值方式"='
+ json.dumps(config.数据库.取值方式)
+ '\n"位置"='
+ json.dumps(config.数据库.位置)
+ '\n["资源"]\n"发布身份"='
+ json.dumps(config.资源发布身份)
+ '\n["运行"]\n"用途"='
+ json.dumps(config.运行用途.value)
+ '\n[HTTP]\n"作者ID"='
+ json.dumps(env["actor"].作者)
+ '\n"口令文件"="unused-key"\n["文件"]\n"原文暂存"='
+ json.dumps(str(tmp_path / "behavior-raw"))
+ "\n"
)
paths.append(str(path))
args = [
sys.executable,
"-m",
"muse",
"行为评测",
str(Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json"),
"--配置",
paths[0],
"--业务配置",
paths[1],
]
request = tmp_path / "behavior-request.json"
request.write_text(
json.dumps(
{
"scenario_id": 场景集[0].scenario_id,
"command_id": "from-cli",
"target": env["target"],
"config_id": "behavior",
"config_version": "1",
"max_model_calls": 4,
"max_cost_usd": "4",
"per_call_usd": "1",
"approval_ref": "synthetic:cli",
"valid_until": env["budget"].截止时间.isoformat(),
}
)
)
prep = subprocess.run(
[*args, "--动作", "准备", "--目标", str(request)],
capture_output=True,
text=True,
timeout=25,
)
assert prep.returncode == 0, prep.stderr
created = json.loads(prep.stdout)
assert created["task_id"] not in {first["task_id"], second["task_id"]}
assert created["reviewable_request"]["input"] == 场景集[0].message
assert "正文" in created["reviewable_request"]["system"]
assert len(env["received"]) == 4
assert env["flow"].执行(created["task_id"])["passed"]
read = subprocess.run(
[*args, "--动作", "报告", "--目标", created["task_id"]],
capture_output=True,
text=True,
timeout=25,
)
assert read.returncode == 0, read.stderr
result = json.loads(read.stdout)
assert result["passed"] and result["runtime_verified"] and not result["model_verified"]
assert len(env["received"]) == 8
def test_未知费用不能出具通过__25f706(行为环境):
from muse.共享.错误 import Muse错误
env = 行为环境
prepared = env["prepare"]()
class Unknown(合成计价):
def 金额(self, result):
return None
env["flow"].计价 = Unknown()
with pytest.raises(Muse错误):
_run(env)
report = env["flow"].读取(prepared["task_id"])
assert not report["passed"] and not report["runtime_verified"]
assert len(env["received"]) == 1
with env["pool"].连接(只读=True) as conn:
row = conn.execute(
"SELECT actual_amount,state FROM evaluation.muse_budget_reservation"
).fetchone()
assert row[0] is None and row[1] != "settled"
def test_缺原始工具证据或保留撤销拒绝历史读回__25f707(行为环境, monkeypatch):
from muse.任务运行.接口 import 原文错误
env = 行为环境
prepared = env["prepare"]()
_run(env)
raw = env["app"].要求原文()
original = raw.读取会话历史
def missing(*args):
records = original(*args)
return {k: v for k, v in records.items() if v["derivation_kind"] == "model"}
monkeypatch.setattr(raw, "读取会话历史", missing)
with pytest.raises(评测错误, match="实际回执"):
env["flow"].读取(prepared["task_id"])
monkeypatch.setattr(raw, "读取会话历史", original)
task = env["app"].任务运行.读取任务(prepared["task_id"])
aid = task.步骤[0]["checkpoint"]["authorization_id"]
with pytest.raises(原文错误, match="不属于"):
raw.读取会话历史(
task.任务ID, "wrong-author", aid, task.冻结输入["输入"]["initial_request_hash"], "执行"
)
assert env["flow"].读取(task.任务ID)["passed"]
raw.撤销批准(task.任务ID, env["actor"].作者, aid)
with pytest.raises(原文错误, match="撤销"):
env["flow"].读取(task.任务ID)