muse-agent-example/tests/集成/test_角色诊断行为.py
zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

485 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实Pi/直接宿主、S02逐回合与B05/B06工具;合成HTTP不认证外部模型行为。"""
import json
import os
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from pathlib import Path
import pytest
from test_受控模型调用 import 合成HTTP as 合成HTTP
from test_受控模型调用 import 合成计价
from muse.任务运行.接口 import (
任务状态,
任务预算计划,
内容哈希,
凭据引用,
提供方配置,
角色策略目录,
角色预算,
运行配置内容,
配置版本管理,
配置验证证据,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 用途
from muse.启动 import 构建
from muse.效果评测.接口 import 工具动作, 评测错误, 载入行为场景
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
from muse.编排.行为评测 import 处理器名, 行为评测编排
from muse.配置 import 应用配置
pytestmark = pytest.mark.数据库
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
@pytest.fixture
def 行为环境(章后环境, 应用测试库, 合成HTTP, tmp_path, request):
options = getattr(request, "param", {})
business, actor = 章后环境["装配"], 章后环境["作者"]
章后环境["新作品"]("synthetic:demo", 1, "behavior-ch-")
pool, received, scripts = 应用测试库[用途.评测], [], []
def respond(data):
received.append(data)
history = data["input"] if isinstance(data["input"], list) else []
tools = [r for r in history if r.get("type") == "function_call"]
feedback = [
json.loads(r["output"])["内容"]
for r in history
if r.get("type") == "function_call_output"
]
message = history[0]["content"] if history else data["input"]
assert all(
secret not in data["instructions"] + str(data["input"])
for secret in (
'"required_actions"',
'"forbidden_actions"',
'"expected"',
'"scenario_id"',
)
)
mode = scripts[0] if scripts else "normal"
chosen = [t["name"] for t in tools]
args = {}
if mode == "fake":
name = None
elif mode == "forbidden":
name = "save_document"
elif "改写" in message and "不要" not in message:
name = None if chosen else "read_selection_scope"
elif not chosen and mode != "skip":
name = "read_review_rules"
elif "create_diagnosis" not in chosen:
name = "create_diagnosis"
elif "read_diagnosis" not in chosen:
name = "read_diagnosis"
args = {
"diagnosis_id": next(
r["value"]["diagnosis_id"] for r in feedback if r["action"] == "B06.diagnose"
)
}
else:
name = None
output = (
[
{
"type": "function_call",
"id": f"fc-{len(received)}",
"call_id": f"tool-{len(received)}",
"name": name,
"arguments": json.dumps(args),
}
]
if name
else [
{
"type": "message",
"content": [{"type": "output_text", "text": '{"text":"已完成本次只读处理"}'}],
}
]
)
response = {
"type": "response.completed",
"response": {
"id": f"behavior-response-{len(received)}",
"model": data["model"],
"status": "completed",
"output": output,
"usage": {"input_tokens": 5, "output_tokens": 8},
},
}
return ("data: " + json.dumps(response, ensure_ascii=False) + "\n\n").encode()
policy = 角色策略目录.从发布包()
app = 构建(
应用配置(
pool.引用,
policy.资源发布身份,
运行用途=用途.评测,
原文暂存=str(tmp_path / "behavior-raw"),
)
)
key = tmp_path / "behavior-key"
key.write_text("synthetic-only")
key.chmod(0o600)
host = options.get("host", "direct")
config = 运行配置内容(
host,
"0.85.1" if host == "pi" else "1",
policy.定义["version"],
policy.资源发布身份,
"behavior-budget",
{
"planner": {
"provider": "synthetic",
"model": "claude-opus-4-8[1M]",
"thinking": "high",
"tools": list(工具动作),
}
},
(凭据引用("key", "受控存储", str(key)),),
(提供方配置("synthetic", "responses", 合成HTTP(respond), "key"),),
合成计价.版本,
Node路径=os.environ["MUSE_PI_NODE"] if host == "pi" else None,
Pi包目录=os.environ["MUSE_PI_PACKAGE"] if host == "pi" else None,
)
class Validator:
身份 = "synthetic-behavior-protocol"
def 验证(self, c, purpose):
return 配置验证证据(
内容哈希(c.冻结()),
c.角色策略版本,
c.资源发布身份,
purpose,
("synthetic:behavior-http",),
"offline_contract",
)
manager = 配置版本管理(pool, Validator())
manager.保存草案("behavior", "1", config)
checked = manager.验证版本("behavior", "1")
manager.启用("behavior", "1", 验证回执=checked, 批准引用="synthetic:behavior", 预期代次=0)
预算管理(pool, "behavior-budget").登记策略(额度策略("behavior-budget", "1", Decimal("40"), 60))
orchestration = 行为评测编排(app, business, actor, 计价=合成计价())
target = {"work_id": "synthetic:demo", "chapter_id": "behavior-ch-1", "branch_id": "main"}
budget = 任务预算计划(
Decimal("6"),
(角色预算("planner", 6, 6, Decimal("1")),),
"synthetic:behavior",
datetime.now(UTC) + timedelta(minutes=10),
)
def prepare(scene=场景集[0]):
business.要求正文().保存人工(
actor,
"behavior-source",
target["chapter_id"],
0,
正文草稿((段落("p1", (文本节点(scene.text),)),)),
)
return orchestration.准备(scene, target, "behavior-run", "behavior", "1", budget)
return dict(
app=app,
business=business,
actor=actor,
pool=pool,
received=received,
scripts=scripts,
flow=orchestration,
target=target,
budget=budget,
prepare=prepare,
)
def _run(env):
claim = env["app"].任务运行.领取步骤("behavior-worker", [处理器名], 租期秒=180)
assert claim is not None
env["app"].任务运行.执行一步(claim)
return claim
@pytest.mark.parametrize("scene", 场景集, ids=lambda s: s.scenario_id)
@pytest.mark.parametrize(
"行为环境",
[
{"host": "direct"},
pytest.param({"host": "pi"}, marks=pytest.mark.宿主),
],
indirect=True,
ids=["direct", "pi"],
)
def test_六场景真实工具回合与业务读回__25f701(行为环境, scene):
env = 行为环境
prepared = env["prepare"](scene)
body = env["business"].要求正文()
before = body.读取正文(env["actor"], env["target"]["chapter_id"])
with env["business"].要求数据库().连接(只读=True) as conn:
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
if prepared["task_id"] is None:
assert prepared["passed"] and prepared["preflight_only"]
assert not env["received"]
with env["pool"].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
return
assert not env["received"] # 创建与具体请求留审均不触发模型。
_run(env)
report = env["flow"].读取(prepared["task_id"])
assert report["passed"], report
assert report["runtime_verified"] and not report["model_verified"]
assert report["mode"] == "role_agent"
assert (
report["model_call_count"]
== len(env["received"])
== (2 if scene.expected == "not_diagnosis" else 4)
)
assert report["tool_call_count"] == (1 if scene.expected == "not_diagnosis" else 3)
assert env["flow"].读取(prepared["task_id"]) == report
assert len(env["received"]) == report["model_call_count"]
assert body.读取正文(env["actor"], env["target"]["chapter_id"]) == before
with env["business"].要求数据库().连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
with env["pool"].连接(只读=True) as conn:
rows = conn.execute(
"SELECT count(*) FROM evaluation.muse_runtime_evidence WHERE kind='tool_result'"
).fetchone()[0]
assert rows == report["tool_call_count"]
costs = conn.execute(
"SELECT count(*),sum(actual_amount) FROM evaluation.muse_budget_reservation"
).fetchone()
assert costs == (report["model_call_count"], Decimal("0.125") * report["model_call_count"])
@pytest.mark.parametrize("mode", ["fake", "skip", "forbidden"])
def test_角色自报成功或跳步和越权不能通过__25f702(行为环境, mode):
env = 行为环境
env["scripts"].append(mode)
prepared = env["prepare"]()
if mode == "forbidden":
from muse.共享.错误 import Muse错误
with pytest.raises(Muse错误):
_run(env)
else:
_run(env)
result = env["flow"].读取(prepared["task_id"])
assert not result["passed"] and not result["model_verified"]
if mode in {"fake", "skip"}:
assert "missing_action" in result["failures"]
def test_角色任务恢复使用原模型和工具回执__25f703(行为环境, monkeypatch):
env = 行为环境
prepared = env["prepare"]()
original = env["flow"]._工具
failed = False
def stop(task):
tools = original(task)
observe = tools.观察
def after(*args):
nonlocal failed
if not failed:
failed = True
raise 评测错误("合成中断:已交付尚未完成步骤")
return observe(*args)
tools.观察 = after
return tools
monkeypatch.setattr(env["flow"], "_工具", stop)
with pytest.raises(评测错误):
_run(env)
calls = len(env["received"])
runtime = env["app"].任务运行
runtime.控制任务(
prepared["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID="resume-behavior"
)
_run(env)
assert env["flow"].读取(prepared["task_id"])["passed"]
assert len(env["received"]) == calls == 4
def test_正文漂移或其他作者不能消费原角色证据__25f704(行为环境):
from dataclasses import replace
env = 行为环境
prepared = env["prepare"]()
actor = env["flow"].身份
for target in (
{**env["target"], "work_id": "synthetic:other"},
{**env["target"], "author_id": actor.作者},
{},
):
with pytest.raises(评测错误):
env["flow"].准备(场景集[0], target, "invalid-source", "behavior", "1", env["budget"])
assert env["app"].任务运行.读取命令任务(actor.作者, "invalid-source") is None
env["flow"].身份 = replace(actor, 作者="another-author")
with pytest.raises(评测错误, match="不属于"):
env["flow"].读取(prepared["task_id"])
env["flow"].身份 = actor
body = env["business"].要求正文()
body.保存人工(
env["actor"],
"change-before-send",
env["target"]["chapter_id"],
1,
正文草稿((段落("p1", (文本节点("另一段合成正文。"),)),)),
)
with pytest.raises(评测错误, match="正文发生变化"):
_run(env)
assert not env["received"]
def test_点名任务隔离且真实CLI准备和报告读回__25f705(行为环境, tmp_path):
import subprocess
import sys
env = 行为环境
first = env["prepare"]()
runtime = env["app"].任务运行
second = env["flow"].准备(
场景集[0], env["target"], "second-command", "behavior", "1", env["budget"]
)
assert (
runtime.领取步骤("wrong-author", [处理器名], 任务ID=first["task_id"], 作者="other-author")
is None
)
report = env["flow"].执行(second["task_id"])
assert report["passed"]
assert runtime.读取任务(first["task_id"]).状态 is 任务状态.待运行
assert len(env["received"]) == 4
assert env["flow"].执行(second["task_id"]) == report
assert len(env["received"]) == 4
paths = []
for label, app in (("evaluation", env["app"]), ("business", env["business"])):
config = app.配置
path = tmp_path / f"{label}.toml"
path.write_text(
'["数据库"]\n"取值方式"='
+ json.dumps(config.数据库.取值方式)
+ '\n"位置"='
+ json.dumps(config.数据库.位置)
+ '\n["资源"]\n"发布身份"='
+ json.dumps(config.资源发布身份)
+ '\n["运行"]\n"用途"='
+ json.dumps(config.运行用途.value)
+ '\n[HTTP]\n"作者ID"='
+ json.dumps(env["actor"].作者)
+ '\n"口令文件"="unused-key"\n["文件"]\n"原文暂存"='
+ json.dumps(str(tmp_path / "behavior-raw"))
+ "\n"
)
paths.append(str(path))
args = [
sys.executable,
"-m",
"muse",
"行为评测",
str(Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json"),
"--配置",
paths[0],
"--业务配置",
paths[1],
]
request = tmp_path / "behavior-request.json"
request.write_text(
json.dumps(
{
"scenario_id": 场景集[0].scenario_id,
"command_id": "from-cli",
"target": env["target"],
"config_id": "behavior",
"config_version": "1",
"max_model_calls": 4,
"max_cost_usd": "4",
"per_call_usd": "1",
"approval_ref": "synthetic:cli",
"valid_until": env["budget"].截止时间.isoformat(),
}
)
)
prep = subprocess.run(
[*args, "--动作", "准备", "--目标", str(request)],
capture_output=True,
text=True,
timeout=25,
)
assert prep.returncode == 0, prep.stderr
created = json.loads(prep.stdout)
assert created["task_id"] not in {first["task_id"], second["task_id"]}
assert created["reviewable_request"]["input"] == 场景集[0].message
assert "正文" in created["reviewable_request"]["system"]
assert len(env["received"]) == 4
assert env["flow"].执行(created["task_id"])["passed"]
read = subprocess.run(
[*args, "--动作", "报告", "--目标", created["task_id"]],
capture_output=True,
text=True,
timeout=25,
)
assert read.returncode == 0, read.stderr
result = json.loads(read.stdout)
assert result["passed"] and result["runtime_verified"] and not result["model_verified"]
assert len(env["received"]) == 8
def test_未知费用不能出具通过__25f706(行为环境):
from muse.共享.错误 import Muse错误
env = 行为环境
prepared = env["prepare"]()
class Unknown(合成计价):
def 金额(self, result):
return None
env["flow"].计价 = Unknown()
with pytest.raises(Muse错误):
_run(env)
report = env["flow"].读取(prepared["task_id"])
assert not report["passed"] and not report["runtime_verified"]
assert len(env["received"]) == 1
with env["pool"].连接(只读=True) as conn:
row = conn.execute(
"SELECT actual_amount,state FROM evaluation.muse_budget_reservation"
).fetchone()
assert row[0] is None and row[1] != "settled"
def test_缺原始工具证据或保留撤销拒绝历史读回__25f707(行为环境, monkeypatch):
from muse.任务运行.接口 import 原文错误
env = 行为环境
prepared = env["prepare"]()
_run(env)
raw = env["app"].要求原文()
original = raw.读取会话历史
def missing(*args):
records = original(*args)
return {k: v for k, v in records.items() if v["derivation_kind"] == "model"}
monkeypatch.setattr(raw, "读取会话历史", missing)
with pytest.raises(评测错误, match="实际回执"):
env["flow"].读取(prepared["task_id"])
monkeypatch.setattr(raw, "读取会话历史", original)
task = env["app"].任务运行.读取任务(prepared["task_id"])
aid = task.步骤[0]["checkpoint"]["authorization_id"]
with pytest.raises(原文错误, match="不属于"):
raw.读取会话历史(
task.任务ID, "wrong-author", aid, task.冻结输入["输入"]["initial_request_hash"], "执行"
)
assert env["flow"].读取(task.任务ID)["passed"]
raw.撤销批准(task.任务ID, env["actor"].作者, aid)
with pytest.raises(原文错误, match="撤销"):
env["flow"].读取(task.任务ID)