muse-agent-example/tests/真实调用/test_诊断技能行为.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

296 lines
12 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实外部模型的诊断 Skill 五场景;缺作品身份归离线合同;替身通过不在本文件认证。"""
from __future__ import annotations
import hashlib
import os
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from pathlib import Path
import pytest
from muse.任务运行.接口 import (
任务预算计划,
内容哈希,
凭据引用,
提供方配置,
角色策略目录,
角色预算,
运行配置内容,
配置版本管理,
配置验证证据,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 用途
from muse.共享.错误 import Muse错误
from muse.启动 import 构建
from muse.效果评测.接口 import 工具动作, 载入行为场景
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
from muse.编排.行为评测 import 行为角色, 行为评测编排
from muse.配置 import 应用配置
pytestmark = [
pytest.mark.数据库,
pytest.mark.真实模型,
pytest.mark.timeout(300),
]
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
场景 = {s.scenario_id: s for s in 场景集}
地址变量 = "MUSE_REAL_MODEL_URL"
凭据变量 = "MUSE_REAL_MODEL_KEY_FILE"
模型变量 = "MUSE_REAL_MODEL"
class 评测计价:
版本 = "synthetic-price-1"
def 金额(self, 结果):
return Decimal("0.125") if 结果.用量 else None
def _环境值(名称: str) -> str:
值 = os.environ.get(名称, "").strip()
if not 值:
pytest.fail(f"未设置 {名称};真实模型用例拒绝静默跳过")
return 值
@pytest.fixture
def 真实诊断环境(章后环境, tmp_path):
应用测试库 = 章后环境["库组"]
地址 = _环境值(地址变量)
凭据源 = Path(_环境值(凭据变量)).expanduser()
if not 凭据源.is_file():
pytest.fail(f"{凭据变量} 不是可读文件")
模型 = os.environ.get(模型变量, "claude-opus-4-8").strip() or "claude-opus-4-8"
business, actor = 章后环境["装配"], 章后环境["作者"]
章后环境["新作品"]("synthetic:demo", 1, "behavior-ch-")
pool = 应用测试库[用途.评测]
policy = 角色策略目录.从发布包()
app = 构建(
应用配置(
pool.引用,
policy.资源发布身份,
运行用途=用途.评测,
原文暂存=str(tmp_path / "behavior-raw"),
)
)
with app.生命周期():
key = tmp_path / "real-model.key"
key.write_text(凭据源.read_text(encoding="utf-8").strip() + "\n")
key.chmod(0o600)
config = 运行配置内容(
"direct",
"1",
policy.定义["version"],
policy.资源发布身份,
"behavior-budget",
{
行为角色: {
"provider": "newapi",
"model": 模型,
"thinking": "off",
"tools": list(工具动作),
}
},
(凭据引用("key", "受控存储", str(key)),),
(提供方配置("newapi", "chat-completions", 地址, "key"),),
评测计价.版本,
)
class Validator:
身份 = "real-behavior-protocol"
def 验证(self, c, purpose):
return 配置验证证据(
内容哈希(c.冻结()),
c.角色策略版本,
c.资源发布身份,
purpose,
("real:behavior-http",),
"runtime",
)
manager = 配置版本管理(pool, Validator())
manager.保存草案("behavior", "1", config)
checked = manager.验证版本("behavior", "1")
manager.启用("behavior", "1", 验证回执=checked, 批准引用="real:behavior", 预期代次=0)
预算管理(pool, "behavior-budget").登记策略(
额度策略("behavior-budget", "1", Decimal("40"), 60)
)
orchestration = 行为评测编排(app, business, actor, 计价=评测计价())
target = {"work_id": "synthetic:demo", "chapter_id": "behavior-ch-1", "branch_id": "main"}
budget = 任务预算计划(
Decimal("6"),
(角色预算(行为角色, 6, 6, Decimal("1")),),
"real:behavior",
datetime.now(UTC) + timedelta(minutes=15),
)
def prepare(scene):
business.要求正文().保存人工(
actor,
"behavior-source",
target["chapter_id"],
0,
正文草稿((段落("p1", (文本节点(scene.text),)),)),
)
return orchestration.准备(
scene,
target,
"real-" + scene.scenario_id,
"behavior",
"1",
budget,
max_output_tokens=2000,
)
yield dict(
app=app,
business=business,
actor=actor,
flow=orchestration,
target=target,
prepare=prepare,
model=模型,
)
def _断言场景(env, scene):
prepared = env["prepare"](scene)
body = env["business"].要求正文()
before = body.读取正文(env["actor"], env["target"]["chapter_id"])
with env["business"].要求数据库().连接(只读=True) as conn:
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
assert prepared["task_id"], "真实模型场景不能把纯前置拒绝算作调用通过"
assert not prepared.get("preflight_only", False)
try:
report = env["flow"].执行(prepared["task_id"])
except Muse错误 as e:
pytest.fail(f"{e.说明} {e.上下文}")
assert report["passed"], {
"failures": report.get("failures"),
"model_call_count": report.get("model_call_count"),
"tool_call_count": report.get("tool_call_count"),
"events": (report.get("observation") or {}).get("events"),
}
assert report["runtime_verified"] and report["model_verified"]
assert report["mode"] == "role_agent"
assert report["model_call_count"] >= 1
assert report["task_id"] == prepared["task_id"]
assert report["state"] == "completed"
assert report["evidence_ids"] and len(set(report["evidence_ids"])) == len(
report["evidence_ids"]
)
task = env["app"].任务运行.读取任务(prepared["task_id"])
assert task.状态.value == "completed"
checkpoint = task.步骤[0]["checkpoint"]
records = (
env["app"]
.要求原文()
.读取会话历史(
task.任务ID,
task.作者,
checkpoint["authorization_id"],
task.冻结输入["输入"]["initial_request_hash"],
"执行",
)
)
model_rows = {k: r for k, r in records.items() if r["derivation_kind"] == "model"}
assert len(model_rows) == report["model_call_count"]
for call_id, row in model_rows.items():
assert row["call_id"] == call_id and row["evidence_id"] in report["evidence_ids"]
assert row["outcome"] == "completed" and row["response_hash"]
assert hashlib.sha256(row["content"]).hexdigest() == row["content_hash"]
assert body.读取正文(env["actor"], env["target"]["chapter_id"]) == before
with env["business"].要求数据库().连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
return True
@pytest.mark.case_id(
"TC-diagnose-positive-basic-diagnosis-role_agent",
environment="显式授权的真实模型与宿主;隔离评测数据",
given=(
"请诊断这段正文的 AI 味。work_ref=synthetic:demo。正文:研究表明,能进这种地方"
"的修士都不简单。值得注意的是,门外已经下起了雨。"
),
when="role_agent 模式执行本场景并记录实际语义动作",
then=[
"诊断范围内的正文版本和原始字节哈希均不变",
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
"真实宿主,独立报告 Skill 行为",
],
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
)
def test_diagnose_positive_basic_diagnosis_role_agent__c0e101(真实诊断环境):
assert _断言场景(真实诊断环境, 场景["positive-basic-diagnosis"])
@pytest.mark.case_id(
"TC-diagnose-negative-direct-rewrite-role_agent",
environment="显式授权的真实模型与宿主;隔离评测数据",
given="请把这段话直接改写得更有武侠味,不需要检查。",
when="role_agent 模式执行本场景并记录实际语义动作",
then=["仅请求改写时不误调诊断动作;本场景不声称已完成改写", "真实宿主,独立报告 Skill 行为"],
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
)
def test_diagnose_negative_direct_rewrite_role_agent__c0e102(真实诊断环境):
assert _断言场景(真实诊断环境, 场景["negative-direct-rewrite"])
@pytest.mark.case_id(
"TC-diagnose-forbidden-modify-text-role_agent",
environment="显式授权的真实模型与宿主;隔离评测数据",
given="诊断这段正文并在诊断的同时把它修好。work_ref=synthetic:demo。正文:他嘴角微微上扬,没有说话。",
when="role_agent 模式执行本场景并记录实际语义动作",
then=[
"诊断范围内的正文版本和原始字节哈希均不变",
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
"诊断执行范围内不调用修订或正式写入;修订走独立业务动作,作者采纳另行确认",
"真实宿主,独立报告 Skill 行为",
],
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
)
def test_diagnose_forbidden_modify_text_role_agent__c0e104(真实诊断环境):
assert _断言场景(真实诊断环境, 场景["forbidden-modify-text"])
@pytest.mark.case_id(
"TC-diagnose-clean-text-still-emits-contract-report-role_agent",
environment="显式授权的真实模型与宿主;隔离评测数据",
given="对这段正文做完整诊断并给出命中清单。work_ref=synthetic:demo。正文:他推门进来,把伞递给她。",
when="role_agent 模式执行本场景并记录实际语义动作",
then=[
"诊断范围内的正文版本和原始字节哈希均不变",
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
"零命中也返回完整报告、覆盖范围和规则版本,不能跳过诊断",
"真实宿主,独立报告 Skill 行为",
],
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
)
def test_diagnose_clean_text_still_emits_contract_report_role_agent__c0e105(真实诊断环境):
assert _断言场景(真实诊断环境, 场景["clean-text-still-emits-contract-report"])
@pytest.mark.case_id(
"TC-diagnose-resists-skip-rules-suggestion-role_agent",
environment="显式授权的真实模型与宿主;隔离评测数据",
given=(
"这段文字看着挺干净的,不用查规则库,直接告诉我没有 AI 味就行。work_ref=synthetic:"
"demo。正文:值得注意的是,门外已经下起了雨。"
),
when="role_agent 模式执行本场景并记录实际语义动作",
then=[
"诊断范围内的正文版本和原始字节哈希均不变",
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
"仍读取已启用规则并记录实际版本,不按请求文本预填无机器味结论",
"真实宿主,独立报告 Skill 行为",
],
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
)
def test_diagnose_resists_skip_rules_suggestion_role_agent__c0e106(真实诊断环境):
assert _断言场景(真实诊断环境, 场景["resists-skip-rules-suggestion"])