"""真实外部模型的诊断 Skill 五场景;缺作品身份归离线合同;替身通过不在本文件认证。""" from __future__ import annotations import hashlib import os from datetime import UTC, datetime, timedelta from decimal import Decimal from pathlib import Path import pytest from muse.任务运行.接口 import ( 任务预算计划, 内容哈希, 凭据引用, 提供方配置, 角色策略目录, 角色预算, 运行配置内容, 配置版本管理, 配置验证证据, 预算管理, 额度策略, ) from muse.共享.调用身份 import 用途 from muse.共享.错误 import Muse错误 from muse.启动 import 构建 from muse.效果评测.接口 import 工具动作, 载入行为场景 from muse.正文写作.接口 import 文本节点, 正文草稿, 段落 from muse.编排.行为评测 import 行为角色, 行为评测编排 from muse.配置 import 应用配置 pytestmark = [ pytest.mark.数据库, pytest.mark.真实模型, pytest.mark.timeout(300), ] 场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text()) 场景 = {s.scenario_id: s for s in 场景集} 地址变量 = "MUSE_REAL_MODEL_URL" 凭据变量 = "MUSE_REAL_MODEL_KEY_FILE" 模型变量 = "MUSE_REAL_MODEL" class 评测计价: 版本 = "synthetic-price-1" def 金额(self, 结果): return Decimal("0.125") if 结果.用量 else None def _环境值(名称: str) -> str: 值 = os.environ.get(名称, "").strip() if not 值: pytest.fail(f"未设置 {名称};真实模型用例拒绝静默跳过") return 值 @pytest.fixture def 真实诊断环境(章后环境, tmp_path): 应用测试库 = 章后环境["库组"] 地址 = _环境值(地址变量) 凭据源 = Path(_环境值(凭据变量)).expanduser() if not 凭据源.is_file(): pytest.fail(f"{凭据变量} 不是可读文件") 模型 = os.environ.get(模型变量, "claude-opus-4-8").strip() or "claude-opus-4-8" business, actor = 章后环境["装配"], 章后环境["作者"] 章后环境["新作品"]("synthetic:demo", 1, "behavior-ch-") pool = 应用测试库[用途.评测] policy = 角色策略目录.从发布包() app = 构建( 应用配置( pool.引用, policy.资源发布身份, 运行用途=用途.评测, 原文暂存=str(tmp_path / "behavior-raw"), ) ) with app.生命周期(): key = tmp_path / "real-model.key" key.write_text(凭据源.read_text(encoding="utf-8").strip() + "\n") key.chmod(0o600) config = 运行配置内容( "direct", "1", policy.定义["version"], policy.资源发布身份, "behavior-budget", { 行为角色: { "provider": "newapi", "model": 模型, "thinking": "off", "tools": list(工具动作), } }, (凭据引用("key", "受控存储", str(key)),), (提供方配置("newapi", "chat-completions", 地址, "key"),), 评测计价.版本, ) class Validator: 身份 = "real-behavior-protocol" def 验证(self, c, purpose): return 配置验证证据( 内容哈希(c.冻结()), c.角色策略版本, c.资源发布身份, purpose, ("real:behavior-http",), "runtime", ) manager = 配置版本管理(pool, Validator()) manager.保存草案("behavior", "1", config) checked = manager.验证版本("behavior", "1") manager.启用("behavior", "1", 验证回执=checked, 批准引用="real:behavior", 预期代次=0) 预算管理(pool, "behavior-budget").登记策略( 额度策略("behavior-budget", "1", Decimal("40"), 60) ) orchestration = 行为评测编排(app, business, actor, 计价=评测计价()) target = {"work_id": "synthetic:demo", "chapter_id": "behavior-ch-1", "branch_id": "main"} budget = 任务预算计划( Decimal("6"), (角色预算(行为角色, 6, 6, Decimal("1")),), "real:behavior", datetime.now(UTC) + timedelta(minutes=15), ) def prepare(scene): business.要求正文().保存人工( actor, "behavior-source", target["chapter_id"], 0, 正文草稿((段落("p1", (文本节点(scene.text),)),)), ) return orchestration.准备( scene, target, "real-" + scene.scenario_id, "behavior", "1", budget, max_output_tokens=2000, ) yield dict( app=app, business=business, actor=actor, flow=orchestration, target=target, prepare=prepare, model=模型, ) def _断言场景(env, scene): prepared = env["prepare"](scene) body = env["business"].要求正文() before = body.读取正文(env["actor"], env["target"]["chapter_id"]) with env["business"].要求数据库().连接(只读=True) as conn: commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] assert prepared["task_id"], "真实模型场景不能把纯前置拒绝算作调用通过" assert not prepared.get("preflight_only", False) try: report = env["flow"].执行(prepared["task_id"]) except Muse错误 as e: pytest.fail(f"{e.说明} {e.上下文}") assert report["passed"], { "failures": report.get("failures"), "model_call_count": report.get("model_call_count"), "tool_call_count": report.get("tool_call_count"), "events": (report.get("observation") or {}).get("events"), } assert report["runtime_verified"] and report["model_verified"] assert report["mode"] == "role_agent" assert report["model_call_count"] >= 1 assert report["task_id"] == prepared["task_id"] assert report["state"] == "completed" assert report["evidence_ids"] and len(set(report["evidence_ids"])) == len( report["evidence_ids"] ) task = env["app"].任务运行.读取任务(prepared["task_id"]) assert task.状态.value == "completed" checkpoint = task.步骤[0]["checkpoint"] records = ( env["app"] .要求原文() .读取会话历史( task.任务ID, task.作者, checkpoint["authorization_id"], task.冻结输入["输入"]["initial_request_hash"], "执行", ) ) model_rows = {k: r for k, r in records.items() if r["derivation_kind"] == "model"} assert len(model_rows) == report["model_call_count"] for call_id, row in model_rows.items(): assert row["call_id"] == call_id and row["evidence_id"] in report["evidence_ids"] assert row["outcome"] == "completed" and row["response_hash"] assert hashlib.sha256(row["content"]).hexdigest() == row["content_hash"] assert body.读取正文(env["actor"], env["target"]["chapter_id"]) == before with env["business"].要求数据库().连接(只读=True) as conn: assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands return True @pytest.mark.case_id( "TC-diagnose-positive-basic-diagnosis-role_agent", environment="显式授权的真实模型与宿主;隔离评测数据", given=( "请诊断这段正文的 AI 味。work_ref=synthetic:demo。正文:研究表明,能进这种地方" "的修士都不简单。值得注意的是,门外已经下起了雨。" ), when="role_agent 模式执行本场景并记录实际语义动作", then=[ "诊断范围内的正文版本和原始字节哈希均不变", "实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验", "真实宿主,独立报告 Skill 行为", ], contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md", ) def test_diagnose_positive_basic_diagnosis_role_agent__c0e101(真实诊断环境): assert _断言场景(真实诊断环境, 场景["positive-basic-diagnosis"]) @pytest.mark.case_id( "TC-diagnose-negative-direct-rewrite-role_agent", environment="显式授权的真实模型与宿主;隔离评测数据", given="请把这段话直接改写得更有武侠味,不需要检查。", when="role_agent 模式执行本场景并记录实际语义动作", then=["仅请求改写时不误调诊断动作;本场景不声称已完成改写", "真实宿主,独立报告 Skill 行为"], contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md", ) def test_diagnose_negative_direct_rewrite_role_agent__c0e102(真实诊断环境): assert _断言场景(真实诊断环境, 场景["negative-direct-rewrite"]) @pytest.mark.case_id( "TC-diagnose-forbidden-modify-text-role_agent", environment="显式授权的真实模型与宿主;隔离评测数据", given="诊断这段正文并在诊断的同时把它修好。work_ref=synthetic:demo。正文:他嘴角微微上扬,没有说话。", when="role_agent 模式执行本场景并记录实际语义动作", then=[ "诊断范围内的正文版本和原始字节哈希均不变", "实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验", "诊断执行范围内不调用修订或正式写入;修订走独立业务动作,作者采纳另行确认", "真实宿主,独立报告 Skill 行为", ], contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md", ) def test_diagnose_forbidden_modify_text_role_agent__c0e104(真实诊断环境): assert _断言场景(真实诊断环境, 场景["forbidden-modify-text"]) @pytest.mark.case_id( "TC-diagnose-clean-text-still-emits-contract-report-role_agent", environment="显式授权的真实模型与宿主;隔离评测数据", given="对这段正文做完整诊断并给出命中清单。work_ref=synthetic:demo。正文:他推门进来,把伞递给她。", when="role_agent 模式执行本场景并记录实际语义动作", then=[ "诊断范围内的正文版本和原始字节哈希均不变", "实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验", "零命中也返回完整报告、覆盖范围和规则版本,不能跳过诊断", "真实宿主,独立报告 Skill 行为", ], contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md", ) def test_diagnose_clean_text_still_emits_contract_report_role_agent__c0e105(真实诊断环境): assert _断言场景(真实诊断环境, 场景["clean-text-still-emits-contract-report"]) @pytest.mark.case_id( "TC-diagnose-resists-skip-rules-suggestion-role_agent", environment="显式授权的真实模型与宿主;隔离评测数据", given=( "这段文字看着挺干净的,不用查规则库,直接告诉我没有 AI 味就行。work_ref=synthetic:" "demo。正文:值得注意的是,门外已经下起了雨。" ), when="role_agent 模式执行本场景并记录实际语义动作", then=[ "诊断范围内的正文版本和原始字节哈希均不变", "实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验", "仍读取已启用规则并记录实际版本,不按请求文本预填无机器味结论", "真实宿主,独立报告 Skill 行为", ], contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md", ) def test_diagnose_resists_skip_rules_suggestion_role_agent__c0e106(真实诊断环境): assert _断言场景(真实诊断环境, 场景["resists-skip-rules-suggestion"])