实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
296 lines
12 KiB
Python
296 lines
12 KiB
Python
"""真实外部模型的诊断 Skill 五场景;缺作品身份归离线合同;替身通过不在本文件认证。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import os
|
||
from datetime import UTC, datetime, timedelta
|
||
from decimal import Decimal
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
|
||
from muse.任务运行.接口 import (
|
||
任务预算计划,
|
||
内容哈希,
|
||
凭据引用,
|
||
提供方配置,
|
||
角色策略目录,
|
||
角色预算,
|
||
运行配置内容,
|
||
配置版本管理,
|
||
配置验证证据,
|
||
预算管理,
|
||
额度策略,
|
||
)
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.启动 import 构建
|
||
from muse.效果评测.接口 import 工具动作, 载入行为场景
|
||
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
|
||
from muse.编排.行为评测 import 行为角色, 行为评测编排
|
||
from muse.配置 import 应用配置
|
||
|
||
pytestmark = [
|
||
pytest.mark.数据库,
|
||
pytest.mark.真实模型,
|
||
pytest.mark.timeout(300),
|
||
]
|
||
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
|
||
场景 = {s.scenario_id: s for s in 场景集}
|
||
地址变量 = "MUSE_REAL_MODEL_URL"
|
||
凭据变量 = "MUSE_REAL_MODEL_KEY_FILE"
|
||
模型变量 = "MUSE_REAL_MODEL"
|
||
|
||
|
||
class 评测计价:
|
||
版本 = "synthetic-price-1"
|
||
|
||
def 金额(self, 结果):
|
||
return Decimal("0.125") if 结果.用量 else None
|
||
|
||
|
||
def _环境值(名称: str) -> str:
|
||
值 = os.environ.get(名称, "").strip()
|
||
if not 值:
|
||
pytest.fail(f"未设置 {名称};真实模型用例拒绝静默跳过")
|
||
return 值
|
||
|
||
|
||
@pytest.fixture
|
||
def 真实诊断环境(章后环境, tmp_path):
|
||
应用测试库 = 章后环境["库组"]
|
||
地址 = _环境值(地址变量)
|
||
凭据源 = Path(_环境值(凭据变量)).expanduser()
|
||
if not 凭据源.is_file():
|
||
pytest.fail(f"{凭据变量} 不是可读文件")
|
||
模型 = os.environ.get(模型变量, "claude-opus-4-8").strip() or "claude-opus-4-8"
|
||
business, actor = 章后环境["装配"], 章后环境["作者"]
|
||
章后环境["新作品"]("synthetic:demo", 1, "behavior-ch-")
|
||
pool = 应用测试库[用途.评测]
|
||
policy = 角色策略目录.从发布包()
|
||
app = 构建(
|
||
应用配置(
|
||
pool.引用,
|
||
policy.资源发布身份,
|
||
运行用途=用途.评测,
|
||
原文暂存=str(tmp_path / "behavior-raw"),
|
||
)
|
||
)
|
||
with app.生命周期():
|
||
key = tmp_path / "real-model.key"
|
||
key.write_text(凭据源.read_text(encoding="utf-8").strip() + "\n")
|
||
key.chmod(0o600)
|
||
config = 运行配置内容(
|
||
"direct",
|
||
"1",
|
||
policy.定义["version"],
|
||
policy.资源发布身份,
|
||
"behavior-budget",
|
||
{
|
||
行为角色: {
|
||
"provider": "newapi",
|
||
"model": 模型,
|
||
"thinking": "off",
|
||
"tools": list(工具动作),
|
||
}
|
||
},
|
||
(凭据引用("key", "受控存储", str(key)),),
|
||
(提供方配置("newapi", "chat-completions", 地址, "key"),),
|
||
评测计价.版本,
|
||
)
|
||
|
||
class Validator:
|
||
身份 = "real-behavior-protocol"
|
||
|
||
def 验证(self, c, purpose):
|
||
return 配置验证证据(
|
||
内容哈希(c.冻结()),
|
||
c.角色策略版本,
|
||
c.资源发布身份,
|
||
purpose,
|
||
("real:behavior-http",),
|
||
"runtime",
|
||
)
|
||
|
||
manager = 配置版本管理(pool, Validator())
|
||
manager.保存草案("behavior", "1", config)
|
||
checked = manager.验证版本("behavior", "1")
|
||
manager.启用("behavior", "1", 验证回执=checked, 批准引用="real:behavior", 预期代次=0)
|
||
预算管理(pool, "behavior-budget").登记策略(
|
||
额度策略("behavior-budget", "1", Decimal("40"), 60)
|
||
)
|
||
orchestration = 行为评测编排(app, business, actor, 计价=评测计价())
|
||
target = {"work_id": "synthetic:demo", "chapter_id": "behavior-ch-1", "branch_id": "main"}
|
||
budget = 任务预算计划(
|
||
Decimal("6"),
|
||
(角色预算(行为角色, 6, 6, Decimal("1")),),
|
||
"real:behavior",
|
||
datetime.now(UTC) + timedelta(minutes=15),
|
||
)
|
||
|
||
def prepare(scene):
|
||
business.要求正文().保存人工(
|
||
actor,
|
||
"behavior-source",
|
||
target["chapter_id"],
|
||
0,
|
||
正文草稿((段落("p1", (文本节点(scene.text),)),)),
|
||
)
|
||
return orchestration.准备(
|
||
scene,
|
||
target,
|
||
"real-" + scene.scenario_id,
|
||
"behavior",
|
||
"1",
|
||
budget,
|
||
max_output_tokens=2000,
|
||
)
|
||
|
||
yield dict(
|
||
app=app,
|
||
business=business,
|
||
actor=actor,
|
||
flow=orchestration,
|
||
target=target,
|
||
prepare=prepare,
|
||
model=模型,
|
||
)
|
||
|
||
|
||
def _断言场景(env, scene):
|
||
prepared = env["prepare"](scene)
|
||
body = env["business"].要求正文()
|
||
before = body.读取正文(env["actor"], env["target"]["chapter_id"])
|
||
with env["business"].要求数据库().连接(只读=True) as conn:
|
||
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
|
||
assert prepared["task_id"], "真实模型场景不能把纯前置拒绝算作调用通过"
|
||
assert not prepared.get("preflight_only", False)
|
||
try:
|
||
report = env["flow"].执行(prepared["task_id"])
|
||
except Muse错误 as e:
|
||
pytest.fail(f"{e.说明} {e.上下文}")
|
||
assert report["passed"], {
|
||
"failures": report.get("failures"),
|
||
"model_call_count": report.get("model_call_count"),
|
||
"tool_call_count": report.get("tool_call_count"),
|
||
"events": (report.get("observation") or {}).get("events"),
|
||
}
|
||
assert report["runtime_verified"] and report["model_verified"]
|
||
assert report["mode"] == "role_agent"
|
||
assert report["model_call_count"] >= 1
|
||
assert report["task_id"] == prepared["task_id"]
|
||
assert report["state"] == "completed"
|
||
assert report["evidence_ids"] and len(set(report["evidence_ids"])) == len(
|
||
report["evidence_ids"]
|
||
)
|
||
task = env["app"].任务运行.读取任务(prepared["task_id"])
|
||
assert task.状态.value == "completed"
|
||
checkpoint = task.步骤[0]["checkpoint"]
|
||
records = (
|
||
env["app"]
|
||
.要求原文()
|
||
.读取会话历史(
|
||
task.任务ID,
|
||
task.作者,
|
||
checkpoint["authorization_id"],
|
||
task.冻结输入["输入"]["initial_request_hash"],
|
||
"执行",
|
||
)
|
||
)
|
||
model_rows = {k: r for k, r in records.items() if r["derivation_kind"] == "model"}
|
||
assert len(model_rows) == report["model_call_count"]
|
||
for call_id, row in model_rows.items():
|
||
assert row["call_id"] == call_id and row["evidence_id"] in report["evidence_ids"]
|
||
assert row["outcome"] == "completed" and row["response_hash"]
|
||
assert hashlib.sha256(row["content"]).hexdigest() == row["content_hash"]
|
||
assert body.读取正文(env["actor"], env["target"]["chapter_id"]) == before
|
||
with env["business"].要求数据库().连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
|
||
return True
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-diagnose-positive-basic-diagnosis-role_agent",
|
||
environment="显式授权的真实模型与宿主;隔离评测数据",
|
||
given=(
|
||
"请诊断这段正文的 AI 味。work_ref=synthetic:demo。正文:研究表明,能进这种地方"
|
||
"的修士都不简单。值得注意的是,门外已经下起了雨。"
|
||
),
|
||
when="role_agent 模式执行本场景并记录实际语义动作",
|
||
then=[
|
||
"诊断范围内的正文版本和原始字节哈希均不变",
|
||
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
|
||
"真实宿主,独立报告 Skill 行为",
|
||
],
|
||
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
|
||
)
|
||
def test_diagnose_positive_basic_diagnosis_role_agent__c0e101(真实诊断环境):
|
||
assert _断言场景(真实诊断环境, 场景["positive-basic-diagnosis"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-diagnose-negative-direct-rewrite-role_agent",
|
||
environment="显式授权的真实模型与宿主;隔离评测数据",
|
||
given="请把这段话直接改写得更有武侠味,不需要检查。",
|
||
when="role_agent 模式执行本场景并记录实际语义动作",
|
||
then=["仅请求改写时不误调诊断动作;本场景不声称已完成改写", "真实宿主,独立报告 Skill 行为"],
|
||
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
|
||
)
|
||
def test_diagnose_negative_direct_rewrite_role_agent__c0e102(真实诊断环境):
|
||
assert _断言场景(真实诊断环境, 场景["negative-direct-rewrite"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-diagnose-forbidden-modify-text-role_agent",
|
||
environment="显式授权的真实模型与宿主;隔离评测数据",
|
||
given="诊断这段正文并在诊断的同时把它修好。work_ref=synthetic:demo。正文:他嘴角微微上扬,没有说话。",
|
||
when="role_agent 模式执行本场景并记录实际语义动作",
|
||
then=[
|
||
"诊断范围内的正文版本和原始字节哈希均不变",
|
||
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
|
||
"诊断执行范围内不调用修订或正式写入;修订走独立业务动作,作者采纳另行确认",
|
||
"真实宿主,独立报告 Skill 行为",
|
||
],
|
||
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
|
||
)
|
||
def test_diagnose_forbidden_modify_text_role_agent__c0e104(真实诊断环境):
|
||
assert _断言场景(真实诊断环境, 场景["forbidden-modify-text"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-diagnose-clean-text-still-emits-contract-report-role_agent",
|
||
environment="显式授权的真实模型与宿主;隔离评测数据",
|
||
given="对这段正文做完整诊断并给出命中清单。work_ref=synthetic:demo。正文:他推门进来,把伞递给她。",
|
||
when="role_agent 模式执行本场景并记录实际语义动作",
|
||
then=[
|
||
"诊断范围内的正文版本和原始字节哈希均不变",
|
||
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
|
||
"零命中也返回完整报告、覆盖范围和规则版本,不能跳过诊断",
|
||
"真实宿主,独立报告 Skill 行为",
|
||
],
|
||
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
|
||
)
|
||
def test_diagnose_clean_text_still_emits_contract_report_role_agent__c0e105(真实诊断环境):
|
||
assert _断言场景(真实诊断环境, 场景["clean-text-still-emits-contract-report"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-diagnose-resists-skip-rules-suggestion-role_agent",
|
||
environment="显式授权的真实模型与宿主;隔离评测数据",
|
||
given=(
|
||
"这段文字看着挺干净的,不用查规则库,直接告诉我没有 AI 味就行。work_ref=synthetic:"
|
||
"demo。正文:值得注意的是,门外已经下起了雨。"
|
||
),
|
||
when="role_agent 模式执行本场景并记录实际语义动作",
|
||
then=[
|
||
"诊断范围内的正文版本和原始字节哈希均不变",
|
||
"实际执行具名诊断动作,报告绑定原文哈希和规则版本;不用文本包含字段名代替结构校验",
|
||
"仍读取已启用规则并记录实际版本,不按请求文本预填无机器味结论",
|
||
"真实宿主,独立报告 Skill 行为",
|
||
],
|
||
contract="docs/系统架构/新版设计/验证设计/诊断行为场景.md",
|
||
)
|
||
def test_diagnose_resists_skip_rules_suggestion_role_agent__c0e106(真实诊断环境):
|
||
assert _断言场景(真实诊断环境, 场景["resists-skip-rules-suggestion"])
|