muse-agent-example/tests/集成/test_标定凭据消费.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

472 lines
19 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""资格评测消费真实标定回执;合成HTTP只验证管线,不证明模型文学能力。"""
import copy
from dataclasses import replace
import pytest
import test_文学评分执行与条件第三 as 文学测试
import test_标定金标准与凭据 as 标定测试
import test_评测执行与失败收敛 as 运行测试
from muse.共享.调用身份 import 用途
from muse.效果评测.接口 import (
公开评测输入,
实验请求,
数据样本,
数据集发布,
标定使用,
评测服务,
评测错误,
)
from muse.正式变更.接口 import 固定哈希
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
参数 = 标定测试.参数
def _校准(env, *, full_panel=True, biased=False, failed=False):
标定测试._生成(env)
raw = 标定测试._请求(env)
score = 7.0 if failed else (7.5 if full_panel else 8.0)
for row in raw["annotations"]:
row["scores"] = {d: score for d in 文学测试.维度}
标定测试._发布(env, raw)
文学测试._推进(env)
if full_panel:
env["scripted"].extend(
["calibration_low", "calibration_high"] if biased else ["ok", "calibration_gap"]
)
运行测试._运行就绪(env)
文学测试._推进(env)
运行测试._运行就绪(env)
return 标定测试._封存(env)
def _请求(
env, certificates, *, group="qualification-book", original="另一部作品的资格底稿。", basis=None
):
if basis is None:
basis = {
**文学测试.依据,
"sources": [
{"source_id": "outline", "kind": "fine_outline", "text": "旅者在山口等待同伴。"},
{
"source_id": "history",
"kind": "historical_prose",
"text": "前卷写过雨后的山道。",
},
],
}
source = 数据集发布(
dataset_id="qualification-fixture",
revision=1,
samples=(
数据样本(
sample_id="qualification-sample",
source_ref="synthetic:qualification",
license_ref="synthetic:owned-fixture",
source_groups=(group,),
split="holdout",
input=公开评测输入(instruction="比较两篇合成叙述。", original=original, context={}),
answer={"judging_basis": basis},
),
),
)
data = 评测服务(env["pools"][用途.维护]).发布数据集(
replace(env["actor"], 用途=用途.维护), source
)
return 实验请求.model_validate(
{
**env["request"].model_dump(mode="json"),
"dataset_version_id": data["version_id"],
"dataset_hash": data["public_hash"],
"split": "holdout",
"calibration_policy": None,
"evaluation_goal": "qualification",
"calibration_use": {
"policy": 标定测试.策略,
"references": [
{
"experiment_id": c["result"]["experiment_id"],
"receipt_hash": c["receipt_hash"],
}
for c in certificates
],
},
}
)
def _新实验(env, req):
exp = env["app"].要求评测().创建实验(env["actor"], "qualification-command", req)
return {**env, "exp": exp, "request": req}
def _资格执行(env):
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
运行测试._运行就绪(env)
文学测试._推进(env)
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
@pytest.mark.case_id(
"NC-w25-25f001",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["三位实际评委完整标定进入资格派发元数据和报告,模型输入不含回执;启用仍未评定"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_完整实际标定约束资格派发并进入读回__25f001(执行环境):
env = 执行环境
certificate = _校准(env)
assert certificate["result"]["status"] == "passed"
dest = _新实验(env, _请求(env, [certificate]))
report = _资格执行(dest)
binding = dest["exp"]["conditions"]["calibration_binding"]
assert len(binding["reviewers"]) == 3
assert report["state"] == "completed" and len(env["received"]) == 9
assert report["evaluation_goal"] == "qualification"
assert report["calibration_use"]["binding"] == binding
assert report["activation_status"] == "not_evaluated"
for unit in 文学测试._读(dest)["units"]:
if unit["evidence"]:
assert unit["evidence"]["runtime"]["dispatch_metadata"][
"calibration_use_hash"
] == 固定哈希(binding)
assert all(certificate["receipt_hash"] not in r["input"] for r in env["received"])
@pytest.mark.case_id(
"TC-75676077ad55",
environment="隔离PG、真实S02及合成HTTP",
given="真实隔离标定、原金标准与资格请求",
when="经公开实验创建入口请求资格消费",
then=["原缺stamp守卫由资格实验缺少明确标定引用拒绝承接,在任何资格调用前拒绝。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_缺失标定凭据在资格调用前拒绝__756760(执行环境):
env = 执行环境
cert = _校准(env)
req = _请求(env, [cert]).model_copy(update={"calibration_use": None})
with pytest.raises(评测错误, match="标定"):
_新实验(env, req)
assert len(env["received"]) == 5
@pytest.mark.case_id(
"TC-1c0f450f8945",
environment="隔离PG、真实S02及合成HTTP",
given="真实隔离标定、原金标准与资格请求",
when="经公开实验创建入口请求资格消费",
then=["原标准指纹不符由真实策略及场景评分标准变化拒绝承接,旧回执不能授权新尺度。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
@pytest.mark.parametrize("variant", ["policy", "rubric"])
def test_评分策略改变不能沿用旧标定__1c0f45(执行环境, monkeypatch, variant):
import muse.效果评测.文学执行 as literary
env = 执行环境
cert = _校准(env)
req = _请求(env, [cert])
if variant == "policy":
use = req.calibration_use.model_dump(mode="json")
use["policy"]["max_mae"] = 0.4
req = req.model_copy(update={"calibration_use": 标定使用.model_validate(use)})
else:
read = literary.读取评分标准
monkeypatch.setattr(
literary, "读取评分标准", lambda scenario: {**read(scenario), "version": "changed"}
)
with pytest.raises(评测错误, match="策略指纹|完整实际标定"):
_新实验(env, req)
assert len(env["received"]) == 5
@pytest.mark.case_id(
"TC-a2621d936e34",
environment="隔离PG、真实S02及合成HTTP",
given="真实隔离标定、原金标准与资格请求",
when="经公开实验创建入口请求资格消费",
then=["真实标定失败凭据保留;资格消费不能以有记录代替合格,也不发起后续调用。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_失败标定不能授权资格调用__a2621d(执行环境):
env = 执行环境
cert = _校准(env, failed=True)
assert cert["result"]["status"] == "failed"
with pytest.raises(评测错误, match="合格凭据"):
_新实验(env, _请求(env, [cert]))
assert len(env["received"]) == 5
@pytest.mark.case_id(
"NC-w25-25f002",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["稳定双评未调用的第三评委不算标定覆盖"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_未调用的第三评委不能算已标定__25f002(执行环境):
env = 执行环境
cert = _校准(env, full_panel=False)
assert cert["result"]["status"] == "passed"
with pytest.raises(评测错误, match="未执行候补"):
_新实验(env, _请求(env, [cert]))
assert len(env["received"]) == 4
@pytest.mark.case_id(
"NC-w25-25f003",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["整体MAE合格不能掩盖单个评委失败"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_整体均差不能掩盖单个评委不合格__25f003(执行环境):
env = 执行环境
cert = _校准(env, biased=True)
assert cert["result"]["status"] == "passed"
assert cert["result"]["metrics"]["mae"] == 0.5
with pytest.raises(评测错误, match="整体均值"):
_新实验(env, _请求(env, [cert]))
assert len(env["received"]) == 5
@pytest.mark.case_id(
"NC-w25-25f004",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["留出集不能换数据集或来源标签复用标定组、原文或共同背景"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("variant", ["group", "background", "input"])
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_留出集不能换标识复用标定原文与共同背景__25f004(执行环境, variant):
from muse.效果评测.存储 import 评测存储
env = 执行环境
cert = _校准(env)
with env["pool"].连接(只读=True) as conn:
source = 评测存储(conn).读取数据集(str(env["request"].dataset_version_id))
sample = source["public_manifest"]["samples"][0]
kwargs = {
"group": {"group": sample["source_groups"][0]},
"input": {"original": sample["input"]["original"]},
"background": {"basis": copy.deepcopy(文学测试.依据)},
}[variant]
with pytest.raises(评测错误, match="交叉|重复"):
_新实验(env, _请求(env, [cert], **kwargs))
assert len(env["received"]) == 5
@pytest.mark.case_id(
"NC-w25-25f005",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["多份完整标定承接各自实际参与评委,不虚构候补参与"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_多份独立标定只承接实际参与的评委__25f005(执行环境):
env = 执行环境
first = _校准(env, full_panel=False)
req = env["request"].model_copy(
update={
"judges": (env["request"].judges[0], env["request"].arbitrator),
"arbitrator": env["request"].judges[1],
}
)
exp = env["app"].要求评测().创建实验(env["actor"], "calibrate-other-panel", req)
second = _校准({**env, "exp": exp, "request": req}, full_panel=False)
dest = _新实验(env, _请求(env, [first, second]))
report = _资格执行(dest)
assert report["state"] == "completed" and len(env["received"]) == 12
binding = report["calibration_use"]["binding"]
assert len(binding["references"]) == 2 and len(binding["reviewers"]) == 3
assert {r["experiment_id"] for r in binding["reviewers"].values()} == {
first["result"]["experiment_id"],
second["result"]["experiment_id"],
}
@pytest.mark.case_id(
"NC-w25-25f006",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["重复引用、错误回执哈希及未封存记录拒绝资格消费"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("variant", ["duplicate", "hash", "unsealed"])
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_重复引用错哈希或未封存不能替代标定__25f006(执行环境, variant, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
cert = _校准(env)
if variant == "unsealed":
monkeypatch.setattr(评测存储, "读取标定凭据", lambda *args: None)
if variant == "hash":
cert = {**cert, "receipt_hash": "0" * 64}
refs = [cert, cert] if variant == "duplicate" else [cert]
with pytest.raises(评测错误):
_新实验(env, _请求(env, refs))
assert len(env["received"]) == 5
@pytest.mark.case_id(
"NC-w25-25f007",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["来源停止阻断新调用且保留已经完成的逐例结果与费用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("stage", ["before_start", "after_complete"])
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_标定停止阻断新调用且保留已发生结果__25f007(执行环境, stage):
env = 执行环境
cert = _校准(env)
dest = _新实验(env, _请求(env, [cert]))
before = _资格执行(dest) if stage == "after_complete" else None
svc = env["app"].要求评测()
source_id = cert["result"]["experiment_id"]
svc.取消实验(env["actor"], source_id, "stop-source")
if stage == "before_start":
with pytest.raises(评测错误, match="来源已停止"):
运行测试._启动执行(dest)
assert 文学测试._读(dest)["state"] == "registered"
assert len(env["received"]) == 5
else:
after = svc.读取实验报告(env["actor"], dest["exp"]["experiment_id"])
assert after["calibration_use"]["stopped_references"] == [source_id]
assert after["samples"] == before["samples"] and after["cost"] == before["cost"]
with pytest.raises(评测错误, match="来源已停止"):
文学测试._推进(dest)
assert len(env["received"]) == 9
@pytest.mark.case_id(
"NC-w25-25f008",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["合法发送后来源停止仍保存原交付,后续消费受阻"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_已合法发送后标定停止仍能保存原交付__25f008(执行环境):
from threading import Thread
env = 执行环境
cert = _校准(env)
dest = _新实验(env, _请求(env, [cert]))
运行测试._启动执行(dest)
runtime = env["app"].任务运行
prepare = runtime.领取步骤("qualification-worker", ["eval.prepare"])
runtime.执行一步(prepare)
claim = runtime.领取步骤("qualification-worker", ["eval.call.writer"])
env["scripted"].append("inflight")
failures = []
def execute():
try:
runtime.执行一步(claim)
except Exception as error:
failures.append(error)
thread = Thread(target=execute, daemon=True)
thread.start()
try:
assert env["in_flight"].wait(5)
env["app"].要求评测().取消实验(
env["actor"], cert["result"]["experiment_id"], "stop-during-send"
)
finally:
env["respond"].set()
thread.join(timeout=8)
assert not thread.is_alive() and not failures
work = 文学测试._读(dest)
assert len([u for u in work["units"] if u["output"]]) == 1
assert len(env["received"]) == 6
with pytest.raises(评测错误, match="来源已停止"):
文学测试._推进(dest)
@pytest.mark.case_id(
"NC-w25-25f009",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["发布资源或评分标准变化拒绝继续派发旧实验"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("variant", ["resource", "rubric"])
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_资源或评分标准变化拒绝旧标定继续消费__25f009(执行环境, monkeypatch, variant):
import muse.效果评测.文学执行 as literary
from muse.任务运行.接口 import 角色策略目录
env = 执行环境
cert = _校准(env)
dest = _新实验(env, _请求(env, [cert]))
if variant == "resource":
current = 角色策略目录.从发布包()
monkeypatch.setattr(
角色策略目录,
"从发布包",
classmethod(
lambda cls: 角色策略目录(
current.定义, 资源发布身份="changed-release", 角色资源=current.角色资源
)
),
)
else:
read = literary.读取评分标准
monkeypatch.setattr(
literary, "读取评分标准", lambda scenario: {**read(scenario), "version": "changed"}
)
with pytest.raises(评测错误, match="资源|评分标准"):
运行测试._启动执行(dest)
assert 文学测试._读(dest)["state"] == "registered"
assert len(env["received"]) == 5
@pytest.mark.case_id(
"NC-w25-25f00a",
environment="隔离PG与合成HTTP",
given="已发布隔离资料、固定评委配置与本例真实标定执行",
when="经公开资格入口、S02派发及报告读取",
then=["允许别名范围内的实际模型变化也不得沿旧标定保存有效比较"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{**参数, "allow_model_alias": True}], indirect=True)
def test_实际模型漂移不能沿用配置名消费标定__25f00a(执行环境):
env = 执行环境
cert = _校准(env)
dest = _新实验(env, _请求(env, [cert]))
运行测试._启动执行(dest)
运行测试._运行就绪(dest)
文学测试._推进(dest)
env["scripted"].append("changed_model")
运行测试._运行就绪(dest, allow_failure=True)
work = 文学测试._读(dest)
failed = [u for u in work["units"] if u["kind"] == "comparison" and u["state"] == "failed"]
assert len(failed) == 1 and failed[0]["output"] is None
assert failed[0]["unregistered_deliveries"]
assert len(env["received"]) == 9