实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
472 lines
19 KiB
Python
472 lines
19 KiB
Python
"""资格评测消费真实标定回执;合成HTTP只验证管线,不证明模型文学能力。"""
|
||
|
||
import copy
|
||
from dataclasses import replace
|
||
|
||
import pytest
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_标定金标准与凭据 as 标定测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.效果评测.接口 import (
|
||
公开评测输入,
|
||
实验请求,
|
||
数据样本,
|
||
数据集发布,
|
||
标定使用,
|
||
评测服务,
|
||
评测错误,
|
||
)
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
参数 = 标定测试.参数
|
||
|
||
|
||
def _校准(env, *, full_panel=True, biased=False, failed=False):
|
||
标定测试._生成(env)
|
||
raw = 标定测试._请求(env)
|
||
score = 7.0 if failed else (7.5 if full_panel else 8.0)
|
||
for row in raw["annotations"]:
|
||
row["scores"] = {d: score for d in 文学测试.维度}
|
||
标定测试._发布(env, raw)
|
||
文学测试._推进(env)
|
||
if full_panel:
|
||
env["scripted"].extend(
|
||
["calibration_low", "calibration_high"] if biased else ["ok", "calibration_gap"]
|
||
)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
return 标定测试._封存(env)
|
||
|
||
|
||
def _请求(
|
||
env, certificates, *, group="qualification-book", original="另一部作品的资格底稿。", basis=None
|
||
):
|
||
if basis is None:
|
||
basis = {
|
||
**文学测试.依据,
|
||
"sources": [
|
||
{"source_id": "outline", "kind": "fine_outline", "text": "旅者在山口等待同伴。"},
|
||
{
|
||
"source_id": "history",
|
||
"kind": "historical_prose",
|
||
"text": "前卷写过雨后的山道。",
|
||
},
|
||
],
|
||
}
|
||
source = 数据集发布(
|
||
dataset_id="qualification-fixture",
|
||
revision=1,
|
||
samples=(
|
||
数据样本(
|
||
sample_id="qualification-sample",
|
||
source_ref="synthetic:qualification",
|
||
license_ref="synthetic:owned-fixture",
|
||
source_groups=(group,),
|
||
split="holdout",
|
||
input=公开评测输入(instruction="比较两篇合成叙述。", original=original, context={}),
|
||
answer={"judging_basis": basis},
|
||
),
|
||
),
|
||
)
|
||
data = 评测服务(env["pools"][用途.维护]).发布数据集(
|
||
replace(env["actor"], 用途=用途.维护), source
|
||
)
|
||
return 实验请求.model_validate(
|
||
{
|
||
**env["request"].model_dump(mode="json"),
|
||
"dataset_version_id": data["version_id"],
|
||
"dataset_hash": data["public_hash"],
|
||
"split": "holdout",
|
||
"calibration_policy": None,
|
||
"evaluation_goal": "qualification",
|
||
"calibration_use": {
|
||
"policy": 标定测试.策略,
|
||
"references": [
|
||
{
|
||
"experiment_id": c["result"]["experiment_id"],
|
||
"receipt_hash": c["receipt_hash"],
|
||
}
|
||
for c in certificates
|
||
],
|
||
},
|
||
}
|
||
)
|
||
|
||
|
||
def _新实验(env, req):
|
||
exp = env["app"].要求评测().创建实验(env["actor"], "qualification-command", req)
|
||
return {**env, "exp": exp, "request": req}
|
||
|
||
|
||
def _资格执行(env):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f001",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["三位实际评委完整标定进入资格派发元数据和报告,模型输入不含回执;启用仍未评定"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_完整实际标定约束资格派发并进入读回__25f001(执行环境):
|
||
env = 执行环境
|
||
certificate = _校准(env)
|
||
assert certificate["result"]["status"] == "passed"
|
||
dest = _新实验(env, _请求(env, [certificate]))
|
||
report = _资格执行(dest)
|
||
binding = dest["exp"]["conditions"]["calibration_binding"]
|
||
assert len(binding["reviewers"]) == 3
|
||
assert report["state"] == "completed" and len(env["received"]) == 9
|
||
assert report["evaluation_goal"] == "qualification"
|
||
assert report["calibration_use"]["binding"] == binding
|
||
assert report["activation_status"] == "not_evaluated"
|
||
for unit in 文学测试._读(dest)["units"]:
|
||
if unit["evidence"]:
|
||
assert unit["evidence"]["runtime"]["dispatch_metadata"][
|
||
"calibration_use_hash"
|
||
] == 固定哈希(binding)
|
||
assert all(certificate["receipt_hash"] not in r["input"] for r in env["received"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-75676077ad55",
|
||
environment="隔离PG、真实S02及合成HTTP",
|
||
given="真实隔离标定、原金标准与资格请求",
|
||
when="经公开实验创建入口请求资格消费",
|
||
then=["原缺stamp守卫由资格实验缺少明确标定引用拒绝承接,在任何资格调用前拒绝。"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_缺失标定凭据在资格调用前拒绝__756760(执行环境):
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
req = _请求(env, [cert]).model_copy(update={"calibration_use": None})
|
||
with pytest.raises(评测错误, match="标定"):
|
||
_新实验(env, req)
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-1c0f450f8945",
|
||
environment="隔离PG、真实S02及合成HTTP",
|
||
given="真实隔离标定、原金标准与资格请求",
|
||
when="经公开实验创建入口请求资格消费",
|
||
then=["原标准指纹不符由真实策略及场景评分标准变化拒绝承接,旧回执不能授权新尺度。"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
@pytest.mark.parametrize("variant", ["policy", "rubric"])
|
||
def test_评分策略改变不能沿用旧标定__1c0f45(执行环境, monkeypatch, variant):
|
||
import muse.效果评测.文学执行 as literary
|
||
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
req = _请求(env, [cert])
|
||
if variant == "policy":
|
||
use = req.calibration_use.model_dump(mode="json")
|
||
use["policy"]["max_mae"] = 0.4
|
||
req = req.model_copy(update={"calibration_use": 标定使用.model_validate(use)})
|
||
else:
|
||
read = literary.读取评分标准
|
||
monkeypatch.setattr(
|
||
literary, "读取评分标准", lambda scenario: {**read(scenario), "version": "changed"}
|
||
)
|
||
with pytest.raises(评测错误, match="策略指纹|完整实际标定"):
|
||
_新实验(env, req)
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-a2621d936e34",
|
||
environment="隔离PG、真实S02及合成HTTP",
|
||
given="真实隔离标定、原金标准与资格请求",
|
||
when="经公开实验创建入口请求资格消费",
|
||
then=["真实标定失败凭据保留;资格消费不能以有记录代替合格,也不发起后续调用。"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_失败标定不能授权资格调用__a2621d(执行环境):
|
||
env = 执行环境
|
||
cert = _校准(env, failed=True)
|
||
assert cert["result"]["status"] == "failed"
|
||
with pytest.raises(评测错误, match="合格凭据"):
|
||
_新实验(env, _请求(env, [cert]))
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f002",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["稳定双评未调用的第三评委不算标定覆盖"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_未调用的第三评委不能算已标定__25f002(执行环境):
|
||
env = 执行环境
|
||
cert = _校准(env, full_panel=False)
|
||
assert cert["result"]["status"] == "passed"
|
||
with pytest.raises(评测错误, match="未执行候补"):
|
||
_新实验(env, _请求(env, [cert]))
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f003",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["整体MAE合格不能掩盖单个评委失败"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_整体均差不能掩盖单个评委不合格__25f003(执行环境):
|
||
env = 执行环境
|
||
cert = _校准(env, biased=True)
|
||
assert cert["result"]["status"] == "passed"
|
||
assert cert["result"]["metrics"]["mae"] == 0.5
|
||
with pytest.raises(评测错误, match="整体均值"):
|
||
_新实验(env, _请求(env, [cert]))
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f004",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["留出集不能换数据集或来源标签复用标定组、原文或共同背景"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("variant", ["group", "background", "input"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_留出集不能换标识复用标定原文与共同背景__25f004(执行环境, variant):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
source = 评测存储(conn).读取数据集(str(env["request"].dataset_version_id))
|
||
sample = source["public_manifest"]["samples"][0]
|
||
kwargs = {
|
||
"group": {"group": sample["source_groups"][0]},
|
||
"input": {"original": sample["input"]["original"]},
|
||
"background": {"basis": copy.deepcopy(文学测试.依据)},
|
||
}[variant]
|
||
with pytest.raises(评测错误, match="交叉|重复"):
|
||
_新实验(env, _请求(env, [cert], **kwargs))
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f005",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["多份完整标定承接各自实际参与评委,不虚构候补参与"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_多份独立标定只承接实际参与的评委__25f005(执行环境):
|
||
env = 执行环境
|
||
first = _校准(env, full_panel=False)
|
||
req = env["request"].model_copy(
|
||
update={
|
||
"judges": (env["request"].judges[0], env["request"].arbitrator),
|
||
"arbitrator": env["request"].judges[1],
|
||
}
|
||
)
|
||
exp = env["app"].要求评测().创建实验(env["actor"], "calibrate-other-panel", req)
|
||
second = _校准({**env, "exp": exp, "request": req}, full_panel=False)
|
||
dest = _新实验(env, _请求(env, [first, second]))
|
||
report = _资格执行(dest)
|
||
assert report["state"] == "completed" and len(env["received"]) == 12
|
||
binding = report["calibration_use"]["binding"]
|
||
assert len(binding["references"]) == 2 and len(binding["reviewers"]) == 3
|
||
assert {r["experiment_id"] for r in binding["reviewers"].values()} == {
|
||
first["result"]["experiment_id"],
|
||
second["result"]["experiment_id"],
|
||
}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f006",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["重复引用、错误回执哈希及未封存记录拒绝资格消费"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("variant", ["duplicate", "hash", "unsealed"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_重复引用错哈希或未封存不能替代标定__25f006(执行环境, variant, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
if variant == "unsealed":
|
||
monkeypatch.setattr(评测存储, "读取标定凭据", lambda *args: None)
|
||
if variant == "hash":
|
||
cert = {**cert, "receipt_hash": "0" * 64}
|
||
refs = [cert, cert] if variant == "duplicate" else [cert]
|
||
with pytest.raises(评测错误):
|
||
_新实验(env, _请求(env, refs))
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f007",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["来源停止阻断新调用且保留已经完成的逐例结果与费用"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("stage", ["before_start", "after_complete"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_标定停止阻断新调用且保留已发生结果__25f007(执行环境, stage):
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
dest = _新实验(env, _请求(env, [cert]))
|
||
before = _资格执行(dest) if stage == "after_complete" else None
|
||
svc = env["app"].要求评测()
|
||
source_id = cert["result"]["experiment_id"]
|
||
svc.取消实验(env["actor"], source_id, "stop-source")
|
||
if stage == "before_start":
|
||
with pytest.raises(评测错误, match="来源已停止"):
|
||
运行测试._启动执行(dest)
|
||
assert 文学测试._读(dest)["state"] == "registered"
|
||
assert len(env["received"]) == 5
|
||
else:
|
||
after = svc.读取实验报告(env["actor"], dest["exp"]["experiment_id"])
|
||
assert after["calibration_use"]["stopped_references"] == [source_id]
|
||
assert after["samples"] == before["samples"] and after["cost"] == before["cost"]
|
||
with pytest.raises(评测错误, match="来源已停止"):
|
||
文学测试._推进(dest)
|
||
assert len(env["received"]) == 9
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f008",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["合法发送后来源停止仍保存原交付,后续消费受阻"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_已合法发送后标定停止仍能保存原交付__25f008(执行环境):
|
||
from threading import Thread
|
||
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
dest = _新实验(env, _请求(env, [cert]))
|
||
运行测试._启动执行(dest)
|
||
runtime = env["app"].任务运行
|
||
prepare = runtime.领取步骤("qualification-worker", ["eval.prepare"])
|
||
runtime.执行一步(prepare)
|
||
claim = runtime.领取步骤("qualification-worker", ["eval.call.writer"])
|
||
env["scripted"].append("inflight")
|
||
failures = []
|
||
|
||
def execute():
|
||
try:
|
||
runtime.执行一步(claim)
|
||
except Exception as error:
|
||
failures.append(error)
|
||
|
||
thread = Thread(target=execute, daemon=True)
|
||
thread.start()
|
||
try:
|
||
assert env["in_flight"].wait(5)
|
||
env["app"].要求评测().取消实验(
|
||
env["actor"], cert["result"]["experiment_id"], "stop-during-send"
|
||
)
|
||
finally:
|
||
env["respond"].set()
|
||
thread.join(timeout=8)
|
||
assert not thread.is_alive() and not failures
|
||
work = 文学测试._读(dest)
|
||
assert len([u for u in work["units"] if u["output"]]) == 1
|
||
assert len(env["received"]) == 6
|
||
with pytest.raises(评测错误, match="来源已停止"):
|
||
文学测试._推进(dest)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f009",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["发布资源或评分标准变化拒绝继续派发旧实验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("variant", ["resource", "rubric"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_资源或评分标准变化拒绝旧标定继续消费__25f009(执行环境, monkeypatch, variant):
|
||
import muse.效果评测.文学执行 as literary
|
||
from muse.任务运行.接口 import 角色策略目录
|
||
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
dest = _新实验(env, _请求(env, [cert]))
|
||
if variant == "resource":
|
||
current = 角色策略目录.从发布包()
|
||
monkeypatch.setattr(
|
||
角色策略目录,
|
||
"从发布包",
|
||
classmethod(
|
||
lambda cls: 角色策略目录(
|
||
current.定义, 资源发布身份="changed-release", 角色资源=current.角色资源
|
||
)
|
||
),
|
||
)
|
||
else:
|
||
read = literary.读取评分标准
|
||
monkeypatch.setattr(
|
||
literary, "读取评分标准", lambda scenario: {**read(scenario), "version": "changed"}
|
||
)
|
||
with pytest.raises(评测错误, match="资源|评分标准"):
|
||
运行测试._启动执行(dest)
|
||
assert 文学测试._读(dest)["state"] == "registered"
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f00a",
|
||
environment="隔离PG与合成HTTP",
|
||
given="已发布隔离资料、固定评委配置与本例真实标定执行",
|
||
when="经公开资格入口、S02派发及报告读取",
|
||
then=["允许别名范围内的实际模型变化也不得沿旧标定保存有效比较"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{**参数, "allow_model_alias": True}], indirect=True)
|
||
def test_实际模型漂移不能沿用配置名消费标定__25f00a(执行环境):
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
dest = _新实验(env, _请求(env, [cert]))
|
||
运行测试._启动执行(dest)
|
||
运行测试._运行就绪(dest)
|
||
文学测试._推进(dest)
|
||
env["scripted"].append("changed_model")
|
||
运行测试._运行就绪(dest, allow_failure=True)
|
||
work = 文学测试._读(dest)
|
||
failed = [u for u in work["units"] if u["kind"] == "comparison" and u["state"] == "failed"]
|
||
assert len(failed) == 1 and failed[0]["output"] is None
|
||
assert failed[0]["unregistered_deliveries"]
|
||
assert len(env["received"]) == 9
|