muse-agent-example/tests/单元/test_评测比较与启用判据.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

121 lines
4.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""标定偏差的确定性合同;统计通过不授予正式启用权限。"""
import pytest
from pydantic import ValidationError
from muse.效果评测.接口 import 标定策略, 计算标定偏差, 评测错误
from muse.效果评测.标定 import 独立来源组数
def _策略(**changes):
return 标定策略.model_validate(
{
"minimum_samples": 1,
"minimum_source_groups": 1,
"max_mae": 0.5,
"max_absolute_error": 1.5,
"minimum_verdict_agreement": 1.0,
**changes,
}
)
@pytest.mark.case_id(
"TC-07954ed49dee",
environment="离线数值合同",
given="原观察的独立偏差或实际隔离实验及标注",
when="调用确定性偏差计算或真实标定服务",
then=["保留原三项偏差样例,按显式策略计算n和MAE通过。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_逐项偏差按显式策略得到样本数与合格结果__07954e():
result = 计算标定偏差([0.1, -0.2, 0.0], _策略())
assert result["passed"] and result["n"] == 3 and result["mae"] == 0.1
@pytest.mark.case_id(
"TC-2d83b2dca999",
environment="离线数值合同",
given="原观察的独立偏差或实际隔离实验及标注",
when="调用确定性偏差计算或真实标定服务",
then=["正负偏差不能抵消平均绝对误差,原失败样例仍拒绝。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_平均绝对误差不能被正负抵消__2d83b2():
result = 计算标定偏差([0.9, -0.8, 0.7], _策略())
assert not result["passed"] and result["mae"] == 0.8
@pytest.mark.case_id(
"TC-29fb4b354bf7",
environment="离线数值合同",
given="原观察的独立偏差或实际隔离实验及标注",
when="调用确定性偏差计算或真实标定服务",
then=["均差小不能掩盖单项大偏差,最大绝对偏差上限严格。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_小均差不能掩盖最大偏差且边界严格__29fb4b():
assert not 计算标定偏差([0.0, 1.6], _策略())["passed"]
assert not 计算标定偏差([0.0, 0.0, 1.5], _策略())["passed"]
@pytest.mark.case_id(
"TC-a1fbe5948ef3",
environment="离线数值合同",
given="原观察的独立偏差或实际隔离实验及标注",
when="调用确定性偏差计算或真实标定服务",
then=["空样本返回n=0且指标为空,不能是零误差合格。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_空对照保留空指标而非零误差合格__a1fbe5():
result = 计算标定偏差([], _策略())
assert result == {"n": 0, "mae": None, "mean_bias": None, "max_abs": None, "passed": False}
@pytest.mark.case_id(
"NC-w25-25e008",
environment="离线",
given="预注册策略、独立样本和本例异常输入",
when="通过公开标定接口或确定性数值合同执行",
then=["数值合同拒绝布尔、字符串和非有限值"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", [True, "0.1", float("nan"), float("inf")])
def test_偏差与策略拒绝布尔字符串和非有限值__25e008(bad):
with pytest.raises(评测错误):
计算标定偏差([bad], _策略())
with pytest.raises(ValidationError):
_策略(max_mae=bad)
@pytest.mark.case_id(
"NC-w25-25e009",
environment="离线",
given="预注册策略、独立样本和本例异常输入",
when="通过公开标定接口或确定性数值合同执行",
then=["十进制阈值不因二进制舍入误判"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_十进制阈值不受二进制舍入误判__25e009():
result = 计算标定偏差([0.1, 0.2], _策略(max_mae=0.15))
assert result["mae"] == 0.15 and result["passed"]
@pytest.mark.case_id(
"NC-w25-25e00a",
environment="离线",
given="预注册策略、独立样本和本例异常输入",
when="通过公开标定接口或确定性数值合同执行",
then=["来源标签按共享连通分量计数,多标签不虚增独立性"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_来源标签多不等于独立组多且桥接按连通分量合并__25e00a():
assert 独立来源组数([{"source_groups": ["book", "chapter", "license"]}]) == 1
assert 独立来源组数([{"source_groups": ["a"]}, {"source_groups": ["b"]}]) == 2
assert (
独立来源组数(
[{"source_groups": ["a"]}, {"source_groups": ["b"]}, {"source_groups": ["a", "b"]}]
)
== 1
)