实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
384 lines
15 KiB
Python
384 lines
15 KiB
Python
"""固定的逐例数学观察,不替代公开入口与真实S02证据的集成验证。"""
|
||
|
||
import pytest
|
||
|
||
from muse.效果评测.启用判据 import 判定效果, 读取效果标准
|
||
from muse.效果评测.文学评分 import 维度
|
||
from muse.效果评测.模型 import 评测错误
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
|
||
def _输入(count=10):
|
||
policy = 读取效果标准("writer-effect-v1")
|
||
conditions = {
|
||
"target": {
|
||
"kind": "code",
|
||
"target_ref": "fixture",
|
||
"version": "1",
|
||
"content_hash": "f" * 64,
|
||
},
|
||
"effect_policy": policy["version"],
|
||
"effect_policy_snapshot": policy,
|
||
"effect_policy_hash": 固定哈希(policy),
|
||
"evaluation_goal": "qualification",
|
||
"split": "holdout",
|
||
"literary_basis": {},
|
||
}
|
||
public, samples = [], []
|
||
for i in range(count):
|
||
sid = f"s{i}"
|
||
work = f"work-{i // 5}"
|
||
scenario = policy["scenarios"][i % 5]
|
||
conditions["literary_basis"][sid] = {"rubric": {"scenario": scenario}}
|
||
public.append(
|
||
{
|
||
"sample_id": sid,
|
||
"source_groups": [work],
|
||
"stratification": {
|
||
"work_ref": work,
|
||
"annotation_ref": "fixture",
|
||
"new_character_ratio": 0.0 if i % 5 == 0 else 0.75,
|
||
},
|
||
"input": {"original": sid, "context": {}},
|
||
}
|
||
)
|
||
literary = {}
|
||
for pair in ["A:B", "A:C", "B:C"]:
|
||
arms = pair.split(":")
|
||
literary[pair] = {
|
||
"status": "stable_report",
|
||
"execution_verified": True,
|
||
"candidates": {a: a for a in arms},
|
||
"scores": [
|
||
{
|
||
"candidate_id": a,
|
||
"dimension": dim,
|
||
"score": {"A": 7.5, "B": 2.0, "C": 8.0}[a],
|
||
}
|
||
for a in arms
|
||
for dim in 维度
|
||
],
|
||
}
|
||
samples.append(
|
||
{
|
||
"sample_id": sid,
|
||
"generation_complete": True,
|
||
"comparison_complete": True,
|
||
"detection_complete": True,
|
||
"detections": {"C": {"report": {"status": "passed"}}},
|
||
"literary": literary,
|
||
"comparisons": {"A:B": "A", "A:C": "C", "B:C": "C"},
|
||
}
|
||
)
|
||
dataset = {
|
||
"public_manifest": {"schema_version": "dataset-v1", "samples": public},
|
||
"public_hash": "dataset",
|
||
}
|
||
exp = {
|
||
"experiment_id": "effect-fixture",
|
||
"conditions": conditions,
|
||
"conditions_hash": 固定哈希(conditions),
|
||
"sample_ids": [s["sample_id"] for s in samples],
|
||
}
|
||
plan = {"primary_pair": ["A", "C"], "conditions_hash": exp["conditions_hash"]}
|
||
report = {
|
||
"experiment_id": exp["experiment_id"],
|
||
"conditions_hash": exp["conditions_hash"],
|
||
"dataset_hash": "dataset",
|
||
"plan_hash": 固定哈希(plan),
|
||
"samples": samples,
|
||
"runtime": {"states": {"completed": 100}},
|
||
"validation_modes": ["offline_contract"],
|
||
"calibration_use": {
|
||
"binding": {"version": "fixture"},
|
||
"validation_modes": ["offline_contract"],
|
||
},
|
||
"cost": {"over_budget": False, "has_unknown": False, "has_pending": False},
|
||
}
|
||
return exp, dataset, plan, report
|
||
|
||
|
||
def _下降(data, predicate, dimension):
|
||
exp, _, _, report = data
|
||
for sample in report["samples"]:
|
||
if predicate(sample, exp):
|
||
for row in sample["literary"]["A:C"]["scores"]:
|
||
if row["candidate_id"] == "C" and row["dimension"] == dimension:
|
||
row["score"] = 6.5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-24b284c3b0c3",
|
||
environment="离线,固定逐例数学合同;不认证执行来源",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="调用纯判据数学函数",
|
||
then=[
|
||
"主比较达到固定阈值时B阶段passed",
|
||
"B减A保真均值为负仍保留诊断",
|
||
"C减A保真均值0.5达到0.25阈值",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_主比较保真增益与B诊断分开__25f301():
|
||
result = 判定效果(*_输入())
|
||
assert result["gate_b"]["status"] == "passed"
|
||
assert result["metrics"]["average_deltas"]["setting_entity_fidelity"] == 0.5
|
||
assert result["metrics"]["diagnostic_deltas"]["A:B"]["setting_entity_fidelity"] == -5.5
|
||
assert result["activation_status"] == "not_evaluated"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f302",
|
||
environment="离线合同",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="核对效果数学合同",
|
||
then=["系统失败、未知费用及超预算分别先于样本不足"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("failure", ["system", "cost", "budget"])
|
||
def test_系统或费用失败先于样本不足__25f302(failure):
|
||
data = _输入(1)
|
||
if failure == "system":
|
||
data[3]["runtime"]["states"]["failed"] = 1
|
||
else:
|
||
data[3]["cost"]["has_unknown" if failure == "cost" else "over_budget"] = True
|
||
result = 判定效果(*data)
|
||
assert result["gate_a"]["status"] == result["gate_b"]["status"] == "failed"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f303",
|
||
environment="离线合同",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="核对效果数学合同",
|
||
then=["连通来源组按真实相交去重,作品标签不能增加独立来源数"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_来源同组不能靠作品标签增加独立性__25f303():
|
||
data = _输入()
|
||
for row in data[1]["public_manifest"]["samples"]:
|
||
row["source_groups"].append("shared-history")
|
||
result = 判定效果(*data)
|
||
assert result["metrics"]["independent_sources"] == 1
|
||
assert "independent_sources_insufficient" in result["gate_b"]["reasons"]
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"group",
|
||
[
|
||
pytest.param(
|
||
"stratum",
|
||
id="stratum",
|
||
marks=pytest.mark.case_id(
|
||
"TC-c9ba81e6768b",
|
||
environment="离线,固定逐例数学合同;不认证执行来源",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="调用纯判据数学函数",
|
||
then=[
|
||
"整体风格均值仍正时非空分层退化使B阶段failed",
|
||
"无新角色分组的zero_style_regression不得由总体均值掩盖",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
),
|
||
),
|
||
pytest.param(
|
||
"scenario",
|
||
id="scenario",
|
||
marks=pytest.mark.case_id(
|
||
"TC-8b1942b4d9d6",
|
||
environment="离线,固定逐例数学合同;不认证执行来源",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="调用纯判据数学函数",
|
||
then=[
|
||
"其他场景增益不能掩盖战斗退化,B阶段failed",
|
||
"battle_scenario_regression明确保留",
|
||
"战斗固定样本2",
|
||
"战斗稳定样本2",
|
||
"战斗场景执行差值-1.0",
|
||
"人物对话场景执行差值0.5",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
),
|
||
),
|
||
],
|
||
)
|
||
def test_非空分层及场景退化不能由均值隐藏__25f304(group):
|
||
data = _输入()
|
||
_下降(
|
||
data,
|
||
lambda s, e: (
|
||
e["conditions"]["literary_basis"][s["sample_id"]]["rubric"]["scenario"] == "battle"
|
||
),
|
||
"style_consistency" if group == "stratum" else "narrative_tension",
|
||
)
|
||
result = 判定效果(*data)
|
||
assert result["gate_b"]["status"] == "failed"
|
||
assert (
|
||
"zero_style_regression" if group == "stratum" else "battle_scenario_regression"
|
||
) in result["gate_b"]["reasons"]
|
||
assert (
|
||
result["metrics"]["average_deltas"][
|
||
"style_consistency" if group == "stratum" else "narrative_tension"
|
||
]
|
||
== 0.2
|
||
)
|
||
if group == "scenario":
|
||
assert result["metrics"]["scenarios"]["battle"] == {
|
||
"sample_count": 2,
|
||
"stable_count": 2,
|
||
"average_deltas": {
|
||
"setting_entity_fidelity": 0.5,
|
||
"fine_outline_fidelity": 0.5,
|
||
"style_consistency": 0.5,
|
||
"narrative_tension": -1.0,
|
||
"prose_readability": 0.5,
|
||
},
|
||
}
|
||
assert (
|
||
result["metrics"]["scenarios"]["character_dialogue"]["average_deltas"][
|
||
"narrative_tension"
|
||
]
|
||
== 0.5
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f305",
|
||
environment="离线合同",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="核对效果数学合同",
|
||
then=["无增益与零稳定评分分别返回no_gain及证据不足,未计算不填零"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_没有增益与没有分数分别保留__25f305():
|
||
data = _输入()
|
||
for sample in data[3]["samples"]:
|
||
for score in sample["literary"]["A:C"]["scores"]:
|
||
score["score"] = 7.5
|
||
assert 判定效果(*data)["gate_b"]["status"] == "no_gain"
|
||
for sample in data[3]["samples"]:
|
||
sample["literary"]["A:C"]["status"] = "unstable"
|
||
result = 判定效果(*data)
|
||
assert result["gate_b"]["status"] == "insufficient_evidence"
|
||
assert result["metrics"]["average_deltas"]["setting_entity_fidelity"] is None
|
||
assert result["metrics"]["fidelity_positive_ratio"] is None
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f307",
|
||
environment="离线合同",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="核对效果数学合同",
|
||
then=["NaN、Infinity、布尔不进入效果分数"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("value", [float("nan"), float("inf"), True])
|
||
def test_非有限分数与布尔不能被纳入效果__25f307(value):
|
||
data = _输入()
|
||
data[3]["samples"][0]["literary"]["A:C"]["scores"][0]["score"] = value
|
||
with pytest.raises(评测错误):
|
||
判定效果(*data)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f308",
|
||
environment="离线合同",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="核对效果数学合同",
|
||
then=["缺少固定样本拒绝计算,不能缩小分母制造增益"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_效果判断缺少固定样本不能缩小分母__25f308():
|
||
data = _输入()
|
||
data[3]["samples"].pop()
|
||
with pytest.raises(评测错误, match="全部固定"):
|
||
判定效果(*data)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-2271d9ca3417",
|
||
environment="离线,固定逐例数学合同;不认证执行来源",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="调用纯判据数学函数",
|
||
then=["两作品均衡五场景不报告单作品、作品未定或场景不足偏差"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_两作品完整五场景无选择偏差__25f30b():
|
||
result = 判定效果(*_输入())
|
||
assert not set(result["confounders"]) & {
|
||
"single_work",
|
||
"work_identity_unresolved",
|
||
"scenario_coverage_incomplete",
|
||
}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-bd7218f796d0",
|
||
environment="离线,固定逐例数学合同;不认证执行来源",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="调用纯判据数学函数",
|
||
then=["第五例重复战斗仍报告场景覆盖不完整", "固定五场景中唯独returning_character缺失"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_五例重复场景仍须指出缺少的指定场景__25f30c():
|
||
data = _输入(5)
|
||
exp, _, plan, report = data
|
||
exp["conditions"]["literary_basis"]["s4"]["rubric"]["scenario"] = "battle"
|
||
exp["conditions_hash"] = 固定哈希(exp["conditions"])
|
||
plan["conditions_hash"] = report["conditions_hash"] = exp["conditions_hash"]
|
||
report["plan_hash"] = 固定哈希(plan)
|
||
result = 判定效果(*data)
|
||
assert result["gate_a"]["status"] == "insufficient_evidence"
|
||
assert "scenario_coverage_incomplete" in result["confounders"]
|
||
assert [
|
||
name for name, group in result["metrics"]["scenarios"].items() if group["sample_count"] == 0
|
||
] == ["returning_character"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f30d",
|
||
environment="离线合同",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="核对效果数学合同",
|
||
then=["空分层保持旧数据集哈希,角色比例及标注来源严格校验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_空分层不改变旧数据集字节而声明须有来源__25f30d():
|
||
from pydantic import ValidationError
|
||
|
||
from muse.效果评测.接口 import 冻结数据集, 数据集发布
|
||
|
||
raw = {
|
||
"dataset_id": "legacy",
|
||
"revision": 1,
|
||
"samples": [
|
||
{
|
||
"sample_id": "s",
|
||
"source_ref": "synthetic:s",
|
||
"license_ref": "fixture",
|
||
"source_groups": ["work"],
|
||
"split": "holdout",
|
||
"input": {"instruction": "写一段。", "original": "原文", "context": {}},
|
||
"answer": {},
|
||
}
|
||
],
|
||
}
|
||
fixed = 冻结数据集(数据集发布.model_validate(raw))
|
||
public = {
|
||
"schema_version": "dataset-v1",
|
||
"dataset_id": "legacy",
|
||
"revision": 1,
|
||
"samples": [{k: v for k, v in raw["samples"][0].items() if k != "answer"}],
|
||
}
|
||
assert fixed["public_hash"] == 固定哈希(public)
|
||
raw["samples"][0]["stratification"] = {
|
||
"work_ref": "work",
|
||
"annotation_ref": "source",
|
||
"new_character_ratio": True,
|
||
}
|
||
with pytest.raises(ValidationError):
|
||
数据集发布.model_validate(raw)
|
||
raw["samples"][0]["stratification"] = {"work_ref": "work", "annotation_ref": " "}
|
||
with pytest.raises(评测错误, match="标注来源"):
|
||
冻结数据集(数据集发布.model_validate(raw))
|