muse-agent-example/tests/契约/test_用途资格与效果依赖.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

634 lines
23 KiB
Python

"""目标、独立模型身份与安装闭包反例;合成分数不作为文学标定。"""
import copy
from types import SimpleNamespace
from uuid import UUID
import pytest
from test_效果判据 import _输入
from muse.任务运行.接口 import 角色策略目录
from muse.效果评测 import 实验条件 as 条件模块
from muse.效果评测 import 效果依赖 as 依赖模块
from muse.效果评测.启用判据 import 判定效果, 读取效果标准
from muse.效果评测.实验条件 import 实验请求
from muse.效果评测.文学评分 import 维度
from muse.效果评测.模型 import 评测错误
from muse.效果评测.目标政策 import 核对目标范围
from muse.效果评测.配置预检 import 要求文学配置, 预检文学配置
from muse.正式变更.接口 import 固定哈希
def _policy(models, aliases=None):
original = 角色策略目录.从发布包()
data = copy.deepcopy(original.定义)
data["models"]["judge-fixed"] = list(models)
data["models"]["fixed"] = ["writer"]
data["actual_model_ids"] = aliases or {}
return 角色策略目录(data, 资源发布身份="test-build")
@pytest.mark.case_id("NC-O03-PREFLIGHT-DEFAULT")
def test_现有两模型白名单在读取评委配置之前拒绝冻结(monkeypatch):
calls = []
version = SimpleNamespace(
内容哈希="a" * 64,
内容=SimpleNamespace(
角色配置={"writer": {"provider": "fixture", "model": "writer", "thinking": None}},
资源发布身份="test-build",
角色策略版本="test-policy",
宿主="direct",
宿主版本="1",
计价版本="fixture",
预算策略引用="fixture",
),
)
class Configs:
def 读取版本(self, id_, version_):
calls.append(id_)
assert id_ == "writer", "不可满足时不读取候补配置,更不得创建或派发任务"
return version
monkeypatch.setattr(条件模块, "配置版本管理", lambda _: Configs())
policy = _policy(["judge-a", "judge-b"])
monkeypatch.setattr(角色策略目录, "从发布包", classmethod(lambda _: policy))
request = 实验请求.model_validate(
{
"dataset_version_id": str(UUID(int=1)),
"dataset_hash": "a" * 64,
"split": "holdout",
"target": {
"kind": "prompt",
"target_ref": "B05.generate",
"version": "1",
"content_hash": "b" * 64,
},
"generator_role": "writer",
"generator": {"config_id": "writer", "version": "1"},
"judges": [{"config_id": "j1", "version": "1"}, {"config_id": "j2", "version": "1"}],
"arbitrator": {"config_id": "j3", "version": "1"},
"comparison_profile": "writer_rubric",
"arms": ["control", "treatment"],
"dimensions": list(维度),
"max_cost_usd": "1",
"max_calls_per_sample": 10,
}
)
with pytest.raises(评测错误, match="独立仲裁"):
条件模块.固定实验条件(object(), request)
assert calls == ["writer"]
@pytest.mark.case_id("NC-O03-PREFLIGHT-ALIASES")
@pytest.mark.parametrize(
"aliases",
[
{"judge-a": ["same"], "judge-c": ["same"]},
{"writer": ["shared"], "judge-c": ["shared"]},
],
)
def test_三名称不能把同模型别名当独立仲裁(aliases):
policy = _policy(["judge-a", "judge-b", "judge-c"], aliases)
with pytest.raises(评测错误, match="实际模型身份"):
要求文学配置(policy, generator={"model": "writer"})
@pytest.mark.case_id("NC-O03-PREFLIGHT-INDEPENDENT")
def test_独立三身份可满足但未获准候选不会加入白名单():
policy = _policy(["judge-a", "judge-b", "judge-c"])
assert 预检文学配置(policy)["satisfiable"] is True
assert (
预检文学配置(
policy,
generator={"model": "writer"},
judges=[
{"model": "judge-a"},
{"model": "judge-b"},
{"model": "glm5.3flash"},
],
)["satisfiable"]
is False
)
def _目标输入(monkeypatch, *, approved=True):
data = _输入()
exp, _, plan, report = data
policy = 读取效果标准("writer-goal-tension-v2")
# 仅该数学测试的独立固定阈值,不写生产目录或改运行凭据。
if approved:
policy["goal_policy"].update(
status="approved",
calibration_ref="synthetic-only",
minimum_gain=0.5,
minimum_positive_ratio=0.6,
)
exp["conditions"].update(
effect_policy=policy["version"],
effect_policy_snapshot=policy,
effect_policy_hash=固定哈希(policy),
application_scope={
"consumer": "B04.writer_method_v2",
"content_use": "generation",
"scenarios": policy["scenarios"],
},
)
exp["conditions_hash"] = 固定哈希(exp["conditions"])
plan["conditions_hash"] = report["conditions_hash"] = exp["conditions_hash"]
report["plan_hash"] = 固定哈希(plan)
for sample in report["samples"]:
for score in sample["literary"]["A:C"]["scores"]:
score["score"] = (
8.0
if score["candidate_id"] == "C" and score["dimension"] == "narrative_tension"
else 7.5
)
return data
@pytest.mark.case_id("NC-O03-GOAL-TENSION")
def test_具名张力收益不强求保真正增长但事实不能下降(monkeypatch):
data = _目标输入(monkeypatch)
result = 判定效果(*data)
assert result["version"] == "effect-assessment-v2"
assert result["gate_b"]["status"] == "passed"
assert result["metrics"]["fidelity_positive_ratio"] == 0
assert result["metrics"]["goal_positive_ratio"] == 1
for row in data[3]["samples"][0]["literary"]["A:C"]["scores"]:
if row["candidate_id"] == "C" and row["dimension"] == "setting_entity_fidelity":
row["score"] = 7.4
assert 判定效果(*data)["gate_b"]["status"] == "failed"
@pytest.mark.case_id("NC-O03-GOAL-DRAFT")
def test_未标定目标可诊断但不能冻结资格或通过判据(monkeypatch):
data = _目标输入(monkeypatch, approved=False)
c = data[0]["conditions"]
with pytest.raises(评测错误, match="draft"):
核对目标范围(
c["effect_policy_snapshot"]["goal_policy"], c["application_scope"], qualification=True
)
result = 判定效果(*data)
assert result["gate_b"]["status"] == "insufficient_evidence"
assert "goal_policy_draft" in result["gate_b"]["reasons"]
@pytest.mark.case_id("NC-O03-GOAL-SAMPLE-FLOOR")
def test_分目标仍保留十样本两作品五场景底线(monkeypatch):
data = _目标输入(monkeypatch)
data[3]["samples"].pop()
data[1]["public_manifest"]["samples"].pop()
data[0]["sample_ids"].pop()
result = 判定效果(*data)
assert result["gate_b"]["status"] == "insufficient_evidence"
assert "samples_insufficient" in result["gate_b"]["reasons"]
assert "per_work_samples_insufficient" in result["gate_b"]["reasons"]
def _manifest():
row = {
"roots": ["muse.fixture"],
"code": {"muse.fixture": "a" * 64},
"resources": {"prompt": "b" * 64, "策略/角色.yaml": "c" * 64},
"closed": True,
"unresolved": [],
}
return {
"构建身份": "build-1",
"资源": {"prompt": {"sha256": "b" * 64}, "策略/角色.yaml": {"sha256": "c" * 64}},
"效果依赖": {
"version": "effect-dependencies-v1",
"dependency_lock_sha256": "d" * 64,
"capabilities": {
name: copy.deepcopy(row)
for name in ["writer_generation", "literary_judging", "semantic_detection"]
},
},
}
def _conditions():
def profile(role, model):
return {
"role": role,
"provider": "fixture",
"provider_fingerprint": "e" * 64,
"model": model,
"thinking": None,
"host": "direct",
"host_version": "1",
"resource_release": "build-1",
"config_hash": "e" * 64,
}
return {
"target": {"kind": "method", "version": "1", "content_hash": "f" * 64},
"dataset_hash": "a" * 64,
"generator": profile("writer", "writer"),
"judges": [profile("judge", "j1"), profile("judge", "j2")],
"arbitrator": profile("judge", "j3"),
"detector": profile("detector", "detector"),
"effect_policy_hash": "c" * 64,
}
@pytest.mark.case_id("NC-O03-DEPENDENCY-UI")
def test_工作台变化只改变构建溯源不撤销实际效果资格(monkeypatch):
manifest = _manifest()
monkeypatch.setattr(依赖模块, "加载清单", lambda: manifest)
original = 依赖模块.冻结效果依赖(_conditions())
manifest["构建身份"] = "build-2"
manifest["资源"]["工作台/index.html"] = {"sha256": "c" * 64}
assert 依赖模块.核对效果依赖(original)
newer = 依赖模块.冻结效果依赖(_conditions())
assert newer["effect_dependency_fingerprint"] == original["effect_dependency_fingerprint"]
assert newer["origin_build_id"] != original["origin_build_id"]
@pytest.mark.case_id("NC-O03-DEPENDENCY-SEMANTIC")
@pytest.mark.parametrize(
"change", ["handler", "prompt", "model", "thinking", "projection", "policy"]
)
def test_真实效果输入改变必使指纹变化(monkeypatch, change):
manifest, conditions = _manifest(), _conditions()
monkeypatch.setattr(依赖模块, "加载清单", lambda: manifest)
old = 依赖模块.冻结效果依赖(conditions)
if change == "handler":
manifest["效果依赖"]["capabilities"]["writer_generation"]["code"]["muse.fixture"] = "9" * 64
elif change == "prompt":
manifest["资源"]["prompt"]["sha256"] = "9" * 64
for row in manifest["效果依赖"]["capabilities"].values():
row["resources"]["prompt"] = "9" * 64
elif change in {"model", "thinking"}:
conditions["generator"][change] = "different"
elif change == "projection":
conditions["target"]["content_hash"] = "9" * 64
else:
conditions["effect_policy_hash"] = "9" * 64
assert (
依赖模块.冻结效果依赖(conditions)["effect_dependency_fingerprint"]
!= old["effect_dependency_fingerprint"]
)
if change in {"handler", "prompt"}:
assert 依赖模块.核对效果依赖(old) is False
@pytest.mark.case_id("NC-O03-DEPENDENCY-INCOMPLETE")
@pytest.mark.parametrize("change", ["dynamic", "unregistered", "legacy"])
def test_动态闭包缺口及资源漏登不能伪造旧凭据的v2身份(monkeypatch, change):
manifest = _manifest()
if change == "dynamic":
manifest["效果依赖"]["capabilities"]["writer_generation"].update(
closed=False, unresolved=["dynamic"]
)
elif change == "unregistered":
manifest["资源"].pop("prompt")
else:
manifest.pop("效果依赖")
monkeypatch.setattr(依赖模块, "加载清单", lambda: manifest)
with pytest.raises(评测错误):
依赖模块.冻结效果依赖(_conditions())
@pytest.mark.case_id("NC-O03-ENABLEMENT-SCOPE")
@pytest.mark.parametrize(
"purpose,scenario,allowed",
[
("generation", "battle", True),
("review", "battle", False),
("generation", "turning_point", False),
("generation", None, False),
],
)
def test_消费端拒绝跨用途跨场景及窄范围未知场景(monkeypatch, purpose, scenario, allowed):
from muse.效果评测 import 启用凭据 as receipt
payload = {
"version": "target-enablement-v2",
"consumer": "B04.writer_method_v2",
"target": {"kind": "method"},
"method_material_hash": "fixture",
"effect_policy": "writer-goal-fidelity-v2",
"validation_modes": ["runtime"],
"application_scope": {
"consumer": "B04.writer_method_v2",
"content_use": "generation",
"scenarios": ["battle"],
},
}
row = {"payload": payload, "stopped": False, "expired": False, "current_policy": True}
monkeypatch.setattr(receipt, "读取启用凭据", lambda *a, **kw: row)
if allowed:
assert (
receipt.核对启用凭据(
None,
"author",
"receipt",
payload["target"],
"fixture",
内容用途=purpose,
场景=scenario,
)
== row
)
else:
with pytest.raises(评测错误, match="用途或场景"):
receipt.核对启用凭据(
None,
"author",
"receipt",
payload["target"],
"fixture",
内容用途=purpose,
场景=scenario,
)
@pytest.mark.case_id("NC-O03-LEGACY-RECEIPT")
def test_旧凭据跨构建继续拒绝而新凭据按冻结依赖判断(monkeypatch):
from datetime import UTC, datetime, timedelta
from muse.效果评测 import 启用凭据 as receipt
manifest = _manifest()
monkeypatch.setattr(依赖模块, "加载清单", lambda: manifest)
binding = 依赖模块.冻结效果依赖(_conditions())
policy = 读取效果标准("writer-effect-v1")
payload = {
"version": "target-enablement-v1",
"effect_policy": "writer-effect-v1",
"policy_hash": 固定哈希(policy),
"resource_release": "build-1",
"role_policy": 角色策略目录.从发布包().定义["version"],
}
row = {
"receipt_id": "fixture",
"payload": payload,
"payload_hash": "fixture",
"stopped": False,
"valid_until": datetime.now(UTC) + timedelta(days=1),
}
monkeypatch.setattr(receipt, "_读取", lambda *args: row)
monkeypatch.setattr(receipt, "加载清单", lambda: manifest)
assert receipt.读取启用凭据(None, "author", "fixture")["current_policy"] is True
manifest["构建身份"] = "build-2"
assert receipt.读取启用凭据(None, "author", "fixture")["current_policy"] is False
# 独立的新凭据格式,原记录没有被重写或冒充已转换。
modern = copy.deepcopy(payload)
modern.update(version="target-enablement-v2", effect_dependency=binding)
row["payload"] = modern
assert receipt.读取启用凭据(None, "author", "fixture")["current_policy"] is True
manifest["效果依赖"]["capabilities"]["writer_generation"]["code"]["muse.fixture"] = "9" * 64
assert receipt.读取启用凭据(None, "author", "fixture")["current_policy"] is False
assert payload["version"] == "target-enablement-v1" and "effect_dependency" not in payload
@pytest.mark.case_id("NC-O03-LEGACY-COMMAND")
def test_旧请求的幂等内容及判据没有补造v2默认字段():
request = 实验请求.model_validate(
{
"dataset_version_id": str(UUID(int=2)),
"dataset_hash": "a" * 64,
"split": "holdout",
"target": {
"kind": "prompt",
"target_ref": "B05.generate",
"version": "1",
"content_hash": "b" * 64,
},
"generator_role": "writer",
"generator": {"config_id": "writer", "version": "1"},
"judges": [{"config_id": "judge", "version": "1"}],
"arms": ["control", "treatment"],
"dimensions": ["quality"],
"max_cost_usd": "1",
"max_calls_per_sample": 3,
}
)
assert "application_scope" not in request.命令内容()
assert "effect_dependency_version" not in request.命令内容()
result = 判定效果(*_输入())
assert result["version"] == "effect-assessment-v1"
assert "goal_positive_ratio" not in result["metrics"]
@pytest.mark.case_id("NC-O03-UNSTABLE-STATES")
@pytest.mark.parametrize("status", ["invalid_unstable", "invalid_report", "needs_third_reviewer"])
def test_无效文学报告保留具体原因且不计完成(status):
from muse.效果评测.混淆项 import 汇总评委不稳定性
result = 汇总评委不稳定性({"control:treatment": {"status": status, "execution_verified": True}})
assert result["status"] == "incomplete" and result["finding_count"] is None
assert status in result["reasons"]
def _复用环境(monkeypatch):
import hashlib
from muse.任务运行.接口 import 任务状态
from muse.效果评测 import 单元复用 as reuse
from muse.效果评测 import 逐例执行 as execution
source_id, target_id = str(UUID(int=11)), str(UUID(int=12))
source_unit, target_unit = str(UUID(int=21)), str(UUID(int=22))
manifest = _manifest()
monkeypatch.setattr(依赖模块, "加载清单", lambda: manifest)
conditions = _conditions()
conditions.update(split="holdout", evaluation_goal="diagnostic")
for p in [
conditions["generator"],
*conditions["judges"],
conditions["arbitrator"],
conditions["detector"],
]:
p["role_policy"] = "fixture-policy"
conditions["effect_dependency"] = 依赖模块.冻结效果依赖(conditions)
source = {
"experiment_id": source_id,
"conditions": copy.deepcopy(conditions),
"conditions_hash": "source-conditions",
"sample_ids": ["s1"],
"dataset_version_id": "dataset",
}
target = copy.deepcopy(source)
target.update(experiment_id=target_id, conditions_hash="target-conditions")
target["conditions"]["reuse_experiments"] = [source_id]
spec = {
"unit_id": target_unit,
"sample_id": "s1",
"kind": "generation",
"arm": "treatment",
"role": "writer",
"profile": conditions["generator"],
"dependencies": [],
"user_input": "same synthetic input",
"system_prompt": "same fixed prompt",
"output_schema": {"type": "object"},
"max_output_tokens": 8192,
"timeout_seconds": 600,
}
old_spec = {**copy.deepcopy(spec), "unit_id": source_unit}
plans = {
source_id: {"target": {}, "units": [old_spec], "conditions_hash": "source-conditions"},
target_id: {"target": {}, "units": [spec], "conditions_hash": "target-conditions"},
}
units = {
source_unit: {
"unit_id": source_unit,
"experiment_id": source_id,
"task_id": "paid-task",
"specification": old_spec,
},
target_unit: {
"unit_id": target_unit,
"experiment_id": target_id,
"task_id": None,
"specification": spec,
},
}
delivered = {
"unit_id": source_unit,
"task_id": "paid-task",
"attempt_id": "attempt",
"call_id": "paid-call",
"user_input_hash": hashlib.sha256(spec["user_input"].encode()).hexdigest(),
"output": {"paragraphs": [{"text": "合成输出"}]},
"output_hash": "f" * 64,
"evidence": {
"cost_state": "settled",
"cost": "0.02",
"over_budget": False,
"model": "writer",
"validation": {"mode": "offline_contract"},
},
}
records = {}
stopped = set()
protected = []
class Store:
def 读实验(self, author, eid):
return {source_id: source, target_id: target}.get(eid) if author == "author" else None
def 锁停止保护(self, eid):
protected.append(eid)
def 读取停止(self, eid):
return eid in stopped
def 读取执行(self, eid):
return {"plan": plans[eid]}
def 读取数据集(self, did):
return {"public_hash": conditions["dataset_hash"]}
def 读取单元(self, uid):
return units.get(uid)
def 读取复用(self, uid):
return records.get(uid)
def 保存复用(self, uid, source_uid, eid, payload):
records[uid] = {
"unit_id": uid,
"source_unit_id": source_uid,
"source_experiment_id": eid,
"payload": payload,
}
store = Store()
monkeypatch.setattr(reuse, "评测存储", lambda _: store)
monkeypatch.setattr(execution, "评测存储", lambda _: store)
monkeypatch.setattr(execution, "核对条件第三", lambda *a, **k: None)
reads = []
def read(uid):
reads.append(uid)
return copy.deepcopy(delivered) if uid == source_unit else None
monkeypatch.setattr(execution, "读取核验交付", lambda conn, runtime, author, uid: read(uid))
class Runtime:
def 读取任务于(self, *args):
return SimpleNamespace(状态=任务状态.已完成)
def 创建受控任务于(self, *a, **k):
raise AssertionError("有效复用不得另建模型任务")
return SimpleNamespace(
reuse=reuse,
execution=execution,
source=source,
target=target,
spec=spec,
old_spec=old_spec,
plans=plans,
units=units,
records=records,
delivered=delivered,
stopped=stopped,
protected=protected,
read=read,
runtime=Runtime(),
)
@pytest.mark.case_id("NC-O03-REUSE-NO-DISPATCH")
def test_就绪推进显式复用原调用且不创建新任务(monkeypatch):
env = _复用环境(monkeypatch)
result = env.execution.创建就绪任务(
None, env.runtime, "author", env.target, env.plans[env.target["experiment_id"]]
)
assert result == []
record = env.records[env.spec["unit_id"]]["payload"]
assert record["source_call_id"] == "paid-call"
assert record["source_task_id"] == "paid-task"
assert env.units[env.spec["unit_id"]]["task_id"] is None
assert env.delivered["evidence"]["cost"] == "0.02"
output = env.reuse.读取复用交付(
None, env.runtime, "author", env.units[env.spec["unit_id"]], env.target, read=env.read
)
assert output["unit_id"] == env.spec["unit_id"]
assert output["reuse"] == record
assert env.source["experiment_id"] in env.protected
@pytest.mark.case_id("NC-O03-REUSE-REFUSAL")
@pytest.mark.parametrize(
"change",
[
"source_stopped",
"cost_unknown",
"output_changed",
"input_changed",
"samples_changed",
"code_changed",
],
)
def test_复用原来源停止证据或依赖改变时不能沿用(monkeypatch, change):
env = _复用环境(monkeypatch)
assert env.reuse.尝试复用单元(
None, env.runtime, "author", env.target, env.spec, [], read=env.read
)
if change == "source_stopped":
env.stopped.add(env.source["experiment_id"])
elif change == "cost_unknown":
env.delivered["evidence"]["cost_state"] = "unknown"
elif change == "output_changed":
env.delivered["output_hash"] = "9" * 64
elif change == "input_changed":
env.spec["user_input"] = "different input"
elif change == "samples_changed":
env.source["sample_ids"].append("invented-independent-sample")
else:
binding = env.target["conditions"]["effect_dependency"]
binding["closure"]["capabilities"]["writer_generation"]["code"]["muse.fixture"] = "9" * 64
binding["effect_dependency_fingerprint"] = 固定哈希(
{"inputs": binding["inputs"], "closure": binding["closure"]}
)
with pytest.raises(评测错误):
env.reuse.读取复用交付(
None, env.runtime, "author", env.units[env.spec["unit_id"]], env.target, read=env.read
)