"""从实际S02交付核对判据来源;不接受留存副本自报的运行身份。""" import copy from decimal import Decimal import pytest import test_回放资料封存 as 资料测试 import test_实际效果判据 as 效果测试 import test_文学评分执行与条件第三 as 文学测试 import test_评测执行与失败收敛 as 运行测试 import test_评测语义检测 as 检测测试 from muse.任务运行.接口 import 任务状态 from muse.共享.错误 import Muse错误 from muse.效果评测.接口 import 评测错误 from muse.正式变更.接口 import 固定哈希 pytestmark = pytest.mark.数据库 执行环境 = 运行测试.执行环境 生成环境 = 资料测试.生成环境 资料环境 = 资料测试.资料环境 参数 = {**检测测试.参数, "calls": 21, "budget_calls": 100, "max_steps": 150} def _读(env): return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"]) def _完成(env): 运行测试._启动执行(env) 运行测试._运行就绪(env) env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"]) 运行测试._运行就绪(env) return _读(env) @pytest.mark.parametrize("mutation", ["empty", "wrong_role", "unbound"]) def test_留存副本缺回执错角色或输入不符均拒绝__25f501(执行环境, monkeypatch, mutation): from muse.效果评测.存储 import 评测存储 env = 执行环境 _完成(env) read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "generation": runtime = row["evidence"]["runtime"] if mutation == "empty": row["evidence"]["runtime"] = {} elif mutation == "wrong_role": runtime["processor_id"] = "eval.call.detector" else: runtime["input_metadata"]["user_input_hash"] = "f" * 64 return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(评测错误, match="交付快照"): _读(env) assert len(env["received"]) == 3 @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_命题身份与原候选派生报告不能换绑__25f503(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 检测测试._执行(env) read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "detection": result = row["evidence"]["detection"] result["report"]["assertion_verdicts"][0]["statement_id"] = "forged-id" result["report_hash"] = 固定哈希(result["report"]) return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(评测错误, match="检测报告"): _读(env) assert len(env["received"]) == 6 @pytest.mark.parametrize( "执行环境", [{**参数, "detector_output": lambda material, _: 检测测试._输出(material, "unknown")}], indirect=True, ) def test_合法未知检测进入判据而覆盖不能记满__25f504(执行环境): env = 效果测试._新实验(执行环境) result = 效果测试._完成(env) assert result["assessment"]["gate_a"] == { "status": "insufficient_evidence", "reasons": ["target_semantics_unresolved"], } report = _读(env) assert report["state"] == "completed" and not report["runtime"]["states"].get("failed") for sample in report["samples"]: assert sample["detection_complete"] counts = sample["detections"]["treatment"]["report"]["constraint_counts"] assert counts == {"pass": 0, "fail": 0, "unknown": 1} assert counts["pass"] / sum(counts.values()) == 0 def _恢复(env, unit, command): return env["app"].任务运行.控制任务( unit["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID=command ) def _重试期(env, kind, script, attempts): """只制造本阶段每个臂的真实失败;逐臂原任务恢复,不重建实验或预算。""" env["scripted"].extend([script] * 3) 运行测试._运行就绪(env, allow_failure=True) failed = [u for u in 文学测试._读(env)["units"] if u["state"] == "failed"] assert len(failed) == 3 and {u["kind"] for u in failed} == {kind} for attempt in range(1, attempts): if attempt < attempts - 1: env["scripted"].extend([script] * 3) for unit in failed: _恢复(env, unit, f"retry-{kind}-{unit['unit_id']}-{attempt}") 运行测试._运行就绪(env, allow_failure=attempt < attempts - 1) return {u["unit_id"] for u in failed} @pytest.mark.parametrize("attempts", [2, 3]) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_真实ABC写手与检测重试均绑定各臂最终交付__25f505(执行环境, 资料环境, attempts): env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=45) 运行测试._启动执行(env) writer_ids = _重试期(env, "generation", "bad_output", attempts) 文学测试._推进(env) detector_ids = _重试期(env, "detection", "bad_output", attempts) 文学测试._推进(env) work = 文学测试._读(env) assert work["state"] == "completed" for unit in work["units"]: if unit["unit_id"] not in writer_ids | detector_ids: continue calls = unit["cost"]["calls"] assert len(calls) == attempts and not unit["cost"]["has_unknown"] current = unit["evidence"]["runtime"] call_id = current["response_metadata"]["call_id"] successful = next(c for c in calls if c["call_id"] == call_id) assert current["attempt_id"] == successful["attempt_id"] with env["pool"].连接(只读=True) as conn: saved = env["app"].任务运行.列出已保存结构化交付于( conn, env["actor"].作者, unit["task_id"] ) assert [(r["call_id"], r["attempt_id"]) for r in saved] == [ (call_id, current["attempt_id"]) ] assert len({c["attempt_id"] for c in calls}) == attempts report = _读(env) assert set(report["samples"][0]["detections"]) == {"A", "B", "C"} assert not report["runtime"]["states"].get("failed") assert len(env["received"]) == 6 * attempts + 6 @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_已知费用API失败后成功可用且失败回合不抹除__25f506(执行环境): env = 执行环境 运行测试._启动执行(env) 运行测试._运行就绪(env) 文学测试._推进(env) env["scripted"].append("api_error") 运行测试._运行就绪(env, allow_failure=True) failed = next(u for u in 文学测试._读(env)["units"] if u["state"] == "failed") assert failed["kind"] == "detection" old_call = failed["cost"]["calls"][0] # 本夹具按已登记的固定每调用费率计价;终态失败也不推定免费。 assert old_call["state"] == "settled" and Decimal(old_call["actual_amount"]) == Decimal(".125") _恢复(env, failed, "known-api-retry") 运行测试._运行就绪(env) 文学测试._推进(env) report = _读(env) assert report["state"] == "completed" and not report["runtime"]["states"].get("failed") unit = next(u for u in report["samples"][0]["units"] if u["unit_id"] == failed["unit_id"]) assert len(unit["calls"]) == 2 and old_call in unit["calls"] final = next( d for d in report["samples"][0]["detections"].values() if d["unit_id"] == unit["unit_id"] ) assert final["call_id"] != old_call["call_id"] assert len(env["received"]) == 7 @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_API失败若为最后回合仍是系统失败__25f507(执行环境): env = 效果测试._新实验(执行环境, 1) 运行测试._启动执行(env) 运行测试._运行就绪(env) 文学测试._推进(env) env["scripted"].append("api_error") 运行测试._运行就绪(env, allow_failure=True) result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"]) assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]} assert not _读(env)["samples"][0]["detection_complete"] @pytest.mark.parametrize("执行环境", [{"calls": 9}], indirect=True) def test_三次写手额度耗尽恢复也不发送第四次__25f508(执行环境): env = 执行环境 env["scripted"].append("bad_output") 运行测试._启动执行(env) 运行测试._运行就绪(env, allow_failure=True) failed = next(u for u in 文学测试._读(env)["units"] if u["state"] == "failed") for attempt in (2, 3): env["scripted"].append("bad_output") _恢复(env, failed, f"quota-retry-{attempt}") 运行测试._运行就绪(env, allow_failure=True) before = len(env["received"]) _恢复(env, failed, "quota-retry-4") 运行测试._运行就绪(env, allow_failure=True) work = 文学测试._读(env) final = next(u for u in work["units"] if u["unit_id"] == failed["unit_id"]) assert final["state"] == "failed" and final["output"] is None assert len(final["cost"]["calls"]) == 3 and len(env["received"]) == before == 4 @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_不重签原哈希的越界评分不得进入判据__25f509(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 检测测试._执行(env) read = 评测存储.读取交付 changed_hashes = [] def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "comparison": original_hash = row["output_hash"] row["output"]["candidate_scores"][0]["scores"][0]["score"] = 999 changed_hashes.append((original_hash, row["output_hash"], 固定哈希(row["output"]))) return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(Muse错误, match="真实调用不一致"): _读(env) assert changed_hashes and all(a == b and b != c for a, b, c in changed_hashes) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_文学报告重签不能换掉原评委回合__25f50a(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 检测测试._执行(env) read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "comparison": row["evidence"]["runtime"]["response_metadata"]["call_id"] = "forged-panel-call" row["evidence"]["judgment"]["report_hash"] = 固定哈希( row["evidence"]["judgment"]["report"] ) return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(评测错误, match="交付快照"): _读(env) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_报告不能替换样本预注册场景标准__25f50b(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 from muse.效果评测.文学评分 import 读取评分标准 env = 执行环境 检测测试._执行(env) read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "comparison": judgment = row["evidence"]["judgment"] judgment["literary"]["rubric_hash"] = 固定哈希(读取评分标准("character_dialogue")) judgment["literary_hash"] = 固定哈希(judgment["literary"]) return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(评测错误, match="实际文本及原始模型输出"): _读(env) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_合法评分重签仍不能替代原结构化输出__25f50c(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 检测测试._执行(env) read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "comparison": row["output"]["candidate_scores"][0]["scores"][0]["score"] = 9.5 row["output_hash"] = 固定哈希(row["output"]) row["evidence"]["structured_output_hash"] = row["output_hash"] return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(Muse错误, match="真实调用不一致"): _读(env) @pytest.mark.parametrize("field", ["call_id", "structured_output_hash"]) def test_留存副本的调用与输出哈希不能污染报告__25f502(执行环境, monkeypatch, field): from muse.效果评测.存储 import 评测存储 env = 执行环境 _完成(env) read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and self.读取单元(uid)["kind"] == "comparison": if field == "call_id": row["evidence"]["runtime"]["response_metadata"]["call_id"] = "forged-call" else: row["evidence"]["structured_output_hash"] = "f" * 64 return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(评测错误, match="交付快照"): _读(env) assert len(env["received"]) == 3