muse-agent-example/tests/集成/test_判据来源与执行.py
zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

330 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""从实际S02交付核对判据来源;不接受留存副本自报的运行身份。"""
import copy
from decimal import Decimal
import pytest
import test_回放资料封存 as 资料测试
import test_实际效果判据 as 效果测试
import test_文学评分执行与条件第三 as 文学测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
from muse.任务运行.接口 import 任务状态
from muse.共享.错误 import Muse错误
from muse.效果评测.接口 import 评测错误
from muse.正式变更.接口 import 固定哈希
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
生成环境 = 资料测试.生成环境
资料环境 = 资料测试.资料环境
参数 = {**检测测试.参数, "calls": 21, "budget_calls": 100, "max_steps": 150}
def _读(env):
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
def _完成(env):
运行测试._启动执行(env)
运行测试._运行就绪(env)
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
运行测试._运行就绪(env)
return _读(env)
@pytest.mark.parametrize("mutation", ["empty", "wrong_role", "unbound"])
def test_留存副本缺回执错角色或输入不符均拒绝__25f501(执行环境, monkeypatch, mutation):
from muse.效果评测.存储 import 评测存储
env = 执行环境
_完成(env)
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "generation":
runtime = row["evidence"]["runtime"]
if mutation == "empty":
row["evidence"]["runtime"] = {}
elif mutation == "wrong_role":
runtime["processor_id"] = "eval.call.detector"
else:
runtime["input_metadata"]["user_input_hash"] = "f" * 64
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(评测错误, match="交付快照"):
_读(env)
assert len(env["received"]) == 3
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_命题身份与原候选派生报告不能换绑__25f503(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
检测测试._执行(env)
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "detection":
result = row["evidence"]["detection"]
result["report"]["assertion_verdicts"][0]["statement_id"] = "forged-id"
result["report_hash"] = 固定哈希(result["report"])
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(评测错误, match="检测报告"):
_读(env)
assert len(env["received"]) == 6
@pytest.mark.parametrize(
"执行环境",
[{**参数, "detector_output": lambda material, _: 检测测试._输出(material, "unknown")}],
indirect=True,
)
def test_合法未知检测进入判据而覆盖不能记满__25f504(执行环境):
env = 效果测试._新实验(执行环境)
result = 效果测试._完成(env)
assert result["assessment"]["gate_a"] == {
"status": "insufficient_evidence",
"reasons": ["target_semantics_unresolved"],
}
report = _读(env)
assert report["state"] == "completed" and not report["runtime"]["states"].get("failed")
for sample in report["samples"]:
assert sample["detection_complete"]
counts = sample["detections"]["treatment"]["report"]["constraint_counts"]
assert counts == {"pass": 0, "fail": 0, "unknown": 1}
assert counts["pass"] / sum(counts.values()) == 0
def _恢复(env, unit, command):
return env["app"].任务运行.控制任务(
unit["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID=command
)
def _重试期(env, kind, script, attempts):
"""只制造本阶段每个臂的真实失败;逐臂原任务恢复,不重建实验或预算。"""
env["scripted"].extend([script] * 3)
运行测试._运行就绪(env, allow_failure=True)
failed = [u for u in 文学测试._读(env)["units"] if u["state"] == "failed"]
assert len(failed) == 3 and {u["kind"] for u in failed} == {kind}
for attempt in range(1, attempts):
if attempt < attempts - 1:
env["scripted"].extend([script] * 3)
for unit in failed:
_恢复(env, unit, f"retry-{kind}-{unit['unit_id']}-{attempt}")
运行测试._运行就绪(env, allow_failure=attempt < attempts - 1)
return {u["unit_id"] for u in failed}
@pytest.mark.parametrize("attempts", [2, 3])
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_真实ABC写手与检测重试均绑定各臂最终交付__25f505(执行环境, 资料环境, attempts):
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=45)
运行测试._启动执行(env)
writer_ids = _重试期(env, "generation", "bad_output", attempts)
文学测试._推进(env)
detector_ids = _重试期(env, "detection", "bad_output", attempts)
文学测试._推进(env)
work = 文学测试._读(env)
assert work["state"] == "completed"
for unit in work["units"]:
if unit["unit_id"] not in writer_ids | detector_ids:
continue
calls = unit["cost"]["calls"]
assert len(calls) == attempts and not unit["cost"]["has_unknown"]
current = unit["evidence"]["runtime"]
call_id = current["response_metadata"]["call_id"]
successful = next(c for c in calls if c["call_id"] == call_id)
assert current["attempt_id"] == successful["attempt_id"]
with env["pool"].连接(只读=True) as conn:
saved = env["app"].任务运行.列出已保存结构化交付于(
conn, env["actor"].作者, unit["task_id"]
)
assert [(r["call_id"], r["attempt_id"]) for r in saved] == [
(call_id, current["attempt_id"])
]
assert len({c["attempt_id"] for c in calls}) == attempts
report = _读(env)
assert set(report["samples"][0]["detections"]) == {"A", "B", "C"}
assert not report["runtime"]["states"].get("failed")
assert len(env["received"]) == 6 * attempts + 6
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_已知费用API失败后成功可用且失败回合不抹除__25f506(执行环境):
env = 执行环境
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
env["scripted"].append("api_error")
运行测试._运行就绪(env, allow_failure=True)
failed = next(u for u in 文学测试._读(env)["units"] if u["state"] == "failed")
assert failed["kind"] == "detection"
old_call = failed["cost"]["calls"][0]
# 本夹具按已登记的固定每调用费率计价;终态失败也不推定免费。
assert old_call["state"] == "settled" and Decimal(old_call["actual_amount"]) == Decimal(".125")
_恢复(env, failed, "known-api-retry")
运行测试._运行就绪(env)
文学测试._推进(env)
report = _读(env)
assert report["state"] == "completed" and not report["runtime"]["states"].get("failed")
unit = next(u for u in report["samples"][0]["units"] if u["unit_id"] == failed["unit_id"])
assert len(unit["calls"]) == 2 and old_call in unit["calls"]
final = next(
d for d in report["samples"][0]["detections"].values() if d["unit_id"] == unit["unit_id"]
)
assert final["call_id"] != old_call["call_id"]
assert len(env["received"]) == 7
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_API失败若为最后回合仍是系统失败__25f507(执行环境):
env = 效果测试._新实验(执行环境, 1)
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
env["scripted"].append("api_error")
运行测试._运行就绪(env, allow_failure=True)
result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]}
assert not _读(env)["samples"][0]["detection_complete"]
@pytest.mark.parametrize("执行环境", [{"calls": 9}], indirect=True)
def test_三次写手额度耗尽恢复也不发送第四次__25f508(执行环境):
env = 执行环境
env["scripted"].append("bad_output")
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
failed = next(u for u in 文学测试._读(env)["units"] if u["state"] == "failed")
for attempt in (2, 3):
env["scripted"].append("bad_output")
_恢复(env, failed, f"quota-retry-{attempt}")
运行测试._运行就绪(env, allow_failure=True)
before = len(env["received"])
_恢复(env, failed, "quota-retry-4")
运行测试._运行就绪(env, allow_failure=True)
work = 文学测试._读(env)
final = next(u for u in work["units"] if u["unit_id"] == failed["unit_id"])
assert final["state"] == "failed" and final["output"] is None
assert len(final["cost"]["calls"]) == 3 and len(env["received"]) == before == 4
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_不重签原哈希的越界评分不得进入判据__25f509(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
检测测试._执行(env)
read = 评测存储.读取交付
changed_hashes = []
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "comparison":
original_hash = row["output_hash"]
row["output"]["candidate_scores"][0]["scores"][0]["score"] = 999
changed_hashes.append((original_hash, row["output_hash"], 固定哈希(row["output"])))
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(Muse错误, match="真实调用不一致"):
_读(env)
assert changed_hashes and all(a == b and b != c for a, b, c in changed_hashes)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_文学报告重签不能换掉原评委回合__25f50a(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
检测测试._执行(env)
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "comparison":
row["evidence"]["runtime"]["response_metadata"]["call_id"] = "forged-panel-call"
row["evidence"]["judgment"]["report_hash"] = 固定哈希(
row["evidence"]["judgment"]["report"]
)
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(评测错误, match="交付快照"):
_读(env)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_报告不能替换样本预注册场景标准__25f50b(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
from muse.效果评测.文学评分 import 读取评分标准
env = 执行环境
检测测试._执行(env)
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "comparison":
judgment = row["evidence"]["judgment"]
judgment["literary"]["rubric_hash"] = 固定哈希(读取评分标准("character_dialogue"))
judgment["literary_hash"] = 固定哈希(judgment["literary"])
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(评测错误, match="实际文本及原始模型输出"):
_读(env)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_合法评分重签仍不能替代原结构化输出__25f50c(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
检测测试._执行(env)
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "comparison":
row["output"]["candidate_scores"][0]["scores"][0]["score"] = 9.5
row["output_hash"] = 固定哈希(row["output"])
row["evidence"]["structured_output_hash"] = row["output_hash"]
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(Muse错误, match="真实调用不一致"):
_读(env)
@pytest.mark.parametrize("field", ["call_id", "structured_output_hash"])
def test_留存副本的调用与输出哈希不能污染报告__25f502(执行环境, monkeypatch, field):
from muse.效果评测.存储 import 评测存储
env = 执行环境
_完成(env)
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and self.读取单元(uid)["kind"] == "comparison":
if field == "call_id":
row["evidence"]["runtime"]["response_metadata"]["call_id"] = "forged-call"
else:
row["evidence"]["structured_output_hash"] = "f" * 64
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(评测错误, match="交付快照"):
_读(env)
assert len(env["received"]) == 3