- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
330 lines
14 KiB
Python
330 lines
14 KiB
Python
"""从实际S02交付核对判据来源;不接受留存副本自报的运行身份。"""
|
||
|
||
import copy
|
||
from decimal import Decimal
|
||
|
||
import pytest
|
||
import test_回放资料封存 as 资料测试
|
||
import test_实际效果判据 as 效果测试
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
import test_评测语义检测 as 检测测试
|
||
|
||
from muse.任务运行.接口 import 任务状态
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.效果评测.接口 import 评测错误
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
生成环境 = 资料测试.生成环境
|
||
资料环境 = 资料测试.资料环境
|
||
参数 = {**检测测试.参数, "calls": 21, "budget_calls": 100, "max_steps": 150}
|
||
|
||
|
||
def _读(env):
|
||
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _完成(env):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
|
||
运行测试._运行就绪(env)
|
||
return _读(env)
|
||
|
||
|
||
@pytest.mark.parametrize("mutation", ["empty", "wrong_role", "unbound"])
|
||
def test_留存副本缺回执错角色或输入不符均拒绝__25f501(执行环境, monkeypatch, mutation):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
_完成(env)
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "generation":
|
||
runtime = row["evidence"]["runtime"]
|
||
if mutation == "empty":
|
||
row["evidence"]["runtime"] = {}
|
||
elif mutation == "wrong_role":
|
||
runtime["processor_id"] = "eval.call.detector"
|
||
else:
|
||
runtime["input_metadata"]["user_input_hash"] = "f" * 64
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(评测错误, match="交付快照"):
|
||
_读(env)
|
||
assert len(env["received"]) == 3
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_命题身份与原候选派生报告不能换绑__25f503(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
检测测试._执行(env)
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "detection":
|
||
result = row["evidence"]["detection"]
|
||
result["report"]["assertion_verdicts"][0]["statement_id"] = "forged-id"
|
||
result["report_hash"] = 固定哈希(result["report"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(评测错误, match="检测报告"):
|
||
_读(env)
|
||
assert len(env["received"]) == 6
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"执行环境",
|
||
[{**参数, "detector_output": lambda material, _: 检测测试._输出(material, "unknown")}],
|
||
indirect=True,
|
||
)
|
||
def test_合法未知检测进入判据而覆盖不能记满__25f504(执行环境):
|
||
env = 效果测试._新实验(执行环境)
|
||
result = 效果测试._完成(env)
|
||
assert result["assessment"]["gate_a"] == {
|
||
"status": "insufficient_evidence",
|
||
"reasons": ["target_semantics_unresolved"],
|
||
}
|
||
report = _读(env)
|
||
assert report["state"] == "completed" and not report["runtime"]["states"].get("failed")
|
||
for sample in report["samples"]:
|
||
assert sample["detection_complete"]
|
||
counts = sample["detections"]["treatment"]["report"]["constraint_counts"]
|
||
assert counts == {"pass": 0, "fail": 0, "unknown": 1}
|
||
assert counts["pass"] / sum(counts.values()) == 0
|
||
|
||
|
||
def _恢复(env, unit, command):
|
||
return env["app"].任务运行.控制任务(
|
||
unit["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID=command
|
||
)
|
||
|
||
|
||
def _重试期(env, kind, script, attempts):
|
||
"""只制造本阶段每个臂的真实失败;逐臂原任务恢复,不重建实验或预算。"""
|
||
env["scripted"].extend([script] * 3)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
failed = [u for u in 文学测试._读(env)["units"] if u["state"] == "failed"]
|
||
assert len(failed) == 3 and {u["kind"] for u in failed} == {kind}
|
||
for attempt in range(1, attempts):
|
||
if attempt < attempts - 1:
|
||
env["scripted"].extend([script] * 3)
|
||
for unit in failed:
|
||
_恢复(env, unit, f"retry-{kind}-{unit['unit_id']}-{attempt}")
|
||
运行测试._运行就绪(env, allow_failure=attempt < attempts - 1)
|
||
return {u["unit_id"] for u in failed}
|
||
|
||
|
||
@pytest.mark.parametrize("attempts", [2, 3])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实ABC写手与检测重试均绑定各臂最终交付__25f505(执行环境, 资料环境, attempts):
|
||
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=45)
|
||
运行测试._启动执行(env)
|
||
writer_ids = _重试期(env, "generation", "bad_output", attempts)
|
||
文学测试._推进(env)
|
||
detector_ids = _重试期(env, "detection", "bad_output", attempts)
|
||
文学测试._推进(env)
|
||
work = 文学测试._读(env)
|
||
assert work["state"] == "completed"
|
||
for unit in work["units"]:
|
||
if unit["unit_id"] not in writer_ids | detector_ids:
|
||
continue
|
||
calls = unit["cost"]["calls"]
|
||
assert len(calls) == attempts and not unit["cost"]["has_unknown"]
|
||
current = unit["evidence"]["runtime"]
|
||
call_id = current["response_metadata"]["call_id"]
|
||
successful = next(c for c in calls if c["call_id"] == call_id)
|
||
assert current["attempt_id"] == successful["attempt_id"]
|
||
with env["pool"].连接(只读=True) as conn:
|
||
saved = env["app"].任务运行.列出已保存结构化交付于(
|
||
conn, env["actor"].作者, unit["task_id"]
|
||
)
|
||
assert [(r["call_id"], r["attempt_id"]) for r in saved] == [
|
||
(call_id, current["attempt_id"])
|
||
]
|
||
assert len({c["attempt_id"] for c in calls}) == attempts
|
||
report = _读(env)
|
||
assert set(report["samples"][0]["detections"]) == {"A", "B", "C"}
|
||
assert not report["runtime"]["states"].get("failed")
|
||
assert len(env["received"]) == 6 * attempts + 6
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_已知费用API失败后成功可用且失败回合不抹除__25f506(执行环境):
|
||
env = 执行环境
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
env["scripted"].append("api_error")
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
failed = next(u for u in 文学测试._读(env)["units"] if u["state"] == "failed")
|
||
assert failed["kind"] == "detection"
|
||
old_call = failed["cost"]["calls"][0]
|
||
# 本夹具按已登记的固定每调用费率计价;终态失败也不推定免费。
|
||
assert old_call["state"] == "settled" and Decimal(old_call["actual_amount"]) == Decimal(".125")
|
||
_恢复(env, failed, "known-api-retry")
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
report = _读(env)
|
||
assert report["state"] == "completed" and not report["runtime"]["states"].get("failed")
|
||
unit = next(u for u in report["samples"][0]["units"] if u["unit_id"] == failed["unit_id"])
|
||
assert len(unit["calls"]) == 2 and old_call in unit["calls"]
|
||
final = next(
|
||
d for d in report["samples"][0]["detections"].values() if d["unit_id"] == unit["unit_id"]
|
||
)
|
||
assert final["call_id"] != old_call["call_id"]
|
||
assert len(env["received"]) == 7
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_API失败若为最后回合仍是系统失败__25f507(执行环境):
|
||
env = 效果测试._新实验(执行环境, 1)
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
env["scripted"].append("api_error")
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]}
|
||
assert not _读(env)["samples"][0]["detection_complete"]
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [{"calls": 9}], indirect=True)
|
||
def test_三次写手额度耗尽恢复也不发送第四次__25f508(执行环境):
|
||
env = 执行环境
|
||
env["scripted"].append("bad_output")
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
failed = next(u for u in 文学测试._读(env)["units"] if u["state"] == "failed")
|
||
for attempt in (2, 3):
|
||
env["scripted"].append("bad_output")
|
||
_恢复(env, failed, f"quota-retry-{attempt}")
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
before = len(env["received"])
|
||
_恢复(env, failed, "quota-retry-4")
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
work = 文学测试._读(env)
|
||
final = next(u for u in work["units"] if u["unit_id"] == failed["unit_id"])
|
||
assert final["state"] == "failed" and final["output"] is None
|
||
assert len(final["cost"]["calls"]) == 3 and len(env["received"]) == before == 4
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_不重签原哈希的越界评分不得进入判据__25f509(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
检测测试._执行(env)
|
||
read = 评测存储.读取交付
|
||
changed_hashes = []
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "comparison":
|
||
original_hash = row["output_hash"]
|
||
row["output"]["candidate_scores"][0]["scores"][0]["score"] = 999
|
||
changed_hashes.append((original_hash, row["output_hash"], 固定哈希(row["output"])))
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(Muse错误, match="真实调用不一致"):
|
||
_读(env)
|
||
assert changed_hashes and all(a == b and b != c for a, b, c in changed_hashes)
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_文学报告重签不能换掉原评委回合__25f50a(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
检测测试._执行(env)
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "comparison":
|
||
row["evidence"]["runtime"]["response_metadata"]["call_id"] = "forged-panel-call"
|
||
row["evidence"]["judgment"]["report_hash"] = 固定哈希(
|
||
row["evidence"]["judgment"]["report"]
|
||
)
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(评测错误, match="交付快照"):
|
||
_读(env)
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_报告不能替换样本预注册场景标准__25f50b(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.效果评测.文学评分 import 读取评分标准
|
||
|
||
env = 执行环境
|
||
检测测试._执行(env)
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "comparison":
|
||
judgment = row["evidence"]["judgment"]
|
||
judgment["literary"]["rubric_hash"] = 固定哈希(读取评分标准("character_dialogue"))
|
||
judgment["literary_hash"] = 固定哈希(judgment["literary"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(评测错误, match="实际文本及原始模型输出"):
|
||
_读(env)
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_合法评分重签仍不能替代原结构化输出__25f50c(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
检测测试._执行(env)
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "comparison":
|
||
row["output"]["candidate_scores"][0]["scores"][0]["score"] = 9.5
|
||
row["output_hash"] = 固定哈希(row["output"])
|
||
row["evidence"]["structured_output_hash"] = row["output_hash"]
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(Muse错误, match="真实调用不一致"):
|
||
_读(env)
|
||
|
||
|
||
@pytest.mark.parametrize("field", ["call_id", "structured_output_hash"])
|
||
def test_留存副本的调用与输出哈希不能污染报告__25f502(执行环境, monkeypatch, field):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
_完成(env)
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and self.读取单元(uid)["kind"] == "comparison":
|
||
if field == "call_id":
|
||
row["evidence"]["runtime"]["response_metadata"]["call_id"] = "forged-call"
|
||
else:
|
||
row["evidence"]["structured_output_hash"] = "f" * 64
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(评测错误, match="交付快照"):
|
||
_读(env)
|
||
assert len(env["received"]) == 3
|