muse-agent-example/tests/集成/test_评测执行与失败收敛.py

1197 lines
50 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""评测任务实际使用S02创建、配置与预算事务;不替代后续模型逐例证据。"""
from dataclasses import replace
from datetime import UTC, datetime, timedelta
from decimal import Decimal
import pytest
import test_两阶段写手 as 写手测试
from muse.任务运行.接口 import (
任务请求,
任务预算计划,
执行计划,
步骤处理器,
步骤结果,
步骤计划,
角色策略目录,
角色预算,
配置版本管理,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 内容用途, 用途
from muse.共享.错误 import Muse错误
from muse.启动 import 构建
from muse.配置 import 应用配置
pytestmark = pytest.mark.数据库
生成环境 = 写手测试.生成环境
@pytest.mark.case_id(
"NC-w25-256001",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["真实S02按固定运行表创建并派发,评委两侧按独立oracle表绑定"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{"samples": 2}], indirect=True)
def test_真实生成按运行表且比较依赖独立盲化表__256001(执行环境):
import json
from muse.正文写作.接口 import 生成正文模板
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
ordering = env["exp"]["conditions"]["ordering"]["execution_order"]
expected = [(r["sample_id"], arm) for r in ordering for arm in r["arm_order"]]
inputs = {sid: svc.读取生成输入(env["actor"], eid, sid) for sid in env["exp"]["sample_ids"]}
运行文本 = {
json.dumps(value, ensure_ascii=False, sort_keys=True): sid for sid, value in inputs.items()
}
运行测试输出 = {}
_启动执行(env)
_运行就绪(env)
actual = []
template = 生成正文模板()[0]
for i, request in enumerate(env["received"], start=1):
arm = "treatment" if template in request["instructions"] else "control"
sid = 运行文本[request["input"]]
actual.append((sid, arm))
运行测试输出[(sid, arm)] = f"合成正文{i}。"
assert actual == expected
with env["pool"].连接(只读=True) as conn:
assignments = conn.execute(
"SELECT sample_id,arm_order FROM oracle.muse_blind_assignment WHERE experiment_id=%s",
(eid,),
).fetchall()
expected_pairs = {
(运行测试输出[(sid, arms[0])], 运行测试输出[(sid, arms[1])]) for sid, arms in assignments
}
svc.推进实验(env["actor"], eid)
_运行就绪(env)
pairs = [json.loads(r["input"]) for r in env["received"][len(expected) :]]
assert {(r["left"]["text"], r["right"]["text"]) for r in pairs} == expected_pairs
assert len(pairs) == len(assignments) == 2
@pytest.mark.case_id(
"NC-w25-256002",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["预注册表与读取映射不符时零任务、零外发"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_预注册表与读取映射不一致时零任务零调用__256002(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
from muse.效果评测.接口 import 评测错误
env = 执行环境
original = 评测存储.读分配
def 错序(self, experiment_id):
rows = original(self, experiment_id)
return [{**r, "arm_order": list(reversed(r["arm_order"]))} for r in rows]
monkeypatch.setattr(评测存储, "读分配", 错序)
with pytest.raises(评测错误, match="预注册"):
_启动执行(env)
with env["pool"].连接(只读=True) as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_experiment_execution").fetchone()[0]
== 0
)
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
assert not env["received"]
@pytest.mark.case_id(
"NC-w25-25300d",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="发送前实际请求的系统提示、输入、结构或输出上限被改写",
when="真实S02发送事务复检请求",
then=["全部在外发前拒绝,合成HTTP零请求,未保存业务交付"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("changed", ["system", "input", "schema", "limit"])
def test_实际派发请求偏离冻结单元在外发前拒绝__25300d(执行环境, monkeypatch, changed):
import muse.编排.执行评测 as flow
env = 执行环境
_启动执行(env)
original = flow.模型请求
def 改写(*args, **kwargs):
request = original(*args, **kwargs)
fields = {
"system": {"系统提示": "未经实验登记的其他提示"},
"input": {"用户输入": "未获此实验批准的其他材料"},
"schema": {"输出合同": {"type": "object"}},
"limit": {"最大输出token": request.最大输出token - 1},
}
return replace(request, **fields[changed])
monkeypatch.setattr(flow, "模型请求", 改写)
_运行就绪(env, allow_failure=True)
result = env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
assert env["received"] == []
assert all(u["output"] is None for u in result["units"])
assert all(u["state"] == "failed" for u in result["units"] if u["kind"] == "generation")
@pytest.fixture
def 原子任务环境(应用测试库):
from muse.任务运行.接口 import 内容哈希, 凭据引用, 提供方配置, 运行配置内容, 配置验证证据
pool = 应用测试库[用途.评测]
policy = 角色策略目录.从发布包()
app = 构建(应用配置(pool.引用, policy.资源发布身份, 运行用途=用途.评测))
with app.生命周期():
registry = app.流程登记
registry.登记类型("atomic-eval-fixture", 必需保护=())
registry.登记处理器(
步骤处理器(
"eval.atomic-fixture",
"1",
lambda _: 步骤结果({}),
"eval-input",
"eval-output",
角色="writer",
)
)
# 本夹具只验证创建事务,禁止冒充模型执行;后续运行使用实际评测处理器。
app.任务运行.发布计划(
执行计划(
"atomic-eval-fixture",
"1",
(
步骤计划(
"运行",
"eval.atomic-fixture",
"1",
输入合同="eval-input",
输出合同="eval-output",
角色="writer",
),
),
)
)
class 合同验证:
身份 = "atomic-fixture-offline-contract"
def 验证(self, content, purpose):
return 配置验证证据(
内容哈希(content.冻结()),
content.角色策略版本,
content.资源发布身份,
purpose,
("synthetic:atomic-task-fixture",),
"offline_contract",
)
manager = 配置版本管理(pool, 合同验证())
content = 运行配置内容(
"direct",
"1",
policy.定义["version"],
policy.资源发布身份,
"eval-atomic-budget",
{
"writer": {
"provider": "synthetic",
"model": "claude-opus-4-8[1M]",
"thinking": "high",
"tools": [],
}
},
(凭据引用("key", "环境变量", "MUSE_UNUSED_EVAL_SECRET"),),
(提供方配置("synthetic", "responses", "http://127.0.0.1:9", "key"),),
"test-price",
)
version = manager.保存草案("writer", "1", content)
receipt = manager.验证版本("writer", "1")
manager.启用(
"writer", "1", 验证回执=receipt, 批准引用="synthetic-test-approval", 预期代次=0
)
req = 任务请求(
"atomic-eval-fixture",
"atomic-command",
"eval-author",
用途.评测,
内容用途.生成,
{"sample_id": "synthetic:sample"},
policy.定义["version"],
policy.资源发布身份,
{
"source_scope": {"source_ids": ["synthetic:sample"]},
"schema_versions": {},
"authorization": {"authorized_by": "eval-author"},
"budget": {},
"stop_conditions": {},
},
)
budget = 任务预算计划(
Decimal("2"),
(角色预算("writer", 1, 2, Decimal("1")),),
"synthetic-test-approval",
datetime.now(UTC) + timedelta(minutes=10),
)
yield app, pool, manager, req, budget, version
def _创建(env, **kwargs):
app, _, _, req, budget, version = env
return app.任务运行.创建受控任务(
req,
"atomic-eval-fixture",
"1",
配置ID="writer",
预算计划=budget,
预期配置版本=kwargs.get("version", version.版本),
预期配置哈希=kwargs.get("hash", version.内容哈希),
)
@pytest.mark.case_id(
"NC-w25-253001",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="缺少真实预算账户的已启用评测配置",
when="原子创建后补账户并按原命令重试",
then=["首次任务、配置和预算全部回滚;重试只有一份绑定"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_预算前置失败不留任务配置且可原命令重试__253001(原子任务环境):
app, pool, manager, req, budget, _ = 原子任务环境
with pytest.raises(Muse错误):
_创建(原子任务环境)
assert app.任务运行.读取命令任务(req.作者, req.命令ID) is None
with pool.连接(只读=True) as conn:
for table in ("muse_task_config_binding", "muse_task_budget"):
assert conn.execute("SELECT count(*) FROM evaluation." + table).fetchone()[0] == 0
预算管理(pool, "eval-atomic-budget").登记策略(
额度策略("eval-atomic-budget", "1", Decimal("20"), 20)
)
tid = _创建(原子任务环境)
assert _创建(原子任务环境) == tid
assert manager.读取任务绑定(tid).内容哈希
with pool.连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 1
row = conn.execute(
"SELECT plan FROM evaluation.muse_task_budget WHERE task_id=%s", (tid,)
).fetchone()
assert row[0] == budget.冻结()
@pytest.mark.case_id(
"NC-w25-253002",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="实际配置版本/哈希或预算角色与请求不一致",
when="创建受控评测任务",
then=["拒绝且无可领取任务或预算残留"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("changed", ["version", "hash", "missing_budget", "wrong_budget_role"])
def test_已登记条件与实际配置或预算不一致拒绝创建__253002(原子任务环境, changed):
app, pool, _, req, budget, _ = 原子任务环境
预算管理(pool, "eval-atomic-budget").登记策略(
额度策略("eval-atomic-budget", "1", Decimal("20"), 20)
)
if changed == "missing_budget":
env = (*原子任务环境[:4], None, 原子任务环境[5])
kwargs = {}
elif changed == "wrong_budget_role":
env = (
*原子任务环境[:4],
replace(budget, 角色=(角色预算("judge", 1, 2, Decimal("1")),)),
原子任务环境[5],
)
kwargs = {}
else:
env = 原子任务环境
kwargs = {changed: "2" if changed == "version" else "0" * 64}
with pytest.raises(Muse错误):
_创建(env, **kwargs)
assert app.任务运行.读取命令任务(req.作者, req.命令ID) is None
with pool.连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task_budget").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-253003",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="真实生产生成入口引用未启用配置",
when="发起新章任务",
then=["同事务回滚,不遗留queued孤儿任务"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_生产生成配置缺失不留下可领取的孤儿任务__253003(生成环境):
from muse.编排.生成正文 import 发起生成正文
env = 生成环境
with pytest.raises(Muse错误):
发起生成正文(
env["装配"],
env["作者"],
"missing-runtime-config",
"new_chapter",
work_id="gen-work",
chapter_id="ch-3",
配置ID="not-enabled",
)
assert env["装配"].任务运行.读取命令任务(env["作者"].作者, "missing-runtime-config") is None
@pytest.fixture
def 执行环境(内置结构测试库, tmp_path, monkeypatch, request):
"""第二评委仅在隔离测试策略中存在,不修改正式白名单或认证真实模型。"""
应用测试库 = 内置结构测试库
import copy
import json
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from threading import Event, Thread
import test_生产评测权限隔离 as 数据测试
from muse.任务运行.接口 import 内容哈希, 凭据引用, 提供方配置, 运行配置内容, 配置验证证据
from muse.共享.调用身份 import 调用身份
from muse.效果评测.接口 import 实验请求, 目标版本, 评测服务, 配置选择
from muse.正文写作.接口 import 生成正文模板
from muse.编排.执行评测 import 登记评测执行
options = getattr(request, "param", {})
if options.get("output_factory"):
options = {**options, **options["output_factory"]()}
judge_count = options.get("judges", 1)
judges = ["synthetic-independent-judge"]
if judge_count == 2:
judges.append("synthetic-independent-judge-second")
if options.get("arbitrator"):
judges.append("synthetic-independent-arbitrator")
original = 角色策略目录.从发布包()
definition = copy.deepcopy(original.定义)
definition["version"] = "synthetic-evaluation-policy"
definition["models"]["fixed"].extend(judges)
definition["models"]["judge-fixed"].extend(judges)
detector_model = "synthetic-semantic-detector"
if options.get("detector"):
definition["models"]["governed"].append(detector_model)
if options.get("allow_model_alias"):
definition["actual_model_ids"] = {model: [model, model + "-changed"] for model in judges}
isolated = 角色策略目录(
definition, 资源发布身份=original.资源发布身份, 角色资源=original.角色资源
)
monkeypatch.setattr(角色策略目录, "从发布包", classmethod(lambda cls: isolated))
received, scripted = [], []
in_flight, respond = Event(), Event()
class Handler(BaseHTTPRequestHandler):
def do_POST(self):
data = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
received.append(data)
script = scripted.pop(0) if scripted else "ok"
if script == "inflight":
in_flight.set()
if not respond.wait(8):
self.send_error(504)
return
if data["model"] in judges:
material = json.loads(data["input"])
evidence = [
dict(
side=side,
quote=material[side]["text"],
)
for side in ("left", "right")
]
choice = script.removeprefix("choose_") if script.startswith("choose_") else "tie"
output = dict(
choice=choice,
rationale="合成判断保留平局",
dimensions=[
dict(
dimension=dim,
choice=choice,
rationale="合成引文覆盖两侧",
evidence=evidence,
)
for dim in material["dimensions"]
],
)
elif data["model"] == detector_model and options.get("detector"):
material = json.loads(data["input"])
output = options["detector_output"](material, script)
else:
output = {"paragraphs": [{"text": f"合成正文{len(received)}。"}]}
if options.get("generator_output"):
output = options["generator_output"](data, script)
if data["model"] in judges and options.get("judge_output"):
output = options["judge_output"](material, script)
if script == "bad_quote":
output["dimensions"][0]["evidence"][0]["quote"] = "候选正文中不存在的引文"
if script == "bad_mixed":
output["dimensions"][0]["evidence"][0]["quote"] = "候选正文中不存在的引文"
output["rationale"] = " "
if script == "bad_output":
output = {"paragraphs": []}
response = {
"id": f"synthetic-response-{len(received)}",
"model": data["model"] + "-changed" if script == "changed_model" else data["model"],
"status": "completed",
"output": [
{
"type": "message",
"content": [
{"type": "output_text", "text": json.dumps(output, ensure_ascii=False)}
],
}
],
}
if script != "unknown_cost":
response["usage"] = {"input_tokens": 5, "output_tokens": 7}
if script == "api_error":
response.update(
status="failed",
model=None,
output=[],
usage={"input_tokens": 0, "output_tokens": 0},
error={"code": "synthetic_failure", "message": "合成已知用量终态失败"},
)
payload = (
"data: "
+ json.dumps({"type": "response." + response["status"], "response": response})
+ "\n\n"
).encode()
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Content-Length", str(len(payload)))
self.end_headers()
self.wfile.write(payload)
def log_message(self, *args):
pass
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
thread = Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
pools = 应用测试库
pool = pools[用途.评测]
app = 构建(
应用配置(
pool.引用, isolated.资源发布身份, 运行用途=用途.评测, 原文暂存=str(tmp_path / "raw")
)
)
with app.生命周期():
登记评测执行(app, 写手测试.合成计价())
actor = 调用身份("eval-author", None, 用途.评测, 内容用途.检测)
samples = [数据测试._数据().samples[2]]
if options.get("calibration_policy"):
samples = [s.model_copy(update={"split": "calibration"}) for s in samples]
if options.get("samples") == 2:
samples.append(
samples[0].model_copy(
update={
"sample_id": "sample-extra",
"source_groups": ("independent-source",),
"source_ref": "synthetic:independent-source",
"input": samples[0].input.model_copy(
update={"original": "另一本合成来源的正文。"}
),
}
)
)
source = 数据测试._数据().model_copy(
update={"dataset_id": "runtime-fixture", "samples": tuple(samples)}
)
if options.get("judging_basis"):
source = source.model_copy(
update={
"samples": tuple(
s.model_copy(
update={
"answer": {
**s.answer,
"judging_basis": options["judging_basis"],
}
}
)
for s in source.samples
)
}
)
data = 评测服务(pools[用途.维护]).发布数据集(replace(actor, 用途=用途.维护), source)
class Validator:
身份 = "synthetic-evaluation-offline"
def 验证(self, c, purpose):
return 配置验证证据(
内容哈希(c.冻结()),
c.角色策略版本,
c.资源发布身份,
purpose,
("synthetic:controlled-local-http",),
# 仅隔离测试可模拟已运行的配置验证;不作为真实外部模型证据。
"runtime"
if options.get("runtime_validation_fixture")
else "offline_contract",
)
key = tmp_path / "model-key"
key.write_text("synthetic-only")
key.chmod(0o600)
manager = 配置版本管理(pool, Validator())
configs = [("writer", "writer", "claude-opus-4-8[1M]")]
if options.get("detector"):
configs.append(("detector", "detector", detector_model))
configs.extend(
(("judge", "judge-second", "arbitrator")[i], "judge", model)
for i, model in enumerate(judges)
)
for config_id, role, model in configs:
config = 运行配置内容(
"direct",
"1",
definition["version"],
isolated.资源发布身份,
"runtime-budget",
{
role: {
"provider": "synthetic",
"model": model,
"thinking": "high",
"tools": [],
}
},
(凭据引用("key", "受控存储", str(key)),),
(
提供方配置(
"synthetic",
"responses",
f"http://127.0.0.1:{server.server_port}/v1/responses",
"key",
),
),
写手测试.合成计价.版本,
)
manager.保存草案(config_id, "1", config)
receipt = manager.验证版本(config_id, "1")
manager.启用(
config_id, "1", 验证回执=receipt, 批准引用="synthetic-only", 预期代次=0
)
预算管理(pool, "runtime-budget").登记策略(
额度策略(
"runtime-budget",
"1",
Decimal(str(options.get("budget_amount", "20"))),
options.get("budget_calls", 30),
)
)
request = 实验请求(
dataset_version_id=data["version_id"],
dataset_hash=data["public_hash"],
split="calibration" if options.get("calibration_policy") else "holdout",
target=目标版本(
kind="prompt",
target_ref="B05.generate",
version=isolated.资源发布身份,
content_hash=生成正文模板()[1],
),
generator_role="writer",
generator=配置选择(config_id="writer", version="1"),
detector=配置选择(config_id="detector", version="1")
if options.get("detector")
else None,
judges=tuple(
配置选择(config_id=config_id, version="1")
for config_id, role, _ in configs
if role == "judge" and config_id != "arbitrator"
),
arms=("control", "treatment"),
dimensions=options.get("dimensions", ("清晰度",)),
max_cost_usd=Decimal("6"),
max_calls_per_sample=options.get("calls", 9),
calibration_policy=options.get("calibration_policy"),
**(
{
"comparison_profile": "writer_rubric",
"arbitrator": 配置选择(config_id="arbitrator", version="1"),
}
if options.get("arbitrator")
else {}
),
**(
{"max_judge_corrections": options["corrections"]}
if "corrections" in options
else {}
),
)
exp = app.要求评测().创建实验(actor, "runtime-fixture", request)
yield dict(
app=app,
pool=pool,
pools=pools,
actor=actor,
exp=exp,
received=received,
scripted=scripted,
in_flight=in_flight,
respond=respond,
request=request,
max_steps=options.get("max_steps", 30),
fixture_control=options.get("fixture_control", {}),
)
finally:
server.shutdown()
server.server_close()
thread.join(timeout=5)
def _启动执行(env):
from muse.效果评测.接口 import 实验执行请求
req = 实验执行请求(
approval_ref="synthetic-only", deadline=datetime.now(UTC) + timedelta(minutes=10)
)
return env["app"].要求评测().启动实验(env["actor"], env["exp"]["experiment_id"], req)
def _运行就绪(env, *, allow_failure=False):
app = env["app"]
for _ in range(env.get("max_steps", 30)):
claim = app.任务运行.领取步骤(
"eval-worker",
["eval.prepare", "eval.call.writer", "eval.call.judge", "eval.call.detector"],
)
if claim is None:
return
try:
app.任务运行.执行一步(claim)
except Muse错误:
if not allow_failure:
raise
pytest.fail("评测测试出现未收敛领取")
@pytest.mark.case_id(
"NC-w25-253004",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="固定提示词对照及仅测试有效的双模型策略",
when="实际S02调用合成HTTP、保存和独立比较",
then=["两侧输入隔离,逐例交付可回查;平局、费用和原验证模式保留"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_逐例交付和独立平局经过真实S02且生成读不到答案__253004(执行环境):
import json
env = 执行环境
app = env["app"]
svc = app.要求评测()
eid = env["exp"]["experiment_id"]
start = _启动执行(env)
assert len(start["units"]) == 3
assert sum(u["task_id"] is not None for u in start["units"]) == 2
_运行就绪(env)
svc.推进实验(env["actor"], eid)
_运行就绪(env)
result = svc.读取执行工作面(env["actor"], eid)
assert all(u["state"] == "completed" for u in result["units"])
assert len(env["received"]) == 3
judge = next(u for u in result["units"] if u["kind"] == "comparison")
assert judge["output"]["choice"] == "tie"
assert judge["evidence"]["model"] == "synthetic-independent-judge"
assert all(u["evidence"]["cost_state"] == "settled" for u in result["units"])
assert sum(Decimal(u["evidence"]["cost"]) for u in result["units"]) == Decimal(".375")
for payload in env["received"][:2]:
assert set(json.loads(payload["input"])) == {"instruction", "original", "context"}
assert "ORACLE-SECRET" not in payload["input"] and "sample-2" not in payload["input"]
assert set(json.loads(env["received"][-1]["input"])) == {"left", "right", "dimensions"}
svc.推进实验(env["actor"], eid)
_运行就绪(env)
assert len(env["received"]) == 3
def _恢复失败单元(env):
from muse.任务运行.接口 import 任务状态
svc = env["app"].要求评测()
result = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
failed = [u for u in result["units"] if u["state"] == "failed"]
assert len(failed) == 1
env["app"].任务运行.控制任务(
failed[0]["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID="resume-sample"
)
return failed[0]
@pytest.mark.case_id(
"NC-w25-253005",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="一次生成返回结构错误",
when="保留失败后显式恢复、推进比较",
then=["只补失败调用,账本包含失败成本,未运行样本不删除"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_失败样本和未开始评委保留且只补失败调用__253005(执行环境):
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
env["scripted"].append("bad_output")
_启动执行(env)
_运行就绪(env, allow_failure=True)
before = svc.推进实验(env["actor"], eid)
assert sorted(u["state"] for u in before["units"]) == ["completed", "failed", "not_started"]
assert len(env["received"]) == 2
_恢复失败单元(env)
_运行就绪(env)
svc.推进实验(env["actor"], eid)
_运行就绪(env)
after = svc.读取执行工作面(env["actor"], eid)
assert all(u["state"] == "completed" for u in after["units"])
assert len(env["received"]) == 4
assert sum(Decimal(u["cost"]["total_usd"]) for u in after["units"]) == Decimal(".5")
assert sum(len(u["cost"]["calls"]) for u in after["units"]) == 4
@pytest.mark.case_id(
"NC-w25-253006",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="真实交付提交后步骤发生异常",
when="恢复同一单元",
then=["核原尝试交付,不重发、不重复计费、不伪造新模型回合"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_交付提交后步骤失败恢复不重发且保留原尝试__253006(执行环境, monkeypatch):
from muse.效果评测.接口 import 评测服务, 评测错误
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
original = 评测服务.保存执行交付
failed = []
def interrupted(self, ctx, *args):
result = original(self, ctx, *args)
if not failed:
failed.append(ctx.领取.尝试ID)
raise 评测错误("合成的交付提交后中断")
return result
monkeypatch.setattr(评测服务, "保存执行交付", interrupted)
_启动执行(env)
_运行就绪(env, allow_failure=True)
old = _恢复失败单元(env)
assert old["output"] is not None and len(env["received"]) == 2
_运行就绪(env)
svc.推进实验(env["actor"], eid)
_运行就绪(env)
after = svc.读取执行工作面(env["actor"], eid)
unit = next(u for u in after["units"] if u["unit_id"] == old["unit_id"])
assert unit["state"] == "completed" and unit["output"] == old["output"]
assert unit["evidence"]["runtime"]["attempt_id"] == failed[0]
assert len(env["received"]) == 3
@pytest.mark.case_id(
"NC-w25-253007",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="模型没有用量导致成本未知",
when="推进下游比较",
then=["未知成本仍显示为空总额,不派评委、不推定免费"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_未知成本仍可见且不能派发下游评委__253007(执行环境):
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
env["scripted"].append("unknown_cost")
_启动执行(env)
_运行就绪(env, allow_failure=True)
after = svc.推进实验(env["actor"], eid)
assert next(u for u in after["units"] if u["kind"] == "comparison")["task_id"] is None
assert after["state"] == "reconciling"
assert any(
u["cost"] and u["cost"]["has_unknown"] and u["cost"]["total_usd"] is None
for u in after["units"]
)
assert all(r["model"] != "synthetic-independent-judge" for r in env["received"])
with env["pool"].连接(只读=True) as conn:
assert (
conn.execute(
"SELECT count(*) FROM evaluation.muse_budget_reservation WHERE state='unknown'"
).fetchone()[0]
== 1
)
assert any(
u["state"] != "completed" or (u["evidence"] and u["evidence"]["cost_state"] == "unknown")
for u in after["units"]
)
@pytest.mark.case_id(
"NC-w25-253008",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="有效产物被改写且局部输出与证据哈希重签",
when="从工作面回查",
then=["反查S02原始输出拒绝伪造,生成与比较都受保护"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("kind", ["generation", "comparison"])
def test_篡改产物并重签局部证据仍不能替代S02原交付__253008(执行环境, monkeypatch, kind):
import copy
from muse.效果评测.存储 import 评测存储
from muse.正式变更.接口 import 固定哈希
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
_启动执行(env)
_运行就绪(env)
svc.推进实验(env["actor"], eid)
_运行就绪(env)
real = svc.读取执行工作面(env["actor"], eid)
target = next(u["unit_id"] for u in real["units"] if u["kind"] == kind)
read = 评测存储.读取交付
def tampered(self, uid):
row = read(self, uid)
if row is not None and str(uid) == target:
row = copy.deepcopy(row)
if kind == "generation":
row["output"]["paragraphs"][0]["text"] = "伪造但结构仍合法的正文。"
else:
row["output"]["rationale"] = "事后改写的评判理由。"
digest = 固定哈希(row["output"])
row["output_hash"] = digest
row["evidence"]["structured_output_hash"] = digest
row["evidence"]["runtime"]["response_metadata"]["structured_output_hash"] = digest
return row
monkeypatch.setattr(评测存储, "读取交付", tampered)
with pytest.raises(Muse错误, match="真实调用不一致"):
svc.读取执行工作面(env["actor"], eid)
with pytest.raises(Muse错误, match="真实调用不一致"):
svc.读取实验报告(env["actor"], eid)
@pytest.mark.case_id(
"NC-w25-253009",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="未开始调用的已登记实验",
when="保存停止决定并重复取消",
then=["停止先持久化,任务收敛,不复活未开始评委"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_停止决定先落库且取消重放不会复活未开始评委__253009(执行环境):
from muse.效果评测.接口 import 评测错误
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
_启动执行(env)
result = svc.取消实验(env["actor"], eid, "stop-now")
assert result["state"] == "stopped"
assert sorted(u["state"] for u in result["units"]) == ["cancelled", "cancelled", "not_started"]
assert svc.取消实验(env["actor"], eid, "stop-again") == result
_运行就绪(env)
assert env["received"] == []
with pytest.raises(评测错误, match="已停止"):
svc.推进实验(env["actor"], eid)
@pytest.mark.case_id(
"NC-w25-25300a",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="一个单元完成,另一个只完成准备",
when="取消实验",
then=["已有产物与证据保留,不再发送剩余调用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_停止发生在准备后仍禁止外发且保留已有交付__25300a(执行环境):
env = 执行环境
app = env["app"]
svc = app.要求评测()
eid = env["exp"]["experiment_id"]
_启动执行(env)
# 顺序领取:先完成第一单元,第二单元停在已准备状态。
for _ in range(3):
claim = app.任务运行.领取步骤("eval-worker", ["eval.prepare", "eval.call.writer"])
assert claim is not None
app.任务运行.执行一步(claim)
before = svc.读取执行工作面(env["actor"], eid)
done = [u for u in before["units"] if u["state"] == "completed"]
assert len(done) == 1 and len(env["received"]) == 1
stopped = svc.取消实验(env["actor"], eid, "after-first")
retained = next(u for u in stopped["units"] if u["unit_id"] == done[0]["unit_id"])
assert retained["output"] == done[0]["output"] and retained["evidence"] == done[0]["evidence"]
_运行就绪(env)
assert len(env["received"]) == 1
@pytest.mark.case_id(
"NC-w25-25300b",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="实际HTTP会话与隔离CLI配置",
when="启动、推进、读回及CLI子进程取消",
then=["同一实验和交付,验证模式仍为offline_contract,不因入口变化升级认证"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_HTTP启动推进与实际CLI读取取消沿同一实验__25300b(执行环境, tmp_path):
import json
import subprocess
import sys
import test_生产评测权限隔离 as 数据测试
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
env = 执行环境
app = env["app"]
eid = env["exp"]["experiment_id"]
config = 数据测试._配置文件(env["pool"], tmp_path)
http = 创建应用(读取配置(config))
endpoint = "/api/v1/evaluation/experiments/" + eid
with TestClient(http, headers={"origin": "http://testserver"}) as client:
http.state.装配 = app
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
payload = {
"approval_ref": "synthetic-only",
"deadline": (datetime.now(UTC) + timedelta(minutes=10)).isoformat(),
}
started = client.post(endpoint + "/execution", json=payload)
assert started.status_code == 201, started.text
assert client.post(endpoint + "/execution", json=payload).json() == started.json()
_运行就绪(env)
advanced = client.post(endpoint + "/advance", json={})
assert advanced.status_code == 200, advanced.text
_运行就绪(env)
read = client.get(endpoint + "/execution").json()
assert read["state"] == "completed" and read["activation_status"] == "not_evaluated"
assert all(u["evidence"]["validation"]["mode"] == "offline_contract" for u in read["units"])
report = client.get(endpoint + "/report")
assert report.status_code == 200, report.text
assert report.json()["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
assert "ORACLE-SECRET" not in report.text
command = [sys.executable, "-I", "-m", "muse", "评测", str(config)]
result = subprocess.run(
[*command, "执行工作面", eid], cwd=tmp_path, capture_output=True, text=True, timeout=30
)
assert result.returncode == 0, result.stderr
assert json.loads(result.stdout) == read
report_result = subprocess.run(
[*command, "报告", eid], cwd=tmp_path, capture_output=True, text=True, timeout=30
)
assert report_result.returncode == 0, report_result.stderr
assert json.loads(report_result.stdout) == report.json()
stopped = subprocess.run(
[*command, "取消实验", eid], cwd=tmp_path, capture_output=True, text=True, timeout=30
)
assert stopped.returncode == 0, stopped.stderr
assert json.loads(stopped.stdout)["state"] == "stopped"
assert len(env["received"]) == 3
@pytest.mark.case_id(
"NC-w25-25300c",
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
given="HTTP已接收请求但尚未返回",
when="并发取消后接收迟到结果",
then=["不保存新业务产物;已发送调用的成本保留,不能归零"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_在途取消不保存业务产物也不把已发送成本归零__25300c(执行环境):
from threading import Thread
env = 执行环境
app = env["app"]
svc = app.要求评测()
eid = env["exp"]["experiment_id"]
env["scripted"].append("inflight")
_启动执行(env)
prepare = app.任务运行.领取步骤("eval-worker", ["eval.prepare"])
app.任务运行.执行一步(prepare)
call = app.任务运行.领取步骤("eval-worker", ["eval.call.writer"])
errors = []
def execute():
try:
app.任务运行.执行一步(call)
except Muse错误 as exc:
errors.append(exc)
thread = Thread(target=execute, daemon=True)
thread.start()
try:
assert env["in_flight"].wait(5)
stopped = svc.取消实验(env["actor"], eid, "cancel-inflight")
assert stopped["state"] == "stopped"
finally:
env["respond"].set()
thread.join(timeout=8)
assert not thread.is_alive() and len(errors) == 1
result = svc.读取执行工作面(env["actor"], eid)
assert len(env["received"]) == 1
assert all(u["output"] is None for u in result["units"])
costs = [u["cost"] for u in result["units"] if u["cost"] and u["cost"]["calls"]]
assert len(costs) == 1 and len(costs[0]["calls"]) == 1
assert costs[0]["has_unknown"] or Decimal(costs[0]["total_usd"]) > 0
@pytest.mark.case_id(
"TC-aee3df41e9b3",
environment="隔离PG及受控合成HTTP",
given="隔离PG冻结两名不同模型评委,两侧实际S02生成文本",
when="通过真实逐例任务完成两名评委的反向比较",
then=[
"两次真实评委请求,文本两侧相反、维度相同、任务/调用/模型独立且均完成",
"不发送先前判断、工具、身份哈希或会话历史",
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{"judges": 2}], indirect=True)
def test_双评委反向文本且独立任务调用无共享报告__aee3df(执行环境):
import json
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
_启动执行(env)
_运行就绪(env)
svc.推进实验(env["actor"], eid)
_运行就绪(env)
requests = env["received"][2:]
assert len(requests) == 2
material = [json.loads(r["input"]) for r in requests]
assert material[0]["left"] == material[1]["right"]
assert material[0]["right"] == material[1]["left"]
assert material[0]["dimensions"] == material[1]["dimensions"]
for request, value in zip(requests, material, strict=True):
assert set(value) == {"left", "right", "dimensions"}
assert set(value["left"]) == set(value["right"]) == {"text"}
assert not request.get("tools") and not request.get("previous_response_id")
assert "previousReport" not in value and isinstance(request["input"], str)
work = svc.读取执行工作面(env["actor"], eid)
judges = [u for u in work["units"] if u["kind"] == "comparison"]
assert (
len({u["task_id"] for u in judges})
== len({u["evidence"]["runtime"]["response_metadata"]["call_id"] for u in judges})
== 2
)
assert len({u["evidence"]["model"] for u in judges}) == 2
assert all(u["state"] == "completed" for u in judges)
@pytest.mark.case_id(
"NC-w25-259001",
environment="隔离PG及受控合成HTTP",
given="实际S02生成及独立评委回合",
when="保存原始引文和派生定位报告,读回后篡改定位并重签派生哈希",
then=[
"原始模型输出无位置;码点由代码派生且两类哈希分别可查",
"报告接口返回实际理由和引文,伪造定位即使重签也拒绝且不追加模型调用",
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_原始模型引文及派生位置分别持久化和核验__259001(执行环境, monkeypatch):
from copy import deepcopy
from muse.效果评测.存储 import 评测存储
from muse.效果评测.接口 import 评测错误
from muse.正式变更.接口 import 固定哈希
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
_启动执行(env)
_运行就绪(env)
svc.推进实验(env["actor"], eid)
_运行就绪(env)
work = svc.读取执行工作面(env["actor"], eid)
judge = next(u for u in work["units"] if u["kind"] == "comparison")
assert all(
set(q) == {"side", "quote"} for d in judge["output"]["dimensions"] for q in d["evidence"]
)
derived = judge["evidence"]["judgment"]
assert derived["contract"] == "quote-only-v2"
assert derived["report_hash"] == 固定哈希(derived["report"])
assert derived["raw_output_hash"] == 固定哈希(judge["output"])
assert all(
q["start"] == 0 and q["end"] == len(q["quote"])
for d in derived["report"]["dimensions"]
for q in d["evidence"]
)
report = svc.读取实验报告(env["actor"], eid)
decision = report["samples"][0]["decisions"][0]
assert decision["judgment_hash"] == derived["report_hash"]
assert decision["judgment"] == derived["report"]
original = 评测存储.读取交付
def 篡改(self, unit_id):
row = original(self, unit_id)
if row is not None and str(unit_id) == judge["unit_id"]:
row = deepcopy(row)
changed = row["evidence"]["judgment"]
changed["report"]["dimensions"][0]["evidence"][0]["start"] = 1
changed["report_hash"] = 固定哈希(changed["report"])
return row
monkeypatch.setattr(评测存储, "读取交付", 篡改)
with pytest.raises(评测错误, match="定位报告"):
svc.读取实验报告(env["actor"], eid)
assert len(env["received"]) == 3
# ---- 生产评测权限隔离线:7项盲评原观察 ----
# 未知模式在创建实验前合同拒绝(fail closed,见 tests/契约/test_盲评与引文合同.py);
# 双评委真实派发链由基线 aee3df 承接(独立任务调用、盲化输入、模型各异、不共享报告),不重复写。
@pytest.mark.case_id(
"TC-c7afbc32e342",
environment="离线,匿名样本、独立评委替身",
when="匿名分配样本、校验逐维引文并处理评委分歧。",
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_评委角色缺失的配置在固定实验条件时拒绝__c7afbc(原子任务环境):
"""旧“错误适配器角色拒绝”等价新版:judge 角色未在配置版本声明时拒绝。"""
from uuid import UUID
from muse.效果评测.实验条件 import 固定实验条件
from muse.效果评测.接口 import 实验请求, 目标版本, 评测错误, 配置选择
_, pool, _, _, _, _ = 原子任务环境
请求 = 实验请求(
dataset_version_id=UUID("12345678-1234-5678-8123-123456789abc"),
dataset_hash="0" * 64,
split="holdout",
target=目标版本(
kind="prompt",
target_ref="B05.generate",
version="1",
content_hash="1" * 64,
),
generator_role="writer",
generator=配置选择(config_id="writer", version="1"),
# 原子任务环境的配置版本只声明 writer 角色;把它用作评委即角色缺失。
judges=(配置选择(config_id="writer", version="1"),),
arms=("original", "candidate"),
dimensions=("清晰度",),
max_cost_usd=Decimal("1.0"),
max_calls_per_sample=3,
)
with pytest.raises(评测错误, match="缺少所需角色"):
固定实验条件(pool, 请求)