1197 lines
50 KiB
Python
1197 lines
50 KiB
Python
"""评测任务实际使用S02创建、配置与预算事务;不替代后续模型逐例证据。"""
|
||
|
||
from dataclasses import replace
|
||
from datetime import UTC, datetime, timedelta
|
||
from decimal import Decimal
|
||
|
||
import pytest
|
||
import test_两阶段写手 as 写手测试
|
||
|
||
from muse.任务运行.接口 import (
|
||
任务请求,
|
||
任务预算计划,
|
||
执行计划,
|
||
步骤处理器,
|
||
步骤结果,
|
||
步骤计划,
|
||
角色策略目录,
|
||
角色预算,
|
||
配置版本管理,
|
||
预算管理,
|
||
额度策略,
|
||
)
|
||
from muse.共享.调用身份 import 内容用途, 用途
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.启动 import 构建
|
||
from muse.配置 import 应用配置
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
生成环境 = 写手测试.生成环境
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-256001",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["真实S02按固定运行表创建并派发,评委两侧按独立oracle表绑定"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{"samples": 2}], indirect=True)
|
||
def test_真实生成按运行表且比较依赖独立盲化表__256001(执行环境):
|
||
import json
|
||
|
||
from muse.正文写作.接口 import 生成正文模板
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
ordering = env["exp"]["conditions"]["ordering"]["execution_order"]
|
||
expected = [(r["sample_id"], arm) for r in ordering for arm in r["arm_order"]]
|
||
inputs = {sid: svc.读取生成输入(env["actor"], eid, sid) for sid in env["exp"]["sample_ids"]}
|
||
运行文本 = {
|
||
json.dumps(value, ensure_ascii=False, sort_keys=True): sid for sid, value in inputs.items()
|
||
}
|
||
运行测试输出 = {}
|
||
_启动执行(env)
|
||
_运行就绪(env)
|
||
actual = []
|
||
template = 生成正文模板()[0]
|
||
for i, request in enumerate(env["received"], start=1):
|
||
arm = "treatment" if template in request["instructions"] else "control"
|
||
sid = 运行文本[request["input"]]
|
||
actual.append((sid, arm))
|
||
运行测试输出[(sid, arm)] = f"合成正文{i}。"
|
||
assert actual == expected
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assignments = conn.execute(
|
||
"SELECT sample_id,arm_order FROM oracle.muse_blind_assignment WHERE experiment_id=%s",
|
||
(eid,),
|
||
).fetchall()
|
||
expected_pairs = {
|
||
(运行测试输出[(sid, arms[0])], 运行测试输出[(sid, arms[1])]) for sid, arms in assignments
|
||
}
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
pairs = [json.loads(r["input"]) for r in env["received"][len(expected) :]]
|
||
assert {(r["left"]["text"], r["right"]["text"]) for r in pairs} == expected_pairs
|
||
assert len(pairs) == len(assignments) == 2
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-256002",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["预注册表与读取映射不符时零任务、零外发"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_预注册表与读取映射不一致时零任务零调用__256002(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.效果评测.接口 import 评测错误
|
||
|
||
env = 执行环境
|
||
original = 评测存储.读分配
|
||
|
||
def 错序(self, experiment_id):
|
||
rows = original(self, experiment_id)
|
||
return [{**r, "arm_order": list(reversed(r["arm_order"]))} for r in rows]
|
||
|
||
monkeypatch.setattr(评测存储, "读分配", 错序)
|
||
with pytest.raises(评测错误, match="预注册"):
|
||
_启动执行(env)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_experiment_execution").fetchone()[0]
|
||
== 0
|
||
)
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
assert not env["received"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25300d",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="发送前实际请求的系统提示、输入、结构或输出上限被改写",
|
||
when="真实S02发送事务复检请求",
|
||
then=["全部在外发前拒绝,合成HTTP零请求,未保存业务交付"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("changed", ["system", "input", "schema", "limit"])
|
||
def test_实际派发请求偏离冻结单元在外发前拒绝__25300d(执行环境, monkeypatch, changed):
|
||
import muse.编排.执行评测 as flow
|
||
|
||
env = 执行环境
|
||
_启动执行(env)
|
||
original = flow.模型请求
|
||
|
||
def 改写(*args, **kwargs):
|
||
request = original(*args, **kwargs)
|
||
fields = {
|
||
"system": {"系统提示": "未经实验登记的其他提示"},
|
||
"input": {"用户输入": "未获此实验批准的其他材料"},
|
||
"schema": {"输出合同": {"type": "object"}},
|
||
"limit": {"最大输出token": request.最大输出token - 1},
|
||
}
|
||
return replace(request, **fields[changed])
|
||
|
||
monkeypatch.setattr(flow, "模型请求", 改写)
|
||
_运行就绪(env, allow_failure=True)
|
||
result = env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
assert env["received"] == []
|
||
assert all(u["output"] is None for u in result["units"])
|
||
assert all(u["state"] == "failed" for u in result["units"] if u["kind"] == "generation")
|
||
|
||
|
||
@pytest.fixture
|
||
def 原子任务环境(应用测试库):
|
||
from muse.任务运行.接口 import 内容哈希, 凭据引用, 提供方配置, 运行配置内容, 配置验证证据
|
||
|
||
pool = 应用测试库[用途.评测]
|
||
policy = 角色策略目录.从发布包()
|
||
app = 构建(应用配置(pool.引用, policy.资源发布身份, 运行用途=用途.评测))
|
||
with app.生命周期():
|
||
registry = app.流程登记
|
||
registry.登记类型("atomic-eval-fixture", 必需保护=())
|
||
registry.登记处理器(
|
||
步骤处理器(
|
||
"eval.atomic-fixture",
|
||
"1",
|
||
lambda _: 步骤结果({}),
|
||
"eval-input",
|
||
"eval-output",
|
||
角色="writer",
|
||
)
|
||
)
|
||
# 本夹具只验证创建事务,禁止冒充模型执行;后续运行使用实际评测处理器。
|
||
app.任务运行.发布计划(
|
||
执行计划(
|
||
"atomic-eval-fixture",
|
||
"1",
|
||
(
|
||
步骤计划(
|
||
"运行",
|
||
"eval.atomic-fixture",
|
||
"1",
|
||
输入合同="eval-input",
|
||
输出合同="eval-output",
|
||
角色="writer",
|
||
),
|
||
),
|
||
)
|
||
)
|
||
|
||
class 合同验证:
|
||
身份 = "atomic-fixture-offline-contract"
|
||
|
||
def 验证(self, content, purpose):
|
||
return 配置验证证据(
|
||
内容哈希(content.冻结()),
|
||
content.角色策略版本,
|
||
content.资源发布身份,
|
||
purpose,
|
||
("synthetic:atomic-task-fixture",),
|
||
"offline_contract",
|
||
)
|
||
|
||
manager = 配置版本管理(pool, 合同验证())
|
||
content = 运行配置内容(
|
||
"direct",
|
||
"1",
|
||
policy.定义["version"],
|
||
policy.资源发布身份,
|
||
"eval-atomic-budget",
|
||
{
|
||
"writer": {
|
||
"provider": "synthetic",
|
||
"model": "claude-opus-4-8[1M]",
|
||
"thinking": "high",
|
||
"tools": [],
|
||
}
|
||
},
|
||
(凭据引用("key", "环境变量", "MUSE_UNUSED_EVAL_SECRET"),),
|
||
(提供方配置("synthetic", "responses", "http://127.0.0.1:9", "key"),),
|
||
"test-price",
|
||
)
|
||
version = manager.保存草案("writer", "1", content)
|
||
receipt = manager.验证版本("writer", "1")
|
||
manager.启用(
|
||
"writer", "1", 验证回执=receipt, 批准引用="synthetic-test-approval", 预期代次=0
|
||
)
|
||
req = 任务请求(
|
||
"atomic-eval-fixture",
|
||
"atomic-command",
|
||
"eval-author",
|
||
用途.评测,
|
||
内容用途.生成,
|
||
{"sample_id": "synthetic:sample"},
|
||
policy.定义["version"],
|
||
policy.资源发布身份,
|
||
{
|
||
"source_scope": {"source_ids": ["synthetic:sample"]},
|
||
"schema_versions": {},
|
||
"authorization": {"authorized_by": "eval-author"},
|
||
"budget": {},
|
||
"stop_conditions": {},
|
||
},
|
||
)
|
||
budget = 任务预算计划(
|
||
Decimal("2"),
|
||
(角色预算("writer", 1, 2, Decimal("1")),),
|
||
"synthetic-test-approval",
|
||
datetime.now(UTC) + timedelta(minutes=10),
|
||
)
|
||
yield app, pool, manager, req, budget, version
|
||
|
||
|
||
def _创建(env, **kwargs):
|
||
app, _, _, req, budget, version = env
|
||
return app.任务运行.创建受控任务(
|
||
req,
|
||
"atomic-eval-fixture",
|
||
"1",
|
||
配置ID="writer",
|
||
预算计划=budget,
|
||
预期配置版本=kwargs.get("version", version.版本),
|
||
预期配置哈希=kwargs.get("hash", version.内容哈希),
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253001",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="缺少真实预算账户的已启用评测配置",
|
||
when="原子创建后补账户并按原命令重试",
|
||
then=["首次任务、配置和预算全部回滚;重试只有一份绑定"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_预算前置失败不留任务配置且可原命令重试__253001(原子任务环境):
|
||
app, pool, manager, req, budget, _ = 原子任务环境
|
||
with pytest.raises(Muse错误):
|
||
_创建(原子任务环境)
|
||
assert app.任务运行.读取命令任务(req.作者, req.命令ID) is None
|
||
with pool.连接(只读=True) as conn:
|
||
for table in ("muse_task_config_binding", "muse_task_budget"):
|
||
assert conn.execute("SELECT count(*) FROM evaluation." + table).fetchone()[0] == 0
|
||
预算管理(pool, "eval-atomic-budget").登记策略(
|
||
额度策略("eval-atomic-budget", "1", Decimal("20"), 20)
|
||
)
|
||
tid = _创建(原子任务环境)
|
||
assert _创建(原子任务环境) == tid
|
||
assert manager.读取任务绑定(tid).内容哈希
|
||
with pool.连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 1
|
||
row = conn.execute(
|
||
"SELECT plan FROM evaluation.muse_task_budget WHERE task_id=%s", (tid,)
|
||
).fetchone()
|
||
assert row[0] == budget.冻结()
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253002",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="实际配置版本/哈希或预算角色与请求不一致",
|
||
when="创建受控评测任务",
|
||
then=["拒绝且无可领取任务或预算残留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("changed", ["version", "hash", "missing_budget", "wrong_budget_role"])
|
||
def test_已登记条件与实际配置或预算不一致拒绝创建__253002(原子任务环境, changed):
|
||
app, pool, _, req, budget, _ = 原子任务环境
|
||
预算管理(pool, "eval-atomic-budget").登记策略(
|
||
额度策略("eval-atomic-budget", "1", Decimal("20"), 20)
|
||
)
|
||
if changed == "missing_budget":
|
||
env = (*原子任务环境[:4], None, 原子任务环境[5])
|
||
kwargs = {}
|
||
elif changed == "wrong_budget_role":
|
||
env = (
|
||
*原子任务环境[:4],
|
||
replace(budget, 角色=(角色预算("judge", 1, 2, Decimal("1")),)),
|
||
原子任务环境[5],
|
||
)
|
||
kwargs = {}
|
||
else:
|
||
env = 原子任务环境
|
||
kwargs = {changed: "2" if changed == "version" else "0" * 64}
|
||
with pytest.raises(Muse错误):
|
||
_创建(env, **kwargs)
|
||
assert app.任务运行.读取命令任务(req.作者, req.命令ID) is None
|
||
with pool.连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task_budget").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253003",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="真实生产生成入口引用未启用配置",
|
||
when="发起新章任务",
|
||
then=["同事务回滚,不遗留queued孤儿任务"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_生产生成配置缺失不留下可领取的孤儿任务__253003(生成环境):
|
||
from muse.编排.生成正文 import 发起生成正文
|
||
|
||
env = 生成环境
|
||
with pytest.raises(Muse错误):
|
||
发起生成正文(
|
||
env["装配"],
|
||
env["作者"],
|
||
"missing-runtime-config",
|
||
"new_chapter",
|
||
work_id="gen-work",
|
||
chapter_id="ch-3",
|
||
配置ID="not-enabled",
|
||
)
|
||
assert env["装配"].任务运行.读取命令任务(env["作者"].作者, "missing-runtime-config") is None
|
||
|
||
|
||
@pytest.fixture
|
||
def 执行环境(内置结构测试库, tmp_path, monkeypatch, request):
|
||
"""第二评委仅在隔离测试策略中存在,不修改正式白名单或认证真实模型。"""
|
||
应用测试库 = 内置结构测试库
|
||
import copy
|
||
import json
|
||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||
from threading import Event, Thread
|
||
|
||
import test_生产评测权限隔离 as 数据测试
|
||
|
||
from muse.任务运行.接口 import 内容哈希, 凭据引用, 提供方配置, 运行配置内容, 配置验证证据
|
||
from muse.共享.调用身份 import 调用身份
|
||
from muse.效果评测.接口 import 实验请求, 目标版本, 评测服务, 配置选择
|
||
from muse.正文写作.接口 import 生成正文模板
|
||
from muse.编排.执行评测 import 登记评测执行
|
||
|
||
options = getattr(request, "param", {})
|
||
if options.get("output_factory"):
|
||
options = {**options, **options["output_factory"]()}
|
||
judge_count = options.get("judges", 1)
|
||
judges = ["synthetic-independent-judge"]
|
||
if judge_count == 2:
|
||
judges.append("synthetic-independent-judge-second")
|
||
if options.get("arbitrator"):
|
||
judges.append("synthetic-independent-arbitrator")
|
||
original = 角色策略目录.从发布包()
|
||
definition = copy.deepcopy(original.定义)
|
||
definition["version"] = "synthetic-evaluation-policy"
|
||
definition["models"]["fixed"].extend(judges)
|
||
definition["models"]["judge-fixed"].extend(judges)
|
||
detector_model = "synthetic-semantic-detector"
|
||
if options.get("detector"):
|
||
definition["models"]["governed"].append(detector_model)
|
||
if options.get("allow_model_alias"):
|
||
definition["actual_model_ids"] = {model: [model, model + "-changed"] for model in judges}
|
||
isolated = 角色策略目录(
|
||
definition, 资源发布身份=original.资源发布身份, 角色资源=original.角色资源
|
||
)
|
||
monkeypatch.setattr(角色策略目录, "从发布包", classmethod(lambda cls: isolated))
|
||
received, scripted = [], []
|
||
in_flight, respond = Event(), Event()
|
||
|
||
class Handler(BaseHTTPRequestHandler):
|
||
def do_POST(self):
|
||
data = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
|
||
received.append(data)
|
||
script = scripted.pop(0) if scripted else "ok"
|
||
if script == "inflight":
|
||
in_flight.set()
|
||
if not respond.wait(8):
|
||
self.send_error(504)
|
||
return
|
||
if data["model"] in judges:
|
||
material = json.loads(data["input"])
|
||
evidence = [
|
||
dict(
|
||
side=side,
|
||
quote=material[side]["text"],
|
||
)
|
||
for side in ("left", "right")
|
||
]
|
||
choice = script.removeprefix("choose_") if script.startswith("choose_") else "tie"
|
||
output = dict(
|
||
choice=choice,
|
||
rationale="合成判断保留平局",
|
||
dimensions=[
|
||
dict(
|
||
dimension=dim,
|
||
choice=choice,
|
||
rationale="合成引文覆盖两侧",
|
||
evidence=evidence,
|
||
)
|
||
for dim in material["dimensions"]
|
||
],
|
||
)
|
||
elif data["model"] == detector_model and options.get("detector"):
|
||
material = json.loads(data["input"])
|
||
output = options["detector_output"](material, script)
|
||
else:
|
||
output = {"paragraphs": [{"text": f"合成正文{len(received)}。"}]}
|
||
if options.get("generator_output"):
|
||
output = options["generator_output"](data, script)
|
||
if data["model"] in judges and options.get("judge_output"):
|
||
output = options["judge_output"](material, script)
|
||
if script == "bad_quote":
|
||
output["dimensions"][0]["evidence"][0]["quote"] = "候选正文中不存在的引文"
|
||
if script == "bad_mixed":
|
||
output["dimensions"][0]["evidence"][0]["quote"] = "候选正文中不存在的引文"
|
||
output["rationale"] = " "
|
||
if script == "bad_output":
|
||
output = {"paragraphs": []}
|
||
response = {
|
||
"id": f"synthetic-response-{len(received)}",
|
||
"model": data["model"] + "-changed" if script == "changed_model" else data["model"],
|
||
"status": "completed",
|
||
"output": [
|
||
{
|
||
"type": "message",
|
||
"content": [
|
||
{"type": "output_text", "text": json.dumps(output, ensure_ascii=False)}
|
||
],
|
||
}
|
||
],
|
||
}
|
||
if script != "unknown_cost":
|
||
response["usage"] = {"input_tokens": 5, "output_tokens": 7}
|
||
if script == "api_error":
|
||
response.update(
|
||
status="failed",
|
||
model=None,
|
||
output=[],
|
||
usage={"input_tokens": 0, "output_tokens": 0},
|
||
error={"code": "synthetic_failure", "message": "合成已知用量终态失败"},
|
||
)
|
||
payload = (
|
||
"data: "
|
||
+ json.dumps({"type": "response." + response["status"], "response": response})
|
||
+ "\n\n"
|
||
).encode()
|
||
self.send_response(200)
|
||
self.send_header("Content-Type", "text/event-stream")
|
||
self.send_header("Content-Length", str(len(payload)))
|
||
self.end_headers()
|
||
self.wfile.write(payload)
|
||
|
||
def log_message(self, *args):
|
||
pass
|
||
|
||
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
||
thread = Thread(target=server.serve_forever, daemon=True)
|
||
thread.start()
|
||
try:
|
||
pools = 应用测试库
|
||
pool = pools[用途.评测]
|
||
app = 构建(
|
||
应用配置(
|
||
pool.引用, isolated.资源发布身份, 运行用途=用途.评测, 原文暂存=str(tmp_path / "raw")
|
||
)
|
||
)
|
||
with app.生命周期():
|
||
登记评测执行(app, 写手测试.合成计价())
|
||
actor = 调用身份("eval-author", None, 用途.评测, 内容用途.检测)
|
||
samples = [数据测试._数据().samples[2]]
|
||
if options.get("calibration_policy"):
|
||
samples = [s.model_copy(update={"split": "calibration"}) for s in samples]
|
||
if options.get("samples") == 2:
|
||
samples.append(
|
||
samples[0].model_copy(
|
||
update={
|
||
"sample_id": "sample-extra",
|
||
"source_groups": ("independent-source",),
|
||
"source_ref": "synthetic:independent-source",
|
||
"input": samples[0].input.model_copy(
|
||
update={"original": "另一本合成来源的正文。"}
|
||
),
|
||
}
|
||
)
|
||
)
|
||
source = 数据测试._数据().model_copy(
|
||
update={"dataset_id": "runtime-fixture", "samples": tuple(samples)}
|
||
)
|
||
if options.get("judging_basis"):
|
||
source = source.model_copy(
|
||
update={
|
||
"samples": tuple(
|
||
s.model_copy(
|
||
update={
|
||
"answer": {
|
||
**s.answer,
|
||
"judging_basis": options["judging_basis"],
|
||
}
|
||
}
|
||
)
|
||
for s in source.samples
|
||
)
|
||
}
|
||
)
|
||
data = 评测服务(pools[用途.维护]).发布数据集(replace(actor, 用途=用途.维护), source)
|
||
|
||
class Validator:
|
||
身份 = "synthetic-evaluation-offline"
|
||
|
||
def 验证(self, c, purpose):
|
||
return 配置验证证据(
|
||
内容哈希(c.冻结()),
|
||
c.角色策略版本,
|
||
c.资源发布身份,
|
||
purpose,
|
||
("synthetic:controlled-local-http",),
|
||
# 仅隔离测试可模拟已运行的配置验证;不作为真实外部模型证据。
|
||
"runtime"
|
||
if options.get("runtime_validation_fixture")
|
||
else "offline_contract",
|
||
)
|
||
|
||
key = tmp_path / "model-key"
|
||
key.write_text("synthetic-only")
|
||
key.chmod(0o600)
|
||
manager = 配置版本管理(pool, Validator())
|
||
configs = [("writer", "writer", "claude-opus-4-8[1M]")]
|
||
if options.get("detector"):
|
||
configs.append(("detector", "detector", detector_model))
|
||
configs.extend(
|
||
(("judge", "judge-second", "arbitrator")[i], "judge", model)
|
||
for i, model in enumerate(judges)
|
||
)
|
||
for config_id, role, model in configs:
|
||
config = 运行配置内容(
|
||
"direct",
|
||
"1",
|
||
definition["version"],
|
||
isolated.资源发布身份,
|
||
"runtime-budget",
|
||
{
|
||
role: {
|
||
"provider": "synthetic",
|
||
"model": model,
|
||
"thinking": "high",
|
||
"tools": [],
|
||
}
|
||
},
|
||
(凭据引用("key", "受控存储", str(key)),),
|
||
(
|
||
提供方配置(
|
||
"synthetic",
|
||
"responses",
|
||
f"http://127.0.0.1:{server.server_port}/v1/responses",
|
||
"key",
|
||
),
|
||
),
|
||
写手测试.合成计价.版本,
|
||
)
|
||
manager.保存草案(config_id, "1", config)
|
||
receipt = manager.验证版本(config_id, "1")
|
||
manager.启用(
|
||
config_id, "1", 验证回执=receipt, 批准引用="synthetic-only", 预期代次=0
|
||
)
|
||
预算管理(pool, "runtime-budget").登记策略(
|
||
额度策略(
|
||
"runtime-budget",
|
||
"1",
|
||
Decimal(str(options.get("budget_amount", "20"))),
|
||
options.get("budget_calls", 30),
|
||
)
|
||
)
|
||
request = 实验请求(
|
||
dataset_version_id=data["version_id"],
|
||
dataset_hash=data["public_hash"],
|
||
split="calibration" if options.get("calibration_policy") else "holdout",
|
||
target=目标版本(
|
||
kind="prompt",
|
||
target_ref="B05.generate",
|
||
version=isolated.资源发布身份,
|
||
content_hash=生成正文模板()[1],
|
||
),
|
||
generator_role="writer",
|
||
generator=配置选择(config_id="writer", version="1"),
|
||
detector=配置选择(config_id="detector", version="1")
|
||
if options.get("detector")
|
||
else None,
|
||
judges=tuple(
|
||
配置选择(config_id=config_id, version="1")
|
||
for config_id, role, _ in configs
|
||
if role == "judge" and config_id != "arbitrator"
|
||
),
|
||
arms=("control", "treatment"),
|
||
dimensions=options.get("dimensions", ("清晰度",)),
|
||
max_cost_usd=Decimal("6"),
|
||
max_calls_per_sample=options.get("calls", 9),
|
||
calibration_policy=options.get("calibration_policy"),
|
||
**(
|
||
{
|
||
"comparison_profile": "writer_rubric",
|
||
"arbitrator": 配置选择(config_id="arbitrator", version="1"),
|
||
}
|
||
if options.get("arbitrator")
|
||
else {}
|
||
),
|
||
**(
|
||
{"max_judge_corrections": options["corrections"]}
|
||
if "corrections" in options
|
||
else {}
|
||
),
|
||
)
|
||
exp = app.要求评测().创建实验(actor, "runtime-fixture", request)
|
||
yield dict(
|
||
app=app,
|
||
pool=pool,
|
||
pools=pools,
|
||
actor=actor,
|
||
exp=exp,
|
||
received=received,
|
||
scripted=scripted,
|
||
in_flight=in_flight,
|
||
respond=respond,
|
||
request=request,
|
||
max_steps=options.get("max_steps", 30),
|
||
fixture_control=options.get("fixture_control", {}),
|
||
)
|
||
finally:
|
||
server.shutdown()
|
||
server.server_close()
|
||
thread.join(timeout=5)
|
||
|
||
|
||
def _启动执行(env):
|
||
from muse.效果评测.接口 import 实验执行请求
|
||
|
||
req = 实验执行请求(
|
||
approval_ref="synthetic-only", deadline=datetime.now(UTC) + timedelta(minutes=10)
|
||
)
|
||
return env["app"].要求评测().启动实验(env["actor"], env["exp"]["experiment_id"], req)
|
||
|
||
|
||
def _运行就绪(env, *, allow_failure=False):
|
||
app = env["app"]
|
||
for _ in range(env.get("max_steps", 30)):
|
||
claim = app.任务运行.领取步骤(
|
||
"eval-worker",
|
||
["eval.prepare", "eval.call.writer", "eval.call.judge", "eval.call.detector"],
|
||
)
|
||
if claim is None:
|
||
return
|
||
try:
|
||
app.任务运行.执行一步(claim)
|
||
except Muse错误:
|
||
if not allow_failure:
|
||
raise
|
||
pytest.fail("评测测试出现未收敛领取")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253004",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="固定提示词对照及仅测试有效的双模型策略",
|
||
when="实际S02调用合成HTTP、保存和独立比较",
|
||
then=["两侧输入隔离,逐例交付可回查;平局、费用和原验证模式保留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_逐例交付和独立平局经过真实S02且生成读不到答案__253004(执行环境):
|
||
import json
|
||
|
||
env = 执行环境
|
||
app = env["app"]
|
||
svc = app.要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
start = _启动执行(env)
|
||
assert len(start["units"]) == 3
|
||
assert sum(u["task_id"] is not None for u in start["units"]) == 2
|
||
_运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
result = svc.读取执行工作面(env["actor"], eid)
|
||
assert all(u["state"] == "completed" for u in result["units"])
|
||
assert len(env["received"]) == 3
|
||
judge = next(u for u in result["units"] if u["kind"] == "comparison")
|
||
assert judge["output"]["choice"] == "tie"
|
||
assert judge["evidence"]["model"] == "synthetic-independent-judge"
|
||
assert all(u["evidence"]["cost_state"] == "settled" for u in result["units"])
|
||
assert sum(Decimal(u["evidence"]["cost"]) for u in result["units"]) == Decimal(".375")
|
||
for payload in env["received"][:2]:
|
||
assert set(json.loads(payload["input"])) == {"instruction", "original", "context"}
|
||
assert "ORACLE-SECRET" not in payload["input"] and "sample-2" not in payload["input"]
|
||
assert set(json.loads(env["received"][-1]["input"])) == {"left", "right", "dimensions"}
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
assert len(env["received"]) == 3
|
||
|
||
|
||
def _恢复失败单元(env):
|
||
from muse.任务运行.接口 import 任务状态
|
||
|
||
svc = env["app"].要求评测()
|
||
result = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
failed = [u for u in result["units"] if u["state"] == "failed"]
|
||
assert len(failed) == 1
|
||
env["app"].任务运行.控制任务(
|
||
failed[0]["task_id"], env["actor"].作者, 任务状态.已失败, "恢复", 命令ID="resume-sample"
|
||
)
|
||
return failed[0]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253005",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="一次生成返回结构错误",
|
||
when="保留失败后显式恢复、推进比较",
|
||
then=["只补失败调用,账本包含失败成本,未运行样本不删除"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_失败样本和未开始评委保留且只补失败调用__253005(执行环境):
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
env["scripted"].append("bad_output")
|
||
_启动执行(env)
|
||
_运行就绪(env, allow_failure=True)
|
||
before = svc.推进实验(env["actor"], eid)
|
||
assert sorted(u["state"] for u in before["units"]) == ["completed", "failed", "not_started"]
|
||
assert len(env["received"]) == 2
|
||
_恢复失败单元(env)
|
||
_运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
after = svc.读取执行工作面(env["actor"], eid)
|
||
assert all(u["state"] == "completed" for u in after["units"])
|
||
assert len(env["received"]) == 4
|
||
assert sum(Decimal(u["cost"]["total_usd"]) for u in after["units"]) == Decimal(".5")
|
||
assert sum(len(u["cost"]["calls"]) for u in after["units"]) == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253006",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="真实交付提交后步骤发生异常",
|
||
when="恢复同一单元",
|
||
then=["核原尝试交付,不重发、不重复计费、不伪造新模型回合"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_交付提交后步骤失败恢复不重发且保留原尝试__253006(执行环境, monkeypatch):
|
||
from muse.效果评测.接口 import 评测服务, 评测错误
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
original = 评测服务.保存执行交付
|
||
failed = []
|
||
|
||
def interrupted(self, ctx, *args):
|
||
result = original(self, ctx, *args)
|
||
if not failed:
|
||
failed.append(ctx.领取.尝试ID)
|
||
raise 评测错误("合成的交付提交后中断")
|
||
return result
|
||
|
||
monkeypatch.setattr(评测服务, "保存执行交付", interrupted)
|
||
_启动执行(env)
|
||
_运行就绪(env, allow_failure=True)
|
||
old = _恢复失败单元(env)
|
||
assert old["output"] is not None and len(env["received"]) == 2
|
||
_运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
after = svc.读取执行工作面(env["actor"], eid)
|
||
unit = next(u for u in after["units"] if u["unit_id"] == old["unit_id"])
|
||
assert unit["state"] == "completed" and unit["output"] == old["output"]
|
||
assert unit["evidence"]["runtime"]["attempt_id"] == failed[0]
|
||
assert len(env["received"]) == 3
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253007",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="模型没有用量导致成本未知",
|
||
when="推进下游比较",
|
||
then=["未知成本仍显示为空总额,不派评委、不推定免费"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_未知成本仍可见且不能派发下游评委__253007(执行环境):
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
env["scripted"].append("unknown_cost")
|
||
_启动执行(env)
|
||
_运行就绪(env, allow_failure=True)
|
||
after = svc.推进实验(env["actor"], eid)
|
||
assert next(u for u in after["units"] if u["kind"] == "comparison")["task_id"] is None
|
||
assert after["state"] == "reconciling"
|
||
assert any(
|
||
u["cost"] and u["cost"]["has_unknown"] and u["cost"]["total_usd"] is None
|
||
for u in after["units"]
|
||
)
|
||
assert all(r["model"] != "synthetic-independent-judge" for r in env["received"])
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert (
|
||
conn.execute(
|
||
"SELECT count(*) FROM evaluation.muse_budget_reservation WHERE state='unknown'"
|
||
).fetchone()[0]
|
||
== 1
|
||
)
|
||
assert any(
|
||
u["state"] != "completed" or (u["evidence"] and u["evidence"]["cost_state"] == "unknown")
|
||
for u in after["units"]
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253008",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="有效产物被改写且局部输出与证据哈希重签",
|
||
when="从工作面回查",
|
||
then=["反查S02原始输出拒绝伪造,生成与比较都受保护"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("kind", ["generation", "comparison"])
|
||
def test_篡改产物并重签局部证据仍不能替代S02原交付__253008(执行环境, monkeypatch, kind):
|
||
import copy
|
||
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
_启动执行(env)
|
||
_运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
real = svc.读取执行工作面(env["actor"], eid)
|
||
target = next(u["unit_id"] for u in real["units"] if u["kind"] == kind)
|
||
read = 评测存储.读取交付
|
||
|
||
def tampered(self, uid):
|
||
row = read(self, uid)
|
||
if row is not None and str(uid) == target:
|
||
row = copy.deepcopy(row)
|
||
if kind == "generation":
|
||
row["output"]["paragraphs"][0]["text"] = "伪造但结构仍合法的正文。"
|
||
else:
|
||
row["output"]["rationale"] = "事后改写的评判理由。"
|
||
digest = 固定哈希(row["output"])
|
||
row["output_hash"] = digest
|
||
row["evidence"]["structured_output_hash"] = digest
|
||
row["evidence"]["runtime"]["response_metadata"]["structured_output_hash"] = digest
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", tampered)
|
||
with pytest.raises(Muse错误, match="真实调用不一致"):
|
||
svc.读取执行工作面(env["actor"], eid)
|
||
with pytest.raises(Muse错误, match="真实调用不一致"):
|
||
svc.读取实验报告(env["actor"], eid)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-253009",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="未开始调用的已登记实验",
|
||
when="保存停止决定并重复取消",
|
||
then=["停止先持久化,任务收敛,不复活未开始评委"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_停止决定先落库且取消重放不会复活未开始评委__253009(执行环境):
|
||
from muse.效果评测.接口 import 评测错误
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
_启动执行(env)
|
||
result = svc.取消实验(env["actor"], eid, "stop-now")
|
||
assert result["state"] == "stopped"
|
||
assert sorted(u["state"] for u in result["units"]) == ["cancelled", "cancelled", "not_started"]
|
||
assert svc.取消实验(env["actor"], eid, "stop-again") == result
|
||
_运行就绪(env)
|
||
assert env["received"] == []
|
||
with pytest.raises(评测错误, match="已停止"):
|
||
svc.推进实验(env["actor"], eid)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25300a",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="一个单元完成,另一个只完成准备",
|
||
when="取消实验",
|
||
then=["已有产物与证据保留,不再发送剩余调用"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_停止发生在准备后仍禁止外发且保留已有交付__25300a(执行环境):
|
||
env = 执行环境
|
||
app = env["app"]
|
||
svc = app.要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
_启动执行(env)
|
||
# 顺序领取:先完成第一单元,第二单元停在已准备状态。
|
||
for _ in range(3):
|
||
claim = app.任务运行.领取步骤("eval-worker", ["eval.prepare", "eval.call.writer"])
|
||
assert claim is not None
|
||
app.任务运行.执行一步(claim)
|
||
before = svc.读取执行工作面(env["actor"], eid)
|
||
done = [u for u in before["units"] if u["state"] == "completed"]
|
||
assert len(done) == 1 and len(env["received"]) == 1
|
||
stopped = svc.取消实验(env["actor"], eid, "after-first")
|
||
retained = next(u for u in stopped["units"] if u["unit_id"] == done[0]["unit_id"])
|
||
assert retained["output"] == done[0]["output"] and retained["evidence"] == done[0]["evidence"]
|
||
_运行就绪(env)
|
||
assert len(env["received"]) == 1
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25300b",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="实际HTTP会话与隔离CLI配置",
|
||
when="启动、推进、读回及CLI子进程取消",
|
||
then=["同一实验和交付,验证模式仍为offline_contract,不因入口变化升级认证"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_HTTP启动推进与实际CLI读取取消沿同一实验__25300b(执行环境, tmp_path):
|
||
import json
|
||
import subprocess
|
||
import sys
|
||
|
||
import test_生产评测权限隔离 as 数据测试
|
||
from fastapi.testclient import TestClient
|
||
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.配置 import 读取配置
|
||
|
||
env = 执行环境
|
||
app = env["app"]
|
||
eid = env["exp"]["experiment_id"]
|
||
config = 数据测试._配置文件(env["pool"], tmp_path)
|
||
http = 创建应用(读取配置(config))
|
||
endpoint = "/api/v1/evaluation/experiments/" + eid
|
||
with TestClient(http, headers={"origin": "http://testserver"}) as client:
|
||
http.state.装配 = app
|
||
assert (
|
||
client.post(
|
||
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
|
||
).status_code
|
||
== 200
|
||
)
|
||
payload = {
|
||
"approval_ref": "synthetic-only",
|
||
"deadline": (datetime.now(UTC) + timedelta(minutes=10)).isoformat(),
|
||
}
|
||
started = client.post(endpoint + "/execution", json=payload)
|
||
assert started.status_code == 201, started.text
|
||
assert client.post(endpoint + "/execution", json=payload).json() == started.json()
|
||
_运行就绪(env)
|
||
advanced = client.post(endpoint + "/advance", json={})
|
||
assert advanced.status_code == 200, advanced.text
|
||
_运行就绪(env)
|
||
read = client.get(endpoint + "/execution").json()
|
||
assert read["state"] == "completed" and read["activation_status"] == "not_evaluated"
|
||
assert all(u["evidence"]["validation"]["mode"] == "offline_contract" for u in read["units"])
|
||
report = client.get(endpoint + "/report")
|
||
assert report.status_code == 200, report.text
|
||
assert report.json()["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
|
||
assert "ORACLE-SECRET" not in report.text
|
||
command = [sys.executable, "-I", "-m", "muse", "评测", str(config)]
|
||
result = subprocess.run(
|
||
[*command, "执行工作面", eid], cwd=tmp_path, capture_output=True, text=True, timeout=30
|
||
)
|
||
assert result.returncode == 0, result.stderr
|
||
assert json.loads(result.stdout) == read
|
||
report_result = subprocess.run(
|
||
[*command, "报告", eid], cwd=tmp_path, capture_output=True, text=True, timeout=30
|
||
)
|
||
assert report_result.returncode == 0, report_result.stderr
|
||
assert json.loads(report_result.stdout) == report.json()
|
||
stopped = subprocess.run(
|
||
[*command, "取消实验", eid], cwd=tmp_path, capture_output=True, text=True, timeout=30
|
||
)
|
||
assert stopped.returncode == 0, stopped.stderr
|
||
assert json.loads(stopped.stdout)["state"] == "stopped"
|
||
assert len(env["received"]) == 3
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25300c",
|
||
environment="隔离PG;S02与受控合成HTTP;独立模型仅测试策略",
|
||
given="HTTP已接收请求但尚未返回",
|
||
when="并发取消后接收迟到结果",
|
||
then=["不保存新业务产物;已发送调用的成本保留,不能归零"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_在途取消不保存业务产物也不把已发送成本归零__25300c(执行环境):
|
||
from threading import Thread
|
||
|
||
env = 执行环境
|
||
app = env["app"]
|
||
svc = app.要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
env["scripted"].append("inflight")
|
||
_启动执行(env)
|
||
prepare = app.任务运行.领取步骤("eval-worker", ["eval.prepare"])
|
||
app.任务运行.执行一步(prepare)
|
||
call = app.任务运行.领取步骤("eval-worker", ["eval.call.writer"])
|
||
errors = []
|
||
|
||
def execute():
|
||
try:
|
||
app.任务运行.执行一步(call)
|
||
except Muse错误 as exc:
|
||
errors.append(exc)
|
||
|
||
thread = Thread(target=execute, daemon=True)
|
||
thread.start()
|
||
try:
|
||
assert env["in_flight"].wait(5)
|
||
stopped = svc.取消实验(env["actor"], eid, "cancel-inflight")
|
||
assert stopped["state"] == "stopped"
|
||
finally:
|
||
env["respond"].set()
|
||
thread.join(timeout=8)
|
||
assert not thread.is_alive() and len(errors) == 1
|
||
result = svc.读取执行工作面(env["actor"], eid)
|
||
assert len(env["received"]) == 1
|
||
assert all(u["output"] is None for u in result["units"])
|
||
costs = [u["cost"] for u in result["units"] if u["cost"] and u["cost"]["calls"]]
|
||
assert len(costs) == 1 and len(costs[0]["calls"]) == 1
|
||
assert costs[0]["has_unknown"] or Decimal(costs[0]["total_usd"]) > 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-aee3df41e9b3",
|
||
environment="隔离PG及受控合成HTTP",
|
||
given="隔离PG冻结两名不同模型评委,两侧实际S02生成文本",
|
||
when="通过真实逐例任务完成两名评委的反向比较",
|
||
then=[
|
||
"两次真实评委请求,文本两侧相反、维度相同、任务/调用/模型独立且均完成",
|
||
"不发送先前判断、工具、身份哈希或会话历史",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{"judges": 2}], indirect=True)
|
||
def test_双评委反向文本且独立任务调用无共享报告__aee3df(执行环境):
|
||
import json
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
_启动执行(env)
|
||
_运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
requests = env["received"][2:]
|
||
assert len(requests) == 2
|
||
material = [json.loads(r["input"]) for r in requests]
|
||
assert material[0]["left"] == material[1]["right"]
|
||
assert material[0]["right"] == material[1]["left"]
|
||
assert material[0]["dimensions"] == material[1]["dimensions"]
|
||
for request, value in zip(requests, material, strict=True):
|
||
assert set(value) == {"left", "right", "dimensions"}
|
||
assert set(value["left"]) == set(value["right"]) == {"text"}
|
||
assert not request.get("tools") and not request.get("previous_response_id")
|
||
assert "previousReport" not in value and isinstance(request["input"], str)
|
||
work = svc.读取执行工作面(env["actor"], eid)
|
||
judges = [u for u in work["units"] if u["kind"] == "comparison"]
|
||
assert (
|
||
len({u["task_id"] for u in judges})
|
||
== len({u["evidence"]["runtime"]["response_metadata"]["call_id"] for u in judges})
|
||
== 2
|
||
)
|
||
assert len({u["evidence"]["model"] for u in judges}) == 2
|
||
assert all(u["state"] == "completed" for u in judges)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-259001",
|
||
environment="隔离PG及受控合成HTTP",
|
||
given="实际S02生成及独立评委回合",
|
||
when="保存原始引文和派生定位报告,读回后篡改定位并重签派生哈希",
|
||
then=[
|
||
"原始模型输出无位置;码点由代码派生且两类哈希分别可查",
|
||
"报告接口返回实际理由和引文,伪造定位即使重签也拒绝且不追加模型调用",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_原始模型引文及派生位置分别持久化和核验__259001(执行环境, monkeypatch):
|
||
from copy import deepcopy
|
||
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.效果评测.接口 import 评测错误
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
_启动执行(env)
|
||
_运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
_运行就绪(env)
|
||
work = svc.读取执行工作面(env["actor"], eid)
|
||
judge = next(u for u in work["units"] if u["kind"] == "comparison")
|
||
assert all(
|
||
set(q) == {"side", "quote"} for d in judge["output"]["dimensions"] for q in d["evidence"]
|
||
)
|
||
derived = judge["evidence"]["judgment"]
|
||
assert derived["contract"] == "quote-only-v2"
|
||
assert derived["report_hash"] == 固定哈希(derived["report"])
|
||
assert derived["raw_output_hash"] == 固定哈希(judge["output"])
|
||
assert all(
|
||
q["start"] == 0 and q["end"] == len(q["quote"])
|
||
for d in derived["report"]["dimensions"]
|
||
for q in d["evidence"]
|
||
)
|
||
report = svc.读取实验报告(env["actor"], eid)
|
||
decision = report["samples"][0]["decisions"][0]
|
||
assert decision["judgment_hash"] == derived["report_hash"]
|
||
assert decision["judgment"] == derived["report"]
|
||
original = 评测存储.读取交付
|
||
|
||
def 篡改(self, unit_id):
|
||
row = original(self, unit_id)
|
||
if row is not None and str(unit_id) == judge["unit_id"]:
|
||
row = deepcopy(row)
|
||
changed = row["evidence"]["judgment"]
|
||
changed["report"]["dimensions"][0]["evidence"][0]["start"] = 1
|
||
changed["report_hash"] = 固定哈希(changed["report"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", 篡改)
|
||
with pytest.raises(评测错误, match="定位报告"):
|
||
svc.读取实验报告(env["actor"], eid)
|
||
assert len(env["received"]) == 3
|
||
|
||
|
||
# ---- 生产评测权限隔离线:7项盲评原观察 ----
|
||
# 未知模式在创建实验前合同拒绝(fail closed,见 tests/契约/test_盲评与引文合同.py);
|
||
# 双评委真实派发链由基线 aee3df 承接(独立任务调用、盲化输入、模型各异、不共享报告),不重复写。
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-c7afbc32e342",
|
||
environment="离线,匿名样本、独立评委替身",
|
||
when="匿名分配样本、校验逐维引文并处理评委分歧。",
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_评委角色缺失的配置在固定实验条件时拒绝__c7afbc(原子任务环境):
|
||
"""旧“错误适配器角色拒绝”等价新版:judge 角色未在配置版本声明时拒绝。"""
|
||
from uuid import UUID
|
||
|
||
from muse.效果评测.实验条件 import 固定实验条件
|
||
from muse.效果评测.接口 import 实验请求, 目标版本, 评测错误, 配置选择
|
||
|
||
_, pool, _, _, _, _ = 原子任务环境
|
||
请求 = 实验请求(
|
||
dataset_version_id=UUID("12345678-1234-5678-8123-123456789abc"),
|
||
dataset_hash="0" * 64,
|
||
split="holdout",
|
||
target=目标版本(
|
||
kind="prompt",
|
||
target_ref="B05.generate",
|
||
version="1",
|
||
content_hash="1" * 64,
|
||
),
|
||
generator_role="writer",
|
||
generator=配置选择(config_id="writer", version="1"),
|
||
# 原子任务环境的配置版本只声明 writer 角色;把它用作评委即角色缺失。
|
||
judges=(配置选择(config_id="writer", version="1"),),
|
||
arms=("original", "candidate"),
|
||
dimensions=("清晰度",),
|
||
max_cost_usd=Decimal("1.0"),
|
||
max_calls_per_sample=3,
|
||
)
|
||
with pytest.raises(评测错误, match="缺少所需角色"):
|
||
固定实验条件(pool, 请求)
|