muse-agent-example/tests/集成/test_生产评测权限隔离.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

593 lines
26 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实隔离PG的数据集与实验封存;不调用外部模型或写生产内容。"""
import json
from concurrent.futures import ThreadPoolExecutor
from dataclasses import replace
from decimal import Decimal
from threading import Barrier
import psycopg
import pytest
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
from muse.效果评测.接口 import (
公开评测输入,
实验请求,
数据样本,
数据集发布,
目标版本,
评测服务,
评测错误,
配置选择,
)
pytestmark = pytest.mark.数据库
def _身份(purpose):
return 调用身份("eval-author", None, purpose, 内容用途.检测)
def _数据():
samples = []
for i, split in enumerate(("discovery", "calibration", "holdout", "holdout", "holdout")):
samples.append(
数据样本(
sample_id=f"sample-{i}",
source_ref=f"synthetic:source-{i}",
license_ref="synthetic:owned-fixture",
source_groups=(f"synthetic:book-{i}",),
split=split,
input=公开评测输入(
instruction="比较叙述是否清楚。",
original=f"第{i}段合成正文。",
context={"事实": "已知背景"},
),
answer={"secret": f"ORACLE-SECRET-{i}", "expected": "保留未公开的评判依据"},
)
)
return 数据集发布(dataset_id="synthetic-dataset", revision=1, samples=tuple(samples))
@pytest.fixture
def 数据集环境(应用测试库):
pools = 应用测试库
maintained = 评测服务(pools[用途.维护])
evaluated = 评测服务(pools[用途.评测])
result = maintained.发布数据集(_身份(用途.维护), _数据())
return pools, maintained, evaluated, result
@pytest.fixture
def 实验环境(数据集环境):
from muse.任务运行.接口 import 凭据引用, 提供方配置, 运行配置内容, 配置版本管理
pools, maintained, evaluated, data = 数据集环境
manager = 配置版本管理(pools[用途.评测])
for role, model in (("writer", "synthetic-writer"), ("judge", "synthetic-judge")):
content = 运行配置内容(
"direct",
"test-host",
"test-policy",
"muse-foundation-r2",
"test-budget",
{role: {"provider": "synthetic", "model": model, "thinking": "high", "tools": []}},
(凭据引用("token", "环境变量", "MUSE_UNUSED_EVAL_SECRET"),),
(提供方配置("synthetic", "responses", "http://127.0.0.1:9", "token"),),
"test-price",
)
manager.保存草案(role, "v1", content)
request = 实验请求(
dataset_version_id=data["version_id"],
dataset_hash=data["public_hash"],
split="holdout",
target=目标版本(
kind="rule", target_ref="candidate-rule", version="1", content_hash="a" * 64
),
generator_role="writer",
generator=配置选择(config_id="writer", version="v1"),
judges=(配置选择(config_id="judge", version="v1"),),
arms=("control", "treatment"),
dimensions=("叙事", "声音"),
max_cost_usd=Decimal("3.5"),
max_calls_per_sample=10,
)
return pools, evaluated, data, request
@pytest.mark.case_id(
"NC-w25-251001",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="真实生产和评测数据库角色",
when="读取与修改封存数据及答案",
then=["生产不能读评测,评测不能改输入、答案或正式命令"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_生产不能读取答案或评测数据且评测不能修改封存__251001(数据集环境):
pools, _, svc, data = 数据集环境
visible = svc.读取数据集(_身份(用途.评测), data["version_id"])
assert len(visible["public_manifest"]["samples"]) == 5
assert "ORACLE-SECRET" not in json.dumps(visible, ensure_ascii=False)
assert "answer_hash" not in visible
for sql in (
"SELECT * FROM oracle.muse_dataset_answers",
"SELECT * FROM evaluation.muse_dataset_version",
"SELECT * FROM oracle.muse_blind_assignment",
):
with pools[用途.生产].连接() as conn:
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
conn.execute(sql).fetchall()
with pools[用途.评测].连接() as conn:
for sql in (
"SELECT * FROM public.muse_document",
"SELECT * FROM public.muse_world_object",
"DELETE FROM evaluation.muse_dataset_version",
"UPDATE oracle.muse_dataset_answers SET answer_hash='x'",
"INSERT INTO public.muse_change_command(author_id,run_purpose,command_id,request_hash) "
"VALUES ('x','production','x','x')",
):
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
conn.execute(sql)
answers = conn.execute("SELECT answers FROM oracle.muse_dataset_answers").fetchone()[0]
assert answers[0]["answer"]["secret"] == "ORACLE-SECRET-0"
@pytest.mark.case_id(
"NC-w25-251002",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="已发布的数据与答案",
when="原样重放、改变答案及发布下一版本",
then=["完全一致才幂等、答案变更不覆盖历史、后继版本独立"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_同版本幂等但答案变化不能冒充同一数据集__251002(数据集环境):
pools, maint, _, result = 数据集环境
same = maint.发布数据集(_身份(用途.维护), _数据())
assert same["duplicate"] and same["version_id"] == result["version_id"]
req = _数据()
changed = req.model_copy(
update={
"samples": (
req.samples[0].model_copy(update={"answer": {"secret": "changed"}}),
*req.samples[1:],
)
}
)
with pytest.raises(评测错误, match="不能覆盖"):
maint.发布数据集(_身份(用途.维护), changed)
second = maint.发布数据集(_身份(用途.维护), changed.model_copy(update={"revision": 2}))
assert second["version_id"] != same["version_id"]
with pools[用途.维护].连接() as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 2
)
assert (
conn.execute(
"SELECT answers FROM oracle.muse_dataset_answers WHERE version_id=%s",
(result["version_id"],),
).fetchone()[0][0]["answer"]["secret"]
== "ORACLE-SECRET-0"
)
with pytest.raises(psycopg.Error), conn.transaction():
conn.execute("UPDATE evaluation.muse_dataset_version SET public_hash=%s", ("0" * 64,))
@pytest.mark.case_id(
"NC-w25-251003",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="答案写入中注入数据库错误",
when="发布数据集",
then=["公开数据和答案全部回滚且不回显私有驱动原文"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_答案写入失败回滚整个发布__251003(应用测试库):
pool = 应用测试库[用途.维护]
with pool.连接() as conn, conn.transaction():
conn.execute(
"CREATE FUNCTION oracle.fail_answer() RETURNS trigger LANGUAGE plpgsql AS "
"$$ BEGIN RAISE EXCEPTION 'synthetic failure'; END $$"
)
conn.execute(
"CREATE TRIGGER fail_answer BEFORE INSERT ON oracle.muse_dataset_answers "
"FOR EACH ROW EXECUTE FUNCTION oracle.fail_answer()"
)
with pytest.raises(评测错误, match="评测存储操作失败"):
评测服务(pool).发布数据集(_身份(用途.维护), _数据())
with pool.连接(只读=True) as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
)
assert conn.execute("SELECT count(*) FROM oracle.muse_dataset_answers").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-251004",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="真实S02配置草案与保留集",
when="登记实验并读取生成输入",
then=["固定全部样本和实际配置哈希,生成只见公开输入,没有模型任务"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_实验冻结实际配置与全分割且模型输入无答案映射__251004(实验环境):
pools, svc, data, req = 实验环境
actor = _身份(用途.评测)
result = svc.创建实验(actor, "experiment-one", req)
assert result == svc.创建实验(actor, "experiment-one", req)
assert result["registration"] == "registered"
assert set(result["sample_ids"]) == {"sample-2", "sample-3", "sample-4"}
assert result["conditions"]["generator"]["model"] == "synthetic-writer"
assert result["conditions"]["judges"][0]["model"] == "synthetic-judge"
assert result["conditions"]["generator"]["config_hash"]
assert "MUSE_UNUSED_EVAL_SECRET" not in json.dumps(result)
payload = svc.读取生成输入(actor, result["experiment_id"], "sample-2")
assert set(payload) == {"instruction", "original", "context"}
assert payload == _数据().samples[2].input.model_dump()
assert "ORACLE" not in json.dumps(payload) and "control" not in json.dumps(payload)
with pytest.raises(评测错误):
svc.读取生成输入(actor, result["experiment_id"], "sample-0")
with pytest.raises(评测错误):
svc.读取实验(replace(actor, 作者="other-author"), result["experiment_id"])
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 1
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 3
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_budget_reservation").fetchone()[0]
== 0
)
@pytest.mark.case_id(
"NC-w25-251005",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="错数据哈希、缺配置、错角色或生成同模型评委",
when="创建实验",
then=["前置拒绝且无实验与匿名映射残留"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", ["hash", "missing_profile", "wrong_role", "same_model"])
def test_实验前置错误不留下条件或匿名映射__251005(实验环境, bad):
from muse.共享.错误 import Muse错误
pools, svc, _, req = 实验环境
if bad == "hash":
req = req.model_copy(update={"dataset_hash": "0" * 64})
elif bad == "missing_profile":
req = req.model_copy(update={"generator": 配置选择(config_id="absent", version="v1")})
elif bad == "wrong_role":
req = req.model_copy(update={"generator": 配置选择(config_id="judge", version="v1")})
else:
from muse.任务运行.接口 import 配置版本管理
profiles = 配置版本管理(pools[用途.评测])
v = profiles.读取版本("judge", "v1")
changed = replace(
v.内容,
角色配置={
"judge": {
"provider": "synthetic",
"model": "synthetic-writer",
"thinking": "high",
"tools": [],
}
},
)
profiles.保存草案("same-model", "v1", changed)
req = req.model_copy(update={"judges": (配置选择(config_id="same-model", version="v1"),)})
with pytest.raises(Muse错误):
svc.创建实验(_身份(用途.评测), "bad-experiment", req)
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-251006",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="已登记命令和匿名映射",
when="同命令更换目标或更新映射",
then=["拒绝覆盖,原实验不变"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_重复命令不能更换实验目标与匿名映射__251006(实验环境):
pools, svc, _, req = 实验环境
actor = _身份(用途.评测)
old = svc.创建实验(actor, "bound-command", req)
with pytest.raises(评测错误, match="不能更换"):
svc.创建实验(
actor,
"bound-command",
req.model_copy(
update={"target": req.target.model_copy(update={"content_hash": "b" * 64})}
),
)
assert svc.读取实验(actor, old["experiment_id"]) == old
with pools[用途.评测].连接() as conn:
with pytest.raises(psycopg.Error), conn.transaction():
conn.execute("UPDATE oracle.muse_blind_assignment SET arm_order='[]'")
@pytest.mark.case_id(
"NC-w25-251007",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="生产或评测用途",
when="调用数据集发布",
then=["维护外用途拒绝且没有新版本"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测])
def test_非维护用途不能发布数据集__251007(应用测试库, purpose):
with pytest.raises(评测错误):
评测服务(应用测试库[purpose]).发布数据集(_身份(purpose), _数据())
with 应用测试库[用途.维护].连接(只读=True) as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
)
@pytest.mark.case_id(
"NC-w25-251008",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="两个并发调用同一发布或实验命令",
when="使用真实独立PG连接提交",
then=["仅一份数据和匿名映射,重放返回同一实验"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("operation", ["dataset", "experiment"])
def test_并发同一发布或实验命令只保存一份__251008(实验环境, operation):
pools, svc, _, req = 实验环境
barrier = Barrier(2)
def execute():
barrier.wait(timeout=5)
if operation == "dataset":
return 评测服务(pools[用途.维护]).发布数据集(
_身份(用途.维护), _数据().model_copy(update={"dataset_id": "concurrent"})
)
return svc.创建实验(_身份(用途.评测), "concurrent", req)
with ThreadPoolExecutor(max_workers=2) as executor:
futures = [executor.submit(execute) for _ in range(2)]
a, b = [f.result(timeout=15) for f in futures]
with pools[用途.评测].连接(只读=True) as conn:
if operation == "dataset":
assert a["version_id"] == b["version_id"]
assert {a["duplicate"], b["duplicate"]} == {True, False}
assert (
conn.execute(
"SELECT count(*) FROM oracle.muse_dataset_answers WHERE version_id=%s",
(a["version_id"],),
).fetchone()[0]
== 1
)
else:
assert a == b
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 1
)
assert (
conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 3
)
@pytest.mark.case_id(
"NC-w25-251009",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="匿名映射写入中途失败",
when="创建实验后移除故障按原命令重试",
then=["首次全部回滚,重试成功且不重复映射"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_匿名映射中途失败回滚全部实验并能原命令重试__251009(实验环境):
pools, svc, _, req = 实验环境
with pools[用途.维护].连接() as conn, conn.transaction():
conn.execute(
"CREATE FUNCTION oracle.fail_assignment() RETURNS trigger LANGUAGE plpgsql AS "
"$$ BEGIN IF NEW.sample_id='sample-3' THEN RAISE EXCEPTION 'synthetic failure'; "
"END IF; RETURN NEW; END $$"
)
conn.execute(
"CREATE TRIGGER fail_assignment BEFORE INSERT ON oracle.muse_blind_assignment "
"FOR EACH ROW EXECUTE FUNCTION oracle.fail_assignment()"
)
with pytest.raises(评测错误, match="评测存储操作失败"):
svc.创建实验(_身份(用途.评测), "recover", req)
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 0
with pools[用途.维护].连接() as conn, conn.transaction():
conn.execute("DROP TRIGGER fail_assignment ON oracle.muse_blind_assignment")
result = svc.创建实验(_身份(用途.评测), "recover", req)
assert len(result["sample_ids"]) == 3
assert svc.创建实验(_身份(用途.评测), "recover", req) == result
@pytest.mark.case_id(
"NC-w25-25100a",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="不同配置引用指向相同实际评委模型",
when="固定独立评委",
then=["不能冒充多个独立评委,实验不落库"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_相同评委模型不能用不同配置冒充多个独立评委__25100a(实验环境):
from muse.任务运行.接口 import 配置版本管理
pools, svc, _, req = 实验环境
profiles = 配置版本管理(pools[用途.评测])
profiles.保存草案("judge-copy", "v1", profiles.读取版本("judge", "v1").内容)
changed = req.model_copy(
update={"judges": (*req.judges, 配置选择(config_id="judge-copy", version="v1"))}
)
with pytest.raises(评测错误, match="不能重复"):
svc.创建实验(_身份(用途.评测), "duplicate-judge", changed)
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-25100b",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="发布的版本没有保留集",
when="指定保留集创建实验",
then=["不能创建空实验或当成通过"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_空保留集不能登记为可运行实验__25100b(实验环境):
pools, svc, _, req = 实验环境
data = 评测服务(pools[用途.维护]).发布数据集(
_身份(用途.维护),
_数据().model_copy(
update={"dataset_id": "without-holdout", "samples": _数据().samples[:2]}
),
)
changed = 实验请求.model_validate(
{
**req.model_dump(mode="json"),
"dataset_version_id": data["version_id"],
"dataset_hash": data["public_hash"],
}
)
with pytest.raises(评测错误, match="没有样本"):
svc.创建实验(_身份(用途.评测), "empty-holdout", changed)
def _配置文件(pool, tmp_path):
secret = tmp_path / "author-password"
secret.write_text("synthetic-evaluation-only")
secret.chmod(0o600)
config = tmp_path / f"{pool.用途.value}.toml"
config.write_text(
'["数据库"]\n"取值方式"="受控存储"\n"位置"='
+ json.dumps(pool.引用.位置)
+ '\n["运行"]\n"用途"='
+ json.dumps(pool.用途.value)
+ '\n["资源"]\n"发布身份"="test"\n["HTTP"]\n"作者ID"="eval-author"'
+ '\n"口令文件"='
+ json.dumps(str(secret))
+ '\n"公开地址"="http://testserver"\n"允许来源"=["http://testserver"]\n'
)
return config
@pytest.mark.case_id(
"NC-w25-25100c",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="维护/评测配置与合成请求JSON",
when="从隔离cwd调用实际CLI并重放",
then=["发布/创建/读取回查同一版本,用途和坏参数受控拒绝且不回显私有输入"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_实际CLI发布实验与读取沿用途及固定版本重放__25100c(实验环境, tmp_path):
import subprocess
import sys
pools, svc, _, req = 实验环境
configs = {p: _配置文件(pools[p], tmp_path) for p in 用途}
def run(purpose, action, target, *args):
return subprocess.run(
[
sys.executable,
"-I",
"-m",
"muse",
"评测",
str(configs[purpose]),
action,
str(target),
*args,
],
cwd=tmp_path,
capture_output=True,
text=True,
timeout=30,
)
path = tmp_path / "dataset.json"
path.write_text(_数据().model_copy(update={"dataset_id": "from-cli"}).model_dump_json())
first = run(用途.维护, "发布数据集", path)
assert first.returncode == 0, first.stderr
data = json.loads(first.stdout)
assert not data["duplicate"]
replay = run(用途.维护, "发布数据集", path)
assert replay.returncode == 0 and json.loads(replay.stdout)["duplicate"]
denied = run(用途.评测, "发布数据集", path)
assert denied.returncode == 1 and json.loads(denied.stderr)["code"] == "MUSE_PURPOSE_VIOLATION"
body = req.model_dump(mode="json")
body.update(dataset_version_id=data["version_id"], dataset_hash=data["public_hash"])
path = tmp_path / "experiment.json"
path.write_text(json.dumps({"command_id": "cli-experiment", "request": body}))
created = run(用途.评测, "创建实验", path)
assert created.returncode == 0, created.stderr
result = json.loads(created.stdout)
replay = run(用途.评测, "创建实验", path)
assert replay.returncode == 0 and json.loads(replay.stdout) == result
assert svc.读取实验(_身份(用途.评测), result["experiment_id"]) == result
read = run(用途.评测, "生成输入", result["experiment_id"], "--样本", "sample-2")
assert read.returncode == 0 and json.loads(read.stdout) == _数据().samples[2].input.model_dump()
malformed = run(用途.评测, "实验", "not-a-uuid")
assert malformed.returncode == 1
assert json.loads(malformed.stderr)["code"] == "EVALUATION_CONTRACT_FAILED"
path.write_text(json.dumps({"command_id": "bad", "request": body, "oracle": "PRIVATE-MARKER"}))
malformed = run(用途.评测, "创建实验", path)
assert malformed.returncode == 1 and "PRIVATE-MARKER" not in malformed.stderr
assert "Traceback" not in malformed.stderr
@pytest.mark.case_id(
"NC-w25-25100d",
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
given="真实HTTP会话及三类用途",
when="经过来源与身份校验调用发布/实验端点",
then=["维护只发布、评测才建实验,生产拒绝、伪造作者与坏ID拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测, 用途.维护])
def test_HTTP会话来源与数据集实验用途强制生效__25100d(实验环境, tmp_path, purpose):
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
pools, _, data, req = 实验环境
http = 创建应用(读取配置(_配置文件(pools[purpose], tmp_path)))
with TestClient(http, headers={"origin": "http://testserver"}) as client:
endpoint = "/api/v1/evaluation/experiments"
body = {"command_id": "http-experiment", "request": req.model_dump(mode="json")}
assert client.post(endpoint, json=body).status_code == 401
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
assert (
client.post(endpoint, json=body, headers={"origin": "http://invalid"}).status_code
== 403
)
created = client.post(endpoint, json=body)
if purpose is 用途.评测:
assert created.status_code == 201, created.text
result = created.json()
assert client.post(endpoint, json=body).json() == result
eid = result["experiment_id"]
assert client.get(f"{endpoint}/{eid}").json() == result
assert (
client.get(f"{endpoint}/{eid}/samples/sample-2/input").json()
== _数据().samples[2].input.model_dump()
)
public = client.get("/api/v1/evaluation/datasets/" + data["version_id"])
assert public.status_code == 200 and "ORACLE-SECRET" not in public.text
forged = client.post(endpoint, json={**body, "author_id": "PRIVATE-MARKER"})
assert forged.status_code == 422 and "PRIVATE-MARKER" not in forged.text
assert client.get(f"{endpoint}/invalid").status_code == 422
else:
assert created.status_code == 403
assert (
client.get("/api/v1/evaluation/datasets/" + data["version_id"]).status_code == 403
)
published = client.post("/api/v1/evaluation/datasets", json=_数据().model_dump(mode="json"))
assert published.status_code == (201 if purpose is 用途.维护 else 403)
assert "ORACLE-SECRET" not in published.text