实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
593 lines
26 KiB
Python
593 lines
26 KiB
Python
"""真实隔离PG的数据集与实验封存;不调用外部模型或写生产内容。"""
|
||
|
||
import json
|
||
from concurrent.futures import ThreadPoolExecutor
|
||
from dataclasses import replace
|
||
from decimal import Decimal
|
||
from threading import Barrier
|
||
|
||
import psycopg
|
||
import pytest
|
||
|
||
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
|
||
from muse.效果评测.接口 import (
|
||
公开评测输入,
|
||
实验请求,
|
||
数据样本,
|
||
数据集发布,
|
||
目标版本,
|
||
评测服务,
|
||
评测错误,
|
||
配置选择,
|
||
)
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
|
||
|
||
def _身份(purpose):
|
||
return 调用身份("eval-author", None, purpose, 内容用途.检测)
|
||
|
||
|
||
def _数据():
|
||
samples = []
|
||
for i, split in enumerate(("discovery", "calibration", "holdout", "holdout", "holdout")):
|
||
samples.append(
|
||
数据样本(
|
||
sample_id=f"sample-{i}",
|
||
source_ref=f"synthetic:source-{i}",
|
||
license_ref="synthetic:owned-fixture",
|
||
source_groups=(f"synthetic:book-{i}",),
|
||
split=split,
|
||
input=公开评测输入(
|
||
instruction="比较叙述是否清楚。",
|
||
original=f"第{i}段合成正文。",
|
||
context={"事实": "已知背景"},
|
||
),
|
||
answer={"secret": f"ORACLE-SECRET-{i}", "expected": "保留未公开的评判依据"},
|
||
)
|
||
)
|
||
return 数据集发布(dataset_id="synthetic-dataset", revision=1, samples=tuple(samples))
|
||
|
||
|
||
@pytest.fixture
|
||
def 数据集环境(应用测试库):
|
||
pools = 应用测试库
|
||
maintained = 评测服务(pools[用途.维护])
|
||
evaluated = 评测服务(pools[用途.评测])
|
||
result = maintained.发布数据集(_身份(用途.维护), _数据())
|
||
return pools, maintained, evaluated, result
|
||
|
||
|
||
@pytest.fixture
|
||
def 实验环境(数据集环境):
|
||
from muse.任务运行.接口 import 凭据引用, 提供方配置, 运行配置内容, 配置版本管理
|
||
|
||
pools, maintained, evaluated, data = 数据集环境
|
||
manager = 配置版本管理(pools[用途.评测])
|
||
for role, model in (("writer", "synthetic-writer"), ("judge", "synthetic-judge")):
|
||
content = 运行配置内容(
|
||
"direct",
|
||
"test-host",
|
||
"test-policy",
|
||
"muse-foundation-r2",
|
||
"test-budget",
|
||
{role: {"provider": "synthetic", "model": model, "thinking": "high", "tools": []}},
|
||
(凭据引用("token", "环境变量", "MUSE_UNUSED_EVAL_SECRET"),),
|
||
(提供方配置("synthetic", "responses", "http://127.0.0.1:9", "token"),),
|
||
"test-price",
|
||
)
|
||
manager.保存草案(role, "v1", content)
|
||
request = 实验请求(
|
||
dataset_version_id=data["version_id"],
|
||
dataset_hash=data["public_hash"],
|
||
split="holdout",
|
||
target=目标版本(
|
||
kind="rule", target_ref="candidate-rule", version="1", content_hash="a" * 64
|
||
),
|
||
generator_role="writer",
|
||
generator=配置选择(config_id="writer", version="v1"),
|
||
judges=(配置选择(config_id="judge", version="v1"),),
|
||
arms=("control", "treatment"),
|
||
dimensions=("叙事", "声音"),
|
||
max_cost_usd=Decimal("3.5"),
|
||
max_calls_per_sample=10,
|
||
)
|
||
return pools, evaluated, data, request
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251001",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="真实生产和评测数据库角色",
|
||
when="读取与修改封存数据及答案",
|
||
then=["生产不能读评测,评测不能改输入、答案或正式命令"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_生产不能读取答案或评测数据且评测不能修改封存__251001(数据集环境):
|
||
pools, _, svc, data = 数据集环境
|
||
visible = svc.读取数据集(_身份(用途.评测), data["version_id"])
|
||
assert len(visible["public_manifest"]["samples"]) == 5
|
||
assert "ORACLE-SECRET" not in json.dumps(visible, ensure_ascii=False)
|
||
assert "answer_hash" not in visible
|
||
for sql in (
|
||
"SELECT * FROM oracle.muse_dataset_answers",
|
||
"SELECT * FROM evaluation.muse_dataset_version",
|
||
"SELECT * FROM oracle.muse_blind_assignment",
|
||
):
|
||
with pools[用途.生产].连接() as conn:
|
||
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
|
||
conn.execute(sql).fetchall()
|
||
with pools[用途.评测].连接() as conn:
|
||
for sql in (
|
||
"SELECT * FROM public.muse_document",
|
||
"SELECT * FROM public.muse_world_object",
|
||
"DELETE FROM evaluation.muse_dataset_version",
|
||
"UPDATE oracle.muse_dataset_answers SET answer_hash='x'",
|
||
"INSERT INTO public.muse_change_command(author_id,run_purpose,command_id,request_hash) "
|
||
"VALUES ('x','production','x','x')",
|
||
):
|
||
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
|
||
conn.execute(sql)
|
||
answers = conn.execute("SELECT answers FROM oracle.muse_dataset_answers").fetchone()[0]
|
||
assert answers[0]["answer"]["secret"] == "ORACLE-SECRET-0"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251002",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="已发布的数据与答案",
|
||
when="原样重放、改变答案及发布下一版本",
|
||
then=["完全一致才幂等、答案变更不覆盖历史、后继版本独立"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_同版本幂等但答案变化不能冒充同一数据集__251002(数据集环境):
|
||
pools, maint, _, result = 数据集环境
|
||
same = maint.发布数据集(_身份(用途.维护), _数据())
|
||
assert same["duplicate"] and same["version_id"] == result["version_id"]
|
||
req = _数据()
|
||
changed = req.model_copy(
|
||
update={
|
||
"samples": (
|
||
req.samples[0].model_copy(update={"answer": {"secret": "changed"}}),
|
||
*req.samples[1:],
|
||
)
|
||
}
|
||
)
|
||
with pytest.raises(评测错误, match="不能覆盖"):
|
||
maint.发布数据集(_身份(用途.维护), changed)
|
||
second = maint.发布数据集(_身份(用途.维护), changed.model_copy(update={"revision": 2}))
|
||
assert second["version_id"] != same["version_id"]
|
||
with pools[用途.维护].连接() as conn:
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 2
|
||
)
|
||
assert (
|
||
conn.execute(
|
||
"SELECT answers FROM oracle.muse_dataset_answers WHERE version_id=%s",
|
||
(result["version_id"],),
|
||
).fetchone()[0][0]["answer"]["secret"]
|
||
== "ORACLE-SECRET-0"
|
||
)
|
||
with pytest.raises(psycopg.Error), conn.transaction():
|
||
conn.execute("UPDATE evaluation.muse_dataset_version SET public_hash=%s", ("0" * 64,))
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251003",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="答案写入中注入数据库错误",
|
||
when="发布数据集",
|
||
then=["公开数据和答案全部回滚且不回显私有驱动原文"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_答案写入失败回滚整个发布__251003(应用测试库):
|
||
pool = 应用测试库[用途.维护]
|
||
with pool.连接() as conn, conn.transaction():
|
||
conn.execute(
|
||
"CREATE FUNCTION oracle.fail_answer() RETURNS trigger LANGUAGE plpgsql AS "
|
||
"$$ BEGIN RAISE EXCEPTION 'synthetic failure'; END $$"
|
||
)
|
||
conn.execute(
|
||
"CREATE TRIGGER fail_answer BEFORE INSERT ON oracle.muse_dataset_answers "
|
||
"FOR EACH ROW EXECUTE FUNCTION oracle.fail_answer()"
|
||
)
|
||
with pytest.raises(评测错误, match="评测存储操作失败"):
|
||
评测服务(pool).发布数据集(_身份(用途.维护), _数据())
|
||
with pool.连接(只读=True) as conn:
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
|
||
)
|
||
assert conn.execute("SELECT count(*) FROM oracle.muse_dataset_answers").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251004",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="真实S02配置草案与保留集",
|
||
when="登记实验并读取生成输入",
|
||
then=["固定全部样本和实际配置哈希,生成只见公开输入,没有模型任务"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_实验冻结实际配置与全分割且模型输入无答案映射__251004(实验环境):
|
||
pools, svc, data, req = 实验环境
|
||
actor = _身份(用途.评测)
|
||
result = svc.创建实验(actor, "experiment-one", req)
|
||
assert result == svc.创建实验(actor, "experiment-one", req)
|
||
assert result["registration"] == "registered"
|
||
assert set(result["sample_ids"]) == {"sample-2", "sample-3", "sample-4"}
|
||
assert result["conditions"]["generator"]["model"] == "synthetic-writer"
|
||
assert result["conditions"]["judges"][0]["model"] == "synthetic-judge"
|
||
assert result["conditions"]["generator"]["config_hash"]
|
||
assert "MUSE_UNUSED_EVAL_SECRET" not in json.dumps(result)
|
||
payload = svc.读取生成输入(actor, result["experiment_id"], "sample-2")
|
||
assert set(payload) == {"instruction", "original", "context"}
|
||
assert payload == _数据().samples[2].input.model_dump()
|
||
assert "ORACLE" not in json.dumps(payload) and "control" not in json.dumps(payload)
|
||
with pytest.raises(评测错误):
|
||
svc.读取生成输入(actor, result["experiment_id"], "sample-0")
|
||
with pytest.raises(评测错误):
|
||
svc.读取实验(replace(actor, 作者="other-author"), result["experiment_id"])
|
||
with pools[用途.评测].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 1
|
||
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 3
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_budget_reservation").fetchone()[0]
|
||
== 0
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251005",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="错数据哈希、缺配置、错角色或生成同模型评委",
|
||
when="创建实验",
|
||
then=["前置拒绝且无实验与匿名映射残留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("bad", ["hash", "missing_profile", "wrong_role", "same_model"])
|
||
def test_实验前置错误不留下条件或匿名映射__251005(实验环境, bad):
|
||
from muse.共享.错误 import Muse错误
|
||
|
||
pools, svc, _, req = 实验环境
|
||
if bad == "hash":
|
||
req = req.model_copy(update={"dataset_hash": "0" * 64})
|
||
elif bad == "missing_profile":
|
||
req = req.model_copy(update={"generator": 配置选择(config_id="absent", version="v1")})
|
||
elif bad == "wrong_role":
|
||
req = req.model_copy(update={"generator": 配置选择(config_id="judge", version="v1")})
|
||
else:
|
||
from muse.任务运行.接口 import 配置版本管理
|
||
|
||
profiles = 配置版本管理(pools[用途.评测])
|
||
v = profiles.读取版本("judge", "v1")
|
||
changed = replace(
|
||
v.内容,
|
||
角色配置={
|
||
"judge": {
|
||
"provider": "synthetic",
|
||
"model": "synthetic-writer",
|
||
"thinking": "high",
|
||
"tools": [],
|
||
}
|
||
},
|
||
)
|
||
profiles.保存草案("same-model", "v1", changed)
|
||
req = req.model_copy(update={"judges": (配置选择(config_id="same-model", version="v1"),)})
|
||
with pytest.raises(Muse错误):
|
||
svc.创建实验(_身份(用途.评测), "bad-experiment", req)
|
||
with pools[用途.评测].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
|
||
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251006",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="已登记命令和匿名映射",
|
||
when="同命令更换目标或更新映射",
|
||
then=["拒绝覆盖,原实验不变"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_重复命令不能更换实验目标与匿名映射__251006(实验环境):
|
||
pools, svc, _, req = 实验环境
|
||
actor = _身份(用途.评测)
|
||
old = svc.创建实验(actor, "bound-command", req)
|
||
with pytest.raises(评测错误, match="不能更换"):
|
||
svc.创建实验(
|
||
actor,
|
||
"bound-command",
|
||
req.model_copy(
|
||
update={"target": req.target.model_copy(update={"content_hash": "b" * 64})}
|
||
),
|
||
)
|
||
assert svc.读取实验(actor, old["experiment_id"]) == old
|
||
with pools[用途.评测].连接() as conn:
|
||
with pytest.raises(psycopg.Error), conn.transaction():
|
||
conn.execute("UPDATE oracle.muse_blind_assignment SET arm_order='[]'")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251007",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="生产或评测用途",
|
||
when="调用数据集发布",
|
||
then=["维护外用途拒绝且没有新版本"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测])
|
||
def test_非维护用途不能发布数据集__251007(应用测试库, purpose):
|
||
with pytest.raises(评测错误):
|
||
评测服务(应用测试库[purpose]).发布数据集(_身份(purpose), _数据())
|
||
with 应用测试库[用途.维护].连接(只读=True) as conn:
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251008",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="两个并发调用同一发布或实验命令",
|
||
when="使用真实独立PG连接提交",
|
||
then=["仅一份数据和匿名映射,重放返回同一实验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("operation", ["dataset", "experiment"])
|
||
def test_并发同一发布或实验命令只保存一份__251008(实验环境, operation):
|
||
pools, svc, _, req = 实验环境
|
||
barrier = Barrier(2)
|
||
|
||
def execute():
|
||
barrier.wait(timeout=5)
|
||
if operation == "dataset":
|
||
return 评测服务(pools[用途.维护]).发布数据集(
|
||
_身份(用途.维护), _数据().model_copy(update={"dataset_id": "concurrent"})
|
||
)
|
||
return svc.创建实验(_身份(用途.评测), "concurrent", req)
|
||
|
||
with ThreadPoolExecutor(max_workers=2) as executor:
|
||
futures = [executor.submit(execute) for _ in range(2)]
|
||
a, b = [f.result(timeout=15) for f in futures]
|
||
with pools[用途.评测].连接(只读=True) as conn:
|
||
if operation == "dataset":
|
||
assert a["version_id"] == b["version_id"]
|
||
assert {a["duplicate"], b["duplicate"]} == {True, False}
|
||
assert (
|
||
conn.execute(
|
||
"SELECT count(*) FROM oracle.muse_dataset_answers WHERE version_id=%s",
|
||
(a["version_id"],),
|
||
).fetchone()[0]
|
||
== 1
|
||
)
|
||
else:
|
||
assert a == b
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 1
|
||
)
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 3
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-251009",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="匿名映射写入中途失败",
|
||
when="创建实验后移除故障按原命令重试",
|
||
then=["首次全部回滚,重试成功且不重复映射"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_匿名映射中途失败回滚全部实验并能原命令重试__251009(实验环境):
|
||
pools, svc, _, req = 实验环境
|
||
with pools[用途.维护].连接() as conn, conn.transaction():
|
||
conn.execute(
|
||
"CREATE FUNCTION oracle.fail_assignment() RETURNS trigger LANGUAGE plpgsql AS "
|
||
"$$ BEGIN IF NEW.sample_id='sample-3' THEN RAISE EXCEPTION 'synthetic failure'; "
|
||
"END IF; RETURN NEW; END $$"
|
||
)
|
||
conn.execute(
|
||
"CREATE TRIGGER fail_assignment BEFORE INSERT ON oracle.muse_blind_assignment "
|
||
"FOR EACH ROW EXECUTE FUNCTION oracle.fail_assignment()"
|
||
)
|
||
with pytest.raises(评测错误, match="评测存储操作失败"):
|
||
svc.创建实验(_身份(用途.评测), "recover", req)
|
||
with pools[用途.评测].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
|
||
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 0
|
||
with pools[用途.维护].连接() as conn, conn.transaction():
|
||
conn.execute("DROP TRIGGER fail_assignment ON oracle.muse_blind_assignment")
|
||
result = svc.创建实验(_身份(用途.评测), "recover", req)
|
||
assert len(result["sample_ids"]) == 3
|
||
assert svc.创建实验(_身份(用途.评测), "recover", req) == result
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25100a",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="不同配置引用指向相同实际评委模型",
|
||
when="固定独立评委",
|
||
then=["不能冒充多个独立评委,实验不落库"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_相同评委模型不能用不同配置冒充多个独立评委__25100a(实验环境):
|
||
from muse.任务运行.接口 import 配置版本管理
|
||
|
||
pools, svc, _, req = 实验环境
|
||
profiles = 配置版本管理(pools[用途.评测])
|
||
profiles.保存草案("judge-copy", "v1", profiles.读取版本("judge", "v1").内容)
|
||
changed = req.model_copy(
|
||
update={"judges": (*req.judges, 配置选择(config_id="judge-copy", version="v1"))}
|
||
)
|
||
with pytest.raises(评测错误, match="不能重复"):
|
||
svc.创建实验(_身份(用途.评测), "duplicate-judge", changed)
|
||
with pools[用途.评测].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25100b",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="发布的版本没有保留集",
|
||
when="指定保留集创建实验",
|
||
then=["不能创建空实验或当成通过"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_空保留集不能登记为可运行实验__25100b(实验环境):
|
||
pools, svc, _, req = 实验环境
|
||
data = 评测服务(pools[用途.维护]).发布数据集(
|
||
_身份(用途.维护),
|
||
_数据().model_copy(
|
||
update={"dataset_id": "without-holdout", "samples": _数据().samples[:2]}
|
||
),
|
||
)
|
||
changed = 实验请求.model_validate(
|
||
{
|
||
**req.model_dump(mode="json"),
|
||
"dataset_version_id": data["version_id"],
|
||
"dataset_hash": data["public_hash"],
|
||
}
|
||
)
|
||
with pytest.raises(评测错误, match="没有样本"):
|
||
svc.创建实验(_身份(用途.评测), "empty-holdout", changed)
|
||
|
||
|
||
def _配置文件(pool, tmp_path):
|
||
secret = tmp_path / "author-password"
|
||
secret.write_text("synthetic-evaluation-only")
|
||
secret.chmod(0o600)
|
||
config = tmp_path / f"{pool.用途.value}.toml"
|
||
config.write_text(
|
||
'["数据库"]\n"取值方式"="受控存储"\n"位置"='
|
||
+ json.dumps(pool.引用.位置)
|
||
+ '\n["运行"]\n"用途"='
|
||
+ json.dumps(pool.用途.value)
|
||
+ '\n["资源"]\n"发布身份"="test"\n["HTTP"]\n"作者ID"="eval-author"'
|
||
+ '\n"口令文件"='
|
||
+ json.dumps(str(secret))
|
||
+ '\n"公开地址"="http://testserver"\n"允许来源"=["http://testserver"]\n'
|
||
)
|
||
return config
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25100c",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="维护/评测配置与合成请求JSON",
|
||
when="从隔离cwd调用实际CLI并重放",
|
||
then=["发布/创建/读取回查同一版本,用途和坏参数受控拒绝且不回显私有输入"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_实际CLI发布实验与读取沿用途及固定版本重放__25100c(实验环境, tmp_path):
|
||
import subprocess
|
||
import sys
|
||
|
||
pools, svc, _, req = 实验环境
|
||
configs = {p: _配置文件(pools[p], tmp_path) for p in 用途}
|
||
|
||
def run(purpose, action, target, *args):
|
||
return subprocess.run(
|
||
[
|
||
sys.executable,
|
||
"-I",
|
||
"-m",
|
||
"muse",
|
||
"评测",
|
||
str(configs[purpose]),
|
||
action,
|
||
str(target),
|
||
*args,
|
||
],
|
||
cwd=tmp_path,
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=30,
|
||
)
|
||
|
||
path = tmp_path / "dataset.json"
|
||
path.write_text(_数据().model_copy(update={"dataset_id": "from-cli"}).model_dump_json())
|
||
first = run(用途.维护, "发布数据集", path)
|
||
assert first.returncode == 0, first.stderr
|
||
data = json.loads(first.stdout)
|
||
assert not data["duplicate"]
|
||
replay = run(用途.维护, "发布数据集", path)
|
||
assert replay.returncode == 0 and json.loads(replay.stdout)["duplicate"]
|
||
denied = run(用途.评测, "发布数据集", path)
|
||
assert denied.returncode == 1 and json.loads(denied.stderr)["code"] == "MUSE_PURPOSE_VIOLATION"
|
||
body = req.model_dump(mode="json")
|
||
body.update(dataset_version_id=data["version_id"], dataset_hash=data["public_hash"])
|
||
path = tmp_path / "experiment.json"
|
||
path.write_text(json.dumps({"command_id": "cli-experiment", "request": body}))
|
||
created = run(用途.评测, "创建实验", path)
|
||
assert created.returncode == 0, created.stderr
|
||
result = json.loads(created.stdout)
|
||
replay = run(用途.评测, "创建实验", path)
|
||
assert replay.returncode == 0 and json.loads(replay.stdout) == result
|
||
assert svc.读取实验(_身份(用途.评测), result["experiment_id"]) == result
|
||
read = run(用途.评测, "生成输入", result["experiment_id"], "--样本", "sample-2")
|
||
assert read.returncode == 0 and json.loads(read.stdout) == _数据().samples[2].input.model_dump()
|
||
malformed = run(用途.评测, "实验", "not-a-uuid")
|
||
assert malformed.returncode == 1
|
||
assert json.loads(malformed.stderr)["code"] == "EVALUATION_CONTRACT_FAILED"
|
||
path.write_text(json.dumps({"command_id": "bad", "request": body, "oracle": "PRIVATE-MARKER"}))
|
||
malformed = run(用途.评测, "创建实验", path)
|
||
assert malformed.returncode == 1 and "PRIVATE-MARKER" not in malformed.stderr
|
||
assert "Traceback" not in malformed.stderr
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25100d",
|
||
environment="隔离PG与合成资料;含实际CLI子进程和HTTP",
|
||
given="真实HTTP会话及三类用途",
|
||
when="经过来源与身份校验调用发布/实验端点",
|
||
then=["维护只发布、评测才建实验,生产拒绝、伪造作者与坏ID拒绝"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测, 用途.维护])
|
||
def test_HTTP会话来源与数据集实验用途强制生效__25100d(实验环境, tmp_path, purpose):
|
||
from fastapi.testclient import TestClient
|
||
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.配置 import 读取配置
|
||
|
||
pools, _, data, req = 实验环境
|
||
http = 创建应用(读取配置(_配置文件(pools[purpose], tmp_path)))
|
||
with TestClient(http, headers={"origin": "http://testserver"}) as client:
|
||
endpoint = "/api/v1/evaluation/experiments"
|
||
body = {"command_id": "http-experiment", "request": req.model_dump(mode="json")}
|
||
assert client.post(endpoint, json=body).status_code == 401
|
||
assert (
|
||
client.post(
|
||
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
|
||
).status_code
|
||
== 200
|
||
)
|
||
assert (
|
||
client.post(endpoint, json=body, headers={"origin": "http://invalid"}).status_code
|
||
== 403
|
||
)
|
||
created = client.post(endpoint, json=body)
|
||
if purpose is 用途.评测:
|
||
assert created.status_code == 201, created.text
|
||
result = created.json()
|
||
assert client.post(endpoint, json=body).json() == result
|
||
eid = result["experiment_id"]
|
||
assert client.get(f"{endpoint}/{eid}").json() == result
|
||
assert (
|
||
client.get(f"{endpoint}/{eid}/samples/sample-2/input").json()
|
||
== _数据().samples[2].input.model_dump()
|
||
)
|
||
public = client.get("/api/v1/evaluation/datasets/" + data["version_id"])
|
||
assert public.status_code == 200 and "ORACLE-SECRET" not in public.text
|
||
forged = client.post(endpoint, json={**body, "author_id": "PRIVATE-MARKER"})
|
||
assert forged.status_code == 422 and "PRIVATE-MARKER" not in forged.text
|
||
assert client.get(f"{endpoint}/invalid").status_code == 422
|
||
else:
|
||
assert created.status_code == 403
|
||
assert (
|
||
client.get("/api/v1/evaluation/datasets/" + data["version_id"]).status_code == 403
|
||
)
|
||
published = client.post("/api/v1/evaluation/datasets", json=_数据().model_dump(mode="json"))
|
||
assert published.status_code == (201 if purpose is 用途.维护 else 403)
|
||
assert "ORACLE-SECRET" not in published.text
|