muse-agent-example/tests/集成/test_生产评测权限隔离.py
zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

489 lines
21 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实隔离PG的数据集与实验封存;不调用外部模型或写生产内容。"""
import json
from concurrent.futures import ThreadPoolExecutor
from dataclasses import replace
from decimal import Decimal
from threading import Barrier
import psycopg
import pytest
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
from muse.效果评测.接口 import (
公开评测输入,
实验请求,
数据样本,
数据集发布,
目标版本,
评测服务,
评测错误,
配置选择,
)
pytestmark = pytest.mark.数据库
def _身份(purpose):
return 调用身份("eval-author", None, purpose, 内容用途.检测)
def _数据():
samples = []
for i, split in enumerate(("discovery", "calibration", "holdout", "holdout", "holdout")):
samples.append(
数据样本(
sample_id=f"sample-{i}",
source_ref=f"synthetic:source-{i}",
license_ref="synthetic:owned-fixture",
source_groups=(f"synthetic:book-{i}",),
split=split,
input=公开评测输入(
instruction="比较叙述是否清楚。",
original=f"第{i}段合成正文。",
context={"事实": "已知背景"},
),
answer={"secret": f"ORACLE-SECRET-{i}", "expected": "保留未公开的评判依据"},
)
)
return 数据集发布(dataset_id="synthetic-dataset", revision=1, samples=tuple(samples))
@pytest.fixture
def 数据集环境(应用测试库):
pools = 应用测试库
maintained = 评测服务(pools[用途.维护])
evaluated = 评测服务(pools[用途.评测])
result = maintained.发布数据集(_身份(用途.维护), _数据())
return pools, maintained, evaluated, result
@pytest.fixture
def 实验环境(数据集环境):
from muse.任务运行.接口 import 凭据引用, 提供方配置, 运行配置内容, 配置版本管理
pools, maintained, evaluated, data = 数据集环境
manager = 配置版本管理(pools[用途.评测])
for role, model in (("writer", "synthetic-writer"), ("judge", "synthetic-judge")):
content = 运行配置内容(
"direct",
"test-host",
"test-policy",
"muse-foundation-r2",
"test-budget",
{role: {"provider": "synthetic", "model": model, "thinking": "high", "tools": []}},
(凭据引用("token", "环境变量", "MUSE_UNUSED_EVAL_SECRET"),),
(提供方配置("synthetic", "responses", "http://127.0.0.1:9", "token"),),
"test-price",
)
manager.保存草案(role, "v1", content)
request = 实验请求(
dataset_version_id=data["version_id"],
dataset_hash=data["public_hash"],
split="holdout",
target=目标版本(
kind="rule", target_ref="candidate-rule", version="1", content_hash="a" * 64
),
generator_role="writer",
generator=配置选择(config_id="writer", version="v1"),
judges=(配置选择(config_id="judge", version="v1"),),
arms=("control", "treatment"),
dimensions=("叙事", "声音"),
max_cost_usd=Decimal("3.5"),
max_calls_per_sample=10,
)
return pools, evaluated, data, request
def test_生产不能读取答案或评测数据且评测不能修改封存__251001(数据集环境):
pools, _, svc, data = 数据集环境
visible = svc.读取数据集(_身份(用途.评测), data["version_id"])
assert len(visible["public_manifest"]["samples"]) == 5
assert "ORACLE-SECRET" not in json.dumps(visible, ensure_ascii=False)
assert "answer_hash" not in visible
for sql in (
"SELECT * FROM oracle.muse_dataset_answers",
"SELECT * FROM evaluation.muse_dataset_version",
"SELECT * FROM oracle.muse_blind_assignment",
):
with pools[用途.生产].连接() as conn:
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
conn.execute(sql).fetchall()
with pools[用途.评测].连接() as conn:
for sql in (
"SELECT * FROM public.muse_document",
"SELECT * FROM public.muse_world_object",
"DELETE FROM evaluation.muse_dataset_version",
"UPDATE oracle.muse_dataset_answers SET answer_hash='x'",
"INSERT INTO public.muse_change_command(author_id,run_purpose,command_id,request_hash) "
"VALUES ('x','production','x','x')",
):
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
conn.execute(sql)
answers = conn.execute("SELECT answers FROM oracle.muse_dataset_answers").fetchone()[0]
assert answers[0]["answer"]["secret"] == "ORACLE-SECRET-0"
def test_同版本幂等但答案变化不能冒充同一数据集__251002(数据集环境):
pools, maint, _, result = 数据集环境
same = maint.发布数据集(_身份(用途.维护), _数据())
assert same["duplicate"] and same["version_id"] == result["version_id"]
req = _数据()
changed = req.model_copy(
update={
"samples": (
req.samples[0].model_copy(update={"answer": {"secret": "changed"}}),
*req.samples[1:],
)
}
)
with pytest.raises(评测错误, match="不能覆盖"):
maint.发布数据集(_身份(用途.维护), changed)
second = maint.发布数据集(_身份(用途.维护), changed.model_copy(update={"revision": 2}))
assert second["version_id"] != same["version_id"]
with pools[用途.维护].连接() as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 2
)
assert (
conn.execute(
"SELECT answers FROM oracle.muse_dataset_answers WHERE version_id=%s",
(result["version_id"],),
).fetchone()[0][0]["answer"]["secret"]
== "ORACLE-SECRET-0"
)
with pytest.raises(psycopg.Error), conn.transaction():
conn.execute("UPDATE evaluation.muse_dataset_version SET public_hash=%s", ("0" * 64,))
def test_答案写入失败回滚整个发布__251003(应用测试库):
pool = 应用测试库[用途.维护]
with pool.连接() as conn, conn.transaction():
conn.execute(
"CREATE FUNCTION oracle.fail_answer() RETURNS trigger LANGUAGE plpgsql AS "
"$$ BEGIN RAISE EXCEPTION 'synthetic failure'; END $$"
)
conn.execute(
"CREATE TRIGGER fail_answer BEFORE INSERT ON oracle.muse_dataset_answers "
"FOR EACH ROW EXECUTE FUNCTION oracle.fail_answer()"
)
with pytest.raises(评测错误, match="评测存储操作失败"):
评测服务(pool).发布数据集(_身份(用途.维护), _数据())
with pool.连接(只读=True) as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
)
assert conn.execute("SELECT count(*) FROM oracle.muse_dataset_answers").fetchone()[0] == 0
def test_实验冻结实际配置与全分割且模型输入无答案映射__251004(实验环境):
pools, svc, data, req = 实验环境
actor = _身份(用途.评测)
result = svc.创建实验(actor, "experiment-one", req)
assert result == svc.创建实验(actor, "experiment-one", req)
assert result["registration"] == "registered"
assert set(result["sample_ids"]) == {"sample-2", "sample-3", "sample-4"}
assert result["conditions"]["generator"]["model"] == "synthetic-writer"
assert result["conditions"]["judges"][0]["model"] == "synthetic-judge"
assert result["conditions"]["generator"]["config_hash"]
assert "MUSE_UNUSED_EVAL_SECRET" not in json.dumps(result)
payload = svc.读取生成输入(actor, result["experiment_id"], "sample-2")
assert set(payload) == {"instruction", "original", "context"}
assert payload == _数据().samples[2].input.model_dump()
assert "ORACLE" not in json.dumps(payload) and "control" not in json.dumps(payload)
with pytest.raises(评测错误):
svc.读取生成输入(actor, result["experiment_id"], "sample-0")
with pytest.raises(评测错误):
svc.读取实验(replace(actor, 作者="other-author"), result["experiment_id"])
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 1
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 3
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_budget_reservation").fetchone()[0]
== 0
)
@pytest.mark.parametrize("bad", ["hash", "missing_profile", "wrong_role", "same_model"])
def test_实验前置错误不留下条件或匿名映射__251005(实验环境, bad):
from muse.共享.错误 import Muse错误
pools, svc, _, req = 实验环境
if bad == "hash":
req = req.model_copy(update={"dataset_hash": "0" * 64})
elif bad == "missing_profile":
req = req.model_copy(update={"generator": 配置选择(config_id="absent", version="v1")})
elif bad == "wrong_role":
req = req.model_copy(update={"generator": 配置选择(config_id="judge", version="v1")})
else:
from muse.任务运行.接口 import 配置版本管理
profiles = 配置版本管理(pools[用途.评测])
v = profiles.读取版本("judge", "v1")
changed = replace(
v.内容,
角色配置={
"judge": {
"provider": "synthetic",
"model": "synthetic-writer",
"thinking": "high",
"tools": [],
}
},
)
profiles.保存草案("same-model", "v1", changed)
req = req.model_copy(update={"judges": (配置选择(config_id="same-model", version="v1"),)})
with pytest.raises(Muse错误):
svc.创建实验(_身份(用途.评测), "bad-experiment", req)
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 0
def test_重复命令不能更换实验目标与匿名映射__251006(实验环境):
pools, svc, _, req = 实验环境
actor = _身份(用途.评测)
old = svc.创建实验(actor, "bound-command", req)
with pytest.raises(评测错误, match="不能更换"):
svc.创建实验(
actor,
"bound-command",
req.model_copy(
update={"target": req.target.model_copy(update={"content_hash": "b" * 64})}
),
)
assert svc.读取实验(actor, old["experiment_id"]) == old
with pools[用途.评测].连接() as conn:
with pytest.raises(psycopg.Error), conn.transaction():
conn.execute("UPDATE oracle.muse_blind_assignment SET arm_order='[]'")
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测])
def test_非维护用途不能发布数据集__251007(应用测试库, purpose):
with pytest.raises(评测错误):
评测服务(应用测试库[purpose]).发布数据集(_身份(purpose), _数据())
with 应用测试库[用途.维护].连接(只读=True) as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
)
@pytest.mark.parametrize("operation", ["dataset", "experiment"])
def test_并发同一发布或实验命令只保存一份__251008(实验环境, operation):
pools, svc, _, req = 实验环境
barrier = Barrier(2)
def execute():
barrier.wait(timeout=5)
if operation == "dataset":
return 评测服务(pools[用途.维护]).发布数据集(
_身份(用途.维护), _数据().model_copy(update={"dataset_id": "concurrent"})
)
return svc.创建实验(_身份(用途.评测), "concurrent", req)
with ThreadPoolExecutor(max_workers=2) as executor:
futures = [executor.submit(execute) for _ in range(2)]
a, b = [f.result(timeout=15) for f in futures]
with pools[用途.评测].连接(只读=True) as conn:
if operation == "dataset":
assert a["version_id"] == b["version_id"]
assert {a["duplicate"], b["duplicate"]} == {True, False}
assert (
conn.execute(
"SELECT count(*) FROM oracle.muse_dataset_answers WHERE version_id=%s",
(a["version_id"],),
).fetchone()[0]
== 1
)
else:
assert a == b
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 1
)
assert (
conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 3
)
def test_匿名映射中途失败回滚全部实验并能原命令重试__251009(实验环境):
pools, svc, _, req = 实验环境
with pools[用途.维护].连接() as conn, conn.transaction():
conn.execute(
"CREATE FUNCTION oracle.fail_assignment() RETURNS trigger LANGUAGE plpgsql AS "
"$$ BEGIN IF NEW.sample_id='sample-3' THEN RAISE EXCEPTION 'synthetic failure'; "
"END IF; RETURN NEW; END $$"
)
conn.execute(
"CREATE TRIGGER fail_assignment BEFORE INSERT ON oracle.muse_blind_assignment "
"FOR EACH ROW EXECUTE FUNCTION oracle.fail_assignment()"
)
with pytest.raises(评测错误, match="评测存储操作失败"):
svc.创建实验(_身份(用途.评测), "recover", req)
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
assert conn.execute("SELECT count(*) FROM oracle.muse_blind_assignment").fetchone()[0] == 0
with pools[用途.维护].连接() as conn, conn.transaction():
conn.execute("DROP TRIGGER fail_assignment ON oracle.muse_blind_assignment")
result = svc.创建实验(_身份(用途.评测), "recover", req)
assert len(result["sample_ids"]) == 3
assert svc.创建实验(_身份(用途.评测), "recover", req) == result
def test_相同评委模型不能用不同配置冒充多个独立评委__25100a(实验环境):
from muse.任务运行.接口 import 配置版本管理
pools, svc, _, req = 实验环境
profiles = 配置版本管理(pools[用途.评测])
profiles.保存草案("judge-copy", "v1", profiles.读取版本("judge", "v1").内容)
changed = req.model_copy(
update={"judges": (*req.judges, 配置选择(config_id="judge-copy", version="v1"))}
)
with pytest.raises(评测错误, match="不能重复"):
svc.创建实验(_身份(用途.评测), "duplicate-judge", changed)
with pools[用途.评测].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_experiment").fetchone()[0] == 0
def test_空保留集不能登记为可运行实验__25100b(实验环境):
pools, svc, _, req = 实验环境
data = 评测服务(pools[用途.维护]).发布数据集(
_身份(用途.维护),
_数据().model_copy(
update={"dataset_id": "without-holdout", "samples": _数据().samples[:2]}
),
)
changed = 实验请求.model_validate(
{
**req.model_dump(mode="json"),
"dataset_version_id": data["version_id"],
"dataset_hash": data["public_hash"],
}
)
with pytest.raises(评测错误, match="没有样本"):
svc.创建实验(_身份(用途.评测), "empty-holdout", changed)
def _配置文件(pool, tmp_path):
secret = tmp_path / "author-password"
secret.write_text("synthetic-evaluation-only")
secret.chmod(0o600)
config = tmp_path / f"{pool.用途.value}.toml"
config.write_text(
'["数据库"]\n"取值方式"="受控存储"\n"位置"='
+ json.dumps(pool.引用.位置)
+ '\n["运行"]\n"用途"='
+ json.dumps(pool.用途.value)
+ '\n["资源"]\n"发布身份"="test"\n["HTTP"]\n"作者ID"="eval-author"'
+ '\n"口令文件"='
+ json.dumps(str(secret))
+ '\n"公开地址"="http://testserver"\n"允许来源"=["http://testserver"]\n'
)
return config
def test_实际CLI发布实验与读取沿用途及固定版本重放__25100c(实验环境, tmp_path):
import subprocess
import sys
pools, svc, _, req = 实验环境
configs = {p: _配置文件(pools[p], tmp_path) for p in 用途}
def run(purpose, action, target, *args):
return subprocess.run(
[
sys.executable,
"-I",
"-m",
"muse",
"评测",
str(configs[purpose]),
action,
str(target),
*args,
],
cwd=tmp_path,
capture_output=True,
text=True,
timeout=30,
)
path = tmp_path / "dataset.json"
path.write_text(_数据().model_copy(update={"dataset_id": "from-cli"}).model_dump_json())
first = run(用途.维护, "发布数据集", path)
assert first.returncode == 0, first.stderr
data = json.loads(first.stdout)
assert not data["duplicate"]
replay = run(用途.维护, "发布数据集", path)
assert replay.returncode == 0 and json.loads(replay.stdout)["duplicate"]
denied = run(用途.评测, "发布数据集", path)
assert denied.returncode == 1 and json.loads(denied.stderr)["code"] == "MUSE_PURPOSE_VIOLATION"
body = req.model_dump(mode="json")
body.update(dataset_version_id=data["version_id"], dataset_hash=data["public_hash"])
path = tmp_path / "experiment.json"
path.write_text(json.dumps({"command_id": "cli-experiment", "request": body}))
created = run(用途.评测, "创建实验", path)
assert created.returncode == 0, created.stderr
result = json.loads(created.stdout)
replay = run(用途.评测, "创建实验", path)
assert replay.returncode == 0 and json.loads(replay.stdout) == result
assert svc.读取实验(_身份(用途.评测), result["experiment_id"]) == result
read = run(用途.评测, "生成输入", result["experiment_id"], "--样本", "sample-2")
assert read.returncode == 0 and json.loads(read.stdout) == _数据().samples[2].input.model_dump()
malformed = run(用途.评测, "实验", "not-a-uuid")
assert malformed.returncode == 1
assert json.loads(malformed.stderr)["code"] == "EVALUATION_CONTRACT_FAILED"
path.write_text(json.dumps({"command_id": "bad", "request": body, "oracle": "PRIVATE-MARKER"}))
malformed = run(用途.评测, "创建实验", path)
assert malformed.returncode == 1 and "PRIVATE-MARKER" not in malformed.stderr
assert "Traceback" not in malformed.stderr
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测, 用途.维护])
def test_HTTP会话来源与数据集实验用途强制生效__25100d(实验环境, tmp_path, purpose):
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
pools, _, data, req = 实验环境
http = 创建应用(读取配置(_配置文件(pools[purpose], tmp_path)))
with TestClient(http, headers={"origin": "http://testserver"}) as client:
endpoint = "/api/v1/evaluation/experiments"
body = {"command_id": "http-experiment", "request": req.model_dump(mode="json")}
assert client.post(endpoint, json=body).status_code == 401
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
assert (
client.post(endpoint, json=body, headers={"origin": "http://invalid"}).status_code
== 403
)
created = client.post(endpoint, json=body)
if purpose is 用途.评测:
assert created.status_code == 201, created.text
result = created.json()
assert client.post(endpoint, json=body).json() == result
eid = result["experiment_id"]
assert client.get(f"{endpoint}/{eid}").json() == result
assert (
client.get(f"{endpoint}/{eid}/samples/sample-2/input").json()
== _数据().samples[2].input.model_dump()
)
public = client.get("/api/v1/evaluation/datasets/" + data["version_id"])
assert public.status_code == 200 and "ORACLE-SECRET" not in public.text
forged = client.post(endpoint, json={**body, "author_id": "PRIVATE-MARKER"})
assert forged.status_code == 422 and "PRIVATE-MARKER" not in forged.text
assert client.get(f"{endpoint}/invalid").status_code == 422
else:
assert created.status_code == 403
assert (
client.get("/api/v1/evaluation/datasets/" + data["version_id"]).status_code == 403
)
published = client.post("/api/v1/evaluation/datasets", json=_数据().model_dump(mode="json"))
assert published.status_code == (201 if purpose is 用途.维护 else 403)
assert "ORACLE-SECRET" not in published.text