muse-agent-example/tests/集成/test_方法版本评测.py
zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

341 lines
14 KiB
Python

"""方法真实确认版本、S04用途投影、隔离逐例消费与CLI/HTTP封存。"""
import json
from dataclasses import asdict, replace
from datetime import UTC, datetime, timedelta
from uuid import uuid4
import psycopg
import pytest
import test_上下文快照与索引写入 as 方法测试
import test_生产评测权限隔离 as 数据测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
from muse.任务运行.接口 import 任务状态
from muse.元数据.接口 import 元数据服务, 启用命令, 字段限制, 定义哈希, 策略快照
from muse.共享.调用身份 import 用途
from muse.共享.错误 import Muse错误
from muse.效果评测.接口 import 实验执行请求, 实验请求, 方法数据集发布, 评测服务
from muse.知识方法.接口 import 状态命令
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
方法环境 = 方法测试.方法环境
@pytest.fixture
def 方法对照环境(执行环境, 方法环境):
env = 执行环境
business, author, methods, source, _, pools = 方法环境
request = 方法测试._提案(
str(uuid4()),
"动作表现人物迟疑",
str(source.source_id),
owner=author.作者,
)
request = replace(request, content={**request.content, "例证出处": "只许规划读取的字段秘密"})
proposed = methods.提出方法(author, "method-for-eval", request)
方法测试._确认(methods, author, proposed["target_ref"])
detail = methods.读取方法详情(author, request.method_id)
assert detail["state"] == "confirmed"
version = detail["versions"][0]
publisher = 评测服务(pools[用途.维护])
maintained = replace(author, 用途=用途.维护)
publish = 方法数据集发布.model_validate(
{
**数据测试._数据().model_dump(mode="json"),
"samples": [数据测试._数据().samples[-1].model_dump(mode="json")],
"dataset_id": "method-comparison",
"method_version_id": str(version["version_id"]),
"approval_ref": "synthetic:method-eval",
"expires_at": datetime.now(UTC) + timedelta(minutes=30),
}
)
if env["request"].comparison_profile == "writer_rubric":
from test_文学评分执行与条件第三 import 依据
publish = publish.model_copy(
update={
"samples": tuple(
s.model_copy(update={"answer": {**s.answer, "judging_basis": 依据}})
for s in publish.samples
)
}
)
receipt = publisher.发布方法数据集(maintained, publish)
actor = replace(env["actor"], 作者=author.作者)
exp_request = 实验请求.model_validate(
{
**env["request"].model_dump(mode="json"),
"dataset_version_id": receipt["version_id"],
"dataset_hash": receipt["public_hash"],
"target": receipt["method_target"],
}
)
exp = env["app"].要求评测().创建实验(actor, "method-comparison-run", exp_request)
return {
**env,
"actor": actor,
"exp": exp,
"request": exp_request,
"business": business,
"owner": author,
"methods": methods,
"version": version,
"method_id": request.method_id,
"publisher": publisher,
"maintained": maintained,
"publish": publish,
"receipt": receipt,
}
def test_两臂同模板且仅处理组实际使用确切方法__25f801(方法对照环境):
env = 方法对照环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
published = svc.读取数据集(env["actor"], env["receipt"]["version_id"])
material = published["public_manifest"]["method_material"]
assert material["content_hash"] == env["version"]["content_hash"]
assert material["version_id"] == str(env["version"]["version_id"])
assert "只许规划读取的字段秘密" not in material["text"]
assert env["publisher"].发布方法数据集(env["maintained"], env["publish"])["duplicate"]
started = 运行测试._启动执行(env)
assert len(started["units"]) == 3
运行测试._运行就绪(env)
requests = list(env["received"])
assert len(requests) == 2 and requests[0]["instructions"] == requests[1]["instructions"]
inputs = [json.loads(r["input"]) for r in requests]
assert sum("方法材料" in r["context"] for r in inputs) == 1
treated = next(r for r in inputs if "方法材料" in r["context"])
assert treated["context"].pop("方法材料") == material["text"]
assert inputs[0] == inputs[1]
assert "只许规划读取的字段秘密" not in json.dumps(requests, ensure_ascii=False)
assert "ORACLE-SECRET" not in json.dumps(requests, ensure_ascii=False)
svc.推进实验(env["actor"], eid)
运行测试._运行就绪(env)
report = svc.读取实验报告(env["actor"], eid)
assert report["adapter"] == "writer_method_v1"
assert report["primary_pair"] == ["control", "treatment"]
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
assert len(env["received"]) == 3
assert env["methods"].读取方法详情(env["owner"], env["method_id"])["state"] == "confirmed"
assert env["methods"].列出消费(env["owner"], str(env["version"]["version_id"])) == []
with env["pool"].连接() as conn, pytest.raises(psycopg.errors.InsufficientPrivilege):
conn.execute("SELECT * FROM public.muse_method_version")
@pytest.mark.parametrize("bad", ["version", "hash", "target", "author", "raw-dataset"])
def test_错误方法目标或作者不能创建实验__25f802(方法对照环境, bad):
env = 方法对照环境
request = env["request"].model_dump(mode="json")
actor = env["actor"]
if bad == "version":
request["target"]["version"] = "999"
elif bad == "hash":
request["target"]["content_hash"] = "a" * 64
elif bad == "target":
request["target"]["target_ref"] = "B04.method:" + str(uuid4())
elif bad == "author":
actor = replace(actor, 作者="other-author")
else:
dataset = env["publisher"].发布数据集(
env["maintained"], 数据测试._数据().model_copy(update={"dataset_id": "raw-method"})
)
request.update(
dataset_version_id=dataset["version_id"], dataset_hash=dataset["public_hash"]
)
with pytest.raises(Muse错误):
env["app"].要求评测().创建实验(
actor, "invalid-method-target", 实验请求.model_validate(request)
)
assert not env["received"]
def test_撤回与错误身份不发布新方法资料__25f803(方法对照环境):
env = 方法对照环境
for purpose in (用途.生产, 用途.评测):
with pytest.raises(Muse错误):
评测服务(env["pools"][purpose]).发布方法数据集(
replace(env["maintained"], 用途=purpose), env["publish"]
)
with pytest.raises(Muse错误):
env["publisher"].发布方法数据集(
replace(env["maintained"], 作者="other-author"),
env["publish"].model_copy(update={"dataset_id": "foreign-method"}),
)
env["methods"].变更方法状态(
env["owner"],
"withdraw-method",
状态命令(env["method_id"], "withdraw", expected_state_revision=1),
)
with pytest.raises(Muse错误, match="撤回"):
env["publisher"].发布方法数据集(
env["maintained"], env["publish"].model_copy(update={"dataset_id": "withdrawn-method"})
)
def test_执行期限与字段策略改变在发送前拒绝__25f804(方法对照环境):
env = 方法对照环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
with pytest.raises(Muse错误, match="不足"):
svc.启动实验(
env["actor"],
eid,
实验执行请求(approval_ref="synthetic", deadline=datetime.now(UTC) + timedelta(hours=1)),
)
运行测试._启动执行(env)
with env["pools"][用途.维护].连接() as conn:
md = 元数据服务(conn)
old = md.当前策略("craft")
field = next(f.field_id for f in md.读取结构("craft", 1).字段 if f.key == "原理")
changed = 策略快照("craft", old.version + 1, (字段限制(field, ("aiContext:generation",)),))
md.更新策略(
changed,
启用命令(
"restrict-method",
定义哈希(asdict(changed)),
old.version,
env["owner"].作者,
"type:craft",
),
)
运行测试._运行就绪(env, allow_failure=True)
assert not env["received"]
assert not any(u["output"] for u in svc.读取执行工作面(env["actor"], eid)["units"])
def test_方法实际发送输入篡改在外发前拒绝__25f805(方法对照环境, monkeypatch):
import muse.编排.执行评测 as flow
env = 方法对照环境
original = flow.模型请求
def drop(*args, **kwargs):
request = original(*args, **kwargs)
value = json.loads(request.用户输入)
value["context"]["方法材料"] = "其他未评测方法"
return replace(request, 用户输入=json.dumps(value, ensure_ascii=False))
monkeypatch.setattr(flow, "模型请求", drop)
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
assert not env["received"]
def test_方法封存CLI与HTTP重放同一真实版本__25f806(方法对照环境, tmp_path):
import subprocess
import sys
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
env = 方法对照环境
config = 数据测试._配置文件(env["pools"][用途.维护], tmp_path)
config.write_text(config.read_text().replace("eval-author", env["owner"].作者))
request = env["publish"].model_copy(update={"dataset_id": "method-cli"})
path = tmp_path / "method-publish.json"
path.write_text(request.model_dump_json())
result = subprocess.run(
[sys.executable, "-I", "-m", "muse", "评测", str(config), "发布方法数据集", str(path)],
capture_output=True,
text=True,
timeout=30,
)
assert result.returncode == 0, result.stderr
receipt = json.loads(result.stdout)
with TestClient(创建应用(读取配置(config)), headers={"origin": "http://testserver"}) as client:
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
response = client.post(
"/api/v1/evaluation/method-datasets", json=request.model_dump(mode="json")
)
assert response.status_code == 201, response.text
assert response.json() == {**receipt, "duplicate": True}
assert receipt["method_target"] == env["receipt"]["method_target"]
assert not env["received"]
def test_方法单元已交付后恢复不重发且保留来源__25f807(方法对照环境, monkeypatch):
env = 方法对照环境
svc = env["app"].要求评测()
original = svc.保存执行交付
failed = False
def interrupt(*args, **kwargs):
nonlocal failed
result = original(*args, **kwargs)
if not failed:
failed = True
raise Muse错误("合成中断:方法输入及交付已保存")
return result
monkeypatch.setattr(svc, "保存执行交付", interrupt)
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
assert len(env["received"]) == 2
face = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
units = [u for u in face["units"] if u["kind"] == "generation"]
output = {u["unit_id"]: u["output"] for u in units}
assert all(output.values())
runtime = env["app"].任务运行
for unit in units:
task = runtime.读取任务(unit["task_id"])
if task.状态 is 任务状态.已失败:
runtime.控制任务(
task.任务ID, env["actor"].作者, task.状态, "恢复", 命令ID="restore-method"
)
运行测试._运行就绪(env)
assert len(env["received"]) == 2
current = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
assert {
u["unit_id"]: u["output"] for u in current["units"] if u["kind"] == "generation"
} == output
svc.推进实验(env["actor"], env["exp"]["experiment_id"])
运行测试._运行就绪(env)
assert svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])["coverage"]["compared"] == 1
def test_确认版本被改字节不按旧哈希发布__25f808(方法对照环境, monkeypatch):
from muse.知识方法.存储 import 方法存储
env = 方法对照环境
original = 方法存储.读取版本
def changed(self, version_id):
row = original(self, version_id)
return {**row, "content": {**row["content"], "原理": "未确认的另一套原理"}}
monkeypatch.setattr(方法存储, "读取版本", changed)
with pytest.raises(Muse错误, match="确认内容哈希"):
env["publisher"].发布方法数据集(
env["maintained"], env["publish"].model_copy(update={"dataset_id": "corrupt-method"})
)
assert not env["received"]
@pytest.mark.parametrize("执行环境", [检测测试.参数], indirect=True)
def test_方法对照贯通文学评分检测且不足证据不签效果凭据__25f809(方法对照环境):
env = 方法对照环境
svc = env["app"].要求评测()
request = env["request"].model_copy(update={"effect_policy": "writer-effect-v1"})
exp = svc.创建实验(env["actor"], "method-literary-evaluation", request)
env = {**env, "exp": exp}
report = 检测测试._执行(env)
assert len(env["received"]) == 6 # 两写手、两检测、两独立评委;候补不需要执行。
assert report["samples"][0]["detection_complete"]
assert report["samples"][0]["literary"]["control:treatment"]["execution_verified"]
assert report["effect"]["assessment"]["target"] == env["receipt"]["method_target"]
assert report["effect"]["assessment"]["gate_b"]["status"] == "insufficient_evidence"
with pytest.raises(Muse错误, match="效果未通过"):
svc.封存效果判据(env["actor"], exp["experiment_id"])
assert env["methods"].读取方法详情(env["owner"], env["method_id"])["state"] == "confirmed"