muse-agent-example/tests/集成/test_方法版本评测.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

413 lines
18 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""方法真实确认版本、S04用途投影、隔离逐例消费与CLI/HTTP封存。"""
import json
from dataclasses import asdict, replace
from datetime import UTC, datetime, timedelta
from uuid import uuid4
import psycopg
import pytest
import test_上下文快照与索引写入 as 方法测试
import test_生产评测权限隔离 as 数据测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
from muse.任务运行.接口 import 任务状态
from muse.元数据.接口 import 元数据服务, 启用命令, 字段限制, 定义哈希, 策略快照
from muse.共享.调用身份 import 用途
from muse.共享.错误 import Muse错误
from muse.效果评测.接口 import 实验执行请求, 实验请求, 方法数据集发布, 评测服务
from muse.知识方法.接口 import 状态命令
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
方法环境 = 方法测试.方法环境
@pytest.fixture
def 方法对照环境(执行环境, 方法环境):
env = 执行环境
business, author, methods, source, _, pools = 方法环境
request = 方法测试._提案(
str(uuid4()),
"动作表现人物迟疑",
str(source.source_id),
owner=author.作者,
)
request = replace(request, content={**request.content, "例证出处": "只许规划读取的字段秘密"})
proposed = methods.提出方法(author, "method-for-eval", request)
方法测试._确认(methods, author, proposed["target_ref"])
detail = methods.读取方法详情(author, request.method_id)
assert detail["state"] == "confirmed"
version = detail["versions"][0]
publisher = 评测服务(pools[用途.维护])
maintained = replace(author, 用途=用途.维护)
publish = 方法数据集发布.model_validate(
{
**数据测试._数据().model_dump(mode="json"),
"samples": [数据测试._数据().samples[-1].model_dump(mode="json")],
"dataset_id": "method-comparison",
"method_version_id": str(version["version_id"]),
"approval_ref": "synthetic:method-eval",
"expires_at": datetime.now(UTC) + timedelta(minutes=30),
}
)
if env["request"].comparison_profile == "writer_rubric":
from test_文学评分执行与条件第三 import 依据
publish = publish.model_copy(
update={
"samples": tuple(
s.model_copy(update={"answer": {**s.answer, "judging_basis": 依据}})
for s in publish.samples
)
}
)
receipt = publisher.发布方法数据集(maintained, publish)
actor = replace(env["actor"], 作者=author.作者)
exp_request = 实验请求.model_validate(
{
**env["request"].model_dump(mode="json"),
"dataset_version_id": receipt["version_id"],
"dataset_hash": receipt["public_hash"],
"target": receipt["method_target"],
}
)
exp = env["app"].要求评测().创建实验(actor, "method-comparison-run", exp_request)
return {
**env,
"actor": actor,
"exp": exp,
"request": exp_request,
"business": business,
"owner": author,
"methods": methods,
"version": version,
"method_id": request.method_id,
"publisher": publisher,
"maintained": maintained,
"publish": publish,
"receipt": receipt,
}
@pytest.mark.case_id(
"NC-w25-25f801",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["确切确认版本、同模板对照、实际方法输入与生产权限隔离"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_两臂同模板且仅处理组实际使用确切方法__25f801(方法对照环境):
env = 方法对照环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
published = svc.读取数据集(env["actor"], env["receipt"]["version_id"])
material = published["public_manifest"]["method_material"]
assert material["content_hash"] == env["version"]["content_hash"]
assert material["version_id"] == str(env["version"]["version_id"])
assert "只许规划读取的字段秘密" not in material["text"]
assert env["publisher"].发布方法数据集(env["maintained"], env["publish"])["duplicate"]
started = 运行测试._启动执行(env)
assert len(started["units"]) == 3
运行测试._运行就绪(env)
requests = list(env["received"])
assert len(requests) == 2 and requests[0]["instructions"] == requests[1]["instructions"]
inputs = [json.loads(r["input"]) for r in requests]
assert sum("方法材料" in r["context"] for r in inputs) == 1
treated = next(r for r in inputs if "方法材料" in r["context"])
assert treated["context"].pop("方法材料") == material["text"]
assert inputs[0] == inputs[1]
assert "只许规划读取的字段秘密" not in json.dumps(requests, ensure_ascii=False)
assert "ORACLE-SECRET" not in json.dumps(requests, ensure_ascii=False)
svc.推进实验(env["actor"], eid)
运行测试._运行就绪(env)
report = svc.读取实验报告(env["actor"], eid)
assert report["adapter"] == "writer_method_v1"
assert report["primary_pair"] == ["control", "treatment"]
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
assert len(env["received"]) == 3
assert env["methods"].读取方法详情(env["owner"], env["method_id"])["state"] == "confirmed"
assert env["methods"].列出消费(env["owner"], str(env["version"]["version_id"])) == []
with env["pool"].连接() as conn, pytest.raises(psycopg.errors.InsufficientPrivilege):
conn.execute("SELECT * FROM public.muse_method_version")
@pytest.mark.case_id(
"NC-w25-25f802",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["错版本、哈希、目标、作者或手填资料不能创建方法实验"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", ["version", "hash", "target", "author", "raw-dataset"])
def test_错误方法目标或作者不能创建实验__25f802(方法对照环境, bad):
env = 方法对照环境
request = env["request"].model_dump(mode="json")
actor = env["actor"]
if bad == "version":
request["target"]["version"] = "999"
elif bad == "hash":
request["target"]["content_hash"] = "a" * 64
elif bad == "target":
request["target"]["target_ref"] = "B04.method:" + str(uuid4())
elif bad == "author":
actor = replace(actor, 作者="other-author")
else:
dataset = env["publisher"].发布数据集(
env["maintained"], 数据测试._数据().model_copy(update={"dataset_id": "raw-method"})
)
request.update(
dataset_version_id=dataset["version_id"], dataset_hash=dataset["public_hash"]
)
with pytest.raises(Muse错误):
env["app"].要求评测().创建实验(
actor, "invalid-method-target", 实验请求.model_validate(request)
)
assert not env["received"]
@pytest.mark.case_id(
"NC-w25-25f803",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["非维护用途、其他作者或撤回方法不能发布新快照"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_撤回与错误身份不发布新方法资料__25f803(方法对照环境):
env = 方法对照环境
for purpose in (用途.生产, 用途.评测):
with pytest.raises(Muse错误):
评测服务(env["pools"][purpose]).发布方法数据集(
replace(env["maintained"], 用途=purpose), env["publish"]
)
with pytest.raises(Muse错误):
env["publisher"].发布方法数据集(
replace(env["maintained"], 作者="other-author"),
env["publish"].model_copy(update={"dataset_id": "foreign-method"}),
)
env["methods"].变更方法状态(
env["owner"],
"withdraw-method",
状态命令(env["method_id"], "withdraw", expected_state_revision=1),
)
with pytest.raises(Muse错误, match="撤回"):
env["publisher"].发布方法数据集(
env["maintained"], env["publish"].model_copy(update={"dataset_id": "withdrawn-method"})
)
@pytest.mark.case_id(
"NC-w25-25f804",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["批准期限及字段策略变更阻断新派发"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_执行期限与字段策略改变在发送前拒绝__25f804(方法对照环境):
env = 方法对照环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
with pytest.raises(Muse错误, match="不足"):
svc.启动实验(
env["actor"],
eid,
实验执行请求(approval_ref="synthetic", deadline=datetime.now(UTC) + timedelta(hours=1)),
)
运行测试._启动执行(env)
with env["pools"][用途.维护].连接() as conn:
md = 元数据服务(conn)
old = md.当前策略("craft")
field = next(f.field_id for f in md.读取结构("craft", 1).字段 if f.key == "原理")
changed = 策略快照("craft", old.version + 1, (字段限制(field, ("aiContext:generation",)),))
md.更新策略(
changed,
启用命令(
"restrict-method",
定义哈希(asdict(changed)),
old.version,
env["owner"].作者,
"type:craft",
),
)
运行测试._运行就绪(env, allow_failure=True)
assert not env["received"]
assert not any(u["output"] for u in svc.读取执行工作面(env["actor"], eid)["units"])
@pytest.mark.case_id(
"NC-w25-25f805",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["实际方法输入被替换在外发前拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_方法实际发送输入篡改在外发前拒绝__25f805(方法对照环境, monkeypatch):
import muse.编排.执行评测 as flow
env = 方法对照环境
original = flow.模型请求
def drop(*args, **kwargs):
request = original(*args, **kwargs)
value = json.loads(request.用户输入)
value["context"]["方法材料"] = "其他未评测方法"
return replace(request, 用户输入=json.dumps(value, ensure_ascii=False))
monkeypatch.setattr(flow, "模型请求", drop)
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
assert not env["received"]
@pytest.mark.case_id(
"NC-w25-25f806",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["真实CLI与HTTP幂等封存同一方法版本"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_方法封存CLI与HTTP重放同一真实版本__25f806(方法对照环境, tmp_path):
import subprocess
import sys
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
env = 方法对照环境
config = 数据测试._配置文件(env["pools"][用途.维护], tmp_path)
config.write_text(config.read_text().replace("eval-author", env["owner"].作者))
request = env["publish"].model_copy(update={"dataset_id": "method-cli"})
path = tmp_path / "method-publish.json"
path.write_text(request.model_dump_json())
result = subprocess.run(
[sys.executable, "-I", "-m", "muse", "评测", str(config), "发布方法数据集", str(path)],
capture_output=True,
text=True,
timeout=30,
)
assert result.returncode == 0, result.stderr
receipt = json.loads(result.stdout)
with TestClient(创建应用(读取配置(config)), headers={"origin": "http://testserver"}) as client:
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
response = client.post(
"/api/v1/evaluation/method-datasets", json=request.model_dump(mode="json")
)
assert response.status_code == 201, response.text
assert response.json() == {**receipt, "duplicate": True}
assert receipt["method_target"] == env["receipt"]["method_target"]
assert not env["received"]
@pytest.mark.case_id(
"NC-w25-25f807",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["已存方法交付恢复不重发,继续原独立比较"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_方法单元已交付后恢复不重发且保留来源__25f807(方法对照环境, monkeypatch):
env = 方法对照环境
svc = env["app"].要求评测()
original = svc.保存执行交付
failed = False
def interrupt(*args, **kwargs):
nonlocal failed
result = original(*args, **kwargs)
if not failed:
failed = True
raise Muse错误("合成中断:方法输入及交付已保存")
return result
monkeypatch.setattr(svc, "保存执行交付", interrupt)
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
assert len(env["received"]) == 2
face = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
units = [u for u in face["units"] if u["kind"] == "generation"]
output = {u["unit_id"]: u["output"] for u in units}
assert all(output.values())
runtime = env["app"].任务运行
for unit in units:
task = runtime.读取任务(unit["task_id"])
if task.状态 is 任务状态.已失败:
runtime.控制任务(
task.任务ID, env["actor"].作者, task.状态, "恢复", 命令ID="restore-method"
)
运行测试._运行就绪(env)
assert len(env["received"]) == 2
current = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
assert {
u["unit_id"]: u["output"] for u in current["units"] if u["kind"] == "generation"
} == output
svc.推进实验(env["actor"], env["exp"]["experiment_id"])
运行测试._运行就绪(env)
assert svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])["coverage"]["compared"] == 1
@pytest.mark.case_id(
"NC-w25-25f808",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["确认版本内容改变不能沿用原哈希发布"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_确认版本被改字节不按旧哈希发布__25f808(方法对照环境, monkeypatch):
from muse.知识方法.存储 import 方法存储
env = 方法对照环境
original = 方法存储.读取版本
def changed(self, version_id):
row = original(self, version_id)
return {**row, "content": {**row["content"], "原理": "未确认的另一套原理"}}
monkeypatch.setattr(方法存储, "读取版本", changed)
with pytest.raises(Muse错误, match="确认内容哈希"):
env["publisher"].发布方法数据集(
env["maintained"], env["publish"].model_copy(update={"dataset_id": "corrupt-method"})
)
assert not env["received"]
@pytest.mark.case_id(
"NC-w25-25f809",
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
then=["实际文学评分与独立检测衔接,样本/标定不足不得签效果凭据"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [检测测试.参数], indirect=True)
def test_方法对照贯通文学评分检测且不足证据不签效果凭据__25f809(方法对照环境):
env = 方法对照环境
svc = env["app"].要求评测()
request = env["request"].model_copy(update={"effect_policy": "writer-effect-v1"})
exp = svc.创建实验(env["actor"], "method-literary-evaluation", request)
env = {**env, "exp": exp}
report = 检测测试._执行(env)
assert len(env["received"]) == 6 # 两写手、两检测、两独立评委;候补不需要执行。
assert report["samples"][0]["detection_complete"]
assert report["samples"][0]["literary"]["control:treatment"]["execution_verified"]
assert report["effect"]["assessment"]["target"] == env["receipt"]["method_target"]
assert report["effect"]["assessment"]["gate_b"]["status"] == "insufficient_evidence"
with pytest.raises(Muse错误, match="效果未通过"):
svc.封存效果判据(env["actor"], exp["experiment_id"])
assert env["methods"].读取方法详情(env["owner"], env["method_id"])["state"] == "confirmed"