muse-agent-example/tests/集成/test_回放资料封存.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

235 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""维护入口从真实业务owner封存资料;模型与评测账号不读取正式正文。"""
import json
from dataclasses import replace
from datetime import UTC, datetime, timedelta
import pytest
import test_两阶段写手 as 写手测试
from muse.共享.调用身份 import 用途
from muse.共享.错误 import Muse错误
from muse.效果评测.接口 import 回放数据集发布, 评测服务
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
pytestmark = pytest.mark.数据库
生成环境 = 写手测试.生成环境
@pytest.fixture
def 资料环境(生成环境):
env = 生成环境
应用测试库 = env["库组"]
for chapter, text in [
("ch-2", "林深回到渡口,雨渐渐小了。"),
("ch-3", "ORACLE-TARGET-PRIVATE:下一章的实际答案。"),
]:
env["装配"].要求正文().保存人工(
env["作者"],
"replay-body-" + chapter,
chapter,
0,
正文草稿((段落("p-" + chapter, (文本节点(text),)),)),
)
actor = replace(env["作者"], 用途=用途.维护)
request = 回放数据集发布.model_validate(
{
"dataset_id": "source-replay-fixture",
"revision": 1,
"approval_ref": "synthetic-source-approval",
"expires_at": datetime.now(UTC) + timedelta(minutes=30),
"samples": [
{
"sample_id": "target-ch3",
"source_groups": ["synthetic-approved"],
"split": "holdout",
"instruction": "根据已确认细纲续写本章。",
"license_ref": "synthetic-only",
"selection": {
"work_id": "gen-work",
"target_chapter_id": "ch-3",
"history": [
{"chapter_id": "ch-1", "revision": 1},
{"chapter_id": "ch-2", "revision": 1},
],
"recent_count": 1,
"supplemental_codepoints": 4,
"context_bytes": 50000,
},
}
],
}
)
return env, 应用测试库, actor, 评测服务(应用测试库[用途.维护]), request
@pytest.mark.case_id(
"NC-w25-257001",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["B01/B02/B05真实owner封存ABC资料,目标答案只进oracle且不发起模型"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_真实ABC资料封存与目标答案分离且不新增执行__257001(资料环境):
env, pools, actor, svc, req = 资料环境
before = env["装配"].任务运行.列出任务(env["作者"].作者)
receipt = svc.发布回放数据集(actor, req)
assert not receipt["duplicate"] and svc.发布回放数据集(actor, req)["duplicate"]
public = 评测服务(pools[用途.评测]).读取数据集(
replace(actor, 用途=用途.评测), receipt["version_id"]
)
assert "ORACLE-TARGET-PRIVATE" not in json.dumps(public, ensure_ascii=False)
row = public["public_manifest"]["samples"][0]
assert row["input"]["original"] == ""
material = row["replay_materials"]
a, b, c = [material["arms"][arm] for arm in "ABC"]
assert (
a["历史正文"]
== c["历史正文"]
== [{"text": "林深回到渡口,雨渐渐小了。"}, {"text": "林深停在"}]
)
assert a["卡片索引"] == [] and b["历史正文"] == []
assert b["卡片索引"] == c["卡片索引"] and c["卡片索引"][0]["content"]["名称"] == "林深"
assert material["basis"]["retrieval"]["C"][0]["chapter_id"] == "ch-1"
assert material["policies"]
with pools[用途.评测].连接(只读=True) as conn:
answers = conn.execute(
"SELECT answers FROM oracle.muse_dataset_answers WHERE version_id=%s",
(receipt["version_id"],),
).fetchone()[0]
assert answers[0]["answer"]["target_text"].startswith("ORACLE-TARGET-PRIVATE")
assert env["装配"].任务运行.列出任务(env["作者"].作者) == before
assert env["收到"] == []
@pytest.mark.case_id(
"NC-w25-257002",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["目标/未来/跨书/版本/重复/基线/预算七种反例精确拒绝并整份回滚"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize(
"bad", ["target", "future", "foreign", "stale", "duplicate", "baseline", "budget"]
)
def test_越界漂移及不足资料拒绝且整份发布回滚__257002(资料环境, bad):
_, pools, actor, svc, req = 资料环境
data = req.model_dump(mode="json")
selection = data["samples"][0]["selection"]
if bad in {"target", "future", "foreign"}:
selection["history"][0]["chapter_id"] = {
"target": "ch-3",
"future": "ch-4",
"foreign": "foreign",
}[bad]
elif bad == "stale":
selection["history"][0]["revision"] = 2
elif bad == "duplicate":
selection["history"][0] = selection["history"][1]
elif bad == "baseline":
selection["recent_count"] = 3
else:
selection["context_bytes"] = 1
reason = {"stale": "正文版本不存在", "baseline": "连续历史", "budget": "上下文字节预算"}.get(
bad, "重复、跨书或包含目标及未来章"
)
with pytest.raises(Muse错误, match=reason):
svc.发布回放数据集(actor, 回放数据集发布.model_validate(data))
with pools[用途.评测].连接(只读=True) as conn:
assert (
conn.execute("SELECT count(*) FROM evaluation.muse_dataset_version").fetchone()[0] == 0
)
assert conn.execute("SELECT count(*) FROM oracle.muse_dataset_answers").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-257003",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["生产及评测身份不能借维护封存入口读取正式源"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("purpose", [用途.生产, 用途.评测])
def test_非维护身份不能借封存入口读取正式源__257003(资料环境, purpose):
env, pools, actor, _, req = 资料环境
with pytest.raises(Muse错误, match="用途"):
评测服务(pools[purpose]).发布回放数据集(replace(actor, 用途=purpose), req)
assert env["收到"] == []
@pytest.mark.case_id(
"NC-w25-257004",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["同书和同原文自动来源组禁止跨分割"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_同书及同原文派生样本不能跨分割__257004(资料环境):
_, _, actor, svc, req = 资料环境
first = req.samples[0]
second = first.model_copy(
update={"sample_id": "same-source-other-split", "split": "calibration"}
)
changed = req.model_copy(update={"samples": (first, second)})
with pytest.raises(Muse错误, match="跨分割"):
svc.发布回放数据集(actor, changed)
@pytest.mark.case_id(
"NC-w25-257005",
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
when="按该用例触发读取、派发、并发或失败恢复",
then=["实际CLI与HTTP维护入口同版幂等,作者范围复检后读回"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_实际CLI与HTTP维护入口发布同一来源封存__257005(资料环境, tmp_path):
import subprocess
import sys
import test_生产评测权限隔离 as 数据测试
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
env, pools, actor, svc, req = 资料环境
config = 数据测试._配置文件(pools[用途.维护], tmp_path)
config.write_text(config.read_text().replace("eval-author", actor.作者))
payload = tmp_path / "publish-replay.json"
payload.write_text(req.model_dump_json())
process = subprocess.run(
[sys.executable, "-I", "-m", "muse", "评测", str(config), "发布回放数据集", str(payload)],
cwd=tmp_path,
capture_output=True,
text=True,
timeout=30,
)
assert process.returncode == 0, process.stderr
first = json.loads(process.stdout)
assert not first["duplicate"]
with TestClient(创建应用(读取配置(config)), headers={"origin": "http://testserver"}) as client:
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
result = client.post("/api/v1/evaluation/replay-datasets", json=req.model_dump(mode="json"))
assert result.status_code == 201, result.text
assert result.json() == {**first, "duplicate": True}
evaluated = 评测服务(pools[用途.评测])
assert (
evaluated.读取数据集(replace(actor, 用途=用途.评测), first["version_id"])["public_hash"]
== first["public_hash"]
)
with pytest.raises(Muse错误, match="资料批准不属于本作者"):
evaluated.读取数据集(
replace(actor, 用途=用途.评测, 作者="other-author"), first["version_id"]
)
assert env["收到"] == []