实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
413 lines
18 KiB
Python
413 lines
18 KiB
Python
"""方法真实确认版本、S04用途投影、隔离逐例消费与CLI/HTTP封存。"""
|
||
|
||
import json
|
||
from dataclasses import asdict, replace
|
||
from datetime import UTC, datetime, timedelta
|
||
from uuid import uuid4
|
||
|
||
import psycopg
|
||
import pytest
|
||
import test_上下文快照与索引写入 as 方法测试
|
||
import test_生产评测权限隔离 as 数据测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
import test_评测语义检测 as 检测测试
|
||
|
||
from muse.任务运行.接口 import 任务状态
|
||
from muse.元数据.接口 import 元数据服务, 启用命令, 字段限制, 定义哈希, 策略快照
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.效果评测.接口 import 实验执行请求, 实验请求, 方法数据集发布, 评测服务
|
||
from muse.知识方法.接口 import 状态命令
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
方法环境 = 方法测试.方法环境
|
||
|
||
|
||
@pytest.fixture
|
||
def 方法对照环境(执行环境, 方法环境):
|
||
env = 执行环境
|
||
business, author, methods, source, _, pools = 方法环境
|
||
request = 方法测试._提案(
|
||
str(uuid4()),
|
||
"动作表现人物迟疑",
|
||
str(source.source_id),
|
||
owner=author.作者,
|
||
)
|
||
request = replace(request, content={**request.content, "例证出处": "只许规划读取的字段秘密"})
|
||
proposed = methods.提出方法(author, "method-for-eval", request)
|
||
方法测试._确认(methods, author, proposed["target_ref"])
|
||
detail = methods.读取方法详情(author, request.method_id)
|
||
assert detail["state"] == "confirmed"
|
||
version = detail["versions"][0]
|
||
publisher = 评测服务(pools[用途.维护])
|
||
maintained = replace(author, 用途=用途.维护)
|
||
publish = 方法数据集发布.model_validate(
|
||
{
|
||
**数据测试._数据().model_dump(mode="json"),
|
||
"samples": [数据测试._数据().samples[-1].model_dump(mode="json")],
|
||
"dataset_id": "method-comparison",
|
||
"method_version_id": str(version["version_id"]),
|
||
"approval_ref": "synthetic:method-eval",
|
||
"expires_at": datetime.now(UTC) + timedelta(minutes=30),
|
||
}
|
||
)
|
||
if env["request"].comparison_profile == "writer_rubric":
|
||
from test_文学评分执行与条件第三 import 依据
|
||
|
||
publish = publish.model_copy(
|
||
update={
|
||
"samples": tuple(
|
||
s.model_copy(update={"answer": {**s.answer, "judging_basis": 依据}})
|
||
for s in publish.samples
|
||
)
|
||
}
|
||
)
|
||
receipt = publisher.发布方法数据集(maintained, publish)
|
||
actor = replace(env["actor"], 作者=author.作者)
|
||
exp_request = 实验请求.model_validate(
|
||
{
|
||
**env["request"].model_dump(mode="json"),
|
||
"dataset_version_id": receipt["version_id"],
|
||
"dataset_hash": receipt["public_hash"],
|
||
"target": receipt["method_target"],
|
||
}
|
||
)
|
||
exp = env["app"].要求评测().创建实验(actor, "method-comparison-run", exp_request)
|
||
return {
|
||
**env,
|
||
"actor": actor,
|
||
"exp": exp,
|
||
"request": exp_request,
|
||
"business": business,
|
||
"owner": author,
|
||
"methods": methods,
|
||
"version": version,
|
||
"method_id": request.method_id,
|
||
"publisher": publisher,
|
||
"maintained": maintained,
|
||
"publish": publish,
|
||
"receipt": receipt,
|
||
}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f801",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["确切确认版本、同模板对照、实际方法输入与生产权限隔离"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_两臂同模板且仅处理组实际使用确切方法__25f801(方法对照环境):
|
||
env = 方法对照环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
published = svc.读取数据集(env["actor"], env["receipt"]["version_id"])
|
||
material = published["public_manifest"]["method_material"]
|
||
assert material["content_hash"] == env["version"]["content_hash"]
|
||
assert material["version_id"] == str(env["version"]["version_id"])
|
||
assert "只许规划读取的字段秘密" not in material["text"]
|
||
assert env["publisher"].发布方法数据集(env["maintained"], env["publish"])["duplicate"]
|
||
started = 运行测试._启动执行(env)
|
||
assert len(started["units"]) == 3
|
||
运行测试._运行就绪(env)
|
||
requests = list(env["received"])
|
||
assert len(requests) == 2 and requests[0]["instructions"] == requests[1]["instructions"]
|
||
inputs = [json.loads(r["input"]) for r in requests]
|
||
assert sum("方法材料" in r["context"] for r in inputs) == 1
|
||
treated = next(r for r in inputs if "方法材料" in r["context"])
|
||
assert treated["context"].pop("方法材料") == material["text"]
|
||
assert inputs[0] == inputs[1]
|
||
assert "只许规划读取的字段秘密" not in json.dumps(requests, ensure_ascii=False)
|
||
assert "ORACLE-SECRET" not in json.dumps(requests, ensure_ascii=False)
|
||
svc.推进实验(env["actor"], eid)
|
||
运行测试._运行就绪(env)
|
||
report = svc.读取实验报告(env["actor"], eid)
|
||
assert report["adapter"] == "writer_method_v1"
|
||
assert report["primary_pair"] == ["control", "treatment"]
|
||
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
|
||
assert len(env["received"]) == 3
|
||
assert env["methods"].读取方法详情(env["owner"], env["method_id"])["state"] == "confirmed"
|
||
assert env["methods"].列出消费(env["owner"], str(env["version"]["version_id"])) == []
|
||
with env["pool"].连接() as conn, pytest.raises(psycopg.errors.InsufficientPrivilege):
|
||
conn.execute("SELECT * FROM public.muse_method_version")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f802",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["错版本、哈希、目标、作者或手填资料不能创建方法实验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("bad", ["version", "hash", "target", "author", "raw-dataset"])
|
||
def test_错误方法目标或作者不能创建实验__25f802(方法对照环境, bad):
|
||
env = 方法对照环境
|
||
request = env["request"].model_dump(mode="json")
|
||
actor = env["actor"]
|
||
if bad == "version":
|
||
request["target"]["version"] = "999"
|
||
elif bad == "hash":
|
||
request["target"]["content_hash"] = "a" * 64
|
||
elif bad == "target":
|
||
request["target"]["target_ref"] = "B04.method:" + str(uuid4())
|
||
elif bad == "author":
|
||
actor = replace(actor, 作者="other-author")
|
||
else:
|
||
dataset = env["publisher"].发布数据集(
|
||
env["maintained"], 数据测试._数据().model_copy(update={"dataset_id": "raw-method"})
|
||
)
|
||
request.update(
|
||
dataset_version_id=dataset["version_id"], dataset_hash=dataset["public_hash"]
|
||
)
|
||
with pytest.raises(Muse错误):
|
||
env["app"].要求评测().创建实验(
|
||
actor, "invalid-method-target", 实验请求.model_validate(request)
|
||
)
|
||
assert not env["received"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f803",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["非维护用途、其他作者或撤回方法不能发布新快照"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_撤回与错误身份不发布新方法资料__25f803(方法对照环境):
|
||
env = 方法对照环境
|
||
for purpose in (用途.生产, 用途.评测):
|
||
with pytest.raises(Muse错误):
|
||
评测服务(env["pools"][purpose]).发布方法数据集(
|
||
replace(env["maintained"], 用途=purpose), env["publish"]
|
||
)
|
||
with pytest.raises(Muse错误):
|
||
env["publisher"].发布方法数据集(
|
||
replace(env["maintained"], 作者="other-author"),
|
||
env["publish"].model_copy(update={"dataset_id": "foreign-method"}),
|
||
)
|
||
env["methods"].变更方法状态(
|
||
env["owner"],
|
||
"withdraw-method",
|
||
状态命令(env["method_id"], "withdraw", expected_state_revision=1),
|
||
)
|
||
with pytest.raises(Muse错误, match="撤回"):
|
||
env["publisher"].发布方法数据集(
|
||
env["maintained"], env["publish"].model_copy(update={"dataset_id": "withdrawn-method"})
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f804",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["批准期限及字段策略变更阻断新派发"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_执行期限与字段策略改变在发送前拒绝__25f804(方法对照环境):
|
||
env = 方法对照环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
with pytest.raises(Muse错误, match="不足"):
|
||
svc.启动实验(
|
||
env["actor"],
|
||
eid,
|
||
实验执行请求(approval_ref="synthetic", deadline=datetime.now(UTC) + timedelta(hours=1)),
|
||
)
|
||
运行测试._启动执行(env)
|
||
with env["pools"][用途.维护].连接() as conn:
|
||
md = 元数据服务(conn)
|
||
old = md.当前策略("craft")
|
||
field = next(f.field_id for f in md.读取结构("craft", 1).字段 if f.key == "原理")
|
||
changed = 策略快照("craft", old.version + 1, (字段限制(field, ("aiContext:generation",)),))
|
||
md.更新策略(
|
||
changed,
|
||
启用命令(
|
||
"restrict-method",
|
||
定义哈希(asdict(changed)),
|
||
old.version,
|
||
env["owner"].作者,
|
||
"type:craft",
|
||
),
|
||
)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert not env["received"]
|
||
assert not any(u["output"] for u in svc.读取执行工作面(env["actor"], eid)["units"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f805",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["实际方法输入被替换在外发前拒绝"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_方法实际发送输入篡改在外发前拒绝__25f805(方法对照环境, monkeypatch):
|
||
import muse.编排.执行评测 as flow
|
||
|
||
env = 方法对照环境
|
||
original = flow.模型请求
|
||
|
||
def drop(*args, **kwargs):
|
||
request = original(*args, **kwargs)
|
||
value = json.loads(request.用户输入)
|
||
value["context"]["方法材料"] = "其他未评测方法"
|
||
return replace(request, 用户输入=json.dumps(value, ensure_ascii=False))
|
||
|
||
monkeypatch.setattr(flow, "模型请求", drop)
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert not env["received"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f806",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["真实CLI与HTTP幂等封存同一方法版本"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_方法封存CLI与HTTP重放同一真实版本__25f806(方法对照环境, tmp_path):
|
||
import subprocess
|
||
import sys
|
||
|
||
from fastapi.testclient import TestClient
|
||
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.配置 import 读取配置
|
||
|
||
env = 方法对照环境
|
||
config = 数据测试._配置文件(env["pools"][用途.维护], tmp_path)
|
||
config.write_text(config.read_text().replace("eval-author", env["owner"].作者))
|
||
request = env["publish"].model_copy(update={"dataset_id": "method-cli"})
|
||
path = tmp_path / "method-publish.json"
|
||
path.write_text(request.model_dump_json())
|
||
result = subprocess.run(
|
||
[sys.executable, "-I", "-m", "muse", "评测", str(config), "发布方法数据集", str(path)],
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=30,
|
||
)
|
||
assert result.returncode == 0, result.stderr
|
||
receipt = json.loads(result.stdout)
|
||
with TestClient(创建应用(读取配置(config)), headers={"origin": "http://testserver"}) as client:
|
||
assert (
|
||
client.post(
|
||
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
|
||
).status_code
|
||
== 200
|
||
)
|
||
response = client.post(
|
||
"/api/v1/evaluation/method-datasets", json=request.model_dump(mode="json")
|
||
)
|
||
assert response.status_code == 201, response.text
|
||
assert response.json() == {**receipt, "duplicate": True}
|
||
assert receipt["method_target"] == env["receipt"]["method_target"]
|
||
assert not env["received"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f807",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["已存方法交付恢复不重发,继续原独立比较"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_方法单元已交付后恢复不重发且保留来源__25f807(方法对照环境, monkeypatch):
|
||
env = 方法对照环境
|
||
svc = env["app"].要求评测()
|
||
original = svc.保存执行交付
|
||
failed = False
|
||
|
||
def interrupt(*args, **kwargs):
|
||
nonlocal failed
|
||
result = original(*args, **kwargs)
|
||
if not failed:
|
||
failed = True
|
||
raise Muse错误("合成中断:方法输入及交付已保存")
|
||
return result
|
||
|
||
monkeypatch.setattr(svc, "保存执行交付", interrupt)
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert len(env["received"]) == 2
|
||
face = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
units = [u for u in face["units"] if u["kind"] == "generation"]
|
||
output = {u["unit_id"]: u["output"] for u in units}
|
||
assert all(output.values())
|
||
runtime = env["app"].任务运行
|
||
for unit in units:
|
||
task = runtime.读取任务(unit["task_id"])
|
||
if task.状态 is 任务状态.已失败:
|
||
runtime.控制任务(
|
||
task.任务ID, env["actor"].作者, task.状态, "恢复", 命令ID="restore-method"
|
||
)
|
||
运行测试._运行就绪(env)
|
||
assert len(env["received"]) == 2
|
||
current = svc.读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
assert {
|
||
u["unit_id"]: u["output"] for u in current["units"] if u["kind"] == "generation"
|
||
} == output
|
||
svc.推进实验(env["actor"], env["exp"]["experiment_id"])
|
||
运行测试._运行就绪(env)
|
||
assert svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])["coverage"]["compared"] == 1
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f808",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["确认版本内容改变不能沿用原哈希发布"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_确认版本被改字节不按旧哈希发布__25f808(方法对照环境, monkeypatch):
|
||
from muse.知识方法.存储 import 方法存储
|
||
|
||
env = 方法对照环境
|
||
original = 方法存储.读取版本
|
||
|
||
def changed(self, version_id):
|
||
row = original(self, version_id)
|
||
return {**row, "content": {**row["content"], "原理": "未确认的另一套原理"}}
|
||
|
||
monkeypatch.setattr(方法存储, "读取版本", changed)
|
||
with pytest.raises(Muse错误, match="确认内容哈希"):
|
||
env["publisher"].发布方法数据集(
|
||
env["maintained"], env["publish"].model_copy(update={"dataset_id": "corrupt-method"})
|
||
)
|
||
assert not env["received"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f809",
|
||
environment="隔离PostgreSQL与合成HTTP;不证明真实模型文学效果",
|
||
given="隔离PG中真实确认方法、用途投影、固定样本与S02配置",
|
||
when="从B04公开封存入口经B10和S02实际对照、读回及恢复",
|
||
then=["实际文学评分与独立检测衔接,样本/标定不足不得签效果凭据"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [检测测试.参数], indirect=True)
|
||
def test_方法对照贯通文学评分检测且不足证据不签效果凭据__25f809(方法对照环境):
|
||
env = 方法对照环境
|
||
svc = env["app"].要求评测()
|
||
request = env["request"].model_copy(update={"effect_policy": "writer-effect-v1"})
|
||
exp = svc.创建实验(env["actor"], "method-literary-evaluation", request)
|
||
env = {**env, "exp": exp}
|
||
report = 检测测试._执行(env)
|
||
assert len(env["received"]) == 6 # 两写手、两检测、两独立评委;候补不需要执行。
|
||
assert report["samples"][0]["detection_complete"]
|
||
assert report["samples"][0]["literary"]["control:treatment"]["execution_verified"]
|
||
assert report["effect"]["assessment"]["target"] == env["receipt"]["method_target"]
|
||
assert report["effect"]["assessment"]["gate_b"]["status"] == "insufficient_evidence"
|
||
with pytest.raises(Muse错误, match="效果未通过"):
|
||
svc.封存效果判据(env["actor"], exp["experiment_id"])
|
||
assert env["methods"].读取方法详情(env["owner"], env["method_id"])["state"] == "confirmed"
|