实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
250 lines
11 KiB
Python
250 lines
11 KiB
Python
"""正式owner封存→评测三臂→独立三对比较,全部实际使用隔离S02和合成HTTP。"""
|
||
|
||
import json
|
||
from dataclasses import asdict, replace
|
||
from datetime import UTC, datetime, timedelta
|
||
from decimal import Decimal
|
||
|
||
import psycopg
|
||
import pytest
|
||
import test_回放资料封存 as 资料测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
|
||
from muse.任务运行.接口 import 角色策略目录
|
||
from muse.元数据.接口 import 元数据服务, 启用命令, 字段限制, 定义哈希, 策略快照
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.效果评测.接口 import 实验执行请求, 实验请求
|
||
|
||
pytestmark = [
|
||
pytest.mark.数据库,
|
||
pytest.mark.parametrize("执行环境", [{"judges": 2}], indirect=True),
|
||
]
|
||
执行环境 = 运行测试.执行环境
|
||
生成环境 = 资料测试.生成环境
|
||
资料环境 = 资料测试.资料环境
|
||
|
||
|
||
@pytest.fixture
|
||
def ABC环境(执行环境, 资料环境):
|
||
env = 执行环境
|
||
_, pools, author, source_svc, source_req = 资料环境
|
||
dataset = source_svc.发布回放数据集(author, source_req)
|
||
data = env["exp"]["conditions"]
|
||
req = 实验请求.model_validate(
|
||
{
|
||
"dataset_version_id": dataset["version_id"],
|
||
"dataset_hash": dataset["public_hash"],
|
||
"split": "holdout",
|
||
"target": {
|
||
"kind": "code",
|
||
"target_ref": "B09.writer-replay",
|
||
"version": 角色策略目录.从发布包().资源发布身份,
|
||
"content_hash": 角色策略目录.从发布包().资源发布身份,
|
||
},
|
||
"generator_role": "writer",
|
||
"generator": {k: data["generator"][k] for k in ("config_id", "version")},
|
||
"judges": [{k: j[k] for k in ("config_id", "version")} for j in data["judges"]],
|
||
"arms": ["A", "B", "C"],
|
||
"dimensions": ["清晰度"],
|
||
"max_cost_usd": "12",
|
||
"max_calls_per_sample": 18,
|
||
}
|
||
)
|
||
actor = replace(env["actor"], 作者=author.作者)
|
||
exp = env["app"].要求评测().创建实验(actor, "abc-fixture", req)
|
||
return {**env, "actor": actor, "exp": exp, "pools": pools, "request": req}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-258001",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["三臂加两名独立评委九次受控调用,主AC及AB/BC诊断分开且不读正式表"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_三臂实际生成只消费本臂资料且独立比较保留主AC__258001(ABC环境):
|
||
env = ABC环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
started = 运行测试._启动执行(env)
|
||
assert len(started["units"]) == 9
|
||
运行测试._运行就绪(env)
|
||
writers = [json.loads(r["input"]) for r in env["received"]]
|
||
assert len(writers) == 3 and len({r["instructions"] for r in env["received"]}) == 1
|
||
assert sorted(
|
||
(bool(r["context"]["历史正文"]), bool(r["context"]["卡片索引"])) for r in writers
|
||
) == [(False, True), (True, False), (True, True)]
|
||
assert all(r["original"] == "" for r in writers)
|
||
assert "ORACLE-TARGET-PRIVATE" not in json.dumps(env["received"], ensure_ascii=False)
|
||
assert all(
|
||
"arm_order" not in r["input"] and "source_authorization" not in r["input"]
|
||
for r in env["received"]
|
||
)
|
||
svc.推进实验(env["actor"], eid)
|
||
运行测试._运行就绪(env)
|
||
report = svc.读取实验报告(env["actor"], eid)
|
||
assert report["primary_pair"] == ["A", "C"] and report["adapter"] == "writer_source_abc_v1"
|
||
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
|
||
sample = report["samples"][0]
|
||
assert sample["comparisons"] == {"A:B": "tie", "A:C": "tie", "B:C": "tie"}
|
||
assert len(sample["decisions"]) == 6 and len(env["received"]) == 9
|
||
assert Decimal(report["cost"]["total_usd"]) == Decimal("1.125")
|
||
assert report["literary_quality"]["metrics"] is None
|
||
with env["pool"].连接() as conn, pytest.raises(psycopg.errors.InsufficientPrivilege):
|
||
conn.execute("SELECT * FROM public.muse_document")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-258002",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["批准作者或执行期限不符不创建任务"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("bad", ["author", "deadline"])
|
||
def test_资料批准作者或执行期限不符时不创建任务__258002(ABC环境, bad):
|
||
env = ABC环境
|
||
actor = env["actor"]
|
||
eid = env["exp"]["experiment_id"]
|
||
deadline = datetime.now(UTC) + timedelta(minutes=40 if bad == "deadline" else 10)
|
||
with pytest.raises(Muse错误, match="资料批准"):
|
||
if bad == "author":
|
||
actor = replace(actor, 作者="other-eval-author")
|
||
eid = (
|
||
env["app"]
|
||
.要求评测()
|
||
.创建实验(actor, "foreign-source", env["request"])["experiment_id"]
|
||
)
|
||
env["app"].要求评测().启动实验(
|
||
actor, eid, 实验执行请求(approval_ref="synthetic", deadline=deadline)
|
||
)
|
||
assert env["received"] == []
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
|
||
|
||
def _收紧字段(conn, author):
|
||
metadata = 元数据服务(conn)
|
||
current = metadata.当前策略("character")
|
||
field = metadata.读取结构("character", 1).字段[0].field_id
|
||
policy = 策略快照(
|
||
"character", current.version + 1, (字段限制(field, ("aiContext:generation",)),)
|
||
)
|
||
metadata.更新策略(
|
||
policy,
|
||
启用命令(
|
||
"restrict-replay", 定义哈希(asdict(policy)), current.version, author, "type:character"
|
||
),
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-258003",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["字段策略收紧后所有新派发拒绝,固定单元仍保留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_封存后字段用途收紧阻断所有新派发且保留固定实验__258003(ABC环境):
|
||
env = ABC环境
|
||
运行测试._启动执行(env)
|
||
with env["pools"][用途.维护].连接() as conn, conn.transaction():
|
||
_收紧字段(conn, env["actor"].作者)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
assert len(result["units"]) == 9 and all(u["output"] is None for u in result["units"])
|
||
assert all(u["state"] == "failed" for u in result["units"] if u["kind"] == "generation")
|
||
assert env["received"] == []
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-258004",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["评测只读身份真实持锁,维护更新在PG等待并在释放后生效"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_评测只读权限可保护字段且维护变更等待发送事务__258004(ABC环境):
|
||
from concurrent.futures import ThreadPoolExecutor
|
||
from queue import Queue
|
||
from time import monotonic, sleep
|
||
|
||
from muse.上下文.接口 import 核对回放字段策略
|
||
|
||
env = ABC环境
|
||
svc = env["app"].要求评测()
|
||
dataset = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"])
|
||
policies = dataset["public_manifest"]["samples"][0]["replay_materials"]["policies"]
|
||
backend = Queue()
|
||
|
||
def change():
|
||
with env["pools"][用途.维护].连接() as conn, conn.transaction():
|
||
backend.put(conn.info.backend_pid)
|
||
_收紧字段(conn, env["actor"].作者)
|
||
|
||
with ThreadPoolExecutor(max_workers=1) as executor:
|
||
with env["pool"].连接() as conn, conn.transaction():
|
||
核对回放字段策略(conn, policies, 保护到事务结束=True)
|
||
future = executor.submit(change)
|
||
pid = backend.get(timeout=3)
|
||
deadline = monotonic() + 3
|
||
while True:
|
||
with env["pools"][用途.维护].连接(只读=True) as probe:
|
||
blockers = probe.execute("SELECT pg_blocking_pids(%s)", (pid,)).fetchone()[0]
|
||
if conn.info.backend_pid in blockers:
|
||
break
|
||
assert monotonic() < deadline and not future.done(), "维护变更未被实际消费事务保护"
|
||
sleep(0.02)
|
||
future.result(timeout=3)
|
||
with env["pool"].连接(只读=True) as conn, pytest.raises(Muse错误, match="字段用途策略已变化"):
|
||
核对回放字段策略(conn, policies)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-258005",
|
||
environment="隔离PG;受控合成HTTP,未认证真实文学效果",
|
||
given="隔离PG中的真实来源、配置和固定实验;HTTP为合成提供方",
|
||
when="按该用例触发读取、派发、并发或失败恢复",
|
||
then=["合法发送后策略变化仍保存原交付,新消费拒绝,费用与单元不丢失"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_合法派发后策略变化仍保存本次交付但拒绝后续调用__258005(ABC环境):
|
||
from threading import Thread
|
||
|
||
env = ABC环境
|
||
app = env["app"]
|
||
env["scripted"].append("inflight")
|
||
运行测试._启动执行(env)
|
||
prepare = app.任务运行.领取步骤("eval-worker", ["eval.prepare"])
|
||
app.任务运行.执行一步(prepare)
|
||
claim = app.任务运行.领取步骤("eval-worker", ["eval.call.writer"])
|
||
errors = []
|
||
|
||
def execute():
|
||
try:
|
||
app.任务运行.执行一步(claim)
|
||
except Exception as exc:
|
||
errors.append(exc)
|
||
|
||
worker = Thread(target=execute, daemon=True)
|
||
worker.start()
|
||
try:
|
||
assert env["in_flight"].wait(5)
|
||
with env["pools"][用途.维护].连接() as conn, conn.transaction():
|
||
_收紧字段(conn, env["actor"].作者)
|
||
finally:
|
||
env["respond"].set()
|
||
worker.join(timeout=8)
|
||
assert not worker.is_alive() and not errors
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = app.要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
saved = [u for u in result["units"] if u["output"] is not None]
|
||
assert len(saved) == 1 and saved[0]["state"] == "completed"
|
||
assert saved[0]["evidence"]["cost_state"] == "settled"
|
||
assert len(env["received"]) == 1
|