- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
210 lines
8.9 KiB
Python
210 lines
8.9 KiB
Python
"""正式owner封存→评测三臂→独立三对比较,全部实际使用隔离S02和合成HTTP。"""
|
||
|
||
import json
|
||
from dataclasses import asdict, replace
|
||
from datetime import UTC, datetime, timedelta
|
||
from decimal import Decimal
|
||
|
||
import psycopg
|
||
import pytest
|
||
import test_回放资料封存 as 资料测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
|
||
from muse.任务运行.接口 import 角色策略目录
|
||
from muse.元数据.接口 import 元数据服务, 启用命令, 字段限制, 定义哈希, 策略快照
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.效果评测.接口 import 实验执行请求, 实验请求
|
||
|
||
pytestmark = [
|
||
pytest.mark.数据库,
|
||
pytest.mark.parametrize("执行环境", [{"judges": 2}], indirect=True),
|
||
]
|
||
执行环境 = 运行测试.执行环境
|
||
生成环境 = 资料测试.生成环境
|
||
资料环境 = 资料测试.资料环境
|
||
|
||
|
||
@pytest.fixture
|
||
def ABC环境(执行环境, 资料环境):
|
||
env = 执行环境
|
||
_, pools, author, source_svc, source_req = 资料环境
|
||
dataset = source_svc.发布回放数据集(author, source_req)
|
||
data = env["exp"]["conditions"]
|
||
req = 实验请求.model_validate(
|
||
{
|
||
"dataset_version_id": dataset["version_id"],
|
||
"dataset_hash": dataset["public_hash"],
|
||
"split": "holdout",
|
||
"target": {
|
||
"kind": "code",
|
||
"target_ref": "B09.writer-replay",
|
||
"version": 角色策略目录.从发布包().资源发布身份,
|
||
"content_hash": 角色策略目录.从发布包().资源发布身份,
|
||
},
|
||
"generator_role": "writer",
|
||
"generator": {k: data["generator"][k] for k in ("config_id", "version")},
|
||
"judges": [{k: j[k] for k in ("config_id", "version")} for j in data["judges"]],
|
||
"arms": ["A", "B", "C"],
|
||
"dimensions": ["清晰度"],
|
||
"max_cost_usd": "12",
|
||
"max_calls_per_sample": 18,
|
||
}
|
||
)
|
||
actor = replace(env["actor"], 作者=author.作者)
|
||
exp = env["app"].要求评测().创建实验(actor, "abc-fixture", req)
|
||
return {**env, "actor": actor, "exp": exp, "pools": pools, "request": req}
|
||
|
||
|
||
def test_三臂实际生成只消费本臂资料且独立比较保留主AC__258001(ABC环境):
|
||
env = ABC环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
started = 运行测试._启动执行(env)
|
||
assert len(started["units"]) == 9
|
||
运行测试._运行就绪(env)
|
||
writers = [json.loads(r["input"]) for r in env["received"]]
|
||
assert len(writers) == 3 and len({r["instructions"] for r in env["received"]}) == 1
|
||
assert sorted(
|
||
(bool(r["context"]["历史正文"]), bool(r["context"]["卡片索引"])) for r in writers
|
||
) == [(False, True), (True, False), (True, True)]
|
||
assert all(r["original"] == "" for r in writers)
|
||
assert "ORACLE-TARGET-PRIVATE" not in json.dumps(env["received"], ensure_ascii=False)
|
||
assert all(
|
||
"arm_order" not in r["input"] and "source_authorization" not in r["input"]
|
||
for r in env["received"]
|
||
)
|
||
svc.推进实验(env["actor"], eid)
|
||
运行测试._运行就绪(env)
|
||
report = svc.读取实验报告(env["actor"], eid)
|
||
assert report["primary_pair"] == ["A", "C"] and report["adapter"] == "writer_source_abc_v1"
|
||
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 1}
|
||
sample = report["samples"][0]
|
||
assert sample["comparisons"] == {"A:B": "tie", "A:C": "tie", "B:C": "tie"}
|
||
assert len(sample["decisions"]) == 6 and len(env["received"]) == 9
|
||
assert Decimal(report["cost"]["total_usd"]) == Decimal("1.125")
|
||
assert report["literary_quality"]["metrics"] is None
|
||
with env["pool"].连接() as conn, pytest.raises(psycopg.errors.InsufficientPrivilege):
|
||
conn.execute("SELECT * FROM public.muse_document")
|
||
|
||
|
||
@pytest.mark.parametrize("bad", ["author", "deadline"])
|
||
def test_资料批准作者或执行期限不符时不创建任务__258002(ABC环境, bad):
|
||
env = ABC环境
|
||
actor = env["actor"]
|
||
eid = env["exp"]["experiment_id"]
|
||
deadline = datetime.now(UTC) + timedelta(minutes=40 if bad == "deadline" else 10)
|
||
with pytest.raises(Muse错误, match="资料批准"):
|
||
if bad == "author":
|
||
actor = replace(actor, 作者="other-eval-author")
|
||
eid = (
|
||
env["app"]
|
||
.要求评测()
|
||
.创建实验(actor, "foreign-source", env["request"])["experiment_id"]
|
||
)
|
||
env["app"].要求评测().启动实验(
|
||
actor, eid, 实验执行请求(approval_ref="synthetic", deadline=deadline)
|
||
)
|
||
assert env["received"] == []
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
|
||
|
||
def _收紧字段(conn, author):
|
||
metadata = 元数据服务(conn)
|
||
current = metadata.当前策略("character")
|
||
field = metadata.读取结构("character", 1).字段[0].field_id
|
||
policy = 策略快照(
|
||
"character", current.version + 1, (字段限制(field, ("aiContext:generation",)),)
|
||
)
|
||
metadata.更新策略(
|
||
policy,
|
||
启用命令(
|
||
"restrict-replay", 定义哈希(asdict(policy)), current.version, author, "type:character"
|
||
),
|
||
)
|
||
|
||
|
||
def test_封存后字段用途收紧阻断所有新派发且保留固定实验__258003(ABC环境):
|
||
env = ABC环境
|
||
运行测试._启动执行(env)
|
||
with env["pools"][用途.维护].连接() as conn, conn.transaction():
|
||
_收紧字段(conn, env["actor"].作者)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
assert len(result["units"]) == 9 and all(u["output"] is None for u in result["units"])
|
||
assert all(u["state"] == "failed" for u in result["units"] if u["kind"] == "generation")
|
||
assert env["received"] == []
|
||
|
||
|
||
def test_评测只读权限可保护字段且维护变更等待发送事务__258004(ABC环境):
|
||
from concurrent.futures import ThreadPoolExecutor
|
||
from queue import Queue
|
||
from time import monotonic, sleep
|
||
|
||
from muse.上下文.接口 import 核对回放字段策略
|
||
|
||
env = ABC环境
|
||
svc = env["app"].要求评测()
|
||
dataset = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"])
|
||
policies = dataset["public_manifest"]["samples"][0]["replay_materials"]["policies"]
|
||
backend = Queue()
|
||
|
||
def change():
|
||
with env["pools"][用途.维护].连接() as conn, conn.transaction():
|
||
backend.put(conn.info.backend_pid)
|
||
_收紧字段(conn, env["actor"].作者)
|
||
|
||
with ThreadPoolExecutor(max_workers=1) as executor:
|
||
with env["pool"].连接() as conn, conn.transaction():
|
||
核对回放字段策略(conn, policies, 保护到事务结束=True)
|
||
future = executor.submit(change)
|
||
pid = backend.get(timeout=3)
|
||
deadline = monotonic() + 3
|
||
while True:
|
||
with env["pools"][用途.维护].连接(只读=True) as probe:
|
||
blockers = probe.execute("SELECT pg_blocking_pids(%s)", (pid,)).fetchone()[0]
|
||
if conn.info.backend_pid in blockers:
|
||
break
|
||
assert monotonic() < deadline and not future.done(), "维护变更未被实际消费事务保护"
|
||
sleep(0.02)
|
||
future.result(timeout=3)
|
||
with env["pool"].连接(只读=True) as conn, pytest.raises(Muse错误, match="字段用途策略已变化"):
|
||
核对回放字段策略(conn, policies)
|
||
|
||
|
||
def test_合法派发后策略变化仍保存本次交付但拒绝后续调用__258005(ABC环境):
|
||
from threading import Thread
|
||
|
||
env = ABC环境
|
||
app = env["app"]
|
||
env["scripted"].append("inflight")
|
||
运行测试._启动执行(env)
|
||
prepare = app.任务运行.领取步骤("eval-worker", ["eval.prepare"])
|
||
app.任务运行.执行一步(prepare)
|
||
claim = app.任务运行.领取步骤("eval-worker", ["eval.call.writer"])
|
||
errors = []
|
||
|
||
def execute():
|
||
try:
|
||
app.任务运行.执行一步(claim)
|
||
except Exception as exc:
|
||
errors.append(exc)
|
||
|
||
worker = Thread(target=execute, daemon=True)
|
||
worker.start()
|
||
try:
|
||
assert env["in_flight"].wait(5)
|
||
with env["pools"][用途.维护].连接() as conn, conn.transaction():
|
||
_收紧字段(conn, env["actor"].作者)
|
||
finally:
|
||
env["respond"].set()
|
||
worker.join(timeout=8)
|
||
assert not worker.is_alive() and not errors
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = app.要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
saved = [u for u in result["units"] if u["output"] is not None]
|
||
assert len(saved) == 1 and saved[0]["state"] == "completed"
|
||
assert saved[0]["evidence"]["cost_state"] == "settled"
|
||
assert len(env["received"]) == 1
|