- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
276 lines
11 KiB
Python
276 lines
11 KiB
Python
"""报告从隔离PG与S02实际交付计算;固定分母、失败费用和分歧不能消失。"""
|
||
|
||
from decimal import Decimal
|
||
|
||
import pytest
|
||
import test_ABC正文回放 as ABC测试
|
||
import test_实际效果判据 as 效果测试
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_标定凭据消费 as 消费测试
|
||
import test_标定金标准与凭据 as 标定测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
import test_评测语义检测 as 检测测试
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
ABC环境 = ABC测试.ABC环境
|
||
资料环境 = ABC测试.资料环境
|
||
生成环境 = ABC测试.生成环境
|
||
|
||
|
||
def _报告(env):
|
||
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [{"samples": 2}], indirect=True)
|
||
def test_完整分母与分组失败保留且恢复计入全部费用__254001(执行环境):
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
env["scripted"].append("bad_output")
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
svc.推进实验(env["actor"], eid)
|
||
运行测试._运行就绪(env)
|
||
before = _报告(env)
|
||
assert before["coverage"] == {"samples": 2, "generated": 1, "compared": 1}
|
||
assert before["outcomes"]["incomplete"] == before["outcomes"]["tie"] == 1
|
||
assert sorted(g["outcomes"]["incomplete"] for g in before["source_groups"].values()) == [0, 1]
|
||
assert Decimal(before["cost"]["total_usd"]) == Decimal(".625")
|
||
assert before["cost"]["sent_calls"] == 5
|
||
assert before["literary_quality"] == {"calibration": "unverified", "metrics": None}
|
||
运行测试._恢复失败单元(env)
|
||
运行测试._运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
运行测试._运行就绪(env)
|
||
after = _报告(env)
|
||
assert after["coverage"] == {"samples": 2, "generated": 2, "compared": 2}
|
||
assert after["outcomes"]["tie"] == 2 and after["outcomes"]["incomplete"] == 0
|
||
assert Decimal(after["cost"]["total_usd"]) == Decimal(".875")
|
||
calls = [c for sample in after["samples"] for u in sample["units"] for c in u["calls"]]
|
||
assert len(calls) == len({c["call_id"] for c in calls}) == 7
|
||
assert before["report_hash"] != after["report_hash"]
|
||
assert _报告(env) == after
|
||
|
||
|
||
def test_未知成本与未开始比较不能报告为零或平局__254002(执行环境):
|
||
env = 执行环境
|
||
env["scripted"].extend(["ok", "unknown_cost"])
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
report = _报告(env)
|
||
assert report["state"] == "reconciling"
|
||
assert report["cost"]["total_usd"] is None and report["cost"]["has_unknown"]
|
||
assert Decimal(report["cost"]["known_total_usd"]) == Decimal(".125")
|
||
assert report["outcomes"]["incomplete"] == 1 and report["outcomes"]["tie"] == 0
|
||
assert report["coverage"] == {"samples": 1, "generated": 0, "compared": 0}
|
||
assert len(env["received"]) == 2
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [{"judges": 2}], indirect=True)
|
||
def test_独立评委分歧逐维保留且不生成启用许可__254003(执行环境):
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
svc.推进实验(env["actor"], eid)
|
||
# 第二评委已交换文本;同选left代表选择不同的实际实验臂。
|
||
env["scripted"].extend(["choose_left", "choose_left"])
|
||
运行测试._运行就绪(env)
|
||
report = _报告(env)
|
||
assert report["outcomes"]["disagreement"] == 1
|
||
assert report["dimensions"]["清晰度"]["disagreement"] == 1
|
||
sample = report["samples"][0]
|
||
assert {d["choice"] for d in sample["decisions"]} == {"control", "treatment"}
|
||
assert len({d["model"] for d in sample["decisions"]}) == 2
|
||
assert all(d["call_id"] and d["output_hash"] for d in sample["decisions"])
|
||
assert report["validation_modes"] == ["offline_contract"]
|
||
assert report["activation_status"] == "not_evaluated"
|
||
assert _报告(env) == report
|
||
svc.取消实验(env["actor"], eid, "retain-comparison")
|
||
stopped = _报告(env)
|
||
assert stopped["state"] == "stopped" and stopped["samples"] == report["samples"]
|
||
|
||
|
||
@pytest.mark.浏览器
|
||
@pytest.mark.timeout(480)
|
||
@pytest.mark.parametrize(
|
||
"report_kind,执行环境",
|
||
[
|
||
("plain", {}),
|
||
("abc", {}),
|
||
("judgment", {}),
|
||
("corrected", {}),
|
||
("literary", 文学测试.参数),
|
||
("calibration", 标定测试.参数),
|
||
("qualification", 标定测试.参数),
|
||
("detection", 检测测试.参数),
|
||
("effect", 效果测试.参数),
|
||
],
|
||
indirect=["执行环境"],
|
||
)
|
||
def test_浏览器报告来自同一隔离实验且读取不追加调用__254004(
|
||
执行环境, tmp_path, request, report_kind
|
||
):
|
||
import json
|
||
import os
|
||
import socket
|
||
import subprocess
|
||
from contextlib import asynccontextmanager
|
||
from dataclasses import replace
|
||
from pathlib import Path
|
||
from threading import Event, Thread
|
||
|
||
import uvicorn
|
||
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.配置 import 服务配置
|
||
|
||
if os.environ.get("MUSE_RUN_BROWSER") != "1":
|
||
pytest.skip("浏览器未显式启用,不计为通过")
|
||
env = request.getfixturevalue("ABC环境") if report_kind == "abc" else 执行环境
|
||
if report_kind == "effect":
|
||
cert = 效果测试._校准(env)
|
||
env = 效果测试._新实验(env, 10, certificate=cert)
|
||
if report_kind == "qualification":
|
||
cert = 消费测试._校准(env)
|
||
env = 消费测试._新实验(env, 消费测试._请求(env, [cert]))
|
||
if report_kind not in {
|
||
"judgment",
|
||
"corrected",
|
||
"literary",
|
||
"calibration",
|
||
"qualification",
|
||
"detection",
|
||
"effect",
|
||
}:
|
||
env["scripted"].extend(["ok", "unknown_cost"])
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
if report_kind == "calibration":
|
||
标定测试._发布(env)
|
||
if report_kind in {
|
||
"judgment",
|
||
"corrected",
|
||
"literary",
|
||
"calibration",
|
||
"qualification",
|
||
"detection",
|
||
"effect",
|
||
}:
|
||
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
|
||
if report_kind == "corrected":
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
if report_kind == "literary":
|
||
env["scripted"].extend(["ok", "score_gap"])
|
||
if report_kind == "detection":
|
||
env["scripted"].extend(["high", "ok"])
|
||
运行测试._运行就绪(env)
|
||
if report_kind in {"literary", "calibration", "qualification", "detection", "effect"}:
|
||
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
|
||
运行测试._运行就绪(env)
|
||
before = _报告(env)
|
||
calls_before = len(env["received"])
|
||
root = Path(__file__).resolve().parents[2]
|
||
password = tmp_path / "browser-password"
|
||
password.write_text("synthetic-evaluation-browser")
|
||
password.chmod(0o600)
|
||
sock = socket.socket()
|
||
sock.bind(("127.0.0.1", 0))
|
||
port = sock.getsockname()[1]
|
||
base = f"http://127.0.0.1:{port}"
|
||
config = replace(
|
||
env["app"].配置,
|
||
HTTP=服务配置(
|
||
str(password), 作者ID=env["actor"].作者, 端口=port, 公开地址=base, 允许来源=(base,)
|
||
),
|
||
)
|
||
app = 创建应用(config)
|
||
previous = app.router.lifespan_context
|
||
ready = Event()
|
||
|
||
@asynccontextmanager
|
||
async def 生命周期(instance):
|
||
async with previous(instance):
|
||
instance.state.装配 = env["app"]
|
||
ready.set()
|
||
yield
|
||
|
||
app.router.lifespan_context = 生命周期
|
||
server = uvicorn.Server(
|
||
uvicorn.Config(app, log_level="warning", access_log=False, timeout_graceful_shutdown=3)
|
||
)
|
||
thread = Thread(target=server.run, kwargs={"sockets": [sock]}, daemon=True)
|
||
output = (
|
||
Path(os.environ.get("MUSE_BROWSER_ARTIFACTS", str(tmp_path / "browser"))).resolve()
|
||
/ report_kind
|
||
)
|
||
output.mkdir(parents=True, exist_ok=True)
|
||
process_env = {
|
||
**os.environ,
|
||
"MUSE_WORKBENCH_URL": base,
|
||
"MUSE_AUTHOR_PASSWORD_FILE": str(password),
|
||
"MUSE_EVALUATION_ID": env["exp"]["experiment_id"],
|
||
"MUSE_EVALUATION_SAMPLE": before["samples"][0]["sample_id"],
|
||
"MUSE_EVALUATION_COMPLETED": "1"
|
||
if report_kind
|
||
in {
|
||
"judgment",
|
||
"corrected",
|
||
"literary",
|
||
"calibration",
|
||
"qualification",
|
||
"detection",
|
||
"effect",
|
||
}
|
||
else "0",
|
||
}
|
||
chrome = Path("/Applications/Google Chrome.app/Contents/MacOS/Google Chrome")
|
||
if chrome.exists():
|
||
process_env.setdefault("MUSE_BROWSER_EXECUTABLE", str(chrome))
|
||
thread.start()
|
||
try:
|
||
assert ready.wait(10)
|
||
with (output.parent / f"报告浏览器-{report_kind}.json").open("w") as log:
|
||
result = subprocess.run(
|
||
[
|
||
"pnpm",
|
||
"exec",
|
||
"playwright",
|
||
"test",
|
||
"评测报告.spec.ts",
|
||
"--reporter=json",
|
||
"--output",
|
||
str(output),
|
||
],
|
||
cwd=root / "web",
|
||
env=process_env,
|
||
stdout=log,
|
||
stderr=subprocess.STDOUT,
|
||
timeout=90,
|
||
)
|
||
assert result.returncode == 0, (
|
||
f"浏览器失败,见{output.parent / f'报告浏览器-{report_kind}.json'}"
|
||
)
|
||
receipts = list(output.rglob("报告回执.json"))
|
||
assert len(receipts) == 1
|
||
after = _报告(env)
|
||
assert json.loads(receipts[0].read_text())["report_hash"] == after["report_hash"]
|
||
if report_kind == "calibration":
|
||
assert before["calibration"]["receipt_id"] is None
|
||
assert after["calibration"]["receipt_id"]
|
||
assert before["samples"] == after["samples"]
|
||
elif report_kind == "effect":
|
||
assert before["effect"]["receipt_id"] is None and after["effect"]["receipt_id"]
|
||
assert before["effect"]["assessment"] == after["effect"]["assessment"]
|
||
assert before["samples"] == after["samples"]
|
||
else:
|
||
assert after == before
|
||
assert len(env["received"]) == calls_before
|
||
finally:
|
||
server.should_exit = True
|
||
thread.join(timeout=8)
|
||
sock.close()
|
||
assert not thread.is_alive()
|