muse-agent-example/tests/集成/test_评测完整报告.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

314 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""报告从隔离PG与S02实际交付计算;固定分母、失败费用和分歧不能消失。"""
from decimal import Decimal
import pytest
import test_ABC正文回放 as ABC测试
import test_实际效果判据 as 效果测试
import test_文学评分执行与条件第三 as 文学测试
import test_标定凭据消费 as 消费测试
import test_标定金标准与凭据 as 标定测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
ABC环境 = ABC测试.ABC环境
资料环境 = ABC测试.资料环境
生成环境 = ABC测试.生成环境
def _报告(env):
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
@pytest.mark.case_id(
"NC-w25-254001",
environment="隔离PG与受控合成HTTP",
given="已登记的合成实验与实际S02逐例结果",
when="经公开报告入口读取完整结果",
then=["完整分母与来源分组保留失败样本,恢复费用包含全部调用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{"samples": 2}], indirect=True)
def test_完整分母与分组失败保留且恢复计入全部费用__254001(执行环境):
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
env["scripted"].append("bad_output")
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
svc.推进实验(env["actor"], eid)
运行测试._运行就绪(env)
before = _报告(env)
assert before["coverage"] == {"samples": 2, "generated": 1, "compared": 1}
assert before["outcomes"]["incomplete"] == before["outcomes"]["tie"] == 1
assert sorted(g["outcomes"]["incomplete"] for g in before["source_groups"].values()) == [0, 1]
assert Decimal(before["cost"]["total_usd"]) == Decimal(".625")
assert before["cost"]["sent_calls"] == 5
assert before["literary_quality"] == {"calibration": "unverified", "metrics": None}
运行测试._恢复失败单元(env)
运行测试._运行就绪(env)
svc.推进实验(env["actor"], eid)
运行测试._运行就绪(env)
after = _报告(env)
assert after["coverage"] == {"samples": 2, "generated": 2, "compared": 2}
assert after["outcomes"]["tie"] == 2 and after["outcomes"]["incomplete"] == 0
assert Decimal(after["cost"]["total_usd"]) == Decimal(".875")
calls = [c for sample in after["samples"] for u in sample["units"] for c in u["calls"]]
assert len(calls) == len({c["call_id"] for c in calls}) == 7
assert before["report_hash"] != after["report_hash"]
assert _报告(env) == after
@pytest.mark.case_id(
"NC-w25-254002",
environment="隔离PG与受控合成HTTP",
given="已登记的合成实验与实际S02逐例结果",
when="经公开报告入口读取完整结果",
then=["已知小计和未知总额分开;未开始比较保持缺失"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_未知成本与未开始比较不能报告为零或平局__254002(执行环境):
env = 执行环境
env["scripted"].extend(["ok", "unknown_cost"])
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
report = _报告(env)
assert report["state"] == "reconciling"
assert report["cost"]["total_usd"] is None and report["cost"]["has_unknown"]
assert Decimal(report["cost"]["known_total_usd"]) == Decimal(".125")
assert report["outcomes"]["incomplete"] == 1 and report["outcomes"]["tie"] == 0
assert report["coverage"] == {"samples": 1, "generated": 0, "compared": 0}
assert len(env["received"]) == 2
@pytest.mark.case_id(
"NC-w25-254003",
environment="隔离PG与受控合成HTTP",
given="已登记的合成实验与实际S02逐例结果",
when="经公开报告入口读取完整结果",
then=["两名实际独立合成评委分歧逐维保留,停止不改旧判断"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{"judges": 2}], indirect=True)
def test_独立评委分歧逐维保留且不生成启用许可__254003(执行环境):
env = 执行环境
svc = env["app"].要求评测()
eid = env["exp"]["experiment_id"]
运行测试._启动执行(env)
运行测试._运行就绪(env)
svc.推进实验(env["actor"], eid)
# 第二评委已交换文本;同选left代表选择不同的实际实验臂。
env["scripted"].extend(["choose_left", "choose_left"])
运行测试._运行就绪(env)
report = _报告(env)
assert report["outcomes"]["disagreement"] == 1
assert report["dimensions"]["清晰度"]["disagreement"] == 1
sample = report["samples"][0]
assert {d["choice"] for d in sample["decisions"]} == {"control", "treatment"}
assert len({d["model"] for d in sample["decisions"]}) == 2
assert all(d["call_id"] and d["output_hash"] for d in sample["decisions"])
assert report["validation_modes"] == ["offline_contract"]
assert report["activation_status"] == "not_evaluated"
assert _报告(env) == report
svc.取消实验(env["actor"], eid, "retain-comparison")
stopped = _报告(env)
assert stopped["state"] == "stopped" and stopped["samples"] == report["samples"]
@pytest.mark.case_id(
"NC-w25-254004",
environment="隔离PG与真实浏览器;显式MUSE_RUN_BROWSER=1",
given="已登记的合成实验与实际S02逐例结果",
when="经公开报告入口读取完整结果",
then=[
"真实浏览器从同一报告读回,刷新不追加模型调用",
"资格报告显示实际标定覆盖和来源链接,读取不新增模型调用或启用",
"实际检测发现和原字引文在报告页读回,未知与未配置分别显示",
"实际效果分层、判据理由与凭据经工作台读回和封存;正式启用仍未评定",
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.浏览器
@pytest.mark.timeout(900)
@pytest.mark.慢
@pytest.mark.parametrize(
"report_kind,执行环境",
[
("plain", {}),
("abc", {}),
("judgment", {}),
("corrected", {}),
("literary", 文学测试.参数),
("calibration", 标定测试.参数),
("qualification", 标定测试.参数),
("detection", 检测测试.参数),
("effect", 效果测试.参数),
],
indirect=["执行环境"],
)
def test_浏览器报告来自同一隔离实验且读取不追加调用__254004(
执行环境, tmp_path, request, report_kind
):
import json
import os
import socket
import subprocess
from contextlib import asynccontextmanager
from dataclasses import replace
from pathlib import Path
from threading import Event, Thread
import uvicorn
from muse.接入.http.应用 import 创建应用
from muse.配置 import 服务配置
if os.environ.get("MUSE_RUN_BROWSER") != "1":
pytest.skip("浏览器未显式启用,不计为通过")
env = request.getfixturevalue("ABC环境") if report_kind == "abc" else 执行环境
if report_kind == "effect":
cert = 效果测试._校准(env)
env = 效果测试._新实验(env, 10, certificate=cert)
if report_kind == "qualification":
cert = 消费测试._校准(env)
env = 消费测试._新实验(env, 消费测试._请求(env, [cert]))
if report_kind not in {
"judgment",
"corrected",
"literary",
"calibration",
"qualification",
"detection",
"effect",
}:
env["scripted"].extend(["ok", "unknown_cost"])
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
if report_kind == "calibration":
标定测试._发布(env)
if report_kind in {
"judgment",
"corrected",
"literary",
"calibration",
"qualification",
"detection",
"effect",
}:
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
if report_kind == "corrected":
env["scripted"].extend(["bad_quote", "ok"])
if report_kind == "literary":
env["scripted"].extend(["ok", "score_gap"])
if report_kind == "detection":
env["scripted"].extend(["high", "ok"])
运行测试._运行就绪(env)
if report_kind in {"literary", "calibration", "qualification", "detection", "effect"}:
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
运行测试._运行就绪(env)
before = _报告(env)
calls_before = len(env["received"])
root = Path(__file__).resolve().parents[2]
password = tmp_path / "browser-password"
password.write_text("synthetic-evaluation-browser")
password.chmod(0o600)
sock = socket.socket()
sock.bind(("127.0.0.1", 0))
port = sock.getsockname()[1]
base = f"http://127.0.0.1:{port}"
config = replace(
env["app"].配置,
HTTP=服务配置(
str(password), 作者ID=env["actor"].作者, 端口=port, 公开地址=base, 允许来源=(base,)
),
)
app = 创建应用(config)
previous = app.router.lifespan_context
ready = Event()
@asynccontextmanager
async def 生命周期(instance):
async with previous(instance):
instance.state.装配 = env["app"]
ready.set()
yield
app.router.lifespan_context = 生命周期
server = uvicorn.Server(
uvicorn.Config(app, log_level="warning", access_log=False, timeout_graceful_shutdown=3)
)
thread = Thread(target=server.run, kwargs={"sockets": [sock]}, daemon=True)
output = (
Path(os.environ.get("MUSE_BROWSER_ARTIFACTS", str(tmp_path / "browser"))).resolve()
/ report_kind
)
output.mkdir(parents=True, exist_ok=True)
process_env = {
**os.environ,
"MUSE_WORKBENCH_URL": base,
"MUSE_AUTHOR_PASSWORD_FILE": str(password),
"MUSE_EVALUATION_ID": env["exp"]["experiment_id"],
"MUSE_EVALUATION_SAMPLE": before["samples"][0]["sample_id"],
"MUSE_EVALUATION_COMPLETED": "1"
if report_kind
in {
"judgment",
"corrected",
"literary",
"calibration",
"qualification",
"detection",
"effect",
}
else "0",
}
chrome = Path("/Applications/Google Chrome.app/Contents/MacOS/Google Chrome")
if chrome.exists():
process_env.setdefault("MUSE_BROWSER_EXECUTABLE", str(chrome))
thread.start()
try:
assert ready.wait(10)
with (output.parent / f"报告浏览器-{report_kind}.json").open("w") as log:
result = subprocess.run(
[
"pnpm",
"exec",
"playwright",
"test",
"评测报告.spec.ts",
"--reporter=json",
"--output",
str(output),
],
cwd=root / "web",
env=process_env,
stdout=log,
stderr=subprocess.STDOUT,
timeout=90,
)
assert result.returncode == 0, (
f"浏览器失败,见{output.parent / f'报告浏览器-{report_kind}.json'}"
)
receipts = list(output.rglob("报告回执.json"))
assert len(receipts) == 1
after = _报告(env)
assert json.loads(receipts[0].read_text())["report_hash"] == after["report_hash"]
if report_kind == "calibration":
assert before["calibration"]["receipt_id"] is None
assert after["calibration"]["receipt_id"]
assert before["samples"] == after["samples"]
elif report_kind == "effect":
assert before["effect"]["receipt_id"] is None and after["effect"]["receipt_id"]
assert before["effect"]["assessment"] == after["effect"]["assessment"]
assert before["samples"] == after["samples"]
else:
assert after == before
assert len(env["received"]) == calls_before
finally:
server.should_exit = True
thread.join(timeout=8)
sock.close()
assert not thread.is_alive()