muse-agent-example/tests/集成/test_实际效果判据.py
zizi 0eba8e4f34 重构(测试): 删除85次合成调用的资格终态用例
去掉四条用假HTTP循环85次的慢用例,以及浏览器报告中依赖同一路径的effect参数。
资格门槛仍由契约与单元覆盖;生产效果标准不改。
2026-09-18 11:21:41 +08:00

305 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""从完整隔离生成、标定、独立评委和检测得出效果判据,不认证真实模型收益。"""
import json
from collections import Counter
from dataclasses import replace
import pytest
import test_文学评分执行与条件第三 as 文学测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
from muse.共享.调用身份 import 用途
from muse.效果评测.接口 import 实验请求, 数据集发布, 评测服务, 评测错误
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
场景 = [
"battle",
"character_dialogue",
"turning_point",
"information_reveal",
"returning_character",
]
标定策略 = {
"minimum_samples": 5,
"minimum_source_groups": 5,
"max_mae": 0.5,
"max_absolute_error": 1.5,
"minimum_verdict_agreement": 1.0,
}
def _基分(text, control):
return 8.0 if "细雨" in text and control["gain"] else 7.5
def _响应工厂():
control = {"gain": True, "calibrating": False, "judge_index": 0, "high": False}
def generate(request, script):
from muse.正文写作.接口 import 生成正文模板
public = json.loads(request["input"])
action = (
"他迎着细雨走到门前。"
if 生成正文模板()[0] in request["instructions"]
else "他站在山门旁等候。"
)
return {"paragraphs": [{"text": public["instruction"] + "。" + action}]}
def judge(material, script):
result = 文学测试._输出(material, "ok")
offset = 0.0
if control["calibrating"]:
index = control["judge_index"]
if index < 10:
offset = 0.5 if index % 2 == 0 else -0.5
control["judge_index"] += 1
for card in result["candidate_scores"]:
score = _基分(material[card["side"]]["text"], control) + offset
for row in card["scores"]:
row["score"] = score
return result
def detect(material, script):
return 检测测试._输出(
material, "high" if control["high"] and "细雨" in material["candidate"] else script
)
return {
"fixture_control": control,
"generator_output": generate,
"judge_output": judge,
"detector_output": detect,
}
参数 = {
**检测测试.参数,
"output_factory": _响应工厂,
"budget_calls": 300,
"budget_amount": "100",
"max_steps": 300,
}
def _数据样本(count, *, calibration=False):
locale = "林道" if calibration else "渡口"
rows = []
for i in range(count):
rows.append(
{
"sample_id": f"{locale}-{i}",
"source_ref": f"synthetic:{locale}-{i}",
"license_ref": "synthetic:owned",
"source_groups": [f"private:{locale}-work-{i if calibration else i // 5}"],
"stratification": {
"work_ref": f"private:{locale}-work-{i if calibration else i // 5}",
"annotation_ref": "synthetic:annotation",
"new_character_ratio": [0.0, 0.25, 0.75, None, 0.25][i % 5],
},
"split": "calibration" if calibration else "holdout",
"input": {
"instruction": f"{locale}第{i}处的行动",
"original": f"{locale}{i}的原始底稿。",
"context": {},
},
"answer": {
"judging_basis": {
**文学测试.依据,
"scenario": 场景[i % 5],
"sources": [
{
"source_id": "outline",
"kind": "fine_outline",
"text": f"{locale}第{i}处细纲。",
},
{
"source_id": "history",
"kind": "historical_prose",
"text": f"{locale}第{i}处先前正文。",
},
],
}
},
}
)
return rows
def _数据(env, count, *, calibration=False):
locale = "林道" if calibration else "渡口"
return 评测服务(env["pools"][用途.维护]).发布数据集(
replace(env["actor"], 用途=用途.维护),
数据集发布.model_validate(
{
"dataset_id": f"effect-{locale}",
"revision": 1,
"samples": _数据样本(count, calibration=calibration),
}
),
)
def _新实验(env, count=5, *, calibration=False, certificate=None):
data = _数据(env, count, calibration=calibration)
req = 实验请求.model_validate(
{
**env["request"].model_dump(mode="json"),
"dataset_version_id": data["version_id"],
"dataset_hash": data["public_hash"],
"split": "calibration" if calibration else "holdout",
"max_cost_usd": "20",
"detector": None if calibration else env["request"].detector.model_dump(mode="json"),
"calibration_policy": 标定策略 if calibration else None,
"evaluation_goal": "qualification" if certificate else "diagnostic",
"calibration_use": {
"policy": 标定策略,
"references": [
{
"experiment_id": certificate["result"]["experiment_id"],
"receipt_hash": certificate["receipt_hash"],
}
],
}
if certificate
else None,
"effect_policy": None if calibration else "writer-effect-v1",
}
)
exp = (
env["app"]
.要求评测()
.创建实验(env["actor"], "effect-calibration" if calibration else "effect-holdout", req)
)
return {**env, "exp": exp, "request": req}
def _完成(env):
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
运行测试._运行就绪(env)
文学测试._推进(env)
运行测试._运行就绪(env)
return env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
@pytest.mark.case_id(
"TC-64bcf0ac927f",
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
when="经真实生成、独立检测、比较与公开效果入口读回",
then=[
"真实检测高严重度交付导致A阶段failed",
"目标每例高严重度计数1及target_semantic_failure留存,不封存合格凭据",
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_真实检测失败进入A阶段并阻止效果签章__25f401(执行环境):
env = _新实验(执行环境)
env["fixture_control"]["high"] = True
result = _完成(env)
assert result["assessment"]["gate_a"] == {
"status": "failed",
"reasons": ["target_semantic_failure"],
}
assert result["receipt_id"] is None
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
assert all(
s["detections"]["treatment"]["report"]["high_severity_count"] == 1
for s in report["samples"]
)
with pytest.raises(评测错误, match="效果未通过"):
env["app"].要求评测().封存效果判据(env["actor"], env["exp"]["experiment_id"])
assert len(env["received"]) == 30
@pytest.mark.case_id(
"TC-a089d73880d8",
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
when="经真实生成、独立检测、比较与公开效果入口读回",
then=["一例真实生成失败时A阶段为failed", "首要原因system_failure先于样本和场景不足"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_实际系统失败优先于五场景不足__25f402(执行环境):
env = _新实验(执行环境, 1)
env["scripted"].append("bad_output")
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]}
@pytest.mark.case_id(
"TC-e61898aba4e5",
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
when="经真实生成、独立检测、比较与公开效果入口读回",
then=[
"单作品五场景A阶段仍passed",
"single_work单独标注且作品样本数量为唯一5例",
"完整五场景不误报scenario_coverage_incomplete",
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_单作品完整A阶段仍保留选择偏差__25f403(执行环境):
env = _新实验(执行环境)
result = _完成(env)
assert result["assessment"]["gate_a"]["status"] == "passed"
assert "single_work" in result["assessment"]["confounders"]
assert list(result["assessment"]["metrics"]["work_counts"].values()) == [5]
assert "scenario_coverage_incomplete" not in result["assessment"]["confounders"]
assert result["assessment"]["gate_b"]["status"] == "insufficient_evidence"
@pytest.mark.case_id(
"NC-w25-25f405",
environment="隔离PG与合成HTTP",
given="固定分母、独立样本及本例边界输入",
when="经实际S02生成后重复读回效果报告",
then=["单次读回每份已核验交付只验一次;下一次读回全部重新核验"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_单次读取复用已核验交付而新读取仍重验__25f405(执行环境, monkeypatch):
env = _新实验(执行环境)
_完成(env)
runtime = env["app"].任务运行
original = runtime.核对历史结构化交付于
calls = Counter()
def count(*args, **kwargs):
calls[args[4]] += 1
return original(*args, **kwargs)
monkeypatch.setattr(runtime, "核对历史结构化交付于", count)
first = 文学测试._读(env)
assert len(calls) == 30 and set(calls.values()) == {1}
calls.clear()
assert 文学测试._读(env) == first
assert len(calls) == 30 and set(calls.values()) == {1}
@pytest.mark.case_id(
"TC-602d0bb37aa2",
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
when="经真实生成、独立检测、比较与公开效果入口读回",
then=["公开入口只接实验ID;手工汇总对象以UUID合同拒绝,无模型调用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_公开效果入口拒绝手工汇总对象__25f407(执行环境):
env = 执行环境
with pytest.raises(评测错误, match="UUID"):
env["app"].要求评测().读取效果判据(
env["actor"], {"gate": "A", "samples": [], "passed": True}
)
assert not env["received"]