305 lines
12 KiB
Python
305 lines
12 KiB
Python
"""从完整隔离生成、标定、独立评委和检测得出效果判据,不认证真实模型收益。"""
|
||
|
||
import json
|
||
from collections import Counter
|
||
from dataclasses import replace
|
||
|
||
import pytest
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
import test_评测语义检测 as 检测测试
|
||
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.效果评测.接口 import 实验请求, 数据集发布, 评测服务, 评测错误
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
场景 = [
|
||
"battle",
|
||
"character_dialogue",
|
||
"turning_point",
|
||
"information_reveal",
|
||
"returning_character",
|
||
]
|
||
标定策略 = {
|
||
"minimum_samples": 5,
|
||
"minimum_source_groups": 5,
|
||
"max_mae": 0.5,
|
||
"max_absolute_error": 1.5,
|
||
"minimum_verdict_agreement": 1.0,
|
||
}
|
||
|
||
|
||
def _基分(text, control):
|
||
return 8.0 if "细雨" in text and control["gain"] else 7.5
|
||
|
||
|
||
def _响应工厂():
|
||
control = {"gain": True, "calibrating": False, "judge_index": 0, "high": False}
|
||
|
||
def generate(request, script):
|
||
from muse.正文写作.接口 import 生成正文模板
|
||
|
||
public = json.loads(request["input"])
|
||
action = (
|
||
"他迎着细雨走到门前。"
|
||
if 生成正文模板()[0] in request["instructions"]
|
||
else "他站在山门旁等候。"
|
||
)
|
||
return {"paragraphs": [{"text": public["instruction"] + "。" + action}]}
|
||
|
||
def judge(material, script):
|
||
result = 文学测试._输出(material, "ok")
|
||
offset = 0.0
|
||
if control["calibrating"]:
|
||
index = control["judge_index"]
|
||
if index < 10:
|
||
offset = 0.5 if index % 2 == 0 else -0.5
|
||
control["judge_index"] += 1
|
||
for card in result["candidate_scores"]:
|
||
score = _基分(material[card["side"]]["text"], control) + offset
|
||
for row in card["scores"]:
|
||
row["score"] = score
|
||
return result
|
||
|
||
def detect(material, script):
|
||
return 检测测试._输出(
|
||
material, "high" if control["high"] and "细雨" in material["candidate"] else script
|
||
)
|
||
|
||
return {
|
||
"fixture_control": control,
|
||
"generator_output": generate,
|
||
"judge_output": judge,
|
||
"detector_output": detect,
|
||
}
|
||
|
||
|
||
参数 = {
|
||
**检测测试.参数,
|
||
"output_factory": _响应工厂,
|
||
"budget_calls": 300,
|
||
"budget_amount": "100",
|
||
"max_steps": 300,
|
||
}
|
||
|
||
|
||
def _数据样本(count, *, calibration=False):
|
||
locale = "林道" if calibration else "渡口"
|
||
rows = []
|
||
for i in range(count):
|
||
rows.append(
|
||
{
|
||
"sample_id": f"{locale}-{i}",
|
||
"source_ref": f"synthetic:{locale}-{i}",
|
||
"license_ref": "synthetic:owned",
|
||
"source_groups": [f"private:{locale}-work-{i if calibration else i // 5}"],
|
||
"stratification": {
|
||
"work_ref": f"private:{locale}-work-{i if calibration else i // 5}",
|
||
"annotation_ref": "synthetic:annotation",
|
||
"new_character_ratio": [0.0, 0.25, 0.75, None, 0.25][i % 5],
|
||
},
|
||
"split": "calibration" if calibration else "holdout",
|
||
"input": {
|
||
"instruction": f"{locale}第{i}处的行动",
|
||
"original": f"{locale}{i}的原始底稿。",
|
||
"context": {},
|
||
},
|
||
"answer": {
|
||
"judging_basis": {
|
||
**文学测试.依据,
|
||
"scenario": 场景[i % 5],
|
||
"sources": [
|
||
{
|
||
"source_id": "outline",
|
||
"kind": "fine_outline",
|
||
"text": f"{locale}第{i}处细纲。",
|
||
},
|
||
{
|
||
"source_id": "history",
|
||
"kind": "historical_prose",
|
||
"text": f"{locale}第{i}处先前正文。",
|
||
},
|
||
],
|
||
}
|
||
},
|
||
}
|
||
)
|
||
return rows
|
||
|
||
|
||
def _数据(env, count, *, calibration=False):
|
||
locale = "林道" if calibration else "渡口"
|
||
return 评测服务(env["pools"][用途.维护]).发布数据集(
|
||
replace(env["actor"], 用途=用途.维护),
|
||
数据集发布.model_validate(
|
||
{
|
||
"dataset_id": f"effect-{locale}",
|
||
"revision": 1,
|
||
"samples": _数据样本(count, calibration=calibration),
|
||
}
|
||
),
|
||
)
|
||
|
||
|
||
def _新实验(env, count=5, *, calibration=False, certificate=None):
|
||
data = _数据(env, count, calibration=calibration)
|
||
req = 实验请求.model_validate(
|
||
{
|
||
**env["request"].model_dump(mode="json"),
|
||
"dataset_version_id": data["version_id"],
|
||
"dataset_hash": data["public_hash"],
|
||
"split": "calibration" if calibration else "holdout",
|
||
"max_cost_usd": "20",
|
||
"detector": None if calibration else env["request"].detector.model_dump(mode="json"),
|
||
"calibration_policy": 标定策略 if calibration else None,
|
||
"evaluation_goal": "qualification" if certificate else "diagnostic",
|
||
"calibration_use": {
|
||
"policy": 标定策略,
|
||
"references": [
|
||
{
|
||
"experiment_id": certificate["result"]["experiment_id"],
|
||
"receipt_hash": certificate["receipt_hash"],
|
||
}
|
||
],
|
||
}
|
||
if certificate
|
||
else None,
|
||
"effect_policy": None if calibration else "writer-effect-v1",
|
||
}
|
||
)
|
||
exp = (
|
||
env["app"]
|
||
.要求评测()
|
||
.创建实验(env["actor"], "effect-calibration" if calibration else "effect-holdout", req)
|
||
)
|
||
return {**env, "exp": exp, "request": req}
|
||
|
||
|
||
def _完成(env):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
return env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-64bcf0ac927f",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=[
|
||
"真实检测高严重度交付导致A阶段failed",
|
||
"目标每例高严重度计数1及target_semantic_failure留存,不封存合格凭据",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实检测失败进入A阶段并阻止效果签章__25f401(执行环境):
|
||
env = _新实验(执行环境)
|
||
env["fixture_control"]["high"] = True
|
||
result = _完成(env)
|
||
assert result["assessment"]["gate_a"] == {
|
||
"status": "failed",
|
||
"reasons": ["target_semantic_failure"],
|
||
}
|
||
assert result["receipt_id"] is None
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert all(
|
||
s["detections"]["treatment"]["report"]["high_severity_count"] == 1
|
||
for s in report["samples"]
|
||
)
|
||
with pytest.raises(评测错误, match="效果未通过"):
|
||
env["app"].要求评测().封存效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
assert len(env["received"]) == 30
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-a089d73880d8",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=["一例真实生成失败时A阶段为failed", "首要原因system_failure先于样本和场景不足"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_实际系统失败优先于五场景不足__25f402(执行环境):
|
||
env = _新实验(执行环境, 1)
|
||
env["scripted"].append("bad_output")
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-e61898aba4e5",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=[
|
||
"单作品五场景A阶段仍passed",
|
||
"single_work单独标注且作品样本数量为唯一5例",
|
||
"完整五场景不误报scenario_coverage_incomplete",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_单作品完整A阶段仍保留选择偏差__25f403(执行环境):
|
||
env = _新实验(执行环境)
|
||
result = _完成(env)
|
||
assert result["assessment"]["gate_a"]["status"] == "passed"
|
||
assert "single_work" in result["assessment"]["confounders"]
|
||
assert list(result["assessment"]["metrics"]["work_counts"].values()) == [5]
|
||
assert "scenario_coverage_incomplete" not in result["assessment"]["confounders"]
|
||
assert result["assessment"]["gate_b"]["status"] == "insufficient_evidence"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f405",
|
||
environment="隔离PG与合成HTTP",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="经实际S02生成后重复读回效果报告",
|
||
then=["单次读回每份已核验交付只验一次;下一次读回全部重新核验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_单次读取复用已核验交付而新读取仍重验__25f405(执行环境, monkeypatch):
|
||
env = _新实验(执行环境)
|
||
_完成(env)
|
||
runtime = env["app"].任务运行
|
||
original = runtime.核对历史结构化交付于
|
||
calls = Counter()
|
||
|
||
def count(*args, **kwargs):
|
||
calls[args[4]] += 1
|
||
return original(*args, **kwargs)
|
||
|
||
monkeypatch.setattr(runtime, "核对历史结构化交付于", count)
|
||
first = 文学测试._读(env)
|
||
assert len(calls) == 30 and set(calls.values()) == {1}
|
||
calls.clear()
|
||
assert 文学测试._读(env) == first
|
||
assert len(calls) == 30 and set(calls.values()) == {1}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-602d0bb37aa2",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=["公开入口只接实验ID;手工汇总对象以UUID合同拒绝,无模型调用"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_公开效果入口拒绝手工汇总对象__25f407(执行环境):
|
||
env = 执行环境
|
||
with pytest.raises(评测错误, match="UUID"):
|
||
env["app"].要求评测().读取效果判据(
|
||
env["actor"], {"gate": "A", "samples": [], "passed": True}
|
||
)
|
||
assert not env["received"]
|