"""从完整隔离生成、标定、独立评委和检测得出效果判据,不认证真实模型收益。""" import json from collections import Counter from dataclasses import replace import pytest import test_文学评分执行与条件第三 as 文学测试 import test_评测执行与失败收敛 as 运行测试 import test_评测语义检测 as 检测测试 from muse.共享.调用身份 import 用途 from muse.效果评测.接口 import 实验请求, 数据集发布, 评测服务, 评测错误 pytestmark = pytest.mark.数据库 执行环境 = 运行测试.执行环境 场景 = [ "battle", "character_dialogue", "turning_point", "information_reveal", "returning_character", ] 标定策略 = { "minimum_samples": 5, "minimum_source_groups": 5, "max_mae": 0.5, "max_absolute_error": 1.5, "minimum_verdict_agreement": 1.0, } def _基分(text, control): return 8.0 if "细雨" in text and control["gain"] else 7.5 def _响应工厂(): control = {"gain": True, "calibrating": False, "judge_index": 0, "high": False} def generate(request, script): from muse.正文写作.接口 import 生成正文模板 public = json.loads(request["input"]) action = ( "他迎着细雨走到门前。" if 生成正文模板()[0] in request["instructions"] else "他站在山门旁等候。" ) return {"paragraphs": [{"text": public["instruction"] + "。" + action}]} def judge(material, script): result = 文学测试._输出(material, "ok") offset = 0.0 if control["calibrating"]: index = control["judge_index"] if index < 10: offset = 0.5 if index % 2 == 0 else -0.5 control["judge_index"] += 1 for card in result["candidate_scores"]: score = _基分(material[card["side"]]["text"], control) + offset for row in card["scores"]: row["score"] = score return result def detect(material, script): return 检测测试._输出( material, "high" if control["high"] and "细雨" in material["candidate"] else script ) return { "fixture_control": control, "generator_output": generate, "judge_output": judge, "detector_output": detect, } 参数 = { **检测测试.参数, "output_factory": _响应工厂, "budget_calls": 300, "budget_amount": "100", "max_steps": 300, } def _数据样本(count, *, calibration=False): locale = "林道" if calibration else "渡口" rows = [] for i in range(count): rows.append( { "sample_id": f"{locale}-{i}", "source_ref": f"synthetic:{locale}-{i}", "license_ref": "synthetic:owned", "source_groups": [f"private:{locale}-work-{i if calibration else i // 5}"], "stratification": { "work_ref": f"private:{locale}-work-{i if calibration else i // 5}", "annotation_ref": "synthetic:annotation", "new_character_ratio": [0.0, 0.25, 0.75, None, 0.25][i % 5], }, "split": "calibration" if calibration else "holdout", "input": { "instruction": f"{locale}第{i}处的行动", "original": f"{locale}{i}的原始底稿。", "context": {}, }, "answer": { "judging_basis": { **文学测试.依据, "scenario": 场景[i % 5], "sources": [ { "source_id": "outline", "kind": "fine_outline", "text": f"{locale}第{i}处细纲。", }, { "source_id": "history", "kind": "historical_prose", "text": f"{locale}第{i}处先前正文。", }, ], } }, } ) return rows def _数据(env, count, *, calibration=False): locale = "林道" if calibration else "渡口" return 评测服务(env["pools"][用途.维护]).发布数据集( replace(env["actor"], 用途=用途.维护), 数据集发布.model_validate( { "dataset_id": f"effect-{locale}", "revision": 1, "samples": _数据样本(count, calibration=calibration), } ), ) def _新实验(env, count=5, *, calibration=False, certificate=None): data = _数据(env, count, calibration=calibration) req = 实验请求.model_validate( { **env["request"].model_dump(mode="json"), "dataset_version_id": data["version_id"], "dataset_hash": data["public_hash"], "split": "calibration" if calibration else "holdout", "max_cost_usd": "20", "detector": None if calibration else env["request"].detector.model_dump(mode="json"), "calibration_policy": 标定策略 if calibration else None, "evaluation_goal": "qualification" if certificate else "diagnostic", "calibration_use": { "policy": 标定策略, "references": [ { "experiment_id": certificate["result"]["experiment_id"], "receipt_hash": certificate["receipt_hash"], } ], } if certificate else None, "effect_policy": None if calibration else "writer-effect-v1", } ) exp = ( env["app"] .要求评测() .创建实验(env["actor"], "effect-calibration" if calibration else "effect-holdout", req) ) return {**env, "exp": exp, "request": req} def _完成(env): 运行测试._启动执行(env) 运行测试._运行就绪(env) 文学测试._推进(env) 运行测试._运行就绪(env) 文学测试._推进(env) 运行测试._运行就绪(env) return env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"]) @pytest.mark.case_id( "TC-64bcf0ac927f", environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益", given="本例固定样本、场景与独立异常,不共享其他用例的执行结果", when="经真实生成、独立检测、比较与公开效果入口读回", then=[ "真实检测高严重度交付导致A阶段failed", "目标每例高严重度计数1及target_semantic_failure留存,不封存合格凭据", ], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_真实检测失败进入A阶段并阻止效果签章__25f401(执行环境): env = _新实验(执行环境) env["fixture_control"]["high"] = True result = _完成(env) assert result["assessment"]["gate_a"] == { "status": "failed", "reasons": ["target_semantic_failure"], } assert result["receipt_id"] is None report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"]) assert all( s["detections"]["treatment"]["report"]["high_severity_count"] == 1 for s in report["samples"] ) with pytest.raises(评测错误, match="效果未通过"): env["app"].要求评测().封存效果判据(env["actor"], env["exp"]["experiment_id"]) assert len(env["received"]) == 30 @pytest.mark.case_id( "TC-a089d73880d8", environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益", given="本例固定样本、场景与独立异常,不共享其他用例的执行结果", when="经真实生成、独立检测、比较与公开效果入口读回", then=["一例真实生成失败时A阶段为failed", "首要原因system_failure先于样本和场景不足"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_实际系统失败优先于五场景不足__25f402(执行环境): env = _新实验(执行环境, 1) env["scripted"].append("bad_output") 运行测试._启动执行(env) 运行测试._运行就绪(env, allow_failure=True) result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"]) assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]} @pytest.mark.case_id( "TC-e61898aba4e5", environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益", given="本例固定样本、场景与独立异常,不共享其他用例的执行结果", when="经真实生成、独立检测、比较与公开效果入口读回", then=[ "单作品五场景A阶段仍passed", "single_work单独标注且作品样本数量为唯一5例", "完整五场景不误报scenario_coverage_incomplete", ], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_单作品完整A阶段仍保留选择偏差__25f403(执行环境): env = _新实验(执行环境) result = _完成(env) assert result["assessment"]["gate_a"]["status"] == "passed" assert "single_work" in result["assessment"]["confounders"] assert list(result["assessment"]["metrics"]["work_counts"].values()) == [5] assert "scenario_coverage_incomplete" not in result["assessment"]["confounders"] assert result["assessment"]["gate_b"]["status"] == "insufficient_evidence" @pytest.mark.case_id( "NC-w25-25f405", environment="隔离PG与合成HTTP", given="固定分母、独立样本及本例边界输入", when="经实际S02生成后重复读回效果报告", then=["单次读回每份已核验交付只验一次;下一次读回全部重新核验"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_单次读取复用已核验交付而新读取仍重验__25f405(执行环境, monkeypatch): env = _新实验(执行环境) _完成(env) runtime = env["app"].任务运行 original = runtime.核对历史结构化交付于 calls = Counter() def count(*args, **kwargs): calls[args[4]] += 1 return original(*args, **kwargs) monkeypatch.setattr(runtime, "核对历史结构化交付于", count) first = 文学测试._读(env) assert len(calls) == 30 and set(calls.values()) == {1} calls.clear() assert 文学测试._读(env) == first assert len(calls) == 30 and set(calls.values()) == {1} @pytest.mark.case_id( "TC-602d0bb37aa2", environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益", given="本例固定样本、场景与独立异常,不共享其他用例的执行结果", when="经真实生成、独立检测、比较与公开效果入口读回", then=["公开入口只接实验ID;手工汇总对象以UUID合同拒绝,无模型调用"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_公开效果入口拒绝手工汇总对象__25f407(执行环境): env = 执行环境 with pytest.raises(评测错误, match="UUID"): env["app"].要求评测().读取效果判据( env["actor"], {"gate": "A", "samples": [], "passed": True} ) assert not env["received"]