"""隔离S02实际检测回合、拒绝及恢复;合成模型不认证语义准确率。""" import copy import json import pytest import test_文学评分执行与条件第三 as 文学测试 import test_评测执行与失败收敛 as 运行测试 from muse.效果评测.接口 import 交付补登请求, 评测错误 from muse.正式变更.接口 import 固定哈希 pytestmark = pytest.mark.数据库 执行环境 = 运行测试.执行环境 def _输出(material, script): result = { "claims": [ { "claim_id": "observed", "text": "合成角色行动。", "candidate_quote": material["candidate"], "state": "supported", "evidence_refs": ["source:outline"], "reason": "根据共同细纲核对。", } ], "findings": [], "new_setting_candidates": [], } for field, source, prefix in ( ("assertion_verdicts", "required_assertions", "assertion"), ("constraint_verdicts", "required_constraints", "constraint"), ): result[field] = [ { "statement_id": sid, "verdict": "unknown" if script == "unknown" else "pass", "candidate_quote": material["candidate"], "evidence_refs": [prefix + ":" + sid], "reason": "合成判断;未知表示材料不足。", } for sid in material[source] ] if script == "high": result["findings"] = [ { "finding_id": "conflict", "category": "fact", "severity": "high", "candidate_quote": material["candidate"], "evidence_refs": ["assertion:guard"], "message": "此合成候选与已给事实冲突。", } ] if script == "missing": result["constraint_verdicts"] = [] return result 参数 = {**文学测试.参数, "detector": True, "detector_output": _输出, "calls": 7} def _执行(env, script="ok", *, allow_failure=False): 运行测试._启动执行(env) 运行测试._运行就绪(env) 文学测试._推进(env) env["scripted"].extend([script, "ok"]) 运行测试._运行就绪(env, allow_failure=allow_failure) 文学测试._推进(env) return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"]) @pytest.mark.case_id( "NC-w25-25f201", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["实际detector与写手分离,每份候选及报告绑定S02,模型不见另一臂或整个答案"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_真实独立检测绑定原候选且模型不见其他臂与答案__25f201(执行环境): env = 执行环境 report = _执行(env, "high") sample = report["samples"][0] assert sample["detection_complete"] and len(sample["detections"]) == 2 assert sorted(r["report"]["status"] for r in sample["detections"].values()) == [ "failed", "passed", ] assert sum(r["report"]["high_severity_count"] for r in sample["detections"].values()) == 1 assert len(env["received"]) == report["cost"]["sent_calls"] == 6 assert report["activation_status"] == "not_evaluated" requests = [r for r in env["received"] if r["model"] == "synthetic-semantic-detector"] assert len(requests) == 2 assert {json.loads(r["input"])["candidate"] for r in requests} == {"合成正文1。", "合成正文2。"} for request in requests: material = json.loads(request["input"]) assert set(material) == { "candidate", "evidence", "required_assertions", "required_constraints", } assert "ORACLE-SECRET" not in request["input"] and "calibration" not in request["input"] assert not request.get("tools") and not request.get("previous_response_id") for row in sample["detections"].values(): assert row["report_hash"] == 固定哈希(row["report"]) assert row["task_id"] and row["call_id"] and row["candidate_output_hash"] @pytest.mark.case_id( "NC-w25-25f202", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["未知检测保留原因及未决状态,不变为已知通过"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_未知检测保留未决而不是零缺陷通过__25f202(执行环境): report = _执行(执行环境, "unknown") rows = list(report["samples"][0]["detections"].values()) uncertain = next(r for r in rows if r["report"]["status"] == "inconclusive") assert uncertain["report"]["unknown_count"] == 2 assert uncertain["report"]["constraint_counts"] == {"pass": 0, "fail": 0, "unknown": 1} assert report["literary_quality"]["metrics"] is None @pytest.mark.case_id( "NC-w25-25f203", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["缺命题失败保留原S02回合与费用,不自动重发"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_缺少命题检测失败保留原交付费用且不自动纠正__25f203(执行环境): env = 执行环境 report = _执行(env, "missing", allow_failure=True) assert not report["samples"][0]["detection_complete"] work = 文学测试._读(env) failed = [r for r in work["units"] if r["kind"] == "detection" and r["state"] == "failed"] assert len(failed) == 1 and failed[0]["unregistered_deliveries"] assert len(env["received"]) == 6 and report["cost"]["total_usd"] is not None @pytest.mark.case_id( "NC-w25-25f204", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["重签零残留派生报告不能覆盖实际高严重度交付"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_重签零残留报告不能覆盖真实高严重度交付__25f204(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 _执行(env, "high") read = 评测存储.读取交付 def changed(self, uid): row = copy.deepcopy(read(self, uid)) if row and row["evidence"].get("detection"): evidence = row["evidence"]["detection"] evidence["report"].update(findings=[], high_severity_count=0, status="passed") evidence["report_hash"] = 固定哈希(evidence["report"]) return row monkeypatch.setattr(评测存储, "读取交付", changed) with pytest.raises(评测错误, match="检测报告"): 文学测试._读(env) assert len(env["received"]) == 6 @pytest.mark.case_id( "NC-w25-25f205", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["检测登记中断可补原回执,恢复不重发"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_检测登记中断后补原回执且恢复不重发__25f205(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 save = 评测存储.保存交付 broken = [] def fail_once(self, uid, *args): if not broken and self.读取单元(uid)["kind"] == "detection": broken.append(uid) raise 评测错误("注入检测登记中断") return save(self, uid, *args) monkeypatch.setattr(评测存储, "保存交付", fail_once) _执行(env, allow_failure=True) monkeypatch.setattr(评测存储, "保存交付", save) unit = next(r for r in 文学测试._读(env)["units"] if r["unit_id"] == broken[0]) original = next(r for r in env["received"] if r["model"] == "synthetic-semantic-detector") request = 交付补登请求( unit_id=unit["unit_id"], call_id=unit["unregistered_deliveries"][0]["call_id"], output=_输出(json.loads(original["input"]), "ok"), ) svc = env["app"].要求评测() first = svc.补登记交付(env["actor"], request) assert svc.补登记交付(env["actor"], request) == first 运行测试._恢复失败单元(env) 运行测试._运行就绪(env) report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"]) assert report["samples"][0]["detection_complete"] and len(env["received"]) == 6 @pytest.mark.case_id( "NC-w25-25f206", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["检测纳入原预算,预算不足不留孤儿任务"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [{**参数, "calls": 5}], indirect=True) def test_新增检测必须纳入原预算不足时零任务__25f206(执行环境): env = 执行环境 with pytest.raises(评测错误, match="调用上限"): 运行测试._启动执行(env) assert not env["received"] with env["pool"].连接(只读=True) as conn: assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0 @pytest.mark.case_id( "NC-w25-25f207", environment="隔离PG与合成HTTP", given="明确候选、共同依据与本例异常", when="运行B06核验或隔离S02实际检测与读回", then=["生成失败时对应检测不创建,完整分母和缺失仍保留"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_生成失败时该份检测不创建且完整分母保留__25f207(执行环境): env = 执行环境 env["scripted"].append("bad_output") 运行测试._启动执行(env) 运行测试._运行就绪(env, allow_failure=True) 文学测试._推进(env) 运行测试._运行就绪(env) work = 文学测试._读(env) detectors = [u for u in work["units"] if u["kind"] == "detection"] assert len(detectors) == 2 and sum(u["task_id"] is None for u in detectors) == 1 assert len(env["received"]) == 3 report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"]) assert report["coverage"]["samples"] == 1 and not report["samples"][0]["detection_complete"]