"""真实S02引文纠正、失败回执和恢复次数;合成HTTP不认证文学质量。""" import json from decimal import Decimal import pytest import test_评测执行与失败收敛 as 运行测试 from muse.效果评测.接口 import 评测错误 pytestmark = pytest.mark.数据库 执行环境 = 运行测试.执行环境 def _准备评委(env): 运行测试._启动执行(env) 运行测试._运行就绪(env) env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"]) def _工作面(env): return env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"]) def _报告(env): return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"]) @pytest.mark.case_id( "TC-70f0ffe8cb5f", environment="隔离PG与合成HTTP;不认证真实外部模型效果", given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP", when="注入原文中不存在的引文并按原固定上限纠正", then=[ "实际两次评委回合;首轮无纠正信息,次轮原字材料不变且带本评委原始草稿、固定错误和原字提示。旧提示ID改为left/right,实际回执和费用另查S02。" ], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_坏引文只纠正一次且保留原稿错误及原字提示__70f0ff(执行环境): env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) 运行测试._运行就绪(env) requests = [json.loads(r["input"]) for r in env["received"][2:]] assert len(requests) == 2 and "correction" not in requests[0] correction = requests[1].pop("correction") assert requests[1] == requests[0] assert ( correction["previousDraft"]["dimensions"][0]["evidence"][0]["quote"] == "候选正文中不存在的引文" ) assert "引文未在对应候选中出现" in correction["error"] assert correction["validQuoteHints"] == { side: [requests[0][side]["text"]] for side in ("left", "right") } judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert judge["state"] == "completed" and len(judge["rejections"]) == 1 assert len(judge["cost"]["calls"]) == 2 assert _报告(env)["coverage"]["compared"] == 1 assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".5") @pytest.mark.case_id( "TC-883b95d3365f", environment="隔离PG与合成HTTP;不认证真实外部模型效果", given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP", when="注入原文中不存在的引文并按原固定上限纠正", then=[ "一次纠正耗尽后没有第三次调用,状态失败和完整样本分母保留;PAIRWISE_QUOTE_NOT_FOUND承接旧错误码,摘要保留两侧/维度计数与实际回执,不回显错误引文。" ], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_一次纠正耗尽后保留失败诊断且不发第三次__883b95(执行环境): env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "bad_quote", "ok"]) 运行测试._运行就绪(env, allow_failure=True) work = _工作面(env) judge = next(u for u in work["units"] if u["kind"] == "comparison") assert judge["state"] == "failed" and judge["output"] is None assert len(env["received"]) == 4 and env["scripted"] == ["ok"] assert len(judge["rejections"]) == 2 assert all(r["failure_code"] == "PAIRWISE_QUOTE_NOT_FOUND" for r in judge["rejections"]) assert "候选正文中不存在的引文" not in json.dumps(judge["rejections"], ensure_ascii=False) report = _报告(env) assert report["outcomes"]["incomplete"] == 1 assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 0} assert Decimal(report["cost"]["total_usd"]) == Decimal(".5") @pytest.mark.case_id( "TC-5edc5e65298b", environment="隔离PG与合成HTTP;不认证真实外部模型效果", given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP", when="注入原文中不存在的引文并按原固定上限纠正", then=[ "两名反向独立评委各有预注册纠正上限;首位两次纠正后完成,次位首轮不含前一位草稿,总计四次评委调用。" ], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [{"judges": 2, "corrections": 2, "calls": 12}], indirect=True) def test_每名评委独立最多两次纠正且不共享前一报告__5edc5e(执行环境): env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "bad_quote", "ok", "ok"]) 运行测试._运行就绪(env) requests = [json.loads(r["input"]) for r in env["received"][2:]] assert len(requests) == 4 assert ["correction" in r for r in requests] == [False, True, True, False] assert requests[0]["left"] == requests[3]["right"] judges = [u for u in _工作面(env)["units"] if u["kind"] == "comparison"] assert len(judges) == 2 and all(u["state"] == "completed" for u in judges) assert sorted(len(u["rejections"]) for u in judges) == [0, 2] assert _报告(env)["outcomes"]["tie"] == 1 @pytest.mark.case_id( "NC-w25-25b001", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["零纠正或原调用配额不足不多发一次"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize( "执行环境", [{"corrections": 0}, {"corrections": 2, "calls": 3}], indirect=True ) def test_零纠正或调用配额不足不新增一次调用__25b001(执行环境): env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) 运行测试._运行就绪(env, allow_failure=True) judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert len(judge["rejections"]) == 1 and judge["state"] == "failed" assert len(env["received"]) == 3 and env["scripted"] == ["ok"] @pytest.mark.case_id( "NC-w25-25b002", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["拒绝提交后中断恢复只用余下次数且保原尝试"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_拒绝提交后中断恢复只继续余下纠正且保留原尝试__25b002(执行环境, monkeypatch): from muse.共享.错误 import Muse错误 env = 执行环境 svc = env["app"].要求评测() _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) save = type(svc).保存执行交付 def 中断(self, *args, **kwargs): result = save(self, *args, **kwargs) if result.get("outcome") == "comparison_rejected": raise Muse错误("合成拒绝提交后中断") return result monkeypatch.setattr(type(svc), "保存执行交付", 中断) 运行测试._运行就绪(env, allow_failure=True) before = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert len(env["received"]) == 3 and before["state"] == "failed" monkeypatch.setattr(type(svc), "保存执行交付", save) 运行测试._恢复失败单元(env) 运行测试._运行就绪(env) after = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert after["state"] == "completed" and after["rejections"] == before["rejections"] assert after["evidence"]["runtime"]["attempt_id"] != before["rejections"][0]["attempt_id"] assert len(env["received"]) == 4 and "correction" in json.loads(env["received"][-1]["input"]) assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".5") @pytest.mark.case_id( "NC-w25-25b003", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["耗尽后恢复被拒,UPDATE/DELETE不能重置次数"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_纠正耗尽不能经恢复或删除拒绝重置次数__25b003(执行环境): import psycopg env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "bad_quote", "ok"]) 运行测试._运行就绪(env, allow_failure=True) before = _工作面(env) with pytest.raises(评测错误, match="上限"): 运行测试._恢复失败单元(env) for statement in [ "DELETE FROM evaluation.muse_comparison_rejection", "UPDATE evaluation.muse_comparison_rejection SET sequence=1", ]: with env["pool"].连接() as conn, pytest.raises(psycopg.Error), conn.transaction(): conn.execute(statement) assert _工作面(env) == before assert len(env["received"]) == 4 and env["scripted"] == ["ok"] @pytest.mark.case_id( "NC-w25-25b004", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["重签拒绝内容或前序顺序仍过不了S02原回合核验"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("changed", ["raw_output", "previous_hash", "sequence"]) def test_拒绝记录篡改重签仍不能冒原回合或重置链__25b004(执行环境, monkeypatch, changed): from copy import deepcopy from muse.共享.错误 import Muse错误 from muse.效果评测.存储 import 评测存储 from muse.正式变更.接口 import 固定哈希 env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) 运行测试._运行就绪(env) read = 评测存储.读取引文拒绝 def 篡改(self, unit_id): rows = deepcopy(read(self, unit_id)) if rows: row = rows[0] if changed == "raw_output": row["output"]["rationale"] = "不是原调用返回的文字" elif changed == "previous_hash": row["previous_hash"] = "f" * 64 else: row["sequence"] = 2 row["record_hash"] = 固定哈希({k: v for k, v in row.items() if k != "record_hash"}) return rows monkeypatch.setattr(评测存储, "读取引文拒绝", 篡改) with pytest.raises(Muse错误): _报告(env) assert len(env["received"]) == 4 @pytest.mark.case_id( "NC-w25-25b005", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["拒绝后停止不发纠正,保留拒绝和全部费用"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_拒绝后停止实验不派发纠正且保留拒绝费用__25b005(执行环境, monkeypatch): env = 执行环境 svc = env["app"].要求评测() _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) save = type(svc).保存执行交付 def 停止(self, *args, **kwargs): result = save(self, *args, **kwargs) if result.get("outcome") == "comparison_rejected": svc.取消实验(env["actor"], env["exp"]["experiment_id"], "stop-after-rejection") return result monkeypatch.setattr(type(svc), "保存执行交付", 停止) 运行测试._运行就绪(env, allow_failure=True) assert len(env["received"]) == 3 and env["scripted"] == ["ok"] work = _工作面(env) assert work["state"] == "stopped" judge = next(u for u in work["units"] if u["kind"] == "comparison") assert len(judge["rejections"]) == 1 and judge["output"] is None assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".375") @pytest.mark.case_id( "NC-w25-25b006", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["协议、非引文语义错误及未知成本不自动纠正"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("script", ["bad_output", "bad_mixed", "unknown_cost"]) def test_协议其他语义错误及未知费用不自动纠正__25b006(执行环境, script): env = 执行环境 _准备评委(env) env["scripted"].extend([script, "ok"]) 运行测试._运行就绪(env, allow_failure=True) assert len(env["received"]) == 3 and env["scripted"] == ["ok"] judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert judge["output"] is None if script == "bad_mixed": assert [r["failure_code"] for r in judge["rejections"]] == ["PAIRWISE_NOT_EXECUTED"] assert judge["unregistered_deliveries"] == [] else: assert judge["rejections"] == [] if script == "unknown_cost": assert _报告(env)["cost"]["total_usd"] is None assert _工作面(env)["state"] == "reconciling" else: assert judge["state"] == "failed" @pytest.mark.case_id( "NC-w25-25b007", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["纠正前稿偏离原回合在发送前拒绝"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_纠正的上次草稿被改写在发送前拒绝__25b007(执行环境, monkeypatch): from dataclasses import replace import muse.编排.执行评测 as flow env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) create = flow.模型请求 def 改写(*args, **kwargs): request = create(*args, **kwargs) content = json.loads(request.用户输入) if "correction" in content: content["correction"]["previousDraft"]["rationale"] = "其他回合的报告" return replace( request, 用户输入=json.dumps(content, ensure_ascii=False, sort_keys=True) ) return request monkeypatch.setattr(flow, "模型请求", 改写) 运行测试._运行就绪(env, allow_failure=True) assert len(env["received"]) == 3 and env["scripted"] == ["ok"] judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert judge["state"] == "failed" and len(judge["rejections"]) == 1 @pytest.mark.case_id( "NC-w25-25b008", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["S02已保存但B10事务失败时,不得绕过未登记回合重发"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_拒绝事务失败不能直接恢复重置次数__25b008(执行环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 执行环境 _准备评委(env) env["scripted"].extend(["bad_quote", "ok"]) save = 评测存储.保存引文拒绝 def 写入失败(self, record): raise 评测错误("合成拒绝事务失败") monkeypatch.setattr(评测存储, "保存引文拒绝", 写入失败) 运行测试._运行就绪(env, allow_failure=True) monkeypatch.setattr(评测存储, "保存引文拒绝", save) with pytest.raises(评测错误, match="未登记"): 运行测试._恢复失败单元(env) assert len(env["received"]) == 3 and env["scripted"] == ["ok"] judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison") assert len(judge["unregistered_deliveries"]) == 1 and not judge["rejections"] def _漏登记(env, monkeypatch, kind): from copy import deepcopy if kind == "generation": 运行测试._启动执行(env) else: _准备评委(env) env["scripted"].append("bad_quote" if kind == "comparison_bad" else "ok") service = type(env["app"].要求评测()) save = service.保存执行交付 captured = [] def 中断(self, ctx, text, output, call): if not captured: captured.append( { "unit_id": ctx.任务.冻结输入["输入"]["unit_id"], "call_id": call, "output": deepcopy(output), } ) raise 评测错误("合成业务登记前中断") return save(self, ctx, text, output, call) monkeypatch.setattr(service, "保存执行交付", 中断) 运行测试._运行就绪(env, allow_failure=True) monkeypatch.setattr(service, "保存执行交付", save) assert len(captured) == 1 return captured[0] @pytest.mark.case_id( "NC-w25-25b009", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["生成和评委原输出经S02复核后补登,错作者/伪造拒绝、原样幂等、恢复不重发旧回合"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("kind", ["generation", "comparison_valid", "comparison_bad"]) def test_按S02原交付补登后恢复不重发已发生回合__25b009(执行环境, monkeypatch, kind): from copy import deepcopy from dataclasses import replace from muse.共享.错误 import Muse错误 from muse.效果评测.接口 import 交付补登请求 env = 执行环境 svc = env["app"].要求评测() original = _漏登记(env, monkeypatch, kind) count = len(env["received"]) with pytest.raises(评测错误, match="未登记"): 运行测试._恢复失败单元(env) request = 交付补登请求(**original) with pytest.raises(评测错误, match="作者"): svc.补登记交付(replace(env["actor"], 作者="foreign"), request) forged = deepcopy(original) forged["output"]["fabricated"] = True with pytest.raises(Muse错误): svc.补登记交付(env["actor"], 交付补登请求(**forged)) receipt = svc.补登记交付(env["actor"], request) assert svc.补登记交付(env["actor"], request) == receipt assert len(env["received"]) == count unit = next(u for u in _工作面(env)["units"] if u["unit_id"] == original["unit_id"]) assert unit["unregistered_deliveries"] == [] assert bool(unit["rejections"]) == (kind == "comparison_bad") 运行测试._恢复失败单元(env) 运行测试._运行就绪(env) assert len(env["received"]) == count + (kind == "comparison_bad") unit = next(u for u in _工作面(env)["units"] if u["unit_id"] == original["unit_id"]) assert unit["state"] == "completed" @pytest.mark.case_id( "NC-w25-25b00a", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["实际CLI补登与HTTP重放同一回执,零新增模型调用"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_实际CLI补登与HTTP重放不调用模型__25b00a(执行环境, monkeypatch, tmp_path): import subprocess import sys import test_生产评测权限隔离 as 数据测试 from fastapi.testclient import TestClient from muse.接入.http.应用 import 创建应用 from muse.配置 import 读取配置 env = 执行环境 original = _漏登记(env, monkeypatch, "comparison_bad") config = 数据测试._配置文件(env["pool"], tmp_path) request_file = tmp_path / "restore.json" request_file.write_text(json.dumps(original, ensure_ascii=False)) result = subprocess.run( [sys.executable, "-I", "-m", "muse", "评测", str(config), "补登记交付", str(request_file)], cwd=tmp_path, capture_output=True, text=True, timeout=30, ) assert result.returncode == 0, result.stderr receipt = json.loads(result.stdout) http = 创建应用(读取配置(config)) with TestClient(http, headers={"origin": "http://testserver"}) as client: http.state.装配 = env["app"] assert ( client.post( "/api/v1/session", json={"password": "synthetic-evaluation-only"} ).status_code == 200 ) replay = client.post("/api/v1/evaluation/deliveries/restore", json=original) assert replay.status_code == 200 and replay.json() == receipt report = client.get( "/api/v1/evaluation/experiments/" + env["exp"]["experiment_id"] + "/report" ) assert report.status_code == 200 units = report.json()["samples"][0]["units"] assert sum(len(u["rejections"]) for u in units) == 1 assert len(env["received"]) == 3 @pytest.mark.case_id( "NC-w25-25b00b", environment="隔离PG与受控合成HTTP", given="隔离PG、固定任务与实际模型回合", when="触发本用例的纠正、登记失败、重签、恢复或停止边界", then=["停止后的未登记交付不能补写为新业务结果"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_实验停止后拒绝补登新产物__25b00b(执行环境, monkeypatch): from muse.效果评测.接口 import 交付补登请求 env = 执行环境 original = _漏登记(env, monkeypatch, "comparison_valid") svc = env["app"].要求评测() svc.取消实验(env["actor"], env["exp"]["experiment_id"], "stop-before-restore") with pytest.raises(评测错误, match="停止"): svc.补登记交付(env["actor"], 交付补登请求(**original)) assert len(env["received"]) == 3 assert sum(len(u["unregistered_deliveries"]) for u in _工作面(env)["units"]) == 1