实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
512 lines
22 KiB
Python
512 lines
22 KiB
Python
"""真实S02引文纠正、失败回执和恢复次数;合成HTTP不认证文学质量。"""
|
||
|
||
import json
|
||
from decimal import Decimal
|
||
|
||
import pytest
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
|
||
from muse.效果评测.接口 import 评测错误
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
|
||
|
||
def _准备评委(env):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _工作面(env):
|
||
return env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _报告(env):
|
||
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-70f0ffe8cb5f",
|
||
environment="隔离PG与合成HTTP;不认证真实外部模型效果",
|
||
given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP",
|
||
when="注入原文中不存在的引文并按原固定上限纠正",
|
||
then=[
|
||
"实际两次评委回合;首轮无纠正信息,次轮原字材料不变且带本评委原始草稿、固定错误和原字提示。旧提示ID改为left/right,实际回执和费用另查S02。"
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_坏引文只纠正一次且保留原稿错误及原字提示__70f0ff(执行环境):
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
运行测试._运行就绪(env)
|
||
requests = [json.loads(r["input"]) for r in env["received"][2:]]
|
||
assert len(requests) == 2 and "correction" not in requests[0]
|
||
correction = requests[1].pop("correction")
|
||
assert requests[1] == requests[0]
|
||
assert (
|
||
correction["previousDraft"]["dimensions"][0]["evidence"][0]["quote"]
|
||
== "候选正文中不存在的引文"
|
||
)
|
||
assert "引文未在对应候选中出现" in correction["error"]
|
||
assert correction["validQuoteHints"] == {
|
||
side: [requests[0][side]["text"]] for side in ("left", "right")
|
||
}
|
||
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert judge["state"] == "completed" and len(judge["rejections"]) == 1
|
||
assert len(judge["cost"]["calls"]) == 2
|
||
assert _报告(env)["coverage"]["compared"] == 1
|
||
assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".5")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-883b95d3365f",
|
||
environment="隔离PG与合成HTTP;不认证真实外部模型效果",
|
||
given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP",
|
||
when="注入原文中不存在的引文并按原固定上限纠正",
|
||
then=[
|
||
"一次纠正耗尽后没有第三次调用,状态失败和完整样本分母保留;PAIRWISE_QUOTE_NOT_FOUND承接旧错误码,摘要保留两侧/维度计数与实际回执,不回显错误引文。"
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_一次纠正耗尽后保留失败诊断且不发第三次__883b95(执行环境):
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "bad_quote", "ok"])
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
work = _工作面(env)
|
||
judge = next(u for u in work["units"] if u["kind"] == "comparison")
|
||
assert judge["state"] == "failed" and judge["output"] is None
|
||
assert len(env["received"]) == 4 and env["scripted"] == ["ok"]
|
||
assert len(judge["rejections"]) == 2
|
||
assert all(r["failure_code"] == "PAIRWISE_QUOTE_NOT_FOUND" for r in judge["rejections"])
|
||
assert "候选正文中不存在的引文" not in json.dumps(judge["rejections"], ensure_ascii=False)
|
||
report = _报告(env)
|
||
assert report["outcomes"]["incomplete"] == 1
|
||
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 0}
|
||
assert Decimal(report["cost"]["total_usd"]) == Decimal(".5")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-5edc5e65298b",
|
||
environment="隔离PG与合成HTTP;不认证真实外部模型效果",
|
||
given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP",
|
||
when="注入原文中不存在的引文并按原固定上限纠正",
|
||
then=[
|
||
"两名反向独立评委各有预注册纠正上限;首位两次纠正后完成,次位首轮不含前一位草稿,总计四次评委调用。"
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{"judges": 2, "corrections": 2, "calls": 12}], indirect=True)
|
||
def test_每名评委独立最多两次纠正且不共享前一报告__5edc5e(执行环境):
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "bad_quote", "ok", "ok"])
|
||
运行测试._运行就绪(env)
|
||
requests = [json.loads(r["input"]) for r in env["received"][2:]]
|
||
assert len(requests) == 4
|
||
assert ["correction" in r for r in requests] == [False, True, True, False]
|
||
assert requests[0]["left"] == requests[3]["right"]
|
||
judges = [u for u in _工作面(env)["units"] if u["kind"] == "comparison"]
|
||
assert len(judges) == 2 and all(u["state"] == "completed" for u in judges)
|
||
assert sorted(len(u["rejections"]) for u in judges) == [0, 2]
|
||
assert _报告(env)["outcomes"]["tie"] == 1
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b001",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["零纠正或原调用配额不足不多发一次"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize(
|
||
"执行环境", [{"corrections": 0}, {"corrections": 2, "calls": 3}], indirect=True
|
||
)
|
||
def test_零纠正或调用配额不足不新增一次调用__25b001(执行环境):
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert len(judge["rejections"]) == 1 and judge["state"] == "failed"
|
||
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b002",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["拒绝提交后中断恢复只用余下次数且保原尝试"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_拒绝提交后中断恢复只继续余下纠正且保留原尝试__25b002(执行环境, monkeypatch):
|
||
from muse.共享.错误 import Muse错误
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
save = type(svc).保存执行交付
|
||
|
||
def 中断(self, *args, **kwargs):
|
||
result = save(self, *args, **kwargs)
|
||
if result.get("outcome") == "comparison_rejected":
|
||
raise Muse错误("合成拒绝提交后中断")
|
||
return result
|
||
|
||
monkeypatch.setattr(type(svc), "保存执行交付", 中断)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
before = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert len(env["received"]) == 3 and before["state"] == "failed"
|
||
monkeypatch.setattr(type(svc), "保存执行交付", save)
|
||
运行测试._恢复失败单元(env)
|
||
运行测试._运行就绪(env)
|
||
after = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert after["state"] == "completed" and after["rejections"] == before["rejections"]
|
||
assert after["evidence"]["runtime"]["attempt_id"] != before["rejections"][0]["attempt_id"]
|
||
assert len(env["received"]) == 4 and "correction" in json.loads(env["received"][-1]["input"])
|
||
assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".5")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b003",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["耗尽后恢复被拒,UPDATE/DELETE不能重置次数"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_纠正耗尽不能经恢复或删除拒绝重置次数__25b003(执行环境):
|
||
import psycopg
|
||
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "bad_quote", "ok"])
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
before = _工作面(env)
|
||
with pytest.raises(评测错误, match="上限"):
|
||
运行测试._恢复失败单元(env)
|
||
for statement in [
|
||
"DELETE FROM evaluation.muse_comparison_rejection",
|
||
"UPDATE evaluation.muse_comparison_rejection SET sequence=1",
|
||
]:
|
||
with env["pool"].连接() as conn, pytest.raises(psycopg.Error), conn.transaction():
|
||
conn.execute(statement)
|
||
assert _工作面(env) == before
|
||
assert len(env["received"]) == 4 and env["scripted"] == ["ok"]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b004",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["重签拒绝内容或前序顺序仍过不了S02原回合核验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("changed", ["raw_output", "previous_hash", "sequence"])
|
||
def test_拒绝记录篡改重签仍不能冒原回合或重置链__25b004(执行环境, monkeypatch, changed):
|
||
from copy import deepcopy
|
||
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
运行测试._运行就绪(env)
|
||
read = 评测存储.读取引文拒绝
|
||
|
||
def 篡改(self, unit_id):
|
||
rows = deepcopy(read(self, unit_id))
|
||
if rows:
|
||
row = rows[0]
|
||
if changed == "raw_output":
|
||
row["output"]["rationale"] = "不是原调用返回的文字"
|
||
elif changed == "previous_hash":
|
||
row["previous_hash"] = "f" * 64
|
||
else:
|
||
row["sequence"] = 2
|
||
row["record_hash"] = 固定哈希({k: v for k, v in row.items() if k != "record_hash"})
|
||
return rows
|
||
|
||
monkeypatch.setattr(评测存储, "读取引文拒绝", 篡改)
|
||
with pytest.raises(Muse错误):
|
||
_报告(env)
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b005",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["拒绝后停止不发纠正,保留拒绝和全部费用"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_拒绝后停止实验不派发纠正且保留拒绝费用__25b005(执行环境, monkeypatch):
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
save = type(svc).保存执行交付
|
||
|
||
def 停止(self, *args, **kwargs):
|
||
result = save(self, *args, **kwargs)
|
||
if result.get("outcome") == "comparison_rejected":
|
||
svc.取消实验(env["actor"], env["exp"]["experiment_id"], "stop-after-rejection")
|
||
return result
|
||
|
||
monkeypatch.setattr(type(svc), "保存执行交付", 停止)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
|
||
work = _工作面(env)
|
||
assert work["state"] == "stopped"
|
||
judge = next(u for u in work["units"] if u["kind"] == "comparison")
|
||
assert len(judge["rejections"]) == 1 and judge["output"] is None
|
||
assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".375")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b006",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["协议、非引文语义错误及未知成本不自动纠正"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("script", ["bad_output", "bad_mixed", "unknown_cost"])
|
||
def test_协议其他语义错误及未知费用不自动纠正__25b006(执行环境, script):
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend([script, "ok"])
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
|
||
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert judge["output"] is None
|
||
if script == "bad_mixed":
|
||
assert [r["failure_code"] for r in judge["rejections"]] == ["PAIRWISE_NOT_EXECUTED"]
|
||
assert judge["unregistered_deliveries"] == []
|
||
else:
|
||
assert judge["rejections"] == []
|
||
if script == "unknown_cost":
|
||
assert _报告(env)["cost"]["total_usd"] is None
|
||
assert _工作面(env)["state"] == "reconciling"
|
||
else:
|
||
assert judge["state"] == "failed"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b007",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["纠正前稿偏离原回合在发送前拒绝"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_纠正的上次草稿被改写在发送前拒绝__25b007(执行环境, monkeypatch):
|
||
from dataclasses import replace
|
||
|
||
import muse.编排.执行评测 as flow
|
||
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
create = flow.模型请求
|
||
|
||
def 改写(*args, **kwargs):
|
||
request = create(*args, **kwargs)
|
||
content = json.loads(request.用户输入)
|
||
if "correction" in content:
|
||
content["correction"]["previousDraft"]["rationale"] = "其他回合的报告"
|
||
return replace(
|
||
request, 用户输入=json.dumps(content, ensure_ascii=False, sort_keys=True)
|
||
)
|
||
return request
|
||
|
||
monkeypatch.setattr(flow, "模型请求", 改写)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
|
||
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert judge["state"] == "failed" and len(judge["rejections"]) == 1
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b008",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["S02已保存但B10事务失败时,不得绕过未登记回合重发"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_拒绝事务失败不能直接恢复重置次数__25b008(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
_准备评委(env)
|
||
env["scripted"].extend(["bad_quote", "ok"])
|
||
save = 评测存储.保存引文拒绝
|
||
|
||
def 写入失败(self, record):
|
||
raise 评测错误("合成拒绝事务失败")
|
||
|
||
monkeypatch.setattr(评测存储, "保存引文拒绝", 写入失败)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
monkeypatch.setattr(评测存储, "保存引文拒绝", save)
|
||
with pytest.raises(评测错误, match="未登记"):
|
||
运行测试._恢复失败单元(env)
|
||
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
|
||
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
|
||
assert len(judge["unregistered_deliveries"]) == 1 and not judge["rejections"]
|
||
|
||
|
||
def _漏登记(env, monkeypatch, kind):
|
||
from copy import deepcopy
|
||
|
||
if kind == "generation":
|
||
运行测试._启动执行(env)
|
||
else:
|
||
_准备评委(env)
|
||
env["scripted"].append("bad_quote" if kind == "comparison_bad" else "ok")
|
||
service = type(env["app"].要求评测())
|
||
save = service.保存执行交付
|
||
captured = []
|
||
|
||
def 中断(self, ctx, text, output, call):
|
||
if not captured:
|
||
captured.append(
|
||
{
|
||
"unit_id": ctx.任务.冻结输入["输入"]["unit_id"],
|
||
"call_id": call,
|
||
"output": deepcopy(output),
|
||
}
|
||
)
|
||
raise 评测错误("合成业务登记前中断")
|
||
return save(self, ctx, text, output, call)
|
||
|
||
monkeypatch.setattr(service, "保存执行交付", 中断)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
monkeypatch.setattr(service, "保存执行交付", save)
|
||
assert len(captured) == 1
|
||
return captured[0]
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b009",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["生成和评委原输出经S02复核后补登,错作者/伪造拒绝、原样幂等、恢复不重发旧回合"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("kind", ["generation", "comparison_valid", "comparison_bad"])
|
||
def test_按S02原交付补登后恢复不重发已发生回合__25b009(执行环境, monkeypatch, kind):
|
||
from copy import deepcopy
|
||
from dataclasses import replace
|
||
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.效果评测.接口 import 交付补登请求
|
||
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
original = _漏登记(env, monkeypatch, kind)
|
||
count = len(env["received"])
|
||
with pytest.raises(评测错误, match="未登记"):
|
||
运行测试._恢复失败单元(env)
|
||
request = 交付补登请求(**original)
|
||
with pytest.raises(评测错误, match="作者"):
|
||
svc.补登记交付(replace(env["actor"], 作者="foreign"), request)
|
||
forged = deepcopy(original)
|
||
forged["output"]["fabricated"] = True
|
||
with pytest.raises(Muse错误):
|
||
svc.补登记交付(env["actor"], 交付补登请求(**forged))
|
||
receipt = svc.补登记交付(env["actor"], request)
|
||
assert svc.补登记交付(env["actor"], request) == receipt
|
||
assert len(env["received"]) == count
|
||
unit = next(u for u in _工作面(env)["units"] if u["unit_id"] == original["unit_id"])
|
||
assert unit["unregistered_deliveries"] == []
|
||
assert bool(unit["rejections"]) == (kind == "comparison_bad")
|
||
运行测试._恢复失败单元(env)
|
||
运行测试._运行就绪(env)
|
||
assert len(env["received"]) == count + (kind == "comparison_bad")
|
||
unit = next(u for u in _工作面(env)["units"] if u["unit_id"] == original["unit_id"])
|
||
assert unit["state"] == "completed"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b00a",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["实际CLI补登与HTTP重放同一回执,零新增模型调用"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_实际CLI补登与HTTP重放不调用模型__25b00a(执行环境, monkeypatch, tmp_path):
|
||
import subprocess
|
||
import sys
|
||
|
||
import test_生产评测权限隔离 as 数据测试
|
||
from fastapi.testclient import TestClient
|
||
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.配置 import 读取配置
|
||
|
||
env = 执行环境
|
||
original = _漏登记(env, monkeypatch, "comparison_bad")
|
||
config = 数据测试._配置文件(env["pool"], tmp_path)
|
||
request_file = tmp_path / "restore.json"
|
||
request_file.write_text(json.dumps(original, ensure_ascii=False))
|
||
result = subprocess.run(
|
||
[sys.executable, "-I", "-m", "muse", "评测", str(config), "补登记交付", str(request_file)],
|
||
cwd=tmp_path,
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=30,
|
||
)
|
||
assert result.returncode == 0, result.stderr
|
||
receipt = json.loads(result.stdout)
|
||
http = 创建应用(读取配置(config))
|
||
with TestClient(http, headers={"origin": "http://testserver"}) as client:
|
||
http.state.装配 = env["app"]
|
||
assert (
|
||
client.post(
|
||
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
|
||
).status_code
|
||
== 200
|
||
)
|
||
replay = client.post("/api/v1/evaluation/deliveries/restore", json=original)
|
||
assert replay.status_code == 200 and replay.json() == receipt
|
||
report = client.get(
|
||
"/api/v1/evaluation/experiments/" + env["exp"]["experiment_id"] + "/report"
|
||
)
|
||
assert report.status_code == 200
|
||
units = report.json()["samples"][0]["units"]
|
||
assert sum(len(u["rejections"]) for u in units) == 1
|
||
assert len(env["received"]) == 3
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25b00b",
|
||
environment="隔离PG与受控合成HTTP",
|
||
given="隔离PG、固定任务与实际模型回合",
|
||
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
|
||
then=["停止后的未登记交付不能补写为新业务结果"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_实验停止后拒绝补登新产物__25b00b(执行环境, monkeypatch):
|
||
from muse.效果评测.接口 import 交付补登请求
|
||
|
||
env = 执行环境
|
||
original = _漏登记(env, monkeypatch, "comparison_valid")
|
||
svc = env["app"].要求评测()
|
||
svc.取消实验(env["actor"], env["exp"]["experiment_id"], "stop-before-restore")
|
||
with pytest.raises(评测错误, match="停止"):
|
||
svc.补登记交付(env["actor"], 交付补登请求(**original))
|
||
assert len(env["received"]) == 3
|
||
assert sum(len(u["unregistered_deliveries"]) for u in _工作面(env)["units"]) == 1
|