muse-agent-example/tests/集成/test_评委纠正与回执.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

512 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实S02引文纠正、失败回执和恢复次数;合成HTTP不认证文学质量。"""
import json
from decimal import Decimal
import pytest
import test_评测执行与失败收敛 as 运行测试
from muse.效果评测.接口 import 评测错误
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
def _准备评委(env):
运行测试._启动执行(env)
运行测试._运行就绪(env)
env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
def _工作面(env):
return env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
def _报告(env):
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
@pytest.mark.case_id(
"TC-70f0ffe8cb5f",
environment="隔离PG与合成HTTP;不认证真实外部模型效果",
given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP",
when="注入原文中不存在的引文并按原固定上限纠正",
then=[
"实际两次评委回合;首轮无纠正信息,次轮原字材料不变且带本评委原始草稿、固定错误和原字提示。旧提示ID改为left/right,实际回执和费用另查S02。"
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_坏引文只纠正一次且保留原稿错误及原字提示__70f0ff(执行环境):
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
运行测试._运行就绪(env)
requests = [json.loads(r["input"]) for r in env["received"][2:]]
assert len(requests) == 2 and "correction" not in requests[0]
correction = requests[1].pop("correction")
assert requests[1] == requests[0]
assert (
correction["previousDraft"]["dimensions"][0]["evidence"][0]["quote"]
== "候选正文中不存在的引文"
)
assert "引文未在对应候选中出现" in correction["error"]
assert correction["validQuoteHints"] == {
side: [requests[0][side]["text"]] for side in ("left", "right")
}
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert judge["state"] == "completed" and len(judge["rejections"]) == 1
assert len(judge["cost"]["calls"]) == 2
assert _报告(env)["coverage"]["compared"] == 1
assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".5")
@pytest.mark.case_id(
"TC-883b95d3365f",
environment="隔离PG与合成HTTP;不认证真实外部模型效果",
given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP",
when="注入原文中不存在的引文并按原固定上限纠正",
then=[
"一次纠正耗尽后没有第三次调用,状态失败和完整样本分母保留;PAIRWISE_QUOTE_NOT_FOUND承接旧错误码,摘要保留两侧/维度计数与实际回执,不回显错误引文。"
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_一次纠正耗尽后保留失败诊断且不发第三次__883b95(执行环境):
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "bad_quote", "ok"])
运行测试._运行就绪(env, allow_failure=True)
work = _工作面(env)
judge = next(u for u in work["units"] if u["kind"] == "comparison")
assert judge["state"] == "failed" and judge["output"] is None
assert len(env["received"]) == 4 and env["scripted"] == ["ok"]
assert len(judge["rejections"]) == 2
assert all(r["failure_code"] == "PAIRWISE_QUOTE_NOT_FOUND" for r in judge["rejections"])
assert "候选正文中不存在的引文" not in json.dumps(judge["rejections"], ensure_ascii=False)
report = _报告(env)
assert report["outcomes"]["incomplete"] == 1
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 0}
assert Decimal(report["cost"]["total_usd"]) == Decimal(".5")
@pytest.mark.case_id(
"TC-5edc5e65298b",
environment="隔离PG与合成HTTP;不认证真实外部模型效果",
given="隔离PG中固定单元、两侧真实生成交付及受控合成评委HTTP",
when="注入原文中不存在的引文并按原固定上限纠正",
then=[
"两名反向独立评委各有预注册纠正上限;首位两次纠正后完成,次位首轮不含前一位草稿,总计四次评委调用。"
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{"judges": 2, "corrections": 2, "calls": 12}], indirect=True)
def test_每名评委独立最多两次纠正且不共享前一报告__5edc5e(执行环境):
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "bad_quote", "ok", "ok"])
运行测试._运行就绪(env)
requests = [json.loads(r["input"]) for r in env["received"][2:]]
assert len(requests) == 4
assert ["correction" in r for r in requests] == [False, True, True, False]
assert requests[0]["left"] == requests[3]["right"]
judges = [u for u in _工作面(env)["units"] if u["kind"] == "comparison"]
assert len(judges) == 2 and all(u["state"] == "completed" for u in judges)
assert sorted(len(u["rejections"]) for u in judges) == [0, 2]
assert _报告(env)["outcomes"]["tie"] == 1
@pytest.mark.case_id(
"NC-w25-25b001",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["零纠正或原调用配额不足不多发一次"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize(
"执行环境", [{"corrections": 0}, {"corrections": 2, "calls": 3}], indirect=True
)
def test_零纠正或调用配额不足不新增一次调用__25b001(执行环境):
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
运行测试._运行就绪(env, allow_failure=True)
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert len(judge["rejections"]) == 1 and judge["state"] == "failed"
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
@pytest.mark.case_id(
"NC-w25-25b002",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["拒绝提交后中断恢复只用余下次数且保原尝试"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_拒绝提交后中断恢复只继续余下纠正且保留原尝试__25b002(执行环境, monkeypatch):
from muse.共享.错误 import Muse错误
env = 执行环境
svc = env["app"].要求评测()
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
save = type(svc).保存执行交付
def 中断(self, *args, **kwargs):
result = save(self, *args, **kwargs)
if result.get("outcome") == "comparison_rejected":
raise Muse错误("合成拒绝提交后中断")
return result
monkeypatch.setattr(type(svc), "保存执行交付", 中断)
运行测试._运行就绪(env, allow_failure=True)
before = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert len(env["received"]) == 3 and before["state"] == "failed"
monkeypatch.setattr(type(svc), "保存执行交付", save)
运行测试._恢复失败单元(env)
运行测试._运行就绪(env)
after = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert after["state"] == "completed" and after["rejections"] == before["rejections"]
assert after["evidence"]["runtime"]["attempt_id"] != before["rejections"][0]["attempt_id"]
assert len(env["received"]) == 4 and "correction" in json.loads(env["received"][-1]["input"])
assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".5")
@pytest.mark.case_id(
"NC-w25-25b003",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["耗尽后恢复被拒,UPDATE/DELETE不能重置次数"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_纠正耗尽不能经恢复或删除拒绝重置次数__25b003(执行环境):
import psycopg
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "bad_quote", "ok"])
运行测试._运行就绪(env, allow_failure=True)
before = _工作面(env)
with pytest.raises(评测错误, match="上限"):
运行测试._恢复失败单元(env)
for statement in [
"DELETE FROM evaluation.muse_comparison_rejection",
"UPDATE evaluation.muse_comparison_rejection SET sequence=1",
]:
with env["pool"].连接() as conn, pytest.raises(psycopg.Error), conn.transaction():
conn.execute(statement)
assert _工作面(env) == before
assert len(env["received"]) == 4 and env["scripted"] == ["ok"]
@pytest.mark.case_id(
"NC-w25-25b004",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["重签拒绝内容或前序顺序仍过不了S02原回合核验"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("changed", ["raw_output", "previous_hash", "sequence"])
def test_拒绝记录篡改重签仍不能冒原回合或重置链__25b004(执行环境, monkeypatch, changed):
from copy import deepcopy
from muse.共享.错误 import Muse错误
from muse.效果评测.存储 import 评测存储
from muse.正式变更.接口 import 固定哈希
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
运行测试._运行就绪(env)
read = 评测存储.读取引文拒绝
def 篡改(self, unit_id):
rows = deepcopy(read(self, unit_id))
if rows:
row = rows[0]
if changed == "raw_output":
row["output"]["rationale"] = "不是原调用返回的文字"
elif changed == "previous_hash":
row["previous_hash"] = "f" * 64
else:
row["sequence"] = 2
row["record_hash"] = 固定哈希({k: v for k, v in row.items() if k != "record_hash"})
return rows
monkeypatch.setattr(评测存储, "读取引文拒绝", 篡改)
with pytest.raises(Muse错误):
_报告(env)
assert len(env["received"]) == 4
@pytest.mark.case_id(
"NC-w25-25b005",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["拒绝后停止不发纠正,保留拒绝和全部费用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_拒绝后停止实验不派发纠正且保留拒绝费用__25b005(执行环境, monkeypatch):
env = 执行环境
svc = env["app"].要求评测()
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
save = type(svc).保存执行交付
def 停止(self, *args, **kwargs):
result = save(self, *args, **kwargs)
if result.get("outcome") == "comparison_rejected":
svc.取消实验(env["actor"], env["exp"]["experiment_id"], "stop-after-rejection")
return result
monkeypatch.setattr(type(svc), "保存执行交付", 停止)
运行测试._运行就绪(env, allow_failure=True)
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
work = _工作面(env)
assert work["state"] == "stopped"
judge = next(u for u in work["units"] if u["kind"] == "comparison")
assert len(judge["rejections"]) == 1 and judge["output"] is None
assert Decimal(_报告(env)["cost"]["total_usd"]) == Decimal(".375")
@pytest.mark.case_id(
"NC-w25-25b006",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["协议、非引文语义错误及未知成本不自动纠正"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("script", ["bad_output", "bad_mixed", "unknown_cost"])
def test_协议其他语义错误及未知费用不自动纠正__25b006(执行环境, script):
env = 执行环境
_准备评委(env)
env["scripted"].extend([script, "ok"])
运行测试._运行就绪(env, allow_failure=True)
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert judge["output"] is None
if script == "bad_mixed":
assert [r["failure_code"] for r in judge["rejections"]] == ["PAIRWISE_NOT_EXECUTED"]
assert judge["unregistered_deliveries"] == []
else:
assert judge["rejections"] == []
if script == "unknown_cost":
assert _报告(env)["cost"]["total_usd"] is None
assert _工作面(env)["state"] == "reconciling"
else:
assert judge["state"] == "failed"
@pytest.mark.case_id(
"NC-w25-25b007",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["纠正前稿偏离原回合在发送前拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_纠正的上次草稿被改写在发送前拒绝__25b007(执行环境, monkeypatch):
from dataclasses import replace
import muse.编排.执行评测 as flow
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
create = flow.模型请求
def 改写(*args, **kwargs):
request = create(*args, **kwargs)
content = json.loads(request.用户输入)
if "correction" in content:
content["correction"]["previousDraft"]["rationale"] = "其他回合的报告"
return replace(
request, 用户输入=json.dumps(content, ensure_ascii=False, sort_keys=True)
)
return request
monkeypatch.setattr(flow, "模型请求", 改写)
运行测试._运行就绪(env, allow_failure=True)
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert judge["state"] == "failed" and len(judge["rejections"]) == 1
@pytest.mark.case_id(
"NC-w25-25b008",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["S02已保存但B10事务失败时,不得绕过未登记回合重发"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_拒绝事务失败不能直接恢复重置次数__25b008(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
_准备评委(env)
env["scripted"].extend(["bad_quote", "ok"])
save = 评测存储.保存引文拒绝
def 写入失败(self, record):
raise 评测错误("合成拒绝事务失败")
monkeypatch.setattr(评测存储, "保存引文拒绝", 写入失败)
运行测试._运行就绪(env, allow_failure=True)
monkeypatch.setattr(评测存储, "保存引文拒绝", save)
with pytest.raises(评测错误, match="未登记"):
运行测试._恢复失败单元(env)
assert len(env["received"]) == 3 and env["scripted"] == ["ok"]
judge = next(u for u in _工作面(env)["units"] if u["kind"] == "comparison")
assert len(judge["unregistered_deliveries"]) == 1 and not judge["rejections"]
def _漏登记(env, monkeypatch, kind):
from copy import deepcopy
if kind == "generation":
运行测试._启动执行(env)
else:
_准备评委(env)
env["scripted"].append("bad_quote" if kind == "comparison_bad" else "ok")
service = type(env["app"].要求评测())
save = service.保存执行交付
captured = []
def 中断(self, ctx, text, output, call):
if not captured:
captured.append(
{
"unit_id": ctx.任务.冻结输入["输入"]["unit_id"],
"call_id": call,
"output": deepcopy(output),
}
)
raise 评测错误("合成业务登记前中断")
return save(self, ctx, text, output, call)
monkeypatch.setattr(service, "保存执行交付", 中断)
运行测试._运行就绪(env, allow_failure=True)
monkeypatch.setattr(service, "保存执行交付", save)
assert len(captured) == 1
return captured[0]
@pytest.mark.case_id(
"NC-w25-25b009",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["生成和评委原输出经S02复核后补登,错作者/伪造拒绝、原样幂等、恢复不重发旧回合"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("kind", ["generation", "comparison_valid", "comparison_bad"])
def test_按S02原交付补登后恢复不重发已发生回合__25b009(执行环境, monkeypatch, kind):
from copy import deepcopy
from dataclasses import replace
from muse.共享.错误 import Muse错误
from muse.效果评测.接口 import 交付补登请求
env = 执行环境
svc = env["app"].要求评测()
original = _漏登记(env, monkeypatch, kind)
count = len(env["received"])
with pytest.raises(评测错误, match="未登记"):
运行测试._恢复失败单元(env)
request = 交付补登请求(**original)
with pytest.raises(评测错误, match="作者"):
svc.补登记交付(replace(env["actor"], 作者="foreign"), request)
forged = deepcopy(original)
forged["output"]["fabricated"] = True
with pytest.raises(Muse错误):
svc.补登记交付(env["actor"], 交付补登请求(**forged))
receipt = svc.补登记交付(env["actor"], request)
assert svc.补登记交付(env["actor"], request) == receipt
assert len(env["received"]) == count
unit = next(u for u in _工作面(env)["units"] if u["unit_id"] == original["unit_id"])
assert unit["unregistered_deliveries"] == []
assert bool(unit["rejections"]) == (kind == "comparison_bad")
运行测试._恢复失败单元(env)
运行测试._运行就绪(env)
assert len(env["received"]) == count + (kind == "comparison_bad")
unit = next(u for u in _工作面(env)["units"] if u["unit_id"] == original["unit_id"])
assert unit["state"] == "completed"
@pytest.mark.case_id(
"NC-w25-25b00a",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["实际CLI补登与HTTP重放同一回执,零新增模型调用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_实际CLI补登与HTTP重放不调用模型__25b00a(执行环境, monkeypatch, tmp_path):
import subprocess
import sys
import test_生产评测权限隔离 as 数据测试
from fastapi.testclient import TestClient
from muse.接入.http.应用 import 创建应用
from muse.配置 import 读取配置
env = 执行环境
original = _漏登记(env, monkeypatch, "comparison_bad")
config = 数据测试._配置文件(env["pool"], tmp_path)
request_file = tmp_path / "restore.json"
request_file.write_text(json.dumps(original, ensure_ascii=False))
result = subprocess.run(
[sys.executable, "-I", "-m", "muse", "评测", str(config), "补登记交付", str(request_file)],
cwd=tmp_path,
capture_output=True,
text=True,
timeout=30,
)
assert result.returncode == 0, result.stderr
receipt = json.loads(result.stdout)
http = 创建应用(读取配置(config))
with TestClient(http, headers={"origin": "http://testserver"}) as client:
http.state.装配 = env["app"]
assert (
client.post(
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
).status_code
== 200
)
replay = client.post("/api/v1/evaluation/deliveries/restore", json=original)
assert replay.status_code == 200 and replay.json() == receipt
report = client.get(
"/api/v1/evaluation/experiments/" + env["exp"]["experiment_id"] + "/report"
)
assert report.status_code == 200
units = report.json()["samples"][0]["units"]
assert sum(len(u["rejections"]) for u in units) == 1
assert len(env["received"]) == 3
@pytest.mark.case_id(
"NC-w25-25b00b",
environment="隔离PG与受控合成HTTP",
given="隔离PG、固定任务与实际模型回合",
when="触发本用例的纠正、登记失败、重签、恢复或停止边界",
then=["停止后的未登记交付不能补写为新业务结果"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_实验停止后拒绝补登新产物__25b00b(执行环境, monkeypatch):
from muse.效果评测.接口 import 交付补登请求
env = 执行环境
original = _漏登记(env, monkeypatch, "comparison_valid")
svc = env["app"].要求评测()
svc.取消实验(env["actor"], env["exp"]["experiment_id"], "stop-before-restore")
with pytest.raises(评测错误, match="停止"):
svc.补登记交付(env["actor"], 交付补登请求(**original))
assert len(env["received"]) == 3
assert sum(len(u["unregistered_deliveries"]) for u in _工作面(env)["units"]) == 1