muse-agent-example/tests/契约/test_文学评分与仲裁.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

506 lines
21 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""文学评分纯合同与确定性稳定计算;不冒充真实评委或标定凭据。"""
import pytest
from muse.效果评测.接口 import (
仲裁文学评分,
文学输出合同,
核对文学判断,
比较未执行,
组装文学材料,
评测错误,
读取评分标准,
)
from muse.正式变更.接口 import 固定哈希
A = "甲守住城门。守卫始终没有离开岗位。"
B = "甲掩住门缝,继续守着城门。"
def _材料(reverse=False):
return 组装文学材料(
B if reverse else A,
A if reverse else B,
{
"scenario": "turning_point",
"sources": [
{
"source_id": "outline",
"kind": "fine_outline",
"text": "甲守住城门,不离开岗位。",
},
{
"source_id": "history",
"kind": "historical_prose",
"text": "甲话不多,一直守着门。",
},
],
"assertions": [{"statement_id": "fact-one", "text": "甲是守卫。"}],
"constraints": [{"statement_id": "must-stay", "text": "甲不离开城门。"}],
},
)
def _引用(material, side, source_type="candidate", source_id=None, identity="q"):
if source_type == "candidate":
source_id = side
text = material[side]["text"]
elif source_type in {"fine_outline", "historical_prose"}:
source = next(s for s in material["basis"]["sources"] if s["kind"] == source_type)
source_id, text = source["source_id"], source["text"]
else:
source = material["basis"][
"assertions" if source_type == "oracle_assertion" else "constraints"
][0]
source_id, text = source["statement_id"], source["text"]
return {
"evidence_id": identity,
"source_type": source_type,
"source_id": source_id,
"quote": text,
}
def _输出(material, score=8.0):
dimensions = material["dimensions"]
cards = []
for side in ("left", "right"):
scores = []
for dim in dimensions:
evidence = [_引用(material, side)]
if dim != "prose_readability":
evidence.append(
_引用(
material,
side,
"historical_prose" if dim == "style_consistency" else "fine_outline",
identity="context",
)
)
scores.append(
{
"dimension": dim,
"score": score,
"rationale": "按候选及共同依据核对。",
"evidence": evidence,
"inferences": [],
}
)
cards.append({"side": side, "scores": scores})
result = {
"choice": "tie",
"rationale": "两侧按本次材料均可成立。",
"candidate_scores": cards,
"preferences": [
{"dimension": d, "choice": "tie", "rationale": "两侧表现相近。"} for d in dimensions
],
}
for field, kind, defs in [
("assertion_verdicts", "oracle_assertion", "assertions"),
("constraint_verdicts", "preregistered_constraint", "constraints"),
]:
result[field] = [
{
"side": side,
"statement_id": statement["statement_id"],
"verdict": "pass",
"rationale": "根据两侧正文与共同命题核对。",
"evidence": [
_引用(material, side),
_引用(material, side, kind, identity="statement"),
],
"inferences": [],
}
for side in ("left", "right")
for statement in material["basis"][defs]
]
return result
def _核对(material, output):
return 核对文学判断(material, output, writer_model="writer", judge_model="judge")
def _评审(index, change=None, verdict=None):
material = _材料(reverse=index == 2)
raw = _输出(material)
side = "right" if index == 2 else "left"
if change:
dimension, value = change
card = next(c for c in raw["candidate_scores"] if c["side"] == side)
next(s for s in card["scores"] if s["dimension"] == dimension)["score"] = value
if verdict:
next(v for v in raw["assertion_verdicts"] if v["side"] == side)["verdict"] = verdict
return {
"judge_ref": f"call-{index}",
"model": f"judge-{index}",
"scope_hash": 固定哈希([A, B, material["basis"], material["rubric"]]),
"sides": {
"left": "candidate-b" if index == 2 else "candidate-a",
"right": "candidate-a" if index == 2 else "candidate-b",
},
"judgment": _核对(material, raw),
}
@pytest.mark.case_id(
"TC-491ddc923da1",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["五维保留0—10半分步长,0和10合法,8.25拒绝。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_五维评分包含零与十分且拒绝四分之一步长__491ddc():
material = _材料()
assert len(material["dimensions"]) == 5
for score in (0.0, 10.0):
report = _核对(material, _输出(material, score))
assert all(s["score"] == score for c in report["candidate_scores"] for s in c["scores"])
with pytest.raises(比较未执行):
_核对(material, _输出(material, 8.25))
@pytest.mark.case_id(
"TC-8d1f8eb227c1",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=[
"四个通用维度与五种已登记场景各具完整五档锚点;保留narrative_tension身份,合同版本迁为writer-replay-rubric-v3,内容哈希随发布冻结。"
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_通用四维与五种场景分别具备完整评分锚点__8d1f8e():
labels = set()
bands = ["9-10", "7-8.5", "5-6.5", "3-4.5", "0-2.5"]
for scenario in [
"battle",
"character_dialogue",
"turning_point",
"information_reveal",
"returning_character",
]:
policy = 读取评分标准(scenario)
assert policy["version"] == "writer-replay-rubric-v3" and policy["scenario"] == scenario
assert len(policy["dimensions"]) == 5 and len(policy["common_dimensions"]) == 4
for dim in [*policy["common_dimensions"], policy["scenario_dimension"]]:
assert [a["scoreBand"] for a in dim["anchors"]] == bands
assert len(policy["scenario_dimension"]["criteria"]) >= 4
labels.add(policy["scenario_dimension"]["displayName"])
assert len(labels) == 5
with pytest.raises(评测错误, match="未登记场景"):
读取评分标准("unknown")
@pytest.mark.case_id(
"TC-0d05fb2cec95",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["每维必须有当前候选及所需共同材料的原字证据;目标答案和卡片索引不是评分来源。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", ["missing", "target_answer", "card_index"])
def test_每维引用必须具备已给出的来源类型__0d05fb(bad):
material = _材料()
raw = _输出(material)
if bad == "missing":
raw["candidate_scores"][0]["scores"][2]["evidence"] = []
else:
raw["candidate_scores"][0]["scores"][0]["evidence"][0]["source_type"] = bad
with pytest.raises(比较未执行):
_核对(material, raw)
@pytest.mark.case_id(
"TC-ab35938d79db",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=[
"未知事实判定或额外臂字段不能成为合格判断。模型原始字段不收运行身份,运行认证留给实际S02。"
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", ["unknown", "extra_field"])
def test_事实未知或偷渡实验臂不能成为合格文学报告__ab3593(bad):
material = _材料()
raw = _输出(material)
if bad == "unknown":
raw["assertion_verdicts"][0]["verdict"] = "unknown"
else:
raw["arm"] = "C"
with pytest.raises(比较未执行):
_核对(material, raw)
@pytest.mark.case_id(
"NC-w25-25d00d",
environment="离线纯合同;不冒充实际调用",
given="完整但乱序的两侧评分与命题以及缺项/重复变体",
when="调用文学评分公开合同核验",
then=["乱序按固定身份归一,缺项与重复仍拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_完整标识乱序可归一且不放宽任何覆盖__31a33d():
material = _材料()
raw = _输出(material)
expected = _核对(material, raw)
raw["candidate_scores"].reverse()
for c in raw["candidate_scores"]:
c["scores"].reverse()
raw["preferences"].reverse()
raw["assertion_verdicts"].reverse()
raw["constraint_verdicts"].reverse()
assert _核对(material, raw) == expected
raw["assertion_verdicts"].pop()
with pytest.raises(比较未执行):
_核对(material, raw)
@pytest.mark.case_id(
"TC-631102c31f06",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["同一候选同一维度差异超过0.5须第三评,不能先给聚合分数。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_超过半分触发一次第三评而不提前给结果__631102():
first = _评审(1)
dimension = first["judgment"]["candidate_scores"][0]["scores"][0]["dimension"]
result = 仲裁文学评分([first, _评审(2, (dimension, 7.0))])
assert result["status"] == "needs_third_reviewer"
assert result["score_gaps"] == [["candidate-a", dimension]]
assert not result["execution_verified"] and "scores" not in result
@pytest.mark.case_id(
"TC-302274f91e3a",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["第三評只有存在稳定分数对才能以中值汇总;保留原始分歧,纯函数不认证执行。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_第三评在有稳定对时取中值__302274():
first = _评审(1)
dimension = first["judgment"]["candidate_scores"][0]["scores"][1]["dimension"]
result = 仲裁文学评分([first, _评审(2, (dimension, 6.5)), _评审(3, (dimension, 7.5))])
assert result["status"] == "adjudicated_report"
assert (
next(
s["score"]
for s in result["scores"]
if s["candidate_id"] == "candidate-a" and s["dimension"] == dimension
)
== 7.5
)
@pytest.mark.case_id(
"TC-9dd9625b5d23",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=[
"第三评要求相同候选集合、冻结范围和标准且身份独立;原单候选比较改为具名文本对,不能跨组复用。"
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", ["candidate", "scope", "rubric", "reviewer"])
def test_第三评必须绑定相同对象集合与冻结范围__9dd962(bad):
first = _评审(1)
dimension = first["judgment"]["candidate_scores"][0]["scores"][1]["dimension"]
second = _评审(2, (dimension, 6.5))
third = _评审(3, (dimension, 7.5))
if bad == "candidate":
third["sides"]["left"] = "other-candidate"
elif bad == "scope":
third["scope_hash"] = 固定哈希("other-source")
elif bad == "rubric":
third["judgment"]["rubric_hash"] = "f" * 64
else:
third["judge_ref"] = first["judge_ref"]
assert 仲裁文学评分([first, second, third])["status"] == "invalid_report"
@pytest.mark.case_id(
"TC-34990bddae56",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["任一维度没有稳定对即不稳定,无winner或汇总分数。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_任何维度无稳定对则不判胜出__34990b():
first = _评审(1)
dimension = first["judgment"]["candidate_scores"][0]["scores"][4]["dimension"]
result = 仲裁文学评分([first, _评审(2, (dimension, 6.5)), _评审(3, (dimension, 9.5))])
assert result["status"] == "invalid_unstable" and result["score_gaps"] == [
["candidate-a", dimension]
]
assert "winner" not in result and "scores" not in result
@pytest.mark.case_id(
"TC-d8e106126aca",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["事实判定冲突需要第三评;有一致判定对时提供结果但原冲突不消失。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_事实判定冲突必须第三评且保留原分歧__d8e106():
first = _评审(1)
second = _评审(2, verdict="fail")
pending = 仲裁文学评分([first, second])
assert pending["status"] == "needs_third_reviewer" and pending["verdict_gaps"]
result = 仲裁文学评分([first, second, _评审(3)])
assert result["status"] == "adjudicated_report" and result["verdict_gaps"]
assert (
next(
v["verdict"]
for v in result["verdicts"]
if v["candidate_id"] == "candidate-a" and v["kind"] == "assertion_verdicts"
)
== "pass"
)
@pytest.mark.case_id(
"TC-7154000d4741",
environment="离线纯合同;未执行真实模型或标定",
given="独立合成两侧文本、共同依据和已冻结评分标准;由代码构造评审视图",
when="调用公开评分核验或稳定性计算",
then=["即使事实多数一致,分数不稳定仍不得给出合格仲裁结果。"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_多数事实判定不能掩盖评分不稳定__715400():
first = _评审(1)
dimension = first["judgment"]["candidate_scores"][0]["scores"][0]["dimension"]
result = 仲裁文学评分([first, _评审(2, (dimension, 6.5), "fail"), _评审(3, (dimension, 9.5))])
assert result["status"] == "invalid_unstable" and result["score_gaps"]
assert "preferences" not in result
@pytest.mark.case_id(
"NC-w25-25c001",
environment="离线纯合同",
given="已固定合成材料与评分合同",
when="注入该用例的边界输入",
then=["分数拒绝布尔/字符串/非有限和越界"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("bad", [True, "8.0", float("nan"), float("inf"), -0.5, 10.5])
def test_分数不接受布尔字符串非有限或越界值__25c001(bad):
material = _材料()
raw = _输出(material, bad)
with pytest.raises(比较未执行):
_核对(material, raw)
@pytest.mark.case_id(
"NC-w25-25c002",
environment="离线纯合同",
given="已固定合成材料与评分合同",
when="注入该用例的边界输入",
then=["候选、共同来源及推断不能跨范围引用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize(
"bad", ["foreign_candidate", "missing_context", "unknown_source", "inference"]
)
def test_候选及共同来源和推断不得跨范围引用__25c002(bad):
material = _材料()
raw = _输出(material)
score = raw["candidate_scores"][0]["scores"][0]
if bad == "foreign_candidate":
score["evidence"][0]["source_id"] = "right"
elif bad == "missing_context":
score["evidence"] = score["evidence"][:1]
elif bad == "unknown_source":
score["evidence"][1]["source_id"] = "unknown-source"
else:
score["inferences"] = [{"claim": "引用其他维度", "refs": ["other-dimension-ref"]}]
with pytest.raises(比较未执行):
_核对(material, raw)
@pytest.mark.case_id(
"NC-w25-25c003",
environment="离线纯合同",
given="已固定合成材料与评分合同",
when="注入该用例的边界输入",
then=["模型不能自报位置、哈希、运行身份或臂"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_评分模型合同不接受自报位置与身份__25c003():
schema = 文学输出合同()
forbidden = {"start", "end", "hash", "model", "reviewer_ref", "scope_hash", "arm"}
assert all(
not (forbidden & set(n.get("properties", {}))) for n in [schema, *schema["$defs"].values()]
)
material = _材料()
raw = _输出(material)
raw["candidate_scores"][0]["scores"][0]["evidence"][0]["start"] = 0
with pytest.raises(比较未执行):
_核对(material, raw)
@pytest.mark.case_id(
"NC-w25-25c004",
environment="离线纯合同",
given="已固定合成材料与评分合同",
when="注入该用例的边界输入",
then=["评分标准改内容但只保版本号仍被固定哈希拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_评分标准内容漂移不能只靠版本号放行__25c004():
material = _材料()
raw = _输出(material)
frozen = 固定哈希(material["rubric"])
material["rubric"]["common_dimensions"][0]["question"] = "改成另一种要求"
with pytest.raises(比较未执行):
核对文学判断(material, raw, writer_model="writer", judge_model="judge", rubric_hash=frozen)
@pytest.mark.case_id(
"NC-w25-25c005",
environment="离线纯合同",
given="已固定合成材料与评分合同",
when="注入该用例的边界输入",
then=["双评稳定不需要第三,多余第三不忽略;纯计算执行认证仍false"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_双评稳定不需要第三且观察不冒执行认证__25c005():
first = _评审(1)
second = _评审(2)
result = 仲裁文学评分([first, second])
assert result["status"] == "stable_report" and not result["execution_verified"]
assert len(result["scores"]) == 10
assert 仲裁文学评分([first, second, _评审(3)])["status"] == "invalid_report"
@pytest.mark.case_id(
"NC-w25-25c006",
environment="离线纯合同",
given="已固定合成材料与评分合同",
when="注入该用例的边界输入",
then=["全合同结构先核,不能被首项坏引文掩盖成可自动纠正"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("other_error", ["verdict_coverage", "later_dimension"])
def test_不完整合同不能被首项坏引文掩盖为可自动纠正__25c006(other_error):
material = _材料()
raw = _输出(material)
raw["candidate_scores"][0]["scores"][0]["evidence"][0]["quote"] = "不存在的引文"
if other_error == "verdict_coverage":
raw["assertion_verdicts"].pop()
else:
raw["candidate_scores"][1]["scores"][4]["evidence"][0]["source_id"] = "left"
with pytest.raises(比较未执行) as error:
_核对(material, raw)
assert error.value.错误码 == "PAIRWISE_NOT_EXECUTED"