- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
364 lines
14 KiB
Python
364 lines
14 KiB
Python
"""文学评分纯合同与确定性稳定计算;不冒充真实评委或标定凭据。"""
|
|
|
|
import pytest
|
|
|
|
from muse.效果评测.接口 import (
|
|
仲裁文学评分,
|
|
文学输出合同,
|
|
核对文学判断,
|
|
比较未执行,
|
|
组装文学材料,
|
|
评测错误,
|
|
读取评分标准,
|
|
)
|
|
from muse.正式变更.接口 import 固定哈希
|
|
|
|
A = "甲守住城门。守卫始终没有离开岗位。"
|
|
B = "甲掩住门缝,继续守着城门。"
|
|
|
|
|
|
def _材料(reverse=False):
|
|
return 组装文学材料(
|
|
B if reverse else A,
|
|
A if reverse else B,
|
|
{
|
|
"scenario": "turning_point",
|
|
"sources": [
|
|
{
|
|
"source_id": "outline",
|
|
"kind": "fine_outline",
|
|
"text": "甲守住城门,不离开岗位。",
|
|
},
|
|
{
|
|
"source_id": "history",
|
|
"kind": "historical_prose",
|
|
"text": "甲话不多,一直守着门。",
|
|
},
|
|
],
|
|
"assertions": [{"statement_id": "fact-one", "text": "甲是守卫。"}],
|
|
"constraints": [{"statement_id": "must-stay", "text": "甲不离开城门。"}],
|
|
},
|
|
)
|
|
|
|
|
|
def _引用(material, side, source_type="candidate", source_id=None, identity="q"):
|
|
if source_type == "candidate":
|
|
source_id = side
|
|
text = material[side]["text"]
|
|
elif source_type in {"fine_outline", "historical_prose"}:
|
|
source = next(s for s in material["basis"]["sources"] if s["kind"] == source_type)
|
|
source_id, text = source["source_id"], source["text"]
|
|
else:
|
|
source = material["basis"][
|
|
"assertions" if source_type == "oracle_assertion" else "constraints"
|
|
][0]
|
|
source_id, text = source["statement_id"], source["text"]
|
|
return {
|
|
"evidence_id": identity,
|
|
"source_type": source_type,
|
|
"source_id": source_id,
|
|
"quote": text,
|
|
}
|
|
|
|
|
|
def _输出(material, score=8.0):
|
|
dimensions = material["dimensions"]
|
|
cards = []
|
|
for side in ("left", "right"):
|
|
scores = []
|
|
for dim in dimensions:
|
|
evidence = [_引用(material, side)]
|
|
if dim != "prose_readability":
|
|
evidence.append(
|
|
_引用(
|
|
material,
|
|
side,
|
|
"historical_prose" if dim == "style_consistency" else "fine_outline",
|
|
identity="context",
|
|
)
|
|
)
|
|
scores.append(
|
|
{
|
|
"dimension": dim,
|
|
"score": score,
|
|
"rationale": "按候选及共同依据核对。",
|
|
"evidence": evidence,
|
|
"inferences": [],
|
|
}
|
|
)
|
|
cards.append({"side": side, "scores": scores})
|
|
result = {
|
|
"choice": "tie",
|
|
"rationale": "两侧按本次材料均可成立。",
|
|
"candidate_scores": cards,
|
|
"preferences": [
|
|
{"dimension": d, "choice": "tie", "rationale": "两侧表现相近。"} for d in dimensions
|
|
],
|
|
}
|
|
for field, kind, defs in [
|
|
("assertion_verdicts", "oracle_assertion", "assertions"),
|
|
("constraint_verdicts", "preregistered_constraint", "constraints"),
|
|
]:
|
|
result[field] = [
|
|
{
|
|
"side": side,
|
|
"statement_id": statement["statement_id"],
|
|
"verdict": "pass",
|
|
"rationale": "根据两侧正文与共同命题核对。",
|
|
"evidence": [
|
|
_引用(material, side),
|
|
_引用(material, side, kind, identity="statement"),
|
|
],
|
|
"inferences": [],
|
|
}
|
|
for side in ("left", "right")
|
|
for statement in material["basis"][defs]
|
|
]
|
|
return result
|
|
|
|
|
|
def _核对(material, output):
|
|
return 核对文学判断(material, output, writer_model="writer", judge_model="judge")
|
|
|
|
|
|
def _评审(index, change=None, verdict=None):
|
|
material = _材料(reverse=index == 2)
|
|
raw = _输出(material)
|
|
side = "right" if index == 2 else "left"
|
|
if change:
|
|
dimension, value = change
|
|
card = next(c for c in raw["candidate_scores"] if c["side"] == side)
|
|
next(s for s in card["scores"] if s["dimension"] == dimension)["score"] = value
|
|
if verdict:
|
|
next(v for v in raw["assertion_verdicts"] if v["side"] == side)["verdict"] = verdict
|
|
return {
|
|
"judge_ref": f"call-{index}",
|
|
"model": f"judge-{index}",
|
|
"scope_hash": 固定哈希([A, B, material["basis"], material["rubric"]]),
|
|
"sides": {
|
|
"left": "candidate-b" if index == 2 else "candidate-a",
|
|
"right": "candidate-a" if index == 2 else "candidate-b",
|
|
},
|
|
"judgment": _核对(material, raw),
|
|
}
|
|
|
|
|
|
def test_五维评分包含零与十分且拒绝四分之一步长__491ddc():
|
|
material = _材料()
|
|
assert len(material["dimensions"]) == 5
|
|
for score in (0.0, 10.0):
|
|
report = _核对(material, _输出(material, score))
|
|
assert all(s["score"] == score for c in report["candidate_scores"] for s in c["scores"])
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, _输出(material, 8.25))
|
|
|
|
|
|
def test_通用四维与五种场景分别具备完整评分锚点__8d1f8e():
|
|
labels = set()
|
|
bands = ["9-10", "7-8.5", "5-6.5", "3-4.5", "0-2.5"]
|
|
for scenario in [
|
|
"battle",
|
|
"character_dialogue",
|
|
"turning_point",
|
|
"information_reveal",
|
|
"returning_character",
|
|
]:
|
|
policy = 读取评分标准(scenario)
|
|
assert policy["version"] == "writer-replay-rubric-v3" and policy["scenario"] == scenario
|
|
assert len(policy["dimensions"]) == 5 and len(policy["common_dimensions"]) == 4
|
|
for dim in [*policy["common_dimensions"], policy["scenario_dimension"]]:
|
|
assert [a["scoreBand"] for a in dim["anchors"]] == bands
|
|
assert len(policy["scenario_dimension"]["criteria"]) >= 4
|
|
labels.add(policy["scenario_dimension"]["displayName"])
|
|
assert len(labels) == 5
|
|
with pytest.raises(评测错误, match="未登记场景"):
|
|
读取评分标准("unknown")
|
|
|
|
|
|
@pytest.mark.parametrize("bad", ["missing", "target_answer", "card_index"])
|
|
def test_每维引用必须具备已给出的来源类型__0d05fb(bad):
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
if bad == "missing":
|
|
raw["candidate_scores"][0]["scores"][2]["evidence"] = []
|
|
else:
|
|
raw["candidate_scores"][0]["scores"][0]["evidence"][0]["source_type"] = bad
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, raw)
|
|
|
|
|
|
@pytest.mark.parametrize("bad", ["unknown", "extra_field"])
|
|
def test_事实未知或偷渡实验臂不能成为合格文学报告__ab3593(bad):
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
if bad == "unknown":
|
|
raw["assertion_verdicts"][0]["verdict"] = "unknown"
|
|
else:
|
|
raw["arm"] = "C"
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, raw)
|
|
|
|
|
|
def test_完整标识乱序可归一且不放宽任何覆盖__31a33d():
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
expected = _核对(material, raw)
|
|
raw["candidate_scores"].reverse()
|
|
for c in raw["candidate_scores"]:
|
|
c["scores"].reverse()
|
|
raw["preferences"].reverse()
|
|
raw["assertion_verdicts"].reverse()
|
|
raw["constraint_verdicts"].reverse()
|
|
assert _核对(material, raw) == expected
|
|
raw["assertion_verdicts"].pop()
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, raw)
|
|
|
|
|
|
def test_超过半分触发一次第三评而不提前给结果__631102():
|
|
first = _评审(1)
|
|
dimension = first["judgment"]["candidate_scores"][0]["scores"][0]["dimension"]
|
|
result = 仲裁文学评分([first, _评审(2, (dimension, 7.0))])
|
|
assert result["status"] == "needs_third_reviewer"
|
|
assert result["score_gaps"] == [["candidate-a", dimension]]
|
|
assert not result["execution_verified"] and "scores" not in result
|
|
|
|
|
|
def test_第三评在有稳定对时取中值__302274():
|
|
first = _评审(1)
|
|
dimension = first["judgment"]["candidate_scores"][0]["scores"][1]["dimension"]
|
|
result = 仲裁文学评分([first, _评审(2, (dimension, 6.5)), _评审(3, (dimension, 7.5))])
|
|
assert result["status"] == "adjudicated_report"
|
|
assert (
|
|
next(
|
|
s["score"]
|
|
for s in result["scores"]
|
|
if s["candidate_id"] == "candidate-a" and s["dimension"] == dimension
|
|
)
|
|
== 7.5
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize("bad", ["candidate", "scope", "rubric", "reviewer"])
|
|
def test_第三评必须绑定相同对象集合与冻结范围__9dd962(bad):
|
|
first = _评审(1)
|
|
dimension = first["judgment"]["candidate_scores"][0]["scores"][1]["dimension"]
|
|
second = _评审(2, (dimension, 6.5))
|
|
third = _评审(3, (dimension, 7.5))
|
|
if bad == "candidate":
|
|
third["sides"]["left"] = "other-candidate"
|
|
elif bad == "scope":
|
|
third["scope_hash"] = 固定哈希("other-source")
|
|
elif bad == "rubric":
|
|
third["judgment"]["rubric_hash"] = "f" * 64
|
|
else:
|
|
third["judge_ref"] = first["judge_ref"]
|
|
assert 仲裁文学评分([first, second, third])["status"] == "invalid_report"
|
|
|
|
|
|
def test_任何维度无稳定对则不判胜出__34990b():
|
|
first = _评审(1)
|
|
dimension = first["judgment"]["candidate_scores"][0]["scores"][4]["dimension"]
|
|
result = 仲裁文学评分([first, _评审(2, (dimension, 6.5)), _评审(3, (dimension, 9.5))])
|
|
assert result["status"] == "invalid_unstable" and result["score_gaps"] == [
|
|
["candidate-a", dimension]
|
|
]
|
|
assert "winner" not in result and "scores" not in result
|
|
|
|
|
|
def test_事实判定冲突必须第三评且保留原分歧__d8e106():
|
|
first = _评审(1)
|
|
second = _评审(2, verdict="fail")
|
|
pending = 仲裁文学评分([first, second])
|
|
assert pending["status"] == "needs_third_reviewer" and pending["verdict_gaps"]
|
|
result = 仲裁文学评分([first, second, _评审(3)])
|
|
assert result["status"] == "adjudicated_report" and result["verdict_gaps"]
|
|
assert (
|
|
next(
|
|
v["verdict"]
|
|
for v in result["verdicts"]
|
|
if v["candidate_id"] == "candidate-a" and v["kind"] == "assertion_verdicts"
|
|
)
|
|
== "pass"
|
|
)
|
|
|
|
|
|
def test_多数事实判定不能掩盖评分不稳定__715400():
|
|
first = _评审(1)
|
|
dimension = first["judgment"]["candidate_scores"][0]["scores"][0]["dimension"]
|
|
result = 仲裁文学评分([first, _评审(2, (dimension, 6.5), "fail"), _评审(3, (dimension, 9.5))])
|
|
assert result["status"] == "invalid_unstable" and result["score_gaps"]
|
|
assert "preferences" not in result
|
|
|
|
|
|
@pytest.mark.parametrize("bad", [True, "8.0", float("nan"), float("inf"), -0.5, 10.5])
|
|
def test_分数不接受布尔字符串非有限或越界值__25c001(bad):
|
|
material = _材料()
|
|
raw = _输出(material, bad)
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, raw)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"bad", ["foreign_candidate", "missing_context", "unknown_source", "inference"]
|
|
)
|
|
def test_候选及共同来源和推断不得跨范围引用__25c002(bad):
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
score = raw["candidate_scores"][0]["scores"][0]
|
|
if bad == "foreign_candidate":
|
|
score["evidence"][0]["source_id"] = "right"
|
|
elif bad == "missing_context":
|
|
score["evidence"] = score["evidence"][:1]
|
|
elif bad == "unknown_source":
|
|
score["evidence"][1]["source_id"] = "unknown-source"
|
|
else:
|
|
score["inferences"] = [{"claim": "引用其他维度", "refs": ["other-dimension-ref"]}]
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, raw)
|
|
|
|
|
|
def test_评分模型合同不接受自报位置与身份__25c003():
|
|
schema = 文学输出合同()
|
|
forbidden = {"start", "end", "hash", "model", "reviewer_ref", "scope_hash", "arm"}
|
|
assert all(
|
|
not (forbidden & set(n.get("properties", {}))) for n in [schema, *schema["$defs"].values()]
|
|
)
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
raw["candidate_scores"][0]["scores"][0]["evidence"][0]["start"] = 0
|
|
with pytest.raises(比较未执行):
|
|
_核对(material, raw)
|
|
|
|
|
|
def test_评分标准内容漂移不能只靠版本号放行__25c004():
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
frozen = 固定哈希(material["rubric"])
|
|
material["rubric"]["common_dimensions"][0]["question"] = "改成另一种要求"
|
|
with pytest.raises(比较未执行):
|
|
核对文学判断(material, raw, writer_model="writer", judge_model="judge", rubric_hash=frozen)
|
|
|
|
|
|
def test_双评稳定不需要第三且观察不冒执行认证__25c005():
|
|
first = _评审(1)
|
|
second = _评审(2)
|
|
result = 仲裁文学评分([first, second])
|
|
assert result["status"] == "stable_report" and not result["execution_verified"]
|
|
assert len(result["scores"]) == 10
|
|
assert 仲裁文学评分([first, second, _评审(3)])["status"] == "invalid_report"
|
|
|
|
|
|
@pytest.mark.parametrize("other_error", ["verdict_coverage", "later_dimension"])
|
|
def test_不完整合同不能被首项坏引文掩盖为可自动纠正__25c006(other_error):
|
|
material = _材料()
|
|
raw = _输出(material)
|
|
raw["candidate_scores"][0]["scores"][0]["evidence"][0]["quote"] = "不存在的引文"
|
|
if other_error == "verdict_coverage":
|
|
raw["assertion_verdicts"].pop()
|
|
else:
|
|
raw["candidate_scores"][1]["scores"][4]["evidence"][0]["source_id"] = "left"
|
|
with pytest.raises(比较未执行) as error:
|
|
_核对(material, raw)
|
|
assert error.value.错误码 == "PAIRWISE_NOT_EXECUTED"
|