实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
564 lines
25 KiB
Python
564 lines
25 KiB
Python
"""真实S02合成HTTP交付验证文学评分与条件第三;不证明文学质量。"""
|
||
|
||
import json
|
||
|
||
import pytest
|
||
import test_回放资料封存 as 资料测试
|
||
import test_评测执行与失败收敛 as 执行测试
|
||
|
||
from muse.效果评测.文学评分 import 维度
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 执行测试.执行环境
|
||
资料环境 = 资料测试.资料环境
|
||
生成环境 = 资料测试.生成环境
|
||
|
||
依据 = {
|
||
"scenario": "turning_point",
|
||
"sources": [
|
||
{"source_id": "outline", "kind": "fine_outline", "text": "角色守住城门。"},
|
||
{"source_id": "history", "kind": "historical_prose", "text": "先前角色静静守门。"},
|
||
],
|
||
"assertions": [{"statement_id": "guard", "text": "角色是守卫。"}],
|
||
"constraints": [{"statement_id": "stay", "text": "角色不能离开城门。"}],
|
||
}
|
||
|
||
|
||
def _输出(material, script):
|
||
def quote(side, kind="candidate", source=None):
|
||
return {
|
||
"evidence_id": "candidate" if source is None else "source",
|
||
"source_type": kind,
|
||
"source_id": side
|
||
if source is None
|
||
else source.get("source_id", source.get("statement_id")),
|
||
"quote": material[side]["text"] if source is None else source["text"],
|
||
}
|
||
|
||
cards = []
|
||
for side in ("left", "right"):
|
||
scores = []
|
||
for dim in material["dimensions"]:
|
||
evidence = [quote(side)]
|
||
if dim != "prose_readability":
|
||
kind = "historical_prose" if dim == "style_consistency" else "fine_outline"
|
||
source = next(s for s in material["basis"]["sources"] if s["kind"] == kind)
|
||
evidence.append(quote(side, kind, source))
|
||
scores.append(
|
||
{
|
||
"dimension": dim,
|
||
"score": {
|
||
"score_gap": 6.0,
|
||
"unstable": 4.0,
|
||
"calibration_gap": 7.0,
|
||
"calibration_low": 7.5,
|
||
"calibration_high": 8.5,
|
||
}.get(script, 8.0),
|
||
"rationale": "合成评分依据候选与共同背景。",
|
||
"evidence": evidence,
|
||
"inferences": [],
|
||
}
|
||
)
|
||
cards.append({"side": side, "scores": scores})
|
||
result = {
|
||
"choice": "tie",
|
||
"rationale": "合成样本维持平局。",
|
||
"candidate_scores": cards,
|
||
"preferences": [
|
||
{"dimension": d, "choice": "tie", "rationale": "两侧合成表现相近。"}
|
||
for d in material["dimensions"]
|
||
],
|
||
}
|
||
for field, kind, key in (
|
||
("assertion_verdicts", "oracle_assertion", "assertions"),
|
||
("constraint_verdicts", "preregistered_constraint", "constraints"),
|
||
):
|
||
result[field] = [
|
||
{
|
||
"side": side,
|
||
"statement_id": statement["statement_id"],
|
||
"verdict": "fail" if script == "verdict_gap" else "pass",
|
||
"rationale": "按共同命题核对合成文本。",
|
||
"inferences": [],
|
||
"evidence": [quote(side), quote(side, kind, statement)],
|
||
}
|
||
for side in ("left", "right")
|
||
for statement in material["basis"][key]
|
||
]
|
||
if script == "shuffle":
|
||
result["candidate_scores"].reverse()
|
||
for card in result["candidate_scores"]:
|
||
card["scores"].reverse()
|
||
for field in ("preferences", "assertion_verdicts", "constraint_verdicts"):
|
||
result[field].reverse()
|
||
if script == "literary_bad_quote":
|
||
result["candidate_scores"][0]["scores"][0]["evidence"][0]["quote"] = "不存在的原文"
|
||
if script == "literary_duplicate":
|
||
result["candidate_scores"][0]["scores"][0]["dimension"] = result["candidate_scores"][0][
|
||
"scores"
|
||
][1]["dimension"]
|
||
return result
|
||
|
||
|
||
参数 = {
|
||
"judges": 2,
|
||
"arbitrator": True,
|
||
"judge_output": _输出,
|
||
"judging_basis": 依据,
|
||
"dimensions": 维度,
|
||
"calls": 5,
|
||
}
|
||
|
||
|
||
def _读(env):
|
||
return env["app"].要求评测().读取执行工作面(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _推进(env):
|
||
return env["app"].要求评测().推进实验(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _双评(env, script="ok"):
|
||
start = 执行测试._启动执行(env)
|
||
assert len(start["units"]) == 5
|
||
assert sum(u["task_id"] is not None for u in start["units"]) == 2
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
env["scripted"].extend(["ok", script])
|
||
执行测试._运行就绪(env)
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d001",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["双评稳定不建第三任务;共同依据仅供评委,目标答案不外发"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_稳定双评不建第三任务且共同依据不泄露答案__25d001(执行环境):
|
||
env = 执行环境
|
||
_双评(env)
|
||
_推进(env)
|
||
执行测试._运行就绪(env)
|
||
result = _读(env)
|
||
assert result["state"] == "completed"
|
||
unneeded = [u for u in result["units"] if u["state"] == "not_required"]
|
||
assert len(unneeded) == 1 and unneeded[0]["task_id"] is None
|
||
assert len(env["received"]) == 4
|
||
inputs = [json.loads(r["input"]) for r in env["received"]]
|
||
assert all("basis" not in r for r in inputs[:2])
|
||
assert inputs[2]["left"] == inputs[3]["right"]
|
||
assert inputs[2]["basis"] == inputs[3]["basis"] == 依据
|
||
assert all("ORACLE-SECRET" not in r["input"] for r in env["received"])
|
||
assert all(not r.get("tools") and not r.get("previous_response_id") for r in env["received"])
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["coverage"]["compared"] == 1
|
||
literary = report["samples"][0]["literary"]["control:treatment"]
|
||
assert literary["status"] == "stable_report" and literary["execution_verified"]
|
||
assert report["literary_quality"]["metrics"] is None
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d002",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["分数或判定分歧只追加预注册第三评,不提供前评报告;原分歧及费用保留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("script", ["score_gap", "verdict_gap"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_分歧只追加预注册第三评且保留原始分歧__25d002(执行环境, script):
|
||
env = 执行环境
|
||
_双评(env, script)
|
||
_推进(env)
|
||
ready = _读(env)
|
||
assert sum(u["task_id"] is not None for u in ready["units"]) == 5
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
执行测试._运行就绪(env)
|
||
assert len(env["received"]) == 5
|
||
original, third = (json.loads(env["received"][i]["input"]) for i in (2, 4))
|
||
assert original == third # 第三评不读取前两份报告或模型身份。
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
literary = report["samples"][0]["literary"]["control:treatment"]
|
||
assert literary["status"] == "adjudicated_report" and literary["execution_verified"]
|
||
assert literary["score_gaps"] if script == "score_gap" else literary["verdict_gaps"]
|
||
assert report["cost"]["sent_calls"] == 5
|
||
assert report["literary_quality"]["calibration"] == "unverified"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d003",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["初评费用未知不启动第三,固定样本分母和未知总额保留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_初评费用不明时不启动第三也不删失败样本__25d003(执行环境):
|
||
env = 执行环境
|
||
执行测试._启动执行(env)
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
env["scripted"].extend(["ok", "unknown_cost"])
|
||
执行测试._运行就绪(env, allow_failure=True)
|
||
_推进(env)
|
||
result = _读(env)
|
||
assert len(env["received"]) == 4 and result["state"] == "reconciling"
|
||
assert sum(u["task_id"] is not None for u in result["units"]) == 4
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["coverage"] == {"samples": 1, "generated": 1, "compared": 0}
|
||
assert report["cost"]["total_usd"] is None
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-31a33d92bdfa",
|
||
environment="隔离PG、实际S02、合成HTTP;未标定文学效果",
|
||
given="隔离真实S02任务及合成HTTP返回完整但乱序的评分和偏好",
|
||
when="保存实际交付并经公开服务读取报告",
|
||
then=[
|
||
"完整ID乱序的实际S02输出保留原哈希;按固定两侧和五维归一,覆盖条件不放宽。模型不再填写技术候选ID,改用本次匿名侧,控制器绑定实际单元。"
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_实际乱序交付归一同时保留S02原始输出__25d004(执行环境):
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
env = 执行环境
|
||
_双评(env, "shuffle")
|
||
_推进(env)
|
||
work = _读(env)
|
||
reversed_output = next(
|
||
u
|
||
for u in work["units"]
|
||
if u["kind"] == "comparison"
|
||
and u["output"]
|
||
and u["output"]["candidate_scores"][0]["side"] == "right"
|
||
)
|
||
evidence = reversed_output["evidence"]
|
||
assert evidence["structured_output_hash"] == 固定哈希(reversed_output["output"])
|
||
normalized = evidence["judgment"]["literary"]
|
||
assert [c["side"] for c in normalized["candidate_scores"]] == ["left", "right"]
|
||
assert all(
|
||
tuple(s["dimension"] for s in c["scores"]) == 维度 for c in normalized["candidate_scores"]
|
||
)
|
||
assert tuple(r["dimension"] for r in evidence["judgment"]["report"]["dimensions"]) == 维度
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["samples"][0]["literary"]["control:treatment"]["status"] == "stable_report"
|
||
assert report["report_hash"] == 固定哈希(
|
||
{k: v for k, v in report.items() if k != "report_hash"}
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d005",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["第三评仍不稳定不产生汇总分数或赢家,全部原判断保留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_第三仍无稳定对保留全部评审且不产生合格分数__25d005(执行环境):
|
||
env = 执行环境
|
||
_双评(env, "score_gap")
|
||
_推进(env)
|
||
env["scripted"].append("unstable")
|
||
执行测试._运行就绪(env)
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
sample = report["samples"][0]
|
||
assert sample["literary"]["control:treatment"]["status"] == "invalid_unstable"
|
||
assert sample["literary"]["control:treatment"]["scores"] == []
|
||
assert sample["outcome"] == "incomplete" and report["outcomes"]["tie"] == 0
|
||
assert len(sample["decisions"]) == 3
|
||
assert report["literary_quality"]["metrics"] is None
|
||
assert len(env["received"]) == 5
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d006",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["条件决议拒绝数据库改写,局部替换必要性仍被S02原交付重算拒绝"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_第三决议不可覆盖且局部重签不能伪造必要性__25d006(执行环境, monkeypatch):
|
||
import psycopg
|
||
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.效果评测.接口 import 评测错误
|
||
|
||
env = 执行环境
|
||
_双评(env)
|
||
work = _推进(env)
|
||
uid = next(u["unit_id"] for u in work["units"] if u["state"] == "not_required")
|
||
with pytest.raises(psycopg.Error), env["pool"].连接() as conn:
|
||
conn.execute(
|
||
"UPDATE evaluation.muse_evaluation_condition SET decision=decision WHERE unit_id=%s",
|
||
(uid,),
|
||
)
|
||
original = 评测存储.读取条件
|
||
|
||
def forged(self, unit_id):
|
||
row = original(self, unit_id)
|
||
return {**row, "disposition": "required"} if row else row
|
||
|
||
monkeypatch.setattr(评测存储, "读取条件", forged)
|
||
with pytest.raises(评测错误, match="重算结果"):
|
||
_推进(env)
|
||
with pytest.raises(评测错误, match="重算结果"):
|
||
_读(env)
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d007",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["预算未覆盖候补时零任务零调用,不能先执行后扩预算"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{**参数, "calls": 4}], indirect=True)
|
||
def test_总预算未覆盖候补时拒绝启动且不留孤儿任务__25d007(执行环境):
|
||
from muse.效果评测.接口 import 评测错误
|
||
|
||
env = 执行环境
|
||
with pytest.raises(评测错误, match="不足"):
|
||
执行测试._启动执行(env)
|
||
assert _读(env)["state"] == "registered"
|
||
assert not env["received"]
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d008",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["发布评分标准变化阻新发送,原报告仍按原版本回查"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_评分标准变化拒绝新发送但已完成报告仍按原版本回查__25d008(执行环境, monkeypatch):
|
||
import muse.效果评测.文学执行 as adapter
|
||
|
||
env = 执行环境
|
||
_双评(env, "score_gap")
|
||
original = adapter.读取评分标准
|
||
monkeypatch.setattr(
|
||
adapter, "读取评分标准", lambda scenario: {**original(scenario), "common_dimensions": []}
|
||
)
|
||
assert len([u for u in _读(env)["units"] if u["output"]]) == 4
|
||
_推进(env)
|
||
执行测试._运行就绪(env, allow_failure=True)
|
||
assert len(env["received"]) == 4
|
||
assert _读(env)["state"] == "partial_failure"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d009",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["文学引文错误沿相同S02调用链纠正,记录拒绝后再决定是否需要第三"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{**参数, "calls": 10}], indirect=True)
|
||
def test_文学坏引文沿原S02调用链有界纠正后再决定第三__25d009(执行环境):
|
||
env = 执行环境
|
||
执行测试._启动执行(env)
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
env["scripted"].extend(["literary_bad_quote", "ok", "ok"])
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
work = _读(env)
|
||
assert work["state"] == "completed"
|
||
assert sum(len(u["rejections"]) for u in work["units"]) == 1
|
||
assert len(env["received"]) == 5
|
||
corrections = [json.loads(r["input"]) for r in env["received"] if '"correction"' in r["input"]]
|
||
assert len(corrections) == 1
|
||
assert "candidate_scores" in corrections[0]["correction"]["previousDraft"]
|
||
assert corrections[0]["basis"] == 依据
|
||
|
||
|
||
def 创建ABC实验(env, 资料环境, *, detector=False, calls=12, audit=None):
|
||
from dataclasses import replace
|
||
|
||
from muse.任务运行.接口 import 角色策略目录
|
||
from muse.效果评测.接口 import 回放数据集发布, 实验请求
|
||
|
||
env = dict(env)
|
||
_, _, author, publisher, source_req = 资料环境
|
||
raw = source_req.model_dump(mode="json")
|
||
if audit is not None:
|
||
raw["samples"][0]["audit"] = audit
|
||
raw["samples"][0]["judging"] = {k: 依据[k] for k in ("scenario", "assertions", "constraints")}
|
||
receipt = publisher.发布回放数据集(author, 回放数据集发布.model_validate(raw))
|
||
frozen = env["exp"]["conditions"]
|
||
request = 实验请求.model_validate(
|
||
{
|
||
"dataset_version_id": receipt["version_id"],
|
||
"dataset_hash": receipt["public_hash"],
|
||
"split": "holdout",
|
||
"generator_role": "writer",
|
||
"target": {
|
||
"kind": "code",
|
||
"target_ref": "B09.writer-replay",
|
||
"version": 角色策略目录.从发布包().资源发布身份,
|
||
"content_hash": 角色策略目录.从发布包().资源发布身份,
|
||
},
|
||
"generator": {k: frozen["generator"][k] for k in ("config_id", "version")},
|
||
"judges": [{k: p[k] for k in ("config_id", "version")} for p in frozen["judges"]],
|
||
"arbitrator": {k: frozen["arbitrator"][k] for k in ("config_id", "version")},
|
||
"comparison_profile": "writer_rubric",
|
||
"detector": (
|
||
{k: frozen["detector"][k] for k in ("config_id", "version")} if detector else None
|
||
),
|
||
"dimensions": list(维度),
|
||
"arms": ["A", "B", "C"],
|
||
"max_cost_usd": "12",
|
||
"max_calls_per_sample": calls,
|
||
}
|
||
)
|
||
env["actor"] = replace(env["actor"], 作者=author.作者)
|
||
svc = env["app"].要求评测()
|
||
env["exp"] = svc.创建实验(env["actor"], "literary-abc", request)
|
||
return {**env, "dataset_receipt": receipt, "request": request}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d00a",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["真实ABC业务来源封存为共同评判依据,目标正文只留oracle;三对主AC保留"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实ABC封存提供共同文学依据且目标正文仍只留oracle__25d00a(执行环境, 资料环境):
|
||
env = 创建ABC实验(执行环境, 资料环境)
|
||
svc = env["app"].要求评测()
|
||
receipt = env["dataset_receipt"]
|
||
started = 执行测试._启动执行(env)
|
||
assert len(started["units"]) == 12
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["state"] == "completed" and report["primary_pair"] == ["A", "C"]
|
||
assert len(env["received"]) == 9
|
||
assert set(report["samples"][0]["literary"]) == {"A:B", "A:C", "B:C"}
|
||
assert all(x["status"] == "stable_report" for x in report["samples"][0]["literary"].values())
|
||
inputs = [json.loads(r["input"]) for r in env["received"]]
|
||
common = inputs[3]["basis"]
|
||
assert all(r["basis"] == common for r in inputs[3:])
|
||
assert {s["kind"] for s in common["sources"]} == {"fine_outline", "historical_prose"}
|
||
assert any(s["text"] == "林深回到渡口,雨渐渐小了。" for s in common["sources"])
|
||
assert "ORACLE-TARGET-PRIVATE" not in json.dumps(env["received"], ensure_ascii=False)
|
||
public = svc.读取数据集(env["actor"], receipt["version_id"])
|
||
assert "judging_basis" not in json.dumps(public, ensure_ascii=False)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
answer = conn.execute(
|
||
"SELECT answers FROM oracle.muse_dataset_answers WHERE version_id=%s",
|
||
(receipt["version_id"],),
|
||
).fetchone()[0][0]["answer"]
|
||
assert answer["target_text"].startswith("ORACLE-TARGET-PRIVATE")
|
||
assert answer["judging_basis"] == common
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d00b",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["初评后停止实验不再登记第三任务"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_停止后的初评分歧不再创建第三任务__25d00b(执行环境):
|
||
from muse.效果评测.接口 import 评测错误
|
||
|
||
env = 执行环境
|
||
_双评(env, "score_gap")
|
||
env["app"].要求评测().取消实验(env["actor"], env["exp"]["experiment_id"], "stop-before-third")
|
||
with pytest.raises(评测错误, match="停止"):
|
||
_推进(env)
|
||
assert _读(env)["state"] == "stopped"
|
||
assert sum(u["task_id"] is not None for u in _读(env)["units"]) == 4
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d00c",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["维度缺失且重复是非引文错误,不借纠正继续调用或生成有效报告"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [{**参数, "calls": 10}], indirect=True)
|
||
def test_实际重复维度不能借引文纠正继续调用或形成报告__25d00c(执行环境):
|
||
env = 执行环境
|
||
执行测试._启动执行(env)
|
||
执行测试._运行就绪(env)
|
||
_推进(env)
|
||
env["scripted"].extend(["literary_duplicate", "ok"])
|
||
执行测试._运行就绪(env, allow_failure=True)
|
||
_推进(env)
|
||
work = _读(env)
|
||
assert work["state"] == "partial_failure" and len(env["received"]) == 4
|
||
rejected = [r for u in work["units"] for r in u["rejections"]]
|
||
assert len(rejected) == 1 and rejected[0]["failure_code"] == "PAIRWISE_NOT_EXECUTED"
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["samples"][0]["literary"]["control:treatment"]["status"] == "incomplete"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25d00e",
|
||
environment="隔离PG和合成HTTP;真实来源例使用隔离业务owner",
|
||
given="完整预注册实验、独立合成配置和真实S02逐例交付",
|
||
when="经公开入口执行并注入本例条件或异常",
|
||
then=["新增可选文学字段不改变普通实验原命令哈希,相同旧请求保持幂等"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
def test_新增可选文学字段不改变已有普通实验的命令身份__25d00e(执行环境):
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
env = 执行环境
|
||
# 旧客户端没有这两个可选字段;原存储哈希必须能被同一请求命中。
|
||
legacy_request = env["request"].model_dump(mode="json")
|
||
# 旧客户端不会发送这些后加的字段;默认值不构成新的命令身份。
|
||
for 字段 in (
|
||
"comparison_profile",
|
||
"arbitrator",
|
||
"calibration_policy",
|
||
"evaluation_goal",
|
||
"calibration_use",
|
||
"effect_policy",
|
||
"application_scope",
|
||
"effect_dependency_version",
|
||
"reuse_experiments",
|
||
):
|
||
legacy_request.pop(字段, None)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
stored = conn.execute(
|
||
"SELECT request_hash FROM evaluation.muse_experiment WHERE experiment_id=%s",
|
||
(env["exp"]["experiment_id"],),
|
||
).fetchone()[0]
|
||
assert stored == 固定哈希(legacy_request)
|
||
repeated = env["app"].要求评测().创建实验(env["actor"], "runtime-fixture", env["request"])
|
||
assert repeated["experiment_id"] == env["exp"]["experiment_id"]
|
||
assert not env["received"]
|