- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
505 lines
19 KiB
Python
505 lines
19 KiB
Python
"""标定采用真实隔离S02回合及合成标注,不认证外部模型的文学质量。"""
|
||
|
||
import copy
|
||
import json
|
||
from dataclasses import replace
|
||
|
||
import psycopg
|
||
import pytest
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.效果评测.接口 import 评测服务, 评测错误, 金标准发布
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
策略 = {
|
||
"minimum_samples": 1,
|
||
"minimum_source_groups": 1,
|
||
"max_mae": 0.5,
|
||
"max_absolute_error": 1.5,
|
||
"minimum_verdict_agreement": 1.0,
|
||
}
|
||
参数 = {**文学测试.参数, "calibration_policy": 策略}
|
||
|
||
|
||
def _生成(env):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
|
||
|
||
def _请求(env):
|
||
work = 文学测试._读(env)
|
||
return {
|
||
"experiment_id": env["exp"]["experiment_id"],
|
||
"approval_ref": "synthetic-human-label",
|
||
"annotations": [
|
||
{
|
||
"unit_id": u["unit_id"],
|
||
"output_hash": u["evidence"]["structured_output_hash"],
|
||
"scores": {d: 8.0 for d in 文学测试.维度},
|
||
"assertions": {"guard": "pass"},
|
||
"constraints": {"stay": "pass"},
|
||
}
|
||
for u in work["units"]
|
||
if u["kind"] == "generation"
|
||
],
|
||
}
|
||
|
||
|
||
def _发布(env, raw=None):
|
||
return 评测服务(env["pools"][用途.维护]).发布标定金标准(
|
||
replace(env["actor"], 用途=用途.维护), 金标准发布.model_validate(raw or _请求(env))
|
||
)
|
||
|
||
|
||
def _评分(env):
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
|
||
|
||
def _读(env):
|
||
return env["app"].要求评测().读取标定(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _封存(env):
|
||
return env["app"].要求评测().封存标定(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_金标准先于评委冻结且不进入模型输入__25e001(执行环境):
|
||
env = 执行环境
|
||
_生成(env)
|
||
waiting = 文学测试._推进(env)
|
||
assert waiting["state"] == "awaiting_gold" and len(env["received"]) == 2
|
||
assert _读(env)["result"]["reasons"] == ["gold_missing"]
|
||
published = _发布(env)
|
||
assert not published["duplicate"] and _发布(env)["duplicate"]
|
||
_评分(env)
|
||
result = _封存(env)
|
||
assert result["result"]["status"] == "passed"
|
||
assert result["result"]["experiment_id"] == env["exp"]["experiment_id"]
|
||
assert result["result"]["conditions_hash"] == env["exp"]["conditions_hash"]
|
||
assert result["result"]["metrics"]["n"] == 20
|
||
assert result["result"]["metrics"]["mae"] == 0
|
||
assert result["result"]["metrics"]["max_abs"] == 0
|
||
assert result["result"]["metrics"]["verdict_count"] == 8
|
||
assert result["result"]["validation_modes"] == ["offline_contract"]
|
||
assert result["receipt_hash"] == 固定哈希(result["result"])
|
||
assert _封存(env) == _读(env) == result
|
||
assert len(env["received"]) == 4
|
||
for data in env["received"]:
|
||
assert "annotations" not in data["input"] and "synthetic-human-label" not in data["input"]
|
||
assert published["gold_hash"] not in data["input"]
|
||
work = 文学测试._读(env)
|
||
assert all(
|
||
u["evidence"]["runtime"]["dispatch_metadata"]["calibration_gold_hash"]
|
||
== published["gold_hash"]
|
||
for u in work["units"]
|
||
if u["kind"] == "comparison" and u["evidence"]
|
||
)
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["calibration"] == result
|
||
assert report["report_hash"] == 固定哈希(
|
||
{k: v for k, v in report.items() if k != "report_hash"}
|
||
)
|
||
assert report["literary_quality"]["metrics"] is None
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"variant,reason",
|
||
[
|
||
("mae", "score_error_exceeded"),
|
||
("max", "score_error_exceeded"),
|
||
("verdict", "verdict_agreement_low"),
|
||
],
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_均差最坏单项和判定分歧分别保留失败凭据__25e002(执行环境, variant, reason):
|
||
env = 执行环境
|
||
_生成(env)
|
||
raw = _请求(env)
|
||
if variant == "mae":
|
||
for row in raw["annotations"]:
|
||
row["scores"] = {d: 7.0 for d in 文学测试.维度}
|
||
elif variant == "max":
|
||
raw["annotations"][0]["scores"][文学测试.维度[0]] = 6.5
|
||
else:
|
||
raw["annotations"][0]["assertions"]["guard"] = "fail"
|
||
_发布(env, raw)
|
||
_评分(env)
|
||
result = _封存(env)
|
||
assert result["result"]["status"] == "failed"
|
||
assert reason in result["result"]["reasons"]
|
||
assert result["receipt_id"] and _读(env) == result
|
||
if variant == "max":
|
||
assert result["result"]["metrics"]["mae"] < 0.5
|
||
assert result["result"]["metrics"]["max_abs"] == 1.5
|
||
|
||
|
||
@pytest.mark.parametrize("variant", ["missing", "duplicate", "text", "dimension", "assertion"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_金标准必须绑定全部实际文本和完整字段__25e003(执行环境, variant):
|
||
env = 执行环境
|
||
_生成(env)
|
||
raw = _请求(env)
|
||
if variant == "missing":
|
||
raw["annotations"].pop()
|
||
elif variant == "duplicate":
|
||
raw["annotations"].append(copy.deepcopy(raw["annotations"][0]))
|
||
elif variant == "text":
|
||
raw["annotations"][0]["output_hash"] = "0" * 64
|
||
elif variant == "dimension":
|
||
raw["annotations"][0]["scores"].pop(文学测试.维度[0])
|
||
else:
|
||
raw["annotations"][0]["assertions"] = {}
|
||
with pytest.raises(评测错误):
|
||
_发布(env, raw)
|
||
assert _读(env)["result"]["reasons"] == ["gold_missing"]
|
||
assert 文学测试._推进(env)["state"] == "awaiting_gold"
|
||
assert len(env["received"]) == 2
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_标注与凭据不可覆盖且数据库用途不越权__25e004(执行环境):
|
||
env = 执行环境
|
||
_生成(env)
|
||
raw = _请求(env)
|
||
with pytest.raises(评测错误):
|
||
env["app"].要求评测().发布标定金标准(env["actor"], 金标准发布.model_validate(raw))
|
||
for statement in (
|
||
"INSERT INTO oracle.muse_calibration_gold SELECT * FROM oracle.muse_calibration_gold",
|
||
"UPDATE oracle.muse_calibration_gold SET payload=payload",
|
||
):
|
||
with pytest.raises(psycopg.Error), env["pool"].连接() as conn:
|
||
conn.execute(statement)
|
||
_发布(env, raw)
|
||
_评分(env)
|
||
_封存(env)
|
||
raw["annotations"][0]["scores"][文学测试.维度[0]] = 7.0
|
||
with pytest.raises(评测错误, match="不可替换"):
|
||
_发布(env, raw)
|
||
for table in ("oracle.muse_calibration_gold", "evaluation.muse_calibration_receipt"):
|
||
with pytest.raises(psycopg.Error), env["pools"][用途.生产].连接(只读=True) as conn:
|
||
conn.execute("SELECT * FROM " + table)
|
||
with pytest.raises(psycopg.Error), env["pools"][用途.维护].连接() as conn:
|
||
conn.execute("UPDATE " + table + " SET payload=payload")
|
||
|
||
|
||
@pytest.mark.parametrize("variant", ["gold", "receipt"])
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_本地重签不能替代原派发或原标定观察__25e005(执行环境, monkeypatch, variant):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
_生成(env)
|
||
_发布(env)
|
||
_评分(env)
|
||
_封存(env)
|
||
name = "读取金标准" if variant == "gold" else "读取标定凭据"
|
||
original = getattr(评测存储, name)
|
||
|
||
def changed(self, eid):
|
||
row = copy.deepcopy(original(self, eid))
|
||
if row:
|
||
if variant == "gold":
|
||
row["payload"]["request"]["annotations"][0]["scores"][文学测试.维度[0]] = 7.0
|
||
else:
|
||
row["payload"]["metrics"]["mae"] = 0.1
|
||
row["payload_hash"] = 固定哈希(row["payload"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, name, changed)
|
||
with pytest.raises(评测错误):
|
||
_读(env)
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
@pytest.mark.parametrize("sealed", [False, True])
|
||
def test_停止保留已封存标定但不允许新的凭据__25e006(执行环境, sealed):
|
||
env = 执行环境
|
||
_生成(env)
|
||
_发布(env)
|
||
_评分(env)
|
||
before = _封存(env) if sealed else _读(env)
|
||
env["app"].要求评测().取消实验(
|
||
env["actor"], env["exp"]["experiment_id"], "stop-after-calibration"
|
||
)
|
||
after = _读(env)
|
||
assert after["stopped"] and after["result"] == before["result"]
|
||
assert after["receipt_hash"] == before["receipt_hash"]
|
||
assert _发布(env)["duplicate"]
|
||
if not sealed:
|
||
with pytest.raises(评测错误, match="停止"):
|
||
_封存(env)
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"执行环境",
|
||
[{**参数, "calibration_policy": {**策略, "minimum_source_groups": 2}}],
|
||
indirect=True,
|
||
)
|
||
def test_来源不足仍保留在分母且不签标定凭据__25e007(执行环境):
|
||
env = 执行环境
|
||
_生成(env)
|
||
_发布(env)
|
||
_评分(env)
|
||
result = _读(env)
|
||
assert result["result"]["status"] == "incomplete"
|
||
assert "source_groups_insufficient" in result["result"]["reasons"]
|
||
assert result["result"]["metrics"] is None
|
||
with pytest.raises(评测错误, match="未完成"):
|
||
_封存(env)
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_CLI维护标注与HTTP封存读回沿同一标定__25e00b(执行环境, tmp_path):
|
||
import subprocess
|
||
import sys
|
||
|
||
import test_生产评测权限隔离 as 数据测试
|
||
from fastapi.testclient import TestClient
|
||
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.配置 import 读取配置
|
||
|
||
env = 执行环境
|
||
_生成(env)
|
||
configs = {p: 数据测试._配置文件(pool, tmp_path) for p, pool in env["pools"].items()}
|
||
source = tmp_path / "gold.json"
|
||
source.write_text(json.dumps(_请求(env), ensure_ascii=False))
|
||
|
||
def run(purpose, action, target):
|
||
return subprocess.run(
|
||
[
|
||
sys.executable,
|
||
"-I",
|
||
"-m",
|
||
"muse",
|
||
"评测",
|
||
str(configs[purpose]),
|
||
action,
|
||
str(target),
|
||
],
|
||
cwd=tmp_path,
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=30,
|
||
)
|
||
|
||
denied = run(用途.评测, "发布金标准", source)
|
||
assert denied.returncode == 1 and "MUSE_PURPOSE_VIOLATION" in denied.stderr
|
||
written = run(用途.维护, "发布金标准", source)
|
||
assert written.returncode == 0, written.stderr
|
||
assert not json.loads(written.stdout)["duplicate"]
|
||
_评分(env)
|
||
http = 创建应用(读取配置(configs[用途.评测]))
|
||
endpoint = "/api/v1/evaluation/experiments/" + env["exp"]["experiment_id"] + "/calibration"
|
||
with TestClient(http, headers={"origin": "http://testserver"}) as client:
|
||
http.state.装配 = env["app"]
|
||
assert (
|
||
client.post(
|
||
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
|
||
).status_code
|
||
== 200
|
||
)
|
||
before = client.get(endpoint)
|
||
assert before.status_code == 200 and before.json()["receipt_id"] is None
|
||
denied = client.post("/api/v1/evaluation/calibration/gold", json=_请求(env))
|
||
assert denied.status_code >= 400
|
||
sealed = client.post(endpoint + "/seal", json={})
|
||
assert sealed.status_code == 200, sealed.text
|
||
assert client.get(endpoint).json() == sealed.json()
|
||
observed = run(用途.评测, "标定", env["exp"]["experiment_id"])
|
||
assert observed.returncode == 0, observed.stderr
|
||
assert json.loads(observed.stdout) == sealed.json()
|
||
replayed = run(用途.评测, "封存标定", env["exp"]["experiment_id"])
|
||
assert replayed.returncode == 0 and json.loads(replayed.stdout) == sealed.json()
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_显式标定分割与固定策略不能用批次名字或旧命令绕过__25e00c(执行环境):
|
||
from uuid import UUID
|
||
|
||
import test_生产评测权限隔离 as 数据测试
|
||
|
||
from muse.效果评测.接口 import 标定策略
|
||
|
||
env = 执行环境
|
||
assert env["exp"]["conditions"]["split"] == "calibration"
|
||
assert len(env["exp"]["conditions_hash"]) == 64
|
||
assert env["app"].要求评测().读取实验(env["actor"], env["exp"]["experiment_id"]) == env["exp"]
|
||
changed = env["request"].model_copy(
|
||
update={"calibration_policy": 标定策略.model_validate({**策略, "max_mae": 0.4})}
|
||
)
|
||
with pytest.raises(评测错误, match="不能更换"):
|
||
env["app"].要求评测().创建实验(env["actor"], "runtime-fixture", changed)
|
||
receipt = 评测服务(env["pools"][用途.维护]).发布数据集(
|
||
replace(env["actor"], 用途=用途.维护), 数据测试._数据()
|
||
)
|
||
wrong = env["request"].model_copy(
|
||
update={
|
||
"dataset_version_id": UUID(receipt["version_id"]),
|
||
"dataset_hash": receipt["public_hash"],
|
||
"split": "holdout",
|
||
}
|
||
)
|
||
with pytest.raises(评测错误, match="calibration分割"):
|
||
env["app"].要求评测().创建实验(env["actor"], "cal-name-cannot-authorize", wrong)
|
||
assert not env["received"]
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_并发封存相同金标准只产生一份且推进使用原记录__25e00d(执行环境):
|
||
from concurrent.futures import ThreadPoolExecutor
|
||
from threading import Barrier
|
||
|
||
env = 执行环境
|
||
_生成(env)
|
||
raw = _请求(env)
|
||
barrier = Barrier(2)
|
||
|
||
def publish():
|
||
barrier.wait(timeout=5)
|
||
return _发布(env, raw)
|
||
|
||
with ThreadPoolExecutor(max_workers=2) as workers:
|
||
futures = [workers.submit(publish) for _ in range(2)]
|
||
results = [f.result(timeout=10) for f in futures]
|
||
assert sorted(r["duplicate"] for r in results) == [False, True]
|
||
assert len({r["gold_hash"] for r in results}) == 1
|
||
_评分(env)
|
||
assert _封存(env)["result"]["gold_hash"] == results[0]["gold_hash"]
|
||
assert len(env["received"]) == 4
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_金标准事务失败不留下半份标签且可原请求恢复__25e00e(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
_生成(env)
|
||
raw = _请求(env)
|
||
original = 评测存储.保存金标准
|
||
|
||
def failed(self, eid, payload):
|
||
original(self, eid, payload)
|
||
raise psycopg.Error("synthetic-after-insert")
|
||
|
||
with monkeypatch.context() as patch:
|
||
patch.setattr(评测存储, "保存金标准", failed)
|
||
with pytest.raises(评测错误, match="存储操作失败"):
|
||
_发布(env, raw)
|
||
assert _读(env)["result"]["gold_hash"] is None
|
||
assert 文学测试._推进(env)["state"] == "awaiting_gold"
|
||
assert not _发布(env, raw)["duplicate"]
|
||
_评分(env)
|
||
assert _封存(env)["result"]["status"] == "passed"
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_标定资格由实际分割和策略登记__fa0275(执行环境):
|
||
env = 执行环境
|
||
actual = env["app"].要求评测().读取实验(env["actor"], env["exp"]["experiment_id"])
|
||
assert actual["conditions"]["split"] == "calibration"
|
||
assert (
|
||
actual["conditions"]["calibration_policy"] == env["request"].calibration_policy.model_dump()
|
||
)
|
||
_生成(env)
|
||
assert 文学测试._推进(env)["state"] == "awaiting_gold"
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_非标定分割不因批次名字取得标定资格__6534ec(执行环境):
|
||
from uuid import UUID
|
||
|
||
import test_生产评测权限隔离 as 数据测试
|
||
|
||
env = 执行环境
|
||
data = 评测服务(env["pools"][用途.维护]).发布数据集(
|
||
replace(env["actor"], 用途=用途.维护), 数据测试._数据()
|
||
)
|
||
request = env["request"].model_copy(
|
||
update={
|
||
"dataset_version_id": UUID(data["version_id"]),
|
||
"dataset_hash": data["public_hash"],
|
||
"split": "holdout",
|
||
}
|
||
)
|
||
with pytest.raises(评测错误, match="calibration分割"):
|
||
env["app"].要求评测().创建实验(env["actor"], "cal-002", request)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert (
|
||
conn.execute(
|
||
"SELECT count(*) FROM evaluation.muse_experiment WHERE command_id='cal-002'"
|
||
).fetchone()[0]
|
||
== 0
|
||
)
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_标定指纹涵盖实际标准与配置并用SHA256保存__9a4006(执行环境):
|
||
env = 执行环境
|
||
with env["pool"].连接(只读=True) as conn:
|
||
conditions, fingerprint = conn.execute(
|
||
"SELECT conditions,conditions_hash FROM evaluation.muse_experiment "
|
||
"WHERE experiment_id=%s",
|
||
(env["exp"]["experiment_id"],),
|
||
).fetchone()
|
||
assert len(fingerprint) == 64 and fingerprint == 固定哈希(conditions)
|
||
assert conditions["generator"]["config_hash"]
|
||
assert conditions["calibration_policy"]
|
||
assert all(
|
||
v["rubric_hash"] == 固定哈希(v["rubric"]) for v in conditions["literary_basis"].values()
|
||
)
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_标定指纹重复读取稳定且调用方副本不改存储__89dd60(执行环境):
|
||
env = 执行环境
|
||
svc = env["app"].要求评测()
|
||
eid = env["exp"]["experiment_id"]
|
||
before = svc.读取实验(env["actor"], eid)
|
||
copy_of_result = svc.读取实验(env["actor"], eid)
|
||
copy_of_result["conditions"]["calibration_policy"]["max_mae"] = 9.0
|
||
after = svc.读取实验(env["actor"], eid)
|
||
assert after == before and after["conditions_hash"] == env["exp"]["conditions_hash"]
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_持久标定凭据读回相同实验批次__93f97c(执行环境):
|
||
env = 执行环境
|
||
_生成(env)
|
||
_发布(env)
|
||
_评分(env)
|
||
sealed = _封存(env)
|
||
assert _读(env)["result"]["experiment_id"] == env["exp"]["experiment_id"]
|
||
with env["pool"].连接(只读=True) as conn:
|
||
batch = conn.execute(
|
||
"SELECT experiment_id FROM evaluation.muse_calibration_receipt WHERE receipt_id=%s",
|
||
(sealed["receipt_id"],),
|
||
).fetchone()[0]
|
||
assert str(batch) == env["exp"]["experiment_id"]
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_标定凭据绑定封存时的原标准指纹__be8eed(执行环境):
|
||
env = 执行环境
|
||
fingerprint = env["exp"]["conditions_hash"]
|
||
_生成(env)
|
||
_发布(env)
|
||
_评分(env)
|
||
sealed = _封存(env)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
saved = conn.execute(
|
||
"SELECT payload FROM evaluation.muse_calibration_receipt WHERE receipt_id=%s",
|
||
(sealed["receipt_id"],),
|
||
).fetchone()[0]
|
||
assert saved["conditions_hash"] == fingerprint
|
||
assert saved["policy"] == env["exp"]["conditions"]["calibration_policy"]
|