实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
495 lines
20 KiB
Python
495 lines
20 KiB
Python
"""从完整隔离生成、标定、独立评委和检测得出效果判据,不认证真实模型收益。"""
|
||
|
||
import json
|
||
from collections import Counter
|
||
from dataclasses import replace
|
||
|
||
import pytest
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
import test_评测语义检测 as 检测测试
|
||
|
||
from muse.共享.调用身份 import 用途
|
||
from muse.效果评测.接口 import 实验请求, 数据集发布, 评测服务, 评测错误, 金标准发布
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
场景 = [
|
||
"battle",
|
||
"character_dialogue",
|
||
"turning_point",
|
||
"information_reveal",
|
||
"returning_character",
|
||
]
|
||
标定策略 = {
|
||
"minimum_samples": 5,
|
||
"minimum_source_groups": 5,
|
||
"max_mae": 0.5,
|
||
"max_absolute_error": 1.5,
|
||
"minimum_verdict_agreement": 1.0,
|
||
}
|
||
|
||
|
||
def _基分(text, control):
|
||
return 8.0 if "细雨" in text and control["gain"] else 7.5
|
||
|
||
|
||
def _响应工厂():
|
||
control = {"gain": True, "calibrating": False, "judge_index": 0, "high": False}
|
||
|
||
def generate(request, script):
|
||
from muse.正文写作.接口 import 生成正文模板
|
||
|
||
public = json.loads(request["input"])
|
||
action = (
|
||
"他迎着细雨走到门前。"
|
||
if 生成正文模板()[0] in request["instructions"]
|
||
else "他站在山门旁等候。"
|
||
)
|
||
return {"paragraphs": [{"text": public["instruction"] + "。" + action}]}
|
||
|
||
def judge(material, script):
|
||
result = 文学测试._输出(material, "ok")
|
||
offset = 0.0
|
||
if control["calibrating"]:
|
||
index = control["judge_index"]
|
||
if index < 10:
|
||
offset = 0.5 if index % 2 == 0 else -0.5
|
||
control["judge_index"] += 1
|
||
for card in result["candidate_scores"]:
|
||
score = _基分(material[card["side"]]["text"], control) + offset
|
||
for row in card["scores"]:
|
||
row["score"] = score
|
||
return result
|
||
|
||
def detect(material, script):
|
||
return 检测测试._输出(
|
||
material, "high" if control["high"] and "细雨" in material["candidate"] else script
|
||
)
|
||
|
||
return {
|
||
"fixture_control": control,
|
||
"generator_output": generate,
|
||
"judge_output": judge,
|
||
"detector_output": detect,
|
||
}
|
||
|
||
|
||
参数 = {
|
||
**检测测试.参数,
|
||
"output_factory": _响应工厂,
|
||
"budget_calls": 300,
|
||
"budget_amount": "100",
|
||
"max_steps": 300,
|
||
}
|
||
|
||
|
||
def _数据样本(count, *, calibration=False):
|
||
locale = "林道" if calibration else "渡口"
|
||
rows = []
|
||
for i in range(count):
|
||
rows.append(
|
||
{
|
||
"sample_id": f"{locale}-{i}",
|
||
"source_ref": f"synthetic:{locale}-{i}",
|
||
"license_ref": "synthetic:owned",
|
||
"source_groups": [f"private:{locale}-work-{i if calibration else i // 5}"],
|
||
"stratification": {
|
||
"work_ref": f"private:{locale}-work-{i if calibration else i // 5}",
|
||
"annotation_ref": "synthetic:annotation",
|
||
"new_character_ratio": [0.0, 0.25, 0.75, None, 0.25][i % 5],
|
||
},
|
||
"split": "calibration" if calibration else "holdout",
|
||
"input": {
|
||
"instruction": f"{locale}第{i}处的行动",
|
||
"original": f"{locale}{i}的原始底稿。",
|
||
"context": {},
|
||
},
|
||
"answer": {
|
||
"judging_basis": {
|
||
**文学测试.依据,
|
||
"scenario": 场景[i % 5],
|
||
"sources": [
|
||
{
|
||
"source_id": "outline",
|
||
"kind": "fine_outline",
|
||
"text": f"{locale}第{i}处细纲。",
|
||
},
|
||
{
|
||
"source_id": "history",
|
||
"kind": "historical_prose",
|
||
"text": f"{locale}第{i}处先前正文。",
|
||
},
|
||
],
|
||
}
|
||
},
|
||
}
|
||
)
|
||
return rows
|
||
|
||
|
||
def _数据(env, count, *, calibration=False):
|
||
locale = "林道" if calibration else "渡口"
|
||
return 评测服务(env["pools"][用途.维护]).发布数据集(
|
||
replace(env["actor"], 用途=用途.维护),
|
||
数据集发布.model_validate(
|
||
{
|
||
"dataset_id": f"effect-{locale}",
|
||
"revision": 1,
|
||
"samples": _数据样本(count, calibration=calibration),
|
||
}
|
||
),
|
||
)
|
||
|
||
|
||
def _新实验(env, count=5, *, calibration=False, certificate=None):
|
||
data = _数据(env, count, calibration=calibration)
|
||
req = 实验请求.model_validate(
|
||
{
|
||
**env["request"].model_dump(mode="json"),
|
||
"dataset_version_id": data["version_id"],
|
||
"dataset_hash": data["public_hash"],
|
||
"split": "calibration" if calibration else "holdout",
|
||
"max_cost_usd": "20",
|
||
"detector": None if calibration else env["request"].detector.model_dump(mode="json"),
|
||
"calibration_policy": 标定策略 if calibration else None,
|
||
"evaluation_goal": "qualification" if certificate else "diagnostic",
|
||
"calibration_use": {
|
||
"policy": 标定策略,
|
||
"references": [
|
||
{
|
||
"experiment_id": certificate["result"]["experiment_id"],
|
||
"receipt_hash": certificate["receipt_hash"],
|
||
}
|
||
],
|
||
}
|
||
if certificate
|
||
else None,
|
||
"effect_policy": None if calibration else "writer-effect-v1",
|
||
}
|
||
)
|
||
exp = (
|
||
env["app"]
|
||
.要求评测()
|
||
.创建实验(env["actor"], "effect-calibration" if calibration else "effect-holdout", req)
|
||
)
|
||
return {**env, "exp": exp, "request": req}
|
||
|
||
|
||
def _完成(env):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
return env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
def _校准(env):
|
||
cal = _新实验(env, calibration=True)
|
||
control = env["fixture_control"]
|
||
control.update(calibrating=True, judge_index=0)
|
||
运行测试._启动执行(cal)
|
||
运行测试._运行就绪(cal)
|
||
annotations = []
|
||
for unit in 文学测试._读(cal)["units"]:
|
||
if unit["kind"] != "generation":
|
||
continue
|
||
score = _基分("\n".join(p["text"] for p in unit["output"]["paragraphs"]), control)
|
||
annotations.append(
|
||
{
|
||
"unit_id": unit["unit_id"],
|
||
"output_hash": unit["evidence"]["structured_output_hash"],
|
||
"scores": {d: score for d in 文学测试.维度},
|
||
"assertions": {"guard": "pass"},
|
||
"constraints": {"stay": "pass"},
|
||
}
|
||
)
|
||
评测服务(env["pools"][用途.维护]).发布标定金标准(
|
||
replace(env["actor"], 用途=用途.维护),
|
||
金标准发布.model_validate(
|
||
{
|
||
"experiment_id": cal["exp"]["experiment_id"],
|
||
"approval_ref": "synthetic:fixed-labels",
|
||
"annotations": annotations,
|
||
}
|
||
),
|
||
)
|
||
文学测试._推进(cal)
|
||
运行测试._运行就绪(cal)
|
||
文学测试._推进(cal)
|
||
运行测试._运行就绪(cal)
|
||
cert = env["app"].要求评测().封存标定(env["actor"], cal["exp"]["experiment_id"])
|
||
assert cert["result"]["status"] == "passed" and control["judge_index"] == 15
|
||
control["calibrating"] = False
|
||
return cert
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-64bcf0ac927f",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=[
|
||
"真实检测高严重度交付导致A阶段failed",
|
||
"目标每例高严重度计数1及target_semantic_failure留存,不封存合格凭据",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实检测失败进入A阶段并阻止效果签章__25f401(执行环境):
|
||
env = _新实验(执行环境)
|
||
env["fixture_control"]["high"] = True
|
||
result = _完成(env)
|
||
assert result["assessment"]["gate_a"] == {
|
||
"status": "failed",
|
||
"reasons": ["target_semantic_failure"],
|
||
}
|
||
assert result["receipt_id"] is None
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert all(
|
||
s["detections"]["treatment"]["report"]["high_severity_count"] == 1
|
||
for s in report["samples"]
|
||
)
|
||
with pytest.raises(评测错误, match="效果未通过"):
|
||
env["app"].要求评测().封存效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
assert len(env["received"]) == 30
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-a089d73880d8",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=["一例真实生成失败时A阶段为failed", "首要原因system_failure先于样本和场景不足"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_实际系统失败优先于五场景不足__25f402(执行环境):
|
||
env = _新实验(执行环境, 1)
|
||
env["scripted"].append("bad_output")
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
result = env["app"].要求评测().读取效果判据(env["actor"], env["exp"]["experiment_id"])
|
||
assert result["assessment"]["gate_a"] == {"status": "failed", "reasons": ["system_failure"]}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-e61898aba4e5",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=[
|
||
"单作品五场景A阶段仍passed",
|
||
"single_work单独标注且作品样本数量为唯一5例",
|
||
"完整五场景不误报scenario_coverage_incomplete",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_单作品完整A阶段仍保留选择偏差__25f403(执行环境):
|
||
env = _新实验(执行环境)
|
||
result = _完成(env)
|
||
assert result["assessment"]["gate_a"]["status"] == "passed"
|
||
assert "single_work" in result["assessment"]["confounders"]
|
||
assert list(result["assessment"]["metrics"]["work_counts"].values()) == [5]
|
||
assert "scenario_coverage_incomplete" not in result["assessment"]["confounders"]
|
||
assert result["assessment"]["gate_b"]["status"] == "insufficient_evidence"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-1569749609e7",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=[
|
||
"实际标定与两作品留出集效果passed",
|
||
"不可变PG凭据存在并关联原目标",
|
||
"重复读回及HTTP、CLI与原凭据哈希一致",
|
||
"对派生凭据篡改且重签也因逐例重算不符被拒绝",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.timeout(900)
|
||
@pytest.mark.慢
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_完整原标定和留出实验通过并封存可复算凭据__25f404(执行环境, monkeypatch, tmp_path):
|
||
import copy
|
||
import subprocess
|
||
import sys
|
||
|
||
import psycopg
|
||
import test_生产评测权限隔离 as 数据测试
|
||
from fastapi.testclient import TestClient
|
||
|
||
import muse.效果评测.启用凭据 as 启用凭据模块
|
||
import muse.效果评测.启用判据 as effects
|
||
import muse.效果评测.接口 as 评测接口模块
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.效果评测.存储 import 评测存储
|
||
from muse.正式变更.接口 import 固定哈希
|
||
from muse.配置 import 读取配置
|
||
|
||
env = 执行环境
|
||
cert = _校准(env)
|
||
dest = _新实验(env, 10, certificate=cert)
|
||
result = _完成(dest)
|
||
assert result["assessment"]["gate_b"]["status"] == "passed"
|
||
assert result["assessment"]["metrics"]["average_deltas"]["setting_entity_fidelity"] == 0.5
|
||
svc = env["app"].要求评测()
|
||
eid = dest["exp"]["experiment_id"]
|
||
policy_reader = effects.读取效果标准
|
||
|
||
def 改动后标准(version):
|
||
return {**policy_reader(version), "maximum_unstable_ratio": 0.19}
|
||
|
||
# 各消费模块在导入时直接绑定了该函数,改动要同时落到实际使用它的模块。
|
||
for 模块 in (effects, 启用凭据模块, 评测接口模块):
|
||
monkeypatch.setattr(模块, "读取效果标准", 改动后标准)
|
||
assert not svc.读取效果判据(env["actor"], eid)["current_policy"]
|
||
with pytest.raises(评测错误, match="标准过期"):
|
||
svc.封存效果判据(env["actor"], eid)
|
||
for 模块 in (effects, 启用凭据模块, 评测接口模块):
|
||
monkeypatch.setattr(模块, "读取效果标准", policy_reader)
|
||
sealed = svc.封存效果判据(env["actor"], eid)
|
||
assert sealed["receipt_id"] and svc.封存效果判据(env["actor"], eid) == sealed
|
||
assert sealed["assessment"]["target"] == dest["exp"]["conditions"]["target"]
|
||
assert sealed["receipt_hash"] == 固定哈希(sealed["assessment"])
|
||
with (
|
||
env["pools"][用途.生产].连接(只读=True) as conn,
|
||
pytest.raises(psycopg.errors.InsufficientPrivilege),
|
||
):
|
||
conn.execute("SELECT * FROM evaluation.muse_effect_assessment")
|
||
with env["pools"][用途.维护].连接() as conn, pytest.raises(psycopg.Error):
|
||
conn.execute("UPDATE evaluation.muse_effect_assessment SET payload=payload")
|
||
report = svc.读取实验报告(env["actor"], eid)
|
||
assert report["effect"] == sealed and report["activation_status"] == "not_evaluated"
|
||
assert sealed["assessment"]["validation_modes"] == ["offline_contract"]
|
||
config = 数据测试._配置文件(env["pool"], tmp_path)
|
||
http = 创建应用(读取配置(config))
|
||
with TestClient(http, headers={"origin": "http://testserver"}) as client:
|
||
http.state.装配 = env["app"]
|
||
assert (
|
||
client.post(
|
||
"/api/v1/session", json={"password": "synthetic-evaluation-only"}
|
||
).status_code
|
||
== 200
|
||
)
|
||
observed = client.get(f"/api/v1/evaluation/experiments/{eid}/effect")
|
||
assert observed.status_code == 200 and observed.json() == sealed
|
||
command = subprocess.run(
|
||
[sys.executable, "-I", "-m", "muse", "评测", str(config), "效果判据", eid],
|
||
cwd=tmp_path,
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=45,
|
||
)
|
||
assert command.returncode == 0, command.stderr
|
||
assert json.loads(command.stdout) == sealed
|
||
with pytest.raises(评测错误):
|
||
svc.读取效果判据(replace(env["actor"], 作者="foreign-author"), eid)
|
||
read = 评测存储.读取效果凭据
|
||
|
||
def changed(self, experiment_id):
|
||
row = copy.deepcopy(read(self, experiment_id))
|
||
if row:
|
||
row["payload"]["gate_b"]["status"] = "failed"
|
||
row["payload_hash"] = 固定哈希(row["payload"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取效果凭据", changed)
|
||
with pytest.raises(评测错误, match="逐例重算"):
|
||
svc.读取效果判据(env["actor"], eid)
|
||
monkeypatch.setattr(评测存储, "读取效果凭据", read)
|
||
svc.取消实验(env["actor"], eid, "stop-after-effect")
|
||
old = svc.封存效果判据(env["actor"], eid)
|
||
assert old["stopped"] and old["assessment"] == sealed["assessment"]
|
||
svc.取消实验(env["actor"], cert["result"]["experiment_id"], "stop-source-after-effect")
|
||
historical = svc.读取效果判据(env["actor"], eid)
|
||
assert historical["calibration_stopped"] == [cert["result"]["experiment_id"]]
|
||
assert historical["assessment"] == sealed["assessment"]
|
||
assert len(env["received"]) == 85
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f405",
|
||
environment="隔离PG与合成HTTP",
|
||
given="固定分母、独立样本及本例边界输入",
|
||
when="经实际S02生成后重复读回效果报告",
|
||
then=["单次读回每份已核验交付只验一次;下一次读回全部重新核验"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_单次读取复用已核验交付而新读取仍重验__25f405(执行环境, monkeypatch):
|
||
env = _新实验(执行环境)
|
||
_完成(env)
|
||
runtime = env["app"].任务运行
|
||
original = runtime.核对历史结构化交付于
|
||
calls = Counter()
|
||
|
||
def count(*args, **kwargs):
|
||
calls[args[4]] += 1
|
||
return original(*args, **kwargs)
|
||
|
||
monkeypatch.setattr(runtime, "核对历史结构化交付于", count)
|
||
first = 文学测试._读(env)
|
||
assert len(calls) == 30 and set(calls.values()) == {1}
|
||
calls.clear()
|
||
assert 文学测试._读(env) == first
|
||
assert len(calls) == 30 and set(calls.values()) == {1}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-0fe403cb0320",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=[
|
||
"真实合格标定后的无增益留出集返回no_gain",
|
||
"封存拒绝且公开凭据为空",
|
||
"隔离PG效果凭据表无记录,代替旧文件不存在断言",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.timeout(900)
|
||
@pytest.mark.慢
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实无增益留出实验不产生合格凭据__25f406(执行环境):
|
||
env = 执行环境
|
||
env["fixture_control"]["gain"] = False
|
||
cert = _校准(env)
|
||
dest = _新实验(env, 10, certificate=cert)
|
||
result = _完成(dest)
|
||
assert result["assessment"]["gate_b"]["status"] == "no_gain"
|
||
with pytest.raises(评测错误, match="效果未通过"):
|
||
env["app"].要求评测().封存效果判据(env["actor"], dest["exp"]["experiment_id"])
|
||
assert result["receipt_id"] is None and len(env["received"]) == 85
|
||
assert (
|
||
env["app"].要求评测().读取效果判据(env["actor"], dest["exp"]["experiment_id"])["receipt_id"]
|
||
is None
|
||
)
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert (
|
||
conn.execute("SELECT count(*) FROM evaluation.muse_effect_assessment").fetchone()[0]
|
||
== 0
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-602d0bb37aa2",
|
||
environment="隔离PG、实际S02与合成HTTP;不认证外部模型文学收益",
|
||
given="本例固定样本、场景与独立异常,不共享其他用例的执行结果",
|
||
when="经真实生成、独立检测、比较与公开效果入口读回",
|
||
then=["公开入口只接实验ID;手工汇总对象以UUID合同拒绝,无模型调用"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_公开效果入口拒绝手工汇总对象__25f407(执行环境):
|
||
env = 执行环境
|
||
with pytest.raises(评测错误, match="UUID"):
|
||
env["app"].要求评测().读取效果判据(
|
||
env["actor"], {"gate": "A", "samples": [], "passed": True}
|
||
)
|
||
assert not env["received"]
|