- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
206 lines
8.5 KiB
Python
206 lines
8.5 KiB
Python
"""隔离S02实际检测回合、拒绝及恢复;合成模型不认证语义准确率。"""
|
||
|
||
import copy
|
||
import json
|
||
|
||
import pytest
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
|
||
from muse.效果评测.接口 import 交付补登请求, 评测错误
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
|
||
|
||
def _输出(material, script):
|
||
result = {
|
||
"claims": [
|
||
{
|
||
"claim_id": "observed",
|
||
"text": "合成角色行动。",
|
||
"candidate_quote": material["candidate"],
|
||
"state": "supported",
|
||
"evidence_refs": ["source:outline"],
|
||
"reason": "根据共同细纲核对。",
|
||
}
|
||
],
|
||
"findings": [],
|
||
"new_setting_candidates": [],
|
||
}
|
||
for field, source, prefix in (
|
||
("assertion_verdicts", "required_assertions", "assertion"),
|
||
("constraint_verdicts", "required_constraints", "constraint"),
|
||
):
|
||
result[field] = [
|
||
{
|
||
"statement_id": sid,
|
||
"verdict": "unknown" if script == "unknown" else "pass",
|
||
"candidate_quote": material["candidate"],
|
||
"evidence_refs": [prefix + ":" + sid],
|
||
"reason": "合成判断;未知表示材料不足。",
|
||
}
|
||
for sid in material[source]
|
||
]
|
||
if script == "high":
|
||
result["findings"] = [
|
||
{
|
||
"finding_id": "conflict",
|
||
"category": "fact",
|
||
"severity": "high",
|
||
"candidate_quote": material["candidate"],
|
||
"evidence_refs": ["assertion:guard"],
|
||
"message": "此合成候选与已给事实冲突。",
|
||
}
|
||
]
|
||
if script == "missing":
|
||
result["constraint_verdicts"] = []
|
||
return result
|
||
|
||
|
||
参数 = {**文学测试.参数, "detector": True, "detector_output": _输出, "calls": 7}
|
||
|
||
|
||
def _执行(env, script="ok", *, allow_failure=False):
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
env["scripted"].extend([script, "ok"])
|
||
运行测试._运行就绪(env, allow_failure=allow_failure)
|
||
文学测试._推进(env)
|
||
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实独立检测绑定原候选且模型不见其他臂与答案__25f201(执行环境):
|
||
env = 执行环境
|
||
report = _执行(env, "high")
|
||
sample = report["samples"][0]
|
||
assert sample["detection_complete"] and len(sample["detections"]) == 2
|
||
assert sorted(r["report"]["status"] for r in sample["detections"].values()) == [
|
||
"failed",
|
||
"passed",
|
||
]
|
||
assert sum(r["report"]["high_severity_count"] for r in sample["detections"].values()) == 1
|
||
assert len(env["received"]) == report["cost"]["sent_calls"] == 6
|
||
assert report["activation_status"] == "not_evaluated"
|
||
requests = [r for r in env["received"] if r["model"] == "synthetic-semantic-detector"]
|
||
assert len(requests) == 2
|
||
assert {json.loads(r["input"])["candidate"] for r in requests} == {"合成正文1。", "合成正文2。"}
|
||
for request in requests:
|
||
material = json.loads(request["input"])
|
||
assert set(material) == {
|
||
"candidate",
|
||
"evidence",
|
||
"required_assertions",
|
||
"required_constraints",
|
||
}
|
||
assert "ORACLE-SECRET" not in request["input"] and "calibration" not in request["input"]
|
||
assert not request.get("tools") and not request.get("previous_response_id")
|
||
for row in sample["detections"].values():
|
||
assert row["report_hash"] == 固定哈希(row["report"])
|
||
assert row["task_id"] and row["call_id"] and row["candidate_output_hash"]
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_未知检测保留未决而不是零缺陷通过__25f202(执行环境):
|
||
report = _执行(执行环境, "unknown")
|
||
rows = list(report["samples"][0]["detections"].values())
|
||
uncertain = next(r for r in rows if r["report"]["status"] == "inconclusive")
|
||
assert uncertain["report"]["unknown_count"] == 2
|
||
assert uncertain["report"]["constraint_counts"] == {"pass": 0, "fail": 0, "unknown": 1}
|
||
assert report["literary_quality"]["metrics"] is None
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_缺少命题检测失败保留原交付费用且不自动纠正__25f203(执行环境):
|
||
env = 执行环境
|
||
report = _执行(env, "missing", allow_failure=True)
|
||
assert not report["samples"][0]["detection_complete"]
|
||
work = 文学测试._读(env)
|
||
failed = [r for r in work["units"] if r["kind"] == "detection" and r["state"] == "failed"]
|
||
assert len(failed) == 1 and failed[0]["unregistered_deliveries"]
|
||
assert len(env["received"]) == 6 and report["cost"]["total_usd"] is not None
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_重签零残留报告不能覆盖真实高严重度交付__25f204(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
_执行(env, "high")
|
||
read = 评测存储.读取交付
|
||
|
||
def changed(self, uid):
|
||
row = copy.deepcopy(read(self, uid))
|
||
if row and row["evidence"].get("detection"):
|
||
evidence = row["evidence"]["detection"]
|
||
evidence["report"].update(findings=[], high_severity_count=0, status="passed")
|
||
evidence["report_hash"] = 固定哈希(evidence["report"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取交付", changed)
|
||
with pytest.raises(评测错误, match="检测报告"):
|
||
文学测试._读(env)
|
||
assert len(env["received"]) == 6
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_检测登记中断后补原回执且恢复不重发__25f205(执行环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 执行环境
|
||
save = 评测存储.保存交付
|
||
broken = []
|
||
|
||
def fail_once(self, uid, *args):
|
||
if not broken and self.读取单元(uid)["kind"] == "detection":
|
||
broken.append(uid)
|
||
raise 评测错误("注入检测登记中断")
|
||
return save(self, uid, *args)
|
||
|
||
monkeypatch.setattr(评测存储, "保存交付", fail_once)
|
||
_执行(env, allow_failure=True)
|
||
monkeypatch.setattr(评测存储, "保存交付", save)
|
||
unit = next(r for r in 文学测试._读(env)["units"] if r["unit_id"] == broken[0])
|
||
original = next(r for r in env["received"] if r["model"] == "synthetic-semantic-detector")
|
||
request = 交付补登请求(
|
||
unit_id=unit["unit_id"],
|
||
call_id=unit["unregistered_deliveries"][0]["call_id"],
|
||
output=_输出(json.loads(original["input"]), "ok"),
|
||
)
|
||
svc = env["app"].要求评测()
|
||
first = svc.补登记交付(env["actor"], request)
|
||
assert svc.补登记交付(env["actor"], request) == first
|
||
运行测试._恢复失败单元(env)
|
||
运行测试._运行就绪(env)
|
||
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["samples"][0]["detection_complete"] and len(env["received"]) == 6
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [{**参数, "calls": 5}], indirect=True)
|
||
def test_新增检测必须纳入原预算不足时零任务__25f206(执行环境):
|
||
env = 执行环境
|
||
with pytest.raises(评测错误, match="调用上限"):
|
||
运行测试._启动执行(env)
|
||
assert not env["received"]
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_生成失败时该份检测不创建且完整分母保留__25f207(执行环境):
|
||
env = 执行环境
|
||
env["scripted"].append("bad_output")
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
work = 文学测试._读(env)
|
||
detectors = [u for u in work["units"] if u["kind"] == "detection"]
|
||
assert len(detectors) == 2 and sum(u["task_id"] is None for u in detectors) == 1
|
||
assert len(env["received"]) == 3
|
||
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
assert report["coverage"]["samples"] == 1 and not report["samples"][0]["detection_complete"]
|