muse-agent-example/tests/集成/test_评测语义检测.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

262 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""隔离S02实际检测回合、拒绝及恢复;合成模型不认证语义准确率。"""
import copy
import json
import pytest
import test_文学评分执行与条件第三 as 文学测试
import test_评测执行与失败收敛 as 运行测试
from muse.效果评测.接口 import 交付补登请求, 评测错误
from muse.正式变更.接口 import 固定哈希
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
def _输出(material, script):
result = {
"claims": [
{
"claim_id": "observed",
"text": "合成角色行动。",
"candidate_quote": material["candidate"],
"state": "supported",
"evidence_refs": ["source:outline"],
"reason": "根据共同细纲核对。",
}
],
"findings": [],
"new_setting_candidates": [],
}
for field, source, prefix in (
("assertion_verdicts", "required_assertions", "assertion"),
("constraint_verdicts", "required_constraints", "constraint"),
):
result[field] = [
{
"statement_id": sid,
"verdict": "unknown" if script == "unknown" else "pass",
"candidate_quote": material["candidate"],
"evidence_refs": [prefix + ":" + sid],
"reason": "合成判断;未知表示材料不足。",
}
for sid in material[source]
]
if script == "high":
result["findings"] = [
{
"finding_id": "conflict",
"category": "fact",
"severity": "high",
"candidate_quote": material["candidate"],
"evidence_refs": ["assertion:guard"],
"message": "此合成候选与已给事实冲突。",
}
]
if script == "missing":
result["constraint_verdicts"] = []
return result
参数 = {**文学测试.参数, "detector": True, "detector_output": _输出, "calls": 7}
def _执行(env, script="ok", *, allow_failure=False):
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
env["scripted"].extend([script, "ok"])
运行测试._运行就绪(env, allow_failure=allow_failure)
文学测试._推进(env)
return env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
@pytest.mark.case_id(
"NC-w25-25f201",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["实际detector与写手分离,每份候选及报告绑定S02,模型不见另一臂或整个答案"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_真实独立检测绑定原候选且模型不见其他臂与答案__25f201(执行环境):
env = 执行环境
report = _执行(env, "high")
sample = report["samples"][0]
assert sample["detection_complete"] and len(sample["detections"]) == 2
assert sorted(r["report"]["status"] for r in sample["detections"].values()) == [
"failed",
"passed",
]
assert sum(r["report"]["high_severity_count"] for r in sample["detections"].values()) == 1
assert len(env["received"]) == report["cost"]["sent_calls"] == 6
assert report["activation_status"] == "not_evaluated"
requests = [r for r in env["received"] if r["model"] == "synthetic-semantic-detector"]
assert len(requests) == 2
assert {json.loads(r["input"])["candidate"] for r in requests} == {"合成正文1。", "合成正文2。"}
for request in requests:
material = json.loads(request["input"])
assert set(material) == {
"candidate",
"evidence",
"required_assertions",
"required_constraints",
}
assert "ORACLE-SECRET" not in request["input"] and "calibration" not in request["input"]
assert not request.get("tools") and not request.get("previous_response_id")
for row in sample["detections"].values():
assert row["report_hash"] == 固定哈希(row["report"])
assert row["task_id"] and row["call_id"] and row["candidate_output_hash"]
@pytest.mark.case_id(
"NC-w25-25f202",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["未知检测保留原因及未决状态,不变为已知通过"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_未知检测保留未决而不是零缺陷通过__25f202(执行环境):
report = _执行(执行环境, "unknown")
rows = list(report["samples"][0]["detections"].values())
uncertain = next(r for r in rows if r["report"]["status"] == "inconclusive")
assert uncertain["report"]["unknown_count"] == 2
assert uncertain["report"]["constraint_counts"] == {"pass": 0, "fail": 0, "unknown": 1}
assert report["literary_quality"]["metrics"] is None
@pytest.mark.case_id(
"NC-w25-25f203",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["缺命题失败保留原S02回合与费用,不自动重发"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_缺少命题检测失败保留原交付费用且不自动纠正__25f203(执行环境):
env = 执行环境
report = _执行(env, "missing", allow_failure=True)
assert not report["samples"][0]["detection_complete"]
work = 文学测试._读(env)
failed = [r for r in work["units"] if r["kind"] == "detection" and r["state"] == "failed"]
assert len(failed) == 1 and failed[0]["unregistered_deliveries"]
assert len(env["received"]) == 6 and report["cost"]["total_usd"] is not None
@pytest.mark.case_id(
"NC-w25-25f204",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["重签零残留派生报告不能覆盖实际高严重度交付"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_重签零残留报告不能覆盖真实高严重度交付__25f204(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
_执行(env, "high")
read = 评测存储.读取交付
def changed(self, uid):
row = copy.deepcopy(read(self, uid))
if row and row["evidence"].get("detection"):
evidence = row["evidence"]["detection"]
evidence["report"].update(findings=[], high_severity_count=0, status="passed")
evidence["report_hash"] = 固定哈希(evidence["report"])
return row
monkeypatch.setattr(评测存储, "读取交付", changed)
with pytest.raises(评测错误, match="检测报告"):
文学测试._读(env)
assert len(env["received"]) == 6
@pytest.mark.case_id(
"NC-w25-25f205",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["检测登记中断可补原回执,恢复不重发"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_检测登记中断后补原回执且恢复不重发__25f205(执行环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 执行环境
save = 评测存储.保存交付
broken = []
def fail_once(self, uid, *args):
if not broken and self.读取单元(uid)["kind"] == "detection":
broken.append(uid)
raise 评测错误("注入检测登记中断")
return save(self, uid, *args)
monkeypatch.setattr(评测存储, "保存交付", fail_once)
_执行(env, allow_failure=True)
monkeypatch.setattr(评测存储, "保存交付", save)
unit = next(r for r in 文学测试._读(env)["units"] if r["unit_id"] == broken[0])
original = next(r for r in env["received"] if r["model"] == "synthetic-semantic-detector")
request = 交付补登请求(
unit_id=unit["unit_id"],
call_id=unit["unregistered_deliveries"][0]["call_id"],
output=_输出(json.loads(original["input"]), "ok"),
)
svc = env["app"].要求评测()
first = svc.补登记交付(env["actor"], request)
assert svc.补登记交付(env["actor"], request) == first
运行测试._恢复失败单元(env)
运行测试._运行就绪(env)
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
assert report["samples"][0]["detection_complete"] and len(env["received"]) == 6
@pytest.mark.case_id(
"NC-w25-25f206",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["检测纳入原预算,预算不足不留孤儿任务"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [{**参数, "calls": 5}], indirect=True)
def test_新增检测必须纳入原预算不足时零任务__25f206(执行环境):
env = 执行环境
with pytest.raises(评测错误, match="调用上限"):
运行测试._启动执行(env)
assert not env["received"]
with env["pool"].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-25f207",
environment="隔离PG与合成HTTP",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["生成失败时对应检测不创建,完整分母和缺失仍保留"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_生成失败时该份检测不创建且完整分母保留__25f207(执行环境):
env = 执行环境
env["scripted"].append("bad_output")
运行测试._启动执行(env)
运行测试._运行就绪(env, allow_failure=True)
文学测试._推进(env)
运行测试._运行就绪(env)
work = 文学测试._读(env)
detectors = [u for u in work["units"] if u["kind"] == "detection"]
assert len(detectors) == 2 and sum(u["task_id"] is None for u in detectors) == 1
assert len(env["received"]) == 3
report = env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
assert report["coverage"]["samples"] == 1 and not report["samples"][0]["detection_complete"]