muse-agent-example/tests/集成/test_评测混淆观察.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

158 lines
6.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""实际生成、检测、独立评分和B09封存后的混淆观察。"""
import copy
import json
import pytest
import test_回放资料封存 as 资料测试
import test_文学评分执行与条件第三 as 文学测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
from muse.效果评测.接口 import 评测错误
from muse.正式变更.接口 import 固定哈希
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
生成环境 = 资料测试.生成环境
资料环境 = 资料测试.资料环境
参数 = {**检测测试.参数, "budget_calls": 100, "max_steps": 100}
def _声明():
return dict(
version="writer-input-audit-v1",
annotation_ref="synthetic:annotation",
as_of_position=2,
forbidden_facts=[
dict(fact_id="future", text="远方来客交出了不可告人的密约", first_position=3)
],
declared_new_entity_ids=["new-synthetic-entity"],
)
@pytest.mark.case_id(
"TC-cf3f5526ab3a",
environment="隔离PG与合成HTTP,不认证模型真值",
given="实际S02生成、独立检测与双评完成的隔离实验",
when="通过公开报告读取混淆项",
then=[
"evaluation-confounders-v1完整输出",
"writer-replay-rubric-v3冻结标准绑定",
"观察证据哈希由条件、计划和实际逐例结果重算",
"五类混淆观察齐备;疑似分歧、未测和不完整不填假零",
],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_实际报告五类观察及版本均绑定原条件__25f601(执行环境):
report = 检测测试._执行(执行环境)
observed = report["confounders"]
assert observed["version"] == "evaluation-confounders-v1"
assert observed["conditions_hash"] == report["conditions_hash"]
conditions = 执行环境["exp"]["conditions"]
assert all(
r["rubric"]["version"] == "writer-replay-rubric-v3"
for r in conditions["literary_basis"].values()
)
assert observed["conditions_hash"] == 固定哈希(conditions)
assert observed["evidence_hash"] == 固定哈希(
{
"conditions": report["conditions_hash"],
"plan": report["plan_hash"],
"samples": report["samples"],
"observations": observed["samples"],
}
)
r = observed["samples"][report["samples"][0]["sample_id"]]
assert set(r) == {
"false_negative",
"false_positive",
"leakage",
"reviewer_instability",
"new_characters_without_cards",
}
assert r["leakage"]["finding_count"] is None and r["leakage"]["status"] == "not_measured"
assert r["false_positive"]["status"] == "incomplete"
assert r["reviewer_instability"]["finding_count"] == 0
@pytest.mark.case_id(
"NC-w25-25f602",
environment="隔离PG与合成HTTP",
given="本例固定资料与独立边界输入",
when="经实际ABC及S02链公开读回",
then=["实际ABC审计、实体ID缺卡及oracle隔离"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_真实ABC来源审计与缺卡声明只存oracle__25f602(执行环境, 资料环境):
audit = _声明()
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
运行测试._运行就绪(env)
文学测试._推进(env)
svc = env["app"].要求评测()
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
r = report["confounders"]["samples"]["target-ch3"]
assert r["leakage"]["status"] == "measured" and r["leakage"]["finding_count"] == 0
assert r["new_characters_without_cards"]["finding_count"] == 1
assert r["new_characters_without_cards"]["denominator"] == 1
public = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"])
for value in (public, report, env["received"]):
encoded = json.dumps(value, ensure_ascii=False)
assert "不可告人的密约" not in encoded and "new-synthetic-entity" not in encoded
assert len(env["received"]) == 12
@pytest.mark.case_id(
"NC-w25-25f603",
environment="隔离PG与合成HTTP",
given="本例固定资料与独立边界输入",
when="经实际ABC及S02链公开读回",
then=["预注册未来片段在外发和建任务前拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_冻结输入命中未来事实在任何外发前拒绝__25f603(执行环境, 资料环境):
audit = _声明()
audit["forbidden_facts"][0]["text"] = "林深回到渡口,雨渐渐小了。"
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
with pytest.raises(评测错误, match="预注册未来事实") as caught:
运行测试._启动执行(env)
assert "林深回到渡口" not in str(caught.value)
assert not env["received"]
with env["pool"].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
@pytest.mark.case_id(
"NC-w25-25f604",
environment="隔离PG与合成HTTP",
given="本例固定资料与独立边界输入",
when="经实际ABC及S02链公开读回",
then=["重签审计声明不能替换原实验依据"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_审计声明重签不能改变原实验依据__25f604(执行环境, 资料环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=_声明())
运行测试._启动执行(env)
read = 评测存储.读取答案
def changed(self, did):
row = copy.deepcopy(read(self, did))
row["answers"][0]["answer"]["writer_audit"]["forbidden_facts"] = []
row["answer_hash"] = 固定哈希(row["answers"])
return row
monkeypatch.setattr(评测存储, "读取答案", changed)
运行测试._运行就绪(env, allow_failure=True)
assert not env["received"]
with pytest.raises(评测错误, match="审计依据"):
env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])