muse-agent-example/tests/集成/test_评测混淆观察.py
zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

121 lines
5.1 KiB
Python

"""实际生成、检测、独立评分和B09封存后的混淆观察。"""
import copy
import json
import pytest
import test_回放资料封存 as 资料测试
import test_文学评分执行与条件第三 as 文学测试
import test_评测执行与失败收敛 as 运行测试
import test_评测语义检测 as 检测测试
from muse.效果评测.接口 import 评测错误
from muse.正式变更.接口 import 固定哈希
pytestmark = pytest.mark.数据库
执行环境 = 运行测试.执行环境
生成环境 = 资料测试.生成环境
资料环境 = 资料测试.资料环境
参数 = {**检测测试.参数, "budget_calls": 100, "max_steps": 100}
def _声明():
return dict(
version="writer-input-audit-v1",
annotation_ref="synthetic:annotation",
as_of_position=2,
forbidden_facts=[
dict(fact_id="future", text="远方来客交出了不可告人的密约", first_position=3)
],
declared_new_entity_ids=["new-synthetic-entity"],
)
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_实际报告五类观察及版本均绑定原条件__25f601(执行环境):
report = 检测测试._执行(执行环境)
observed = report["confounders"]
assert observed["version"] == "evaluation-confounders-v1"
assert observed["conditions_hash"] == report["conditions_hash"]
conditions = 执行环境["exp"]["conditions"]
assert all(
r["rubric"]["version"] == "writer-replay-rubric-v3"
for r in conditions["literary_basis"].values()
)
assert observed["conditions_hash"] == 固定哈希(conditions)
assert observed["evidence_hash"] == 固定哈希(
{
"conditions": report["conditions_hash"],
"plan": report["plan_hash"],
"samples": report["samples"],
"observations": observed["samples"],
}
)
r = observed["samples"][report["samples"][0]["sample_id"]]
assert set(r) == {
"false_negative",
"false_positive",
"leakage",
"reviewer_instability",
"new_characters_without_cards",
}
assert r["leakage"]["finding_count"] is None and r["leakage"]["status"] == "not_measured"
assert r["false_positive"]["status"] == "incomplete"
assert r["reviewer_instability"]["finding_count"] == 0
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_真实ABC来源审计与缺卡声明只存oracle__25f602(执行环境, 资料环境):
audit = _声明()
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
运行测试._启动执行(env)
运行测试._运行就绪(env)
文学测试._推进(env)
运行测试._运行就绪(env)
文学测试._推进(env)
svc = env["app"].要求评测()
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
r = report["confounders"]["samples"]["target-ch3"]
assert r["leakage"]["status"] == "measured" and r["leakage"]["finding_count"] == 0
assert r["new_characters_without_cards"]["finding_count"] == 1
assert r["new_characters_without_cards"]["denominator"] == 1
public = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"])
for value in (public, report, env["received"]):
encoded = json.dumps(value, ensure_ascii=False)
assert "不可告人的密约" not in encoded and "new-synthetic-entity" not in encoded
assert len(env["received"]) == 12
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_冻结输入命中未来事实在任何外发前拒绝__25f603(执行环境, 资料环境):
audit = _声明()
audit["forbidden_facts"][0]["text"] = "林深回到渡口,雨渐渐小了。"
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
with pytest.raises(评测错误, match="预注册未来事实") as caught:
运行测试._启动执行(env)
assert "林深回到渡口" not in str(caught.value)
assert not env["received"]
with env["pool"].连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
def test_审计声明重签不能改变原实验依据__25f604(执行环境, 资料环境, monkeypatch):
from muse.效果评测.存储 import 评测存储
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=_声明())
运行测试._启动执行(env)
read = 评测存储.读取答案
def changed(self, did):
row = copy.deepcopy(read(self, did))
row["answers"][0]["answer"]["writer_audit"]["forbidden_facts"] = []
row["answer_hash"] = 固定哈希(row["answers"])
return row
monkeypatch.setattr(评测存储, "读取答案", changed)
运行测试._运行就绪(env, allow_failure=True)
assert not env["received"]
with pytest.raises(评测错误, match="审计依据"):
env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])