"""实际生成、检测、独立评分和B09封存后的混淆观察。""" import copy import json import pytest import test_回放资料封存 as 资料测试 import test_文学评分执行与条件第三 as 文学测试 import test_评测执行与失败收敛 as 运行测试 import test_评测语义检测 as 检测测试 from muse.效果评测.接口 import 评测错误 from muse.正式变更.接口 import 固定哈希 pytestmark = pytest.mark.数据库 执行环境 = 运行测试.执行环境 生成环境 = 资料测试.生成环境 资料环境 = 资料测试.资料环境 参数 = {**检测测试.参数, "budget_calls": 100, "max_steps": 100} def _声明(): return dict( version="writer-input-audit-v1", annotation_ref="synthetic:annotation", as_of_position=2, forbidden_facts=[ dict(fact_id="future", text="远方来客交出了不可告人的密约", first_position=3) ], declared_new_entity_ids=["new-synthetic-entity"], ) @pytest.mark.case_id( "TC-cf3f5526ab3a", environment="隔离PG与合成HTTP,不认证模型真值", given="实际S02生成、独立检测与双评完成的隔离实验", when="通过公开报告读取混淆项", then=[ "evaluation-confounders-v1完整输出", "writer-replay-rubric-v3冻结标准绑定", "观察证据哈希由条件、计划和实际逐例结果重算", "五类混淆观察齐备;疑似分歧、未测和不完整不填假零", ], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_实际报告五类观察及版本均绑定原条件__25f601(执行环境): report = 检测测试._执行(执行环境) observed = report["confounders"] assert observed["version"] == "evaluation-confounders-v1" assert observed["conditions_hash"] == report["conditions_hash"] conditions = 执行环境["exp"]["conditions"] assert all( r["rubric"]["version"] == "writer-replay-rubric-v3" for r in conditions["literary_basis"].values() ) assert observed["conditions_hash"] == 固定哈希(conditions) assert observed["evidence_hash"] == 固定哈希( { "conditions": report["conditions_hash"], "plan": report["plan_hash"], "samples": report["samples"], "observations": observed["samples"], } ) r = observed["samples"][report["samples"][0]["sample_id"]] assert set(r) == { "false_negative", "false_positive", "leakage", "reviewer_instability", "new_characters_without_cards", } assert r["leakage"]["finding_count"] is None and r["leakage"]["status"] == "not_measured" assert r["false_positive"]["status"] == "incomplete" assert r["reviewer_instability"]["finding_count"] == 0 @pytest.mark.case_id( "NC-w25-25f602", environment="隔离PG与合成HTTP", given="本例固定资料与独立边界输入", when="经实际ABC及S02链公开读回", then=["实际ABC审计、实体ID缺卡及oracle隔离"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_真实ABC来源审计与缺卡声明只存oracle__25f602(执行环境, 资料环境): audit = _声明() env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit) 运行测试._启动执行(env) 运行测试._运行就绪(env) 文学测试._推进(env) 运行测试._运行就绪(env) 文学测试._推进(env) svc = env["app"].要求评测() report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"]) r = report["confounders"]["samples"]["target-ch3"] assert r["leakage"]["status"] == "measured" and r["leakage"]["finding_count"] == 0 assert r["new_characters_without_cards"]["finding_count"] == 1 assert r["new_characters_without_cards"]["denominator"] == 1 public = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"]) for value in (public, report, env["received"]): encoded = json.dumps(value, ensure_ascii=False) assert "不可告人的密约" not in encoded and "new-synthetic-entity" not in encoded assert len(env["received"]) == 12 @pytest.mark.case_id( "NC-w25-25f603", environment="隔离PG与合成HTTP", given="本例固定资料与独立边界输入", when="经实际ABC及S02链公开读回", then=["预注册未来片段在外发和建任务前拒绝"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_冻结输入命中未来事实在任何外发前拒绝__25f603(执行环境, 资料环境): audit = _声明() audit["forbidden_facts"][0]["text"] = "林深回到渡口,雨渐渐小了。" env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit) with pytest.raises(评测错误, match="预注册未来事实") as caught: 运行测试._启动执行(env) assert "林深回到渡口" not in str(caught.value) assert not env["received"] with env["pool"].连接(只读=True) as conn: assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0 @pytest.mark.case_id( "NC-w25-25f604", environment="隔离PG与合成HTTP", given="本例固定资料与独立边界输入", when="经实际ABC及S02链公开读回", then=["重签审计声明不能替换原实验依据"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) @pytest.mark.parametrize("执行环境", [参数], indirect=True) def test_审计声明重签不能改变原实验依据__25f604(执行环境, 资料环境, monkeypatch): from muse.效果评测.存储 import 评测存储 env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=_声明()) 运行测试._启动执行(env) read = 评测存储.读取答案 def changed(self, did): row = copy.deepcopy(read(self, did)) row["answers"][0]["answer"]["writer_audit"]["forbidden_facts"] = [] row["answer_hash"] = 固定哈希(row["answers"]) return row monkeypatch.setattr(评测存储, "读取答案", changed) 运行测试._运行就绪(env, allow_failure=True) assert not env["received"] with pytest.raises(评测错误, match="审计依据"): env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])