- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
121 lines
5.1 KiB
Python
121 lines
5.1 KiB
Python
"""实际生成、检测、独立评分和B09封存后的混淆观察。"""
|
|
|
|
import copy
|
|
import json
|
|
|
|
import pytest
|
|
import test_回放资料封存 as 资料测试
|
|
import test_文学评分执行与条件第三 as 文学测试
|
|
import test_评测执行与失败收敛 as 运行测试
|
|
import test_评测语义检测 as 检测测试
|
|
|
|
from muse.效果评测.接口 import 评测错误
|
|
from muse.正式变更.接口 import 固定哈希
|
|
|
|
pytestmark = pytest.mark.数据库
|
|
执行环境 = 运行测试.执行环境
|
|
生成环境 = 资料测试.生成环境
|
|
资料环境 = 资料测试.资料环境
|
|
参数 = {**检测测试.参数, "budget_calls": 100, "max_steps": 100}
|
|
|
|
|
|
def _声明():
|
|
return dict(
|
|
version="writer-input-audit-v1",
|
|
annotation_ref="synthetic:annotation",
|
|
as_of_position=2,
|
|
forbidden_facts=[
|
|
dict(fact_id="future", text="远方来客交出了不可告人的密约", first_position=3)
|
|
],
|
|
declared_new_entity_ids=["new-synthetic-entity"],
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
|
def test_实际报告五类观察及版本均绑定原条件__25f601(执行环境):
|
|
report = 检测测试._执行(执行环境)
|
|
observed = report["confounders"]
|
|
assert observed["version"] == "evaluation-confounders-v1"
|
|
assert observed["conditions_hash"] == report["conditions_hash"]
|
|
conditions = 执行环境["exp"]["conditions"]
|
|
assert all(
|
|
r["rubric"]["version"] == "writer-replay-rubric-v3"
|
|
for r in conditions["literary_basis"].values()
|
|
)
|
|
assert observed["conditions_hash"] == 固定哈希(conditions)
|
|
assert observed["evidence_hash"] == 固定哈希(
|
|
{
|
|
"conditions": report["conditions_hash"],
|
|
"plan": report["plan_hash"],
|
|
"samples": report["samples"],
|
|
"observations": observed["samples"],
|
|
}
|
|
)
|
|
r = observed["samples"][report["samples"][0]["sample_id"]]
|
|
assert set(r) == {
|
|
"false_negative",
|
|
"false_positive",
|
|
"leakage",
|
|
"reviewer_instability",
|
|
"new_characters_without_cards",
|
|
}
|
|
assert r["leakage"]["finding_count"] is None and r["leakage"]["status"] == "not_measured"
|
|
assert r["false_positive"]["status"] == "incomplete"
|
|
assert r["reviewer_instability"]["finding_count"] == 0
|
|
|
|
|
|
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
|
def test_真实ABC来源审计与缺卡声明只存oracle__25f602(执行环境, 资料环境):
|
|
audit = _声明()
|
|
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
|
|
运行测试._启动执行(env)
|
|
运行测试._运行就绪(env)
|
|
文学测试._推进(env)
|
|
运行测试._运行就绪(env)
|
|
文学测试._推进(env)
|
|
svc = env["app"].要求评测()
|
|
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
|
r = report["confounders"]["samples"]["target-ch3"]
|
|
assert r["leakage"]["status"] == "measured" and r["leakage"]["finding_count"] == 0
|
|
assert r["new_characters_without_cards"]["finding_count"] == 1
|
|
assert r["new_characters_without_cards"]["denominator"] == 1
|
|
public = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"])
|
|
for value in (public, report, env["received"]):
|
|
encoded = json.dumps(value, ensure_ascii=False)
|
|
assert "不可告人的密约" not in encoded and "new-synthetic-entity" not in encoded
|
|
assert len(env["received"]) == 12
|
|
|
|
|
|
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
|
def test_冻结输入命中未来事实在任何外发前拒绝__25f603(执行环境, 资料环境):
|
|
audit = _声明()
|
|
audit["forbidden_facts"][0]["text"] = "林深回到渡口,雨渐渐小了。"
|
|
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
|
|
with pytest.raises(评测错误, match="预注册未来事实") as caught:
|
|
运行测试._启动执行(env)
|
|
assert "林深回到渡口" not in str(caught.value)
|
|
assert not env["received"]
|
|
with env["pool"].连接(只读=True) as conn:
|
|
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
|
|
|
|
|
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
|
def test_审计声明重签不能改变原实验依据__25f604(执行环境, 资料环境, monkeypatch):
|
|
from muse.效果评测.存储 import 评测存储
|
|
|
|
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=_声明())
|
|
运行测试._启动执行(env)
|
|
read = 评测存储.读取答案
|
|
|
|
def changed(self, did):
|
|
row = copy.deepcopy(read(self, did))
|
|
row["answers"][0]["answer"]["writer_audit"]["forbidden_facts"] = []
|
|
row["answer_hash"] = 固定哈希(row["answers"])
|
|
return row
|
|
|
|
monkeypatch.setattr(评测存储, "读取答案", changed)
|
|
运行测试._运行就绪(env, allow_failure=True)
|
|
assert not env["received"]
|
|
with pytest.raises(评测错误, match="审计依据"):
|
|
env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|