实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
158 lines
6.6 KiB
Python
158 lines
6.6 KiB
Python
"""实际生成、检测、独立评分和B09封存后的混淆观察。"""
|
||
|
||
import copy
|
||
import json
|
||
|
||
import pytest
|
||
import test_回放资料封存 as 资料测试
|
||
import test_文学评分执行与条件第三 as 文学测试
|
||
import test_评测执行与失败收敛 as 运行测试
|
||
import test_评测语义检测 as 检测测试
|
||
|
||
from muse.效果评测.接口 import 评测错误
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
执行环境 = 运行测试.执行环境
|
||
生成环境 = 资料测试.生成环境
|
||
资料环境 = 资料测试.资料环境
|
||
参数 = {**检测测试.参数, "budget_calls": 100, "max_steps": 100}
|
||
|
||
|
||
def _声明():
|
||
return dict(
|
||
version="writer-input-audit-v1",
|
||
annotation_ref="synthetic:annotation",
|
||
as_of_position=2,
|
||
forbidden_facts=[
|
||
dict(fact_id="future", text="远方来客交出了不可告人的密约", first_position=3)
|
||
],
|
||
declared_new_entity_ids=["new-synthetic-entity"],
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-cf3f5526ab3a",
|
||
environment="隔离PG与合成HTTP,不认证模型真值",
|
||
given="实际S02生成、独立检测与双评完成的隔离实验",
|
||
when="通过公开报告读取混淆项",
|
||
then=[
|
||
"evaluation-confounders-v1完整输出",
|
||
"writer-replay-rubric-v3冻结标准绑定",
|
||
"观察证据哈希由条件、计划和实际逐例结果重算",
|
||
"五类混淆观察齐备;疑似分歧、未测和不完整不填假零",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_实际报告五类观察及版本均绑定原条件__25f601(执行环境):
|
||
report = 检测测试._执行(执行环境)
|
||
observed = report["confounders"]
|
||
assert observed["version"] == "evaluation-confounders-v1"
|
||
assert observed["conditions_hash"] == report["conditions_hash"]
|
||
conditions = 执行环境["exp"]["conditions"]
|
||
assert all(
|
||
r["rubric"]["version"] == "writer-replay-rubric-v3"
|
||
for r in conditions["literary_basis"].values()
|
||
)
|
||
assert observed["conditions_hash"] == 固定哈希(conditions)
|
||
assert observed["evidence_hash"] == 固定哈希(
|
||
{
|
||
"conditions": report["conditions_hash"],
|
||
"plan": report["plan_hash"],
|
||
"samples": report["samples"],
|
||
"observations": observed["samples"],
|
||
}
|
||
)
|
||
r = observed["samples"][report["samples"][0]["sample_id"]]
|
||
assert set(r) == {
|
||
"false_negative",
|
||
"false_positive",
|
||
"leakage",
|
||
"reviewer_instability",
|
||
"new_characters_without_cards",
|
||
}
|
||
assert r["leakage"]["finding_count"] is None and r["leakage"]["status"] == "not_measured"
|
||
assert r["false_positive"]["status"] == "incomplete"
|
||
assert r["reviewer_instability"]["finding_count"] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f602",
|
||
environment="隔离PG与合成HTTP",
|
||
given="本例固定资料与独立边界输入",
|
||
when="经实际ABC及S02链公开读回",
|
||
then=["实际ABC审计、实体ID缺卡及oracle隔离"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_真实ABC来源审计与缺卡声明只存oracle__25f602(执行环境, 资料环境):
|
||
audit = _声明()
|
||
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
|
||
运行测试._启动执行(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
运行测试._运行就绪(env)
|
||
文学测试._推进(env)
|
||
svc = env["app"].要求评测()
|
||
report = svc.读取实验报告(env["actor"], env["exp"]["experiment_id"])
|
||
r = report["confounders"]["samples"]["target-ch3"]
|
||
assert r["leakage"]["status"] == "measured" and r["leakage"]["finding_count"] == 0
|
||
assert r["new_characters_without_cards"]["finding_count"] == 1
|
||
assert r["new_characters_without_cards"]["denominator"] == 1
|
||
public = svc.读取数据集(env["actor"], env["exp"]["dataset_version_id"])
|
||
for value in (public, report, env["received"]):
|
||
encoded = json.dumps(value, ensure_ascii=False)
|
||
assert "不可告人的密约" not in encoded and "new-synthetic-entity" not in encoded
|
||
assert len(env["received"]) == 12
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f603",
|
||
environment="隔离PG与合成HTTP",
|
||
given="本例固定资料与独立边界输入",
|
||
when="经实际ABC及S02链公开读回",
|
||
then=["预注册未来片段在外发和建任务前拒绝"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_冻结输入命中未来事实在任何外发前拒绝__25f603(执行环境, 资料环境):
|
||
audit = _声明()
|
||
audit["forbidden_facts"][0]["text"] = "林深回到渡口,雨渐渐小了。"
|
||
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=audit)
|
||
with pytest.raises(评测错误, match="预注册未来事实") as caught:
|
||
运行测试._启动执行(env)
|
||
assert "林深回到渡口" not in str(caught.value)
|
||
assert not env["received"]
|
||
with env["pool"].连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM evaluation.muse_task").fetchone()[0] == 0
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w25-25f604",
|
||
environment="隔离PG与合成HTTP",
|
||
given="本例固定资料与独立边界输入",
|
||
when="经实际ABC及S02链公开读回",
|
||
then=["重签审计声明不能替换原实验依据"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("执行环境", [参数], indirect=True)
|
||
def test_审计声明重签不能改变原实验依据__25f604(执行环境, 资料环境, monkeypatch):
|
||
from muse.效果评测.存储 import 评测存储
|
||
|
||
env = 文学测试.创建ABC实验(执行环境, 资料环境, detector=True, calls=15, audit=_声明())
|
||
运行测试._启动执行(env)
|
||
read = 评测存储.读取答案
|
||
|
||
def changed(self, did):
|
||
row = copy.deepcopy(read(self, did))
|
||
row["answers"][0]["answer"]["writer_audit"]["forbidden_facts"] = []
|
||
row["answer_hash"] = 固定哈希(row["answers"])
|
||
return row
|
||
|
||
monkeypatch.setattr(评测存储, "读取答案", changed)
|
||
运行测试._运行就绪(env, allow_failure=True)
|
||
assert not env["received"]
|
||
with pytest.raises(评测错误, match="审计依据"):
|
||
env["app"].要求评测().读取实验报告(env["actor"], env["exp"]["experiment_id"])
|