muse-agent-example/tests/集成/test_规则诊断资格与消费.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

403 lines
16 KiB
Python

"""B06确定性规则的逐例资格、S01启停与实际诊断消费。"""
from copy import deepcopy
from dataclasses import replace
from datetime import UTC, datetime, timedelta
from typing import Literal
import psycopg
import pytest
from muse.作品规划.接口 import 档案保存
from muse.共享.调用身份 import 内容用途, 用途
from muse.审校修订.接口 import (
审校错误,
核对规则库,
规则状态命令,
读取内置规则,
)
from muse.效果评测.接口 import (
标注区间,
规则诊断数据集发布,
规则诊断样本,
规则诊断金标,
评测服务,
评测错误,
预期发现,
)
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
pytestmark = pytest.mark.数据库
def _单规则库():
all_rules = 读取内置规则()
rule = all_rules["rules"]["l002"]
samples = {ref: all_rules["samples"][ref] for refs in rule["samples"].values() for ref in refs}
return 核对规则库([rule], list(samples.values()))
def _span(text: str, quote: str, disposition: Literal["ask", "retain", "repair"] = "ask"):
start = text.index(quote)
return 预期发现(
start=start,
end=start + len(quote),
quote=quote,
disposition=disposition,
)
def _sample(
sample_id: str,
text: str,
kind: Literal["positive", "negative", "protected"],
source_kind: Literal["owned", "licensed", "public_domain", "synthetic"],
):
quote = "值得注意的是"
findings = (
()
if kind == "negative"
else (_span(text, quote, "retain" if kind == "protected" else "ask"),)
)
protected = ()
if kind == "protected":
start = text.index(quote)
protected = (标注区间(start=start, end=start + len(quote), quote=quote),)
return 规则诊断样本(
sample_id=sample_id,
source_ref=f"source:{sample_id}",
license_ref=f"license:{source_kind}:{sample_id}",
source_groups=(f"group:{sample_id}",),
source_kind=source_kind,
split="holdout",
text=text,
gold=规则诊断金标(
annotation_status="confirmed",
annotation_ref=f"annotation:{sample_id}",
case_kind=kind,
findings=findings,
protected_expressions=protected,
),
)
def _samples(source_kind: Literal["owned", "licensed", "public_domain", "synthetic"], prefix: str):
return (
_sample(
prefix + "-positive", "石阶发冷。值得注意的是,灯还亮着。", "positive", source_kind
),
_sample(prefix + "-negative", "石阶发冷,灯还亮着。", "negative", source_kind),
_sample(
prefix + "-protected",
"值得注意的是,是陈伯每次开场的口头禅。",
"protected",
source_kind,
),
)
@pytest.fixture
def 规则资格环境(章后环境):
# 规则验证、凭据导出与审阅必须同库:复用章后环境所在库。
应用测试库 = 章后环境["库组"]
app = 章后环境["装配"]
author = replace(章后环境["作者"], 内容用途=内容用途.检测)
review = app.要求审校()
review.导入规则候选(author, 规则库=_单规则库())
works = app.要求作品()
structure = works.新档案结构(author, "rule-work", "work_core", 1)
works.保存档案(
author,
"create-rule-work",
档案保存("rule-work", 0, {"名称": "规则诊断作品"}, structure),
)
works.添加章节(author, "add-rule-ch", "rule-work", "rule-ch", "诊断章", 预期目录版本=0)
app.要求正文().保存人工(
author,
"save-rule-body",
"rule-ch",
0,
正文草稿((段落("p1", (文本节点("值得注意的是,门外已经下起了雨。"),)),)),
)
doc = app.要求正文().读取正文(author, "rule-ch")
target = {
"work_id": "rule-work",
"chapter_id": "rule-ch",
"revision": doc["revision"],
"branch_id": "main",
"document_hash": doc["document_hash"],
}
maintain = 评测服务(应用测试库[用途.维护])
evaluate = 评测服务(应用测试库[用途.评测])
return {
"app": app,
"review": review,
"author": author,
"maintain": maintain,
"evaluate": evaluate,
"maintenance_identity": replace(author, 用途=用途.维护),
"evaluation_identity": replace(author, 用途=用途.评测),
"pools": 应用测试库,
"target": target,
}
def _publish(
env,
dataset_id: str,
source_kind: Literal["owned", "licensed", "public_domain", "synthetic"],
):
request = 规则诊断数据集发布(
dataset_id=dataset_id,
revision=1,
rule_id="l002",
rule_version=1,
approval_ref="maintainer-approved:" + dataset_id,
expires_at=datetime.now(UTC) + timedelta(hours=3),
samples=_samples(source_kind, dataset_id),
)
return env["maintain"].发布规则诊断数据集(
env["maintenance_identity"], request, env["review"], env["author"]
)
@pytest.mark.case_id(
"NC-w25-25d101",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["合成逐例保留完整证据但不导出启用凭据"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_合成逐例保留完整证据但不导出启用凭据__25d101(规则资格环境):
env = 规则资格环境
published = _publish(env, "synthetic-rule-diagnostic", "synthetic")
run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
assert run["assessment"]["status"] == "engineering_only"
assert run["assessment"]["activation_eligible"] is False
assert run["assessment"]["totals"] == {
"false_positives": 0,
"false_negatives": 0,
"disposition_mismatches": 0,
"protection_violations": 0,
"unknown_items": 0,
}
assert len(run["results"]) == 3
assert env["evaluate"].读取规则诊断验证(env["evaluation_identity"], run["run_id"]) == run
with pytest.raises(评测错误, match="不导出"):
env["maintain"].导出规则诊断凭据(
env["maintenance_identity"],
run["run_id"],
批准引用="synthetic-must-not-enable",
有效期=datetime.now(UTC) + timedelta(hours=1),
)
training_text = _单规则库()["samples"]["sf-l002-01"]["text"]
request = 规则诊断数据集发布(
dataset_id="training-leak",
revision=1,
rule_id="l002",
rule_version=1,
approval_ref="reject-training",
expires_at=datetime.now(UTC) + timedelta(hours=2),
samples=(_sample("copied-training", training_text, "positive", "synthetic"),),
)
with pytest.raises(评测错误, match="训练例证"):
env["maintain"].发布规则诊断数据集(
env["maintenance_identity"], request, env["review"], env["author"]
)
@pytest.mark.case_id(
"NC-w25-25d102",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["金标缺失和误标保留未完成与硬失败"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_金标缺失和误标保留未完成与硬失败__25d102(规则资格环境):
env = 规则资格环境
samples = list(_samples("owned", "incomplete"))
samples[0] = samples[0].model_copy(
update={
"gold": 规则诊断金标(
annotation_status="unknown", annotation_ref="", case_kind="unknown"
),
}
)
request = 规则诊断数据集发布(
dataset_id="incomplete-rule-diagnostic",
revision=1,
rule_id="l002",
rule_version=1,
approval_ref="annotation-still-open",
expires_at=datetime.now(UTC) + timedelta(hours=2),
samples=tuple(samples),
)
published = env["maintain"].发布规则诊断数据集(
env["maintenance_identity"], request, env["review"], env["author"]
)
result = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
assert result["assessment"]["status"] == "insufficient_evidence"
assert result["assessment"]["totals"]["unknown_items"] == 1
assert any(row["execution_status"] == "unknown_annotation" for row in result["results"])
wrong = list(_samples("owned", "wrong-label"))
wrong[1] = wrong[1].model_copy(update={"text": "值得注意的是,这条负例其实会命中。"})
request = 规则诊断数据集发布(
dataset_id="failed-rule-diagnostic",
revision=1,
rule_id="l002",
rule_version=1,
approval_ref="fixed-before-running",
expires_at=datetime.now(UTC) + timedelta(hours=2),
samples=tuple(wrong),
)
published = env["maintain"].发布规则诊断数据集(
env["maintenance_identity"], request, env["review"], env["author"]
)
failed = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
assert failed["assessment"]["status"] == "failed"
assert failed["assessment"]["totals"]["false_positives"] == 1
@pytest.mark.case_id(
"NC-w25-25d103",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["S01审阅启停与确切规则消费及停止恢复"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_S01审阅启停与确切规则消费及停止恢复__25d103(规则资格环境):
env = 规则资格环境
before = env["review"].生成诊断(env["author"], env["target"])
assert before["report"]["findings"] == []
published = _publish(env, "declared-owned-rule-diagnostic", "owned")
run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
assert run["assessment"]["status"] == "passed"
proof = env["maintain"].导出规则诊断凭据(
env["maintenance_identity"],
run["run_id"],
批准引用="maintainer-exported-exact-proof",
有效期=datetime.now(UTC) + timedelta(hours=1),
)
assert proof["payload"]["target"] == published["target"]
assert "gold" not in str(proof).lower()
for table in ("oracle.muse_dataset_answers", "evaluation.muse_rule_diagnostic_result"):
with (
env["pools"][用途.生产].连接() as conn,
pytest.raises(psycopg.errors.InsufficientPrivilege),
):
conn.execute("SELECT * FROM " + table)
command = 规则状态命令(
"l002", 1, "enable", expected_activation_revision=0, credential_id=proof["receipt_id"]
)
opened = env["review"].打开规则状态审阅(env["author"], command)
receipt = env["review"].决定规则状态(
env["author"],
"enable-l002-v1",
command,
审阅ID=opened["review_id"],
审阅哈希=opened["review_hash"],
)
assert (
env["review"].决定规则状态(
env["author"],
"enable-l002-v1",
command,
审阅ID=opened["review_id"],
审阅哈希=opened["review_hash"],
)
== receipt
)
active = env["review"].生成诊断(env["author"], env["target"])
assert active["report"]["findings"][0]["rule_id"] == "l002"
assert active["report"]["rule_bindings"][0]["version"] == 1
assert active["report"]["rule_bindings"][0]["credential_id"] == proof["receipt_id"]
assert env["review"].读取规则版本(env["author"], "l002", 1)["activation_recorded"]
version_two = deepcopy(_单规则库())
version_two["rules"]["l002"]["version"] = 2
version_two["rules"]["l002"]["fix_hint"] = "新版仍只询问,不自动修改"
version_two = 核对规则库(
list(version_two["rules"].values()), list(version_two["samples"].values())
)
env["review"].导入规则候选(env["author"], 规则库=version_two)
still_v1 = env["review"].生成诊断(env["author"], env["target"])
assert still_v1["report"]["rule_bindings"][0]["version"] == 1
assert (
env["review"].读取诊断(env["author"], active["diagnosis_id"])["report"] == active["report"]
)
with pytest.raises(审校错误, match="旧规则版本"):
env["review"].打开规则状态审阅(
env["author"],
规则状态命令(
"l002",
1,
"enable",
expected_activation_revision=1,
credential_id=proof["receipt_id"],
),
)
env["evaluate"].停止规则诊断验证(env["evaluation_identity"], run["run_id"], "stop-l002-proof")
state = env["review"].读取规则状态(env["author"], "l002")
assert not state["consumable"] and "失效" in state["reason"]
with pytest.raises(审校错误, match="凭据失效"):
env["review"].生成诊断(env["author"], env["target"])
history = env["review"].读取诊断(env["author"], active["diagnosis_id"])
assert history["report"] == active["report"] and history["stale"]
disable = 规则状态命令("l002", 1, "disable", expected_activation_revision=1)
opened = env["review"].打开规则状态审阅(env["author"], disable)
env["review"].决定规则状态(
env["author"],
"disable-l002-v1",
disable,
审阅ID=opened["review_id"],
审阅哈希=opened["review_hash"],
)
after = env["review"].生成诊断(env["author"], env["target"])
assert after["report"]["findings"] == []
state = env["review"].读取规则状态(env["author"], "l002")
assert state["current"]["state"] == "disabled" and len(state["history"]) == 2
assert (
env["review"].读取诊断(env["author"], active["diagnosis_id"])["report"] == active["report"]
)
@pytest.mark.case_id(
"NC-w25-25d106",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["程序资源变化后旧规则凭据不能新导出或启用"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_程序资源变化后旧规则凭据不能新导出或启用__25d106(规则资格环境, monkeypatch):
from muse.效果评测 import 规则诊断
env = 规则资格环境
data = _publish(env, "build-change", "owned")
run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], data["version_id"])
proof = env["maintain"].导出规则诊断凭据(
env["maintenance_identity"],
run["run_id"],
批准引用="fixed-before-build-change",
有效期=datetime.now(UTC) + timedelta(hours=1),
)
monkeypatch.setattr(规则诊断, "加载清单", lambda: {"构建身份": "a" * 64})
with pytest.raises(评测错误, match="程序资源已变化"):
env["maintain"].导出规则诊断凭据(
env["maintenance_identity"],
run["run_id"],
批准引用="new-export",
有效期=datetime.now(UTC) + timedelta(minutes=30),
)
command = 规则状态命令("l002", 1, "enable", credential_id=proof["receipt_id"])
with pytest.raises(审校错误, match="凭据"):
env["review"].打开规则状态审阅(env["author"], command)
assert env["review"].读取规则状态(env["author"], "l002")["current"] is None