实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
403 lines
16 KiB
Python
403 lines
16 KiB
Python
"""B06确定性规则的逐例资格、S01启停与实际诊断消费。"""
|
|
|
|
from copy import deepcopy
|
|
from dataclasses import replace
|
|
from datetime import UTC, datetime, timedelta
|
|
from typing import Literal
|
|
|
|
import psycopg
|
|
import pytest
|
|
|
|
from muse.作品规划.接口 import 档案保存
|
|
from muse.共享.调用身份 import 内容用途, 用途
|
|
from muse.审校修订.接口 import (
|
|
审校错误,
|
|
核对规则库,
|
|
规则状态命令,
|
|
读取内置规则,
|
|
)
|
|
from muse.效果评测.接口 import (
|
|
标注区间,
|
|
规则诊断数据集发布,
|
|
规则诊断样本,
|
|
规则诊断金标,
|
|
评测服务,
|
|
评测错误,
|
|
预期发现,
|
|
)
|
|
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
|
|
|
|
pytestmark = pytest.mark.数据库
|
|
|
|
|
|
def _单规则库():
|
|
all_rules = 读取内置规则()
|
|
rule = all_rules["rules"]["l002"]
|
|
samples = {ref: all_rules["samples"][ref] for refs in rule["samples"].values() for ref in refs}
|
|
return 核对规则库([rule], list(samples.values()))
|
|
|
|
|
|
def _span(text: str, quote: str, disposition: Literal["ask", "retain", "repair"] = "ask"):
|
|
start = text.index(quote)
|
|
return 预期发现(
|
|
start=start,
|
|
end=start + len(quote),
|
|
quote=quote,
|
|
disposition=disposition,
|
|
)
|
|
|
|
|
|
def _sample(
|
|
sample_id: str,
|
|
text: str,
|
|
kind: Literal["positive", "negative", "protected"],
|
|
source_kind: Literal["owned", "licensed", "public_domain", "synthetic"],
|
|
):
|
|
quote = "值得注意的是"
|
|
findings = (
|
|
()
|
|
if kind == "negative"
|
|
else (_span(text, quote, "retain" if kind == "protected" else "ask"),)
|
|
)
|
|
protected = ()
|
|
if kind == "protected":
|
|
start = text.index(quote)
|
|
protected = (标注区间(start=start, end=start + len(quote), quote=quote),)
|
|
return 规则诊断样本(
|
|
sample_id=sample_id,
|
|
source_ref=f"source:{sample_id}",
|
|
license_ref=f"license:{source_kind}:{sample_id}",
|
|
source_groups=(f"group:{sample_id}",),
|
|
source_kind=source_kind,
|
|
split="holdout",
|
|
text=text,
|
|
gold=规则诊断金标(
|
|
annotation_status="confirmed",
|
|
annotation_ref=f"annotation:{sample_id}",
|
|
case_kind=kind,
|
|
findings=findings,
|
|
protected_expressions=protected,
|
|
),
|
|
)
|
|
|
|
|
|
def _samples(source_kind: Literal["owned", "licensed", "public_domain", "synthetic"], prefix: str):
|
|
return (
|
|
_sample(
|
|
prefix + "-positive", "石阶发冷。值得注意的是,灯还亮着。", "positive", source_kind
|
|
),
|
|
_sample(prefix + "-negative", "石阶发冷,灯还亮着。", "negative", source_kind),
|
|
_sample(
|
|
prefix + "-protected",
|
|
"值得注意的是,是陈伯每次开场的口头禅。",
|
|
"protected",
|
|
source_kind,
|
|
),
|
|
)
|
|
|
|
|
|
@pytest.fixture
|
|
def 规则资格环境(章后环境):
|
|
# 规则验证、凭据导出与审阅必须同库:复用章后环境所在库。
|
|
应用测试库 = 章后环境["库组"]
|
|
app = 章后环境["装配"]
|
|
author = replace(章后环境["作者"], 内容用途=内容用途.检测)
|
|
review = app.要求审校()
|
|
review.导入规则候选(author, 规则库=_单规则库())
|
|
works = app.要求作品()
|
|
structure = works.新档案结构(author, "rule-work", "work_core", 1)
|
|
works.保存档案(
|
|
author,
|
|
"create-rule-work",
|
|
档案保存("rule-work", 0, {"名称": "规则诊断作品"}, structure),
|
|
)
|
|
works.添加章节(author, "add-rule-ch", "rule-work", "rule-ch", "诊断章", 预期目录版本=0)
|
|
app.要求正文().保存人工(
|
|
author,
|
|
"save-rule-body",
|
|
"rule-ch",
|
|
0,
|
|
正文草稿((段落("p1", (文本节点("值得注意的是,门外已经下起了雨。"),)),)),
|
|
)
|
|
doc = app.要求正文().读取正文(author, "rule-ch")
|
|
target = {
|
|
"work_id": "rule-work",
|
|
"chapter_id": "rule-ch",
|
|
"revision": doc["revision"],
|
|
"branch_id": "main",
|
|
"document_hash": doc["document_hash"],
|
|
}
|
|
maintain = 评测服务(应用测试库[用途.维护])
|
|
evaluate = 评测服务(应用测试库[用途.评测])
|
|
return {
|
|
"app": app,
|
|
"review": review,
|
|
"author": author,
|
|
"maintain": maintain,
|
|
"evaluate": evaluate,
|
|
"maintenance_identity": replace(author, 用途=用途.维护),
|
|
"evaluation_identity": replace(author, 用途=用途.评测),
|
|
"pools": 应用测试库,
|
|
"target": target,
|
|
}
|
|
|
|
|
|
def _publish(
|
|
env,
|
|
dataset_id: str,
|
|
source_kind: Literal["owned", "licensed", "public_domain", "synthetic"],
|
|
):
|
|
request = 规则诊断数据集发布(
|
|
dataset_id=dataset_id,
|
|
revision=1,
|
|
rule_id="l002",
|
|
rule_version=1,
|
|
approval_ref="maintainer-approved:" + dataset_id,
|
|
expires_at=datetime.now(UTC) + timedelta(hours=3),
|
|
samples=_samples(source_kind, dataset_id),
|
|
)
|
|
return env["maintain"].发布规则诊断数据集(
|
|
env["maintenance_identity"], request, env["review"], env["author"]
|
|
)
|
|
|
|
|
|
@pytest.mark.case_id(
|
|
"NC-w25-25d101",
|
|
environment="隔离PG/合成HTTP及浏览器",
|
|
given="固定请求、来源与作者;按用例环境准备独立测试输入",
|
|
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
|
then=["合成逐例保留完整证据但不导出启用凭据"],
|
|
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
|
)
|
|
def test_合成逐例保留完整证据但不导出启用凭据__25d101(规则资格环境):
|
|
env = 规则资格环境
|
|
published = _publish(env, "synthetic-rule-diagnostic", "synthetic")
|
|
run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
|
|
assert run["assessment"]["status"] == "engineering_only"
|
|
assert run["assessment"]["activation_eligible"] is False
|
|
assert run["assessment"]["totals"] == {
|
|
"false_positives": 0,
|
|
"false_negatives": 0,
|
|
"disposition_mismatches": 0,
|
|
"protection_violations": 0,
|
|
"unknown_items": 0,
|
|
}
|
|
assert len(run["results"]) == 3
|
|
assert env["evaluate"].读取规则诊断验证(env["evaluation_identity"], run["run_id"]) == run
|
|
with pytest.raises(评测错误, match="不导出"):
|
|
env["maintain"].导出规则诊断凭据(
|
|
env["maintenance_identity"],
|
|
run["run_id"],
|
|
批准引用="synthetic-must-not-enable",
|
|
有效期=datetime.now(UTC) + timedelta(hours=1),
|
|
)
|
|
training_text = _单规则库()["samples"]["sf-l002-01"]["text"]
|
|
request = 规则诊断数据集发布(
|
|
dataset_id="training-leak",
|
|
revision=1,
|
|
rule_id="l002",
|
|
rule_version=1,
|
|
approval_ref="reject-training",
|
|
expires_at=datetime.now(UTC) + timedelta(hours=2),
|
|
samples=(_sample("copied-training", training_text, "positive", "synthetic"),),
|
|
)
|
|
with pytest.raises(评测错误, match="训练例证"):
|
|
env["maintain"].发布规则诊断数据集(
|
|
env["maintenance_identity"], request, env["review"], env["author"]
|
|
)
|
|
|
|
|
|
@pytest.mark.case_id(
|
|
"NC-w25-25d102",
|
|
environment="隔离PG/合成HTTP及浏览器",
|
|
given="固定请求、来源与作者;按用例环境准备独立测试输入",
|
|
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
|
then=["金标缺失和误标保留未完成与硬失败"],
|
|
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
|
)
|
|
def test_金标缺失和误标保留未完成与硬失败__25d102(规则资格环境):
|
|
env = 规则资格环境
|
|
samples = list(_samples("owned", "incomplete"))
|
|
samples[0] = samples[0].model_copy(
|
|
update={
|
|
"gold": 规则诊断金标(
|
|
annotation_status="unknown", annotation_ref="", case_kind="unknown"
|
|
),
|
|
}
|
|
)
|
|
request = 规则诊断数据集发布(
|
|
dataset_id="incomplete-rule-diagnostic",
|
|
revision=1,
|
|
rule_id="l002",
|
|
rule_version=1,
|
|
approval_ref="annotation-still-open",
|
|
expires_at=datetime.now(UTC) + timedelta(hours=2),
|
|
samples=tuple(samples),
|
|
)
|
|
published = env["maintain"].发布规则诊断数据集(
|
|
env["maintenance_identity"], request, env["review"], env["author"]
|
|
)
|
|
result = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
|
|
assert result["assessment"]["status"] == "insufficient_evidence"
|
|
assert result["assessment"]["totals"]["unknown_items"] == 1
|
|
assert any(row["execution_status"] == "unknown_annotation" for row in result["results"])
|
|
|
|
wrong = list(_samples("owned", "wrong-label"))
|
|
wrong[1] = wrong[1].model_copy(update={"text": "值得注意的是,这条负例其实会命中。"})
|
|
request = 规则诊断数据集发布(
|
|
dataset_id="failed-rule-diagnostic",
|
|
revision=1,
|
|
rule_id="l002",
|
|
rule_version=1,
|
|
approval_ref="fixed-before-running",
|
|
expires_at=datetime.now(UTC) + timedelta(hours=2),
|
|
samples=tuple(wrong),
|
|
)
|
|
published = env["maintain"].发布规则诊断数据集(
|
|
env["maintenance_identity"], request, env["review"], env["author"]
|
|
)
|
|
failed = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
|
|
assert failed["assessment"]["status"] == "failed"
|
|
assert failed["assessment"]["totals"]["false_positives"] == 1
|
|
|
|
|
|
@pytest.mark.case_id(
|
|
"NC-w25-25d103",
|
|
environment="隔离PG/合成HTTP及浏览器",
|
|
given="固定请求、来源与作者;按用例环境准备独立测试输入",
|
|
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
|
then=["S01审阅启停与确切规则消费及停止恢复"],
|
|
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
|
)
|
|
def test_S01审阅启停与确切规则消费及停止恢复__25d103(规则资格环境):
|
|
env = 规则资格环境
|
|
before = env["review"].生成诊断(env["author"], env["target"])
|
|
assert before["report"]["findings"] == []
|
|
published = _publish(env, "declared-owned-rule-diagnostic", "owned")
|
|
run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"])
|
|
assert run["assessment"]["status"] == "passed"
|
|
proof = env["maintain"].导出规则诊断凭据(
|
|
env["maintenance_identity"],
|
|
run["run_id"],
|
|
批准引用="maintainer-exported-exact-proof",
|
|
有效期=datetime.now(UTC) + timedelta(hours=1),
|
|
)
|
|
assert proof["payload"]["target"] == published["target"]
|
|
assert "gold" not in str(proof).lower()
|
|
for table in ("oracle.muse_dataset_answers", "evaluation.muse_rule_diagnostic_result"):
|
|
with (
|
|
env["pools"][用途.生产].连接() as conn,
|
|
pytest.raises(psycopg.errors.InsufficientPrivilege),
|
|
):
|
|
conn.execute("SELECT * FROM " + table)
|
|
|
|
command = 规则状态命令(
|
|
"l002", 1, "enable", expected_activation_revision=0, credential_id=proof["receipt_id"]
|
|
)
|
|
opened = env["review"].打开规则状态审阅(env["author"], command)
|
|
receipt = env["review"].决定规则状态(
|
|
env["author"],
|
|
"enable-l002-v1",
|
|
command,
|
|
审阅ID=opened["review_id"],
|
|
审阅哈希=opened["review_hash"],
|
|
)
|
|
assert (
|
|
env["review"].决定规则状态(
|
|
env["author"],
|
|
"enable-l002-v1",
|
|
command,
|
|
审阅ID=opened["review_id"],
|
|
审阅哈希=opened["review_hash"],
|
|
)
|
|
== receipt
|
|
)
|
|
active = env["review"].生成诊断(env["author"], env["target"])
|
|
assert active["report"]["findings"][0]["rule_id"] == "l002"
|
|
assert active["report"]["rule_bindings"][0]["version"] == 1
|
|
assert active["report"]["rule_bindings"][0]["credential_id"] == proof["receipt_id"]
|
|
assert env["review"].读取规则版本(env["author"], "l002", 1)["activation_recorded"]
|
|
|
|
version_two = deepcopy(_单规则库())
|
|
version_two["rules"]["l002"]["version"] = 2
|
|
version_two["rules"]["l002"]["fix_hint"] = "新版仍只询问,不自动修改"
|
|
version_two = 核对规则库(
|
|
list(version_two["rules"].values()), list(version_two["samples"].values())
|
|
)
|
|
env["review"].导入规则候选(env["author"], 规则库=version_two)
|
|
still_v1 = env["review"].生成诊断(env["author"], env["target"])
|
|
assert still_v1["report"]["rule_bindings"][0]["version"] == 1
|
|
assert (
|
|
env["review"].读取诊断(env["author"], active["diagnosis_id"])["report"] == active["report"]
|
|
)
|
|
with pytest.raises(审校错误, match="旧规则版本"):
|
|
env["review"].打开规则状态审阅(
|
|
env["author"],
|
|
规则状态命令(
|
|
"l002",
|
|
1,
|
|
"enable",
|
|
expected_activation_revision=1,
|
|
credential_id=proof["receipt_id"],
|
|
),
|
|
)
|
|
|
|
env["evaluate"].停止规则诊断验证(env["evaluation_identity"], run["run_id"], "stop-l002-proof")
|
|
state = env["review"].读取规则状态(env["author"], "l002")
|
|
assert not state["consumable"] and "失效" in state["reason"]
|
|
with pytest.raises(审校错误, match="凭据失效"):
|
|
env["review"].生成诊断(env["author"], env["target"])
|
|
history = env["review"].读取诊断(env["author"], active["diagnosis_id"])
|
|
assert history["report"] == active["report"] and history["stale"]
|
|
|
|
disable = 规则状态命令("l002", 1, "disable", expected_activation_revision=1)
|
|
opened = env["review"].打开规则状态审阅(env["author"], disable)
|
|
env["review"].决定规则状态(
|
|
env["author"],
|
|
"disable-l002-v1",
|
|
disable,
|
|
审阅ID=opened["review_id"],
|
|
审阅哈希=opened["review_hash"],
|
|
)
|
|
after = env["review"].生成诊断(env["author"], env["target"])
|
|
assert after["report"]["findings"] == []
|
|
state = env["review"].读取规则状态(env["author"], "l002")
|
|
assert state["current"]["state"] == "disabled" and len(state["history"]) == 2
|
|
assert (
|
|
env["review"].读取诊断(env["author"], active["diagnosis_id"])["report"] == active["report"]
|
|
)
|
|
|
|
|
|
@pytest.mark.case_id(
|
|
"NC-w25-25d106",
|
|
environment="隔离PG/合成HTTP及浏览器",
|
|
given="固定请求、来源与作者;按用例环境准备独立测试输入",
|
|
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
|
then=["程序资源变化后旧规则凭据不能新导出或启用"],
|
|
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
|
)
|
|
def test_程序资源变化后旧规则凭据不能新导出或启用__25d106(规则资格环境, monkeypatch):
|
|
from muse.效果评测 import 规则诊断
|
|
|
|
env = 规则资格环境
|
|
data = _publish(env, "build-change", "owned")
|
|
run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], data["version_id"])
|
|
proof = env["maintain"].导出规则诊断凭据(
|
|
env["maintenance_identity"],
|
|
run["run_id"],
|
|
批准引用="fixed-before-build-change",
|
|
有效期=datetime.now(UTC) + timedelta(hours=1),
|
|
)
|
|
monkeypatch.setattr(规则诊断, "加载清单", lambda: {"构建身份": "a" * 64})
|
|
with pytest.raises(评测错误, match="程序资源已变化"):
|
|
env["maintain"].导出规则诊断凭据(
|
|
env["maintenance_identity"],
|
|
run["run_id"],
|
|
批准引用="new-export",
|
|
有效期=datetime.now(UTC) + timedelta(minutes=30),
|
|
)
|
|
command = 规则状态命令("l002", 1, "enable", credential_id=proof["receipt_id"])
|
|
with pytest.raises(审校错误, match="凭据"):
|
|
env["review"].打开规则状态审阅(env["author"], command)
|
|
assert env["review"].读取规则状态(env["author"], "l002")["current"] is None
|