"""B06确定性规则的逐例资格、S01启停与实际诊断消费。""" from copy import deepcopy from dataclasses import replace from datetime import UTC, datetime, timedelta from typing import Literal import psycopg import pytest from muse.作品规划.接口 import 档案保存 from muse.共享.调用身份 import 内容用途, 用途 from muse.审校修订.接口 import ( 审校错误, 核对规则库, 规则状态命令, 读取内置规则, ) from muse.效果评测.接口 import ( 标注区间, 规则诊断数据集发布, 规则诊断样本, 规则诊断金标, 评测服务, 评测错误, 预期发现, ) from muse.正文写作.接口 import 文本节点, 正文草稿, 段落 pytestmark = pytest.mark.数据库 def _单规则库(): all_rules = 读取内置规则() rule = all_rules["rules"]["l002"] samples = {ref: all_rules["samples"][ref] for refs in rule["samples"].values() for ref in refs} return 核对规则库([rule], list(samples.values())) def _span(text: str, quote: str, disposition: Literal["ask", "retain", "repair"] = "ask"): start = text.index(quote) return 预期发现( start=start, end=start + len(quote), quote=quote, disposition=disposition, ) def _sample( sample_id: str, text: str, kind: Literal["positive", "negative", "protected"], source_kind: Literal["owned", "licensed", "public_domain", "synthetic"], ): quote = "值得注意的是" findings = ( () if kind == "negative" else (_span(text, quote, "retain" if kind == "protected" else "ask"),) ) protected = () if kind == "protected": start = text.index(quote) protected = (标注区间(start=start, end=start + len(quote), quote=quote),) return 规则诊断样本( sample_id=sample_id, source_ref=f"source:{sample_id}", license_ref=f"license:{source_kind}:{sample_id}", source_groups=(f"group:{sample_id}",), source_kind=source_kind, split="holdout", text=text, gold=规则诊断金标( annotation_status="confirmed", annotation_ref=f"annotation:{sample_id}", case_kind=kind, findings=findings, protected_expressions=protected, ), ) def _samples(source_kind: Literal["owned", "licensed", "public_domain", "synthetic"], prefix: str): return ( _sample( prefix + "-positive", "石阶发冷。值得注意的是,灯还亮着。", "positive", source_kind ), _sample(prefix + "-negative", "石阶发冷,灯还亮着。", "negative", source_kind), _sample( prefix + "-protected", "值得注意的是,是陈伯每次开场的口头禅。", "protected", source_kind, ), ) @pytest.fixture def 规则资格环境(章后环境): # 规则验证、凭据导出与审阅必须同库:复用章后环境所在库。 应用测试库 = 章后环境["库组"] app = 章后环境["装配"] author = replace(章后环境["作者"], 内容用途=内容用途.检测) review = app.要求审校() review.导入规则候选(author, 规则库=_单规则库()) works = app.要求作品() structure = works.新档案结构(author, "rule-work", "work_core", 1) works.保存档案( author, "create-rule-work", 档案保存("rule-work", 0, {"名称": "规则诊断作品"}, structure), ) works.添加章节(author, "add-rule-ch", "rule-work", "rule-ch", "诊断章", 预期目录版本=0) app.要求正文().保存人工( author, "save-rule-body", "rule-ch", 0, 正文草稿((段落("p1", (文本节点("值得注意的是,门外已经下起了雨。"),)),)), ) doc = app.要求正文().读取正文(author, "rule-ch") target = { "work_id": "rule-work", "chapter_id": "rule-ch", "revision": doc["revision"], "branch_id": "main", "document_hash": doc["document_hash"], } maintain = 评测服务(应用测试库[用途.维护]) evaluate = 评测服务(应用测试库[用途.评测]) return { "app": app, "review": review, "author": author, "maintain": maintain, "evaluate": evaluate, "maintenance_identity": replace(author, 用途=用途.维护), "evaluation_identity": replace(author, 用途=用途.评测), "pools": 应用测试库, "target": target, } def _publish( env, dataset_id: str, source_kind: Literal["owned", "licensed", "public_domain", "synthetic"], ): request = 规则诊断数据集发布( dataset_id=dataset_id, revision=1, rule_id="l002", rule_version=1, approval_ref="maintainer-approved:" + dataset_id, expires_at=datetime.now(UTC) + timedelta(hours=3), samples=_samples(source_kind, dataset_id), ) return env["maintain"].发布规则诊断数据集( env["maintenance_identity"], request, env["review"], env["author"] ) @pytest.mark.case_id( "NC-w25-25d101", environment="隔离PG/合成HTTP及浏览器", given="固定请求、来源与作者;按用例环境准备独立测试输入", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["合成逐例保留完整证据但不导出启用凭据"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_合成逐例保留完整证据但不导出启用凭据__25d101(规则资格环境): env = 规则资格环境 published = _publish(env, "synthetic-rule-diagnostic", "synthetic") run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"]) assert run["assessment"]["status"] == "engineering_only" assert run["assessment"]["activation_eligible"] is False assert run["assessment"]["totals"] == { "false_positives": 0, "false_negatives": 0, "disposition_mismatches": 0, "protection_violations": 0, "unknown_items": 0, } assert len(run["results"]) == 3 assert env["evaluate"].读取规则诊断验证(env["evaluation_identity"], run["run_id"]) == run with pytest.raises(评测错误, match="不导出"): env["maintain"].导出规则诊断凭据( env["maintenance_identity"], run["run_id"], 批准引用="synthetic-must-not-enable", 有效期=datetime.now(UTC) + timedelta(hours=1), ) training_text = _单规则库()["samples"]["sf-l002-01"]["text"] request = 规则诊断数据集发布( dataset_id="training-leak", revision=1, rule_id="l002", rule_version=1, approval_ref="reject-training", expires_at=datetime.now(UTC) + timedelta(hours=2), samples=(_sample("copied-training", training_text, "positive", "synthetic"),), ) with pytest.raises(评测错误, match="训练例证"): env["maintain"].发布规则诊断数据集( env["maintenance_identity"], request, env["review"], env["author"] ) @pytest.mark.case_id( "NC-w25-25d102", environment="隔离PG/合成HTTP及浏览器", given="固定请求、来源与作者;按用例环境准备独立测试输入", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["金标缺失和误标保留未完成与硬失败"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_金标缺失和误标保留未完成与硬失败__25d102(规则资格环境): env = 规则资格环境 samples = list(_samples("owned", "incomplete")) samples[0] = samples[0].model_copy( update={ "gold": 规则诊断金标( annotation_status="unknown", annotation_ref="", case_kind="unknown" ), } ) request = 规则诊断数据集发布( dataset_id="incomplete-rule-diagnostic", revision=1, rule_id="l002", rule_version=1, approval_ref="annotation-still-open", expires_at=datetime.now(UTC) + timedelta(hours=2), samples=tuple(samples), ) published = env["maintain"].发布规则诊断数据集( env["maintenance_identity"], request, env["review"], env["author"] ) result = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"]) assert result["assessment"]["status"] == "insufficient_evidence" assert result["assessment"]["totals"]["unknown_items"] == 1 assert any(row["execution_status"] == "unknown_annotation" for row in result["results"]) wrong = list(_samples("owned", "wrong-label")) wrong[1] = wrong[1].model_copy(update={"text": "值得注意的是,这条负例其实会命中。"}) request = 规则诊断数据集发布( dataset_id="failed-rule-diagnostic", revision=1, rule_id="l002", rule_version=1, approval_ref="fixed-before-running", expires_at=datetime.now(UTC) + timedelta(hours=2), samples=tuple(wrong), ) published = env["maintain"].发布规则诊断数据集( env["maintenance_identity"], request, env["review"], env["author"] ) failed = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"]) assert failed["assessment"]["status"] == "failed" assert failed["assessment"]["totals"]["false_positives"] == 1 @pytest.mark.case_id( "NC-w25-25d103", environment="隔离PG/合成HTTP及浏览器", given="固定请求、来源与作者;按用例环境准备独立测试输入", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["S01审阅启停与确切规则消费及停止恢复"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_S01审阅启停与确切规则消费及停止恢复__25d103(规则资格环境): env = 规则资格环境 before = env["review"].生成诊断(env["author"], env["target"]) assert before["report"]["findings"] == [] published = _publish(env, "declared-owned-rule-diagnostic", "owned") run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], published["version_id"]) assert run["assessment"]["status"] == "passed" proof = env["maintain"].导出规则诊断凭据( env["maintenance_identity"], run["run_id"], 批准引用="maintainer-exported-exact-proof", 有效期=datetime.now(UTC) + timedelta(hours=1), ) assert proof["payload"]["target"] == published["target"] assert "gold" not in str(proof).lower() for table in ("oracle.muse_dataset_answers", "evaluation.muse_rule_diagnostic_result"): with ( env["pools"][用途.生产].连接() as conn, pytest.raises(psycopg.errors.InsufficientPrivilege), ): conn.execute("SELECT * FROM " + table) command = 规则状态命令( "l002", 1, "enable", expected_activation_revision=0, credential_id=proof["receipt_id"] ) opened = env["review"].打开规则状态审阅(env["author"], command) receipt = env["review"].决定规则状态( env["author"], "enable-l002-v1", command, 审阅ID=opened["review_id"], 审阅哈希=opened["review_hash"], ) assert ( env["review"].决定规则状态( env["author"], "enable-l002-v1", command, 审阅ID=opened["review_id"], 审阅哈希=opened["review_hash"], ) == receipt ) active = env["review"].生成诊断(env["author"], env["target"]) assert active["report"]["findings"][0]["rule_id"] == "l002" assert active["report"]["rule_bindings"][0]["version"] == 1 assert active["report"]["rule_bindings"][0]["credential_id"] == proof["receipt_id"] assert env["review"].读取规则版本(env["author"], "l002", 1)["activation_recorded"] version_two = deepcopy(_单规则库()) version_two["rules"]["l002"]["version"] = 2 version_two["rules"]["l002"]["fix_hint"] = "新版仍只询问,不自动修改" version_two = 核对规则库( list(version_two["rules"].values()), list(version_two["samples"].values()) ) env["review"].导入规则候选(env["author"], 规则库=version_two) still_v1 = env["review"].生成诊断(env["author"], env["target"]) assert still_v1["report"]["rule_bindings"][0]["version"] == 1 assert ( env["review"].读取诊断(env["author"], active["diagnosis_id"])["report"] == active["report"] ) with pytest.raises(审校错误, match="旧规则版本"): env["review"].打开规则状态审阅( env["author"], 规则状态命令( "l002", 1, "enable", expected_activation_revision=1, credential_id=proof["receipt_id"], ), ) env["evaluate"].停止规则诊断验证(env["evaluation_identity"], run["run_id"], "stop-l002-proof") state = env["review"].读取规则状态(env["author"], "l002") assert not state["consumable"] and "失效" in state["reason"] with pytest.raises(审校错误, match="凭据失效"): env["review"].生成诊断(env["author"], env["target"]) history = env["review"].读取诊断(env["author"], active["diagnosis_id"]) assert history["report"] == active["report"] and history["stale"] disable = 规则状态命令("l002", 1, "disable", expected_activation_revision=1) opened = env["review"].打开规则状态审阅(env["author"], disable) env["review"].决定规则状态( env["author"], "disable-l002-v1", disable, 审阅ID=opened["review_id"], 审阅哈希=opened["review_hash"], ) after = env["review"].生成诊断(env["author"], env["target"]) assert after["report"]["findings"] == [] state = env["review"].读取规则状态(env["author"], "l002") assert state["current"]["state"] == "disabled" and len(state["history"]) == 2 assert ( env["review"].读取诊断(env["author"], active["diagnosis_id"])["report"] == active["report"] ) @pytest.mark.case_id( "NC-w25-25d106", environment="隔离PG/合成HTTP及浏览器", given="固定请求、来源与作者;按用例环境准备独立测试输入", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["程序资源变化后旧规则凭据不能新导出或启用"], contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md", ) def test_程序资源变化后旧规则凭据不能新导出或启用__25d106(规则资格环境, monkeypatch): from muse.效果评测 import 规则诊断 env = 规则资格环境 data = _publish(env, "build-change", "owned") run = env["evaluate"].执行规则诊断验证(env["evaluation_identity"], data["version_id"]) proof = env["maintain"].导出规则诊断凭据( env["maintenance_identity"], run["run_id"], 批准引用="fixed-before-build-change", 有效期=datetime.now(UTC) + timedelta(hours=1), ) monkeypatch.setattr(规则诊断, "加载清单", lambda: {"构建身份": "a" * 64}) with pytest.raises(评测错误, match="程序资源已变化"): env["maintain"].导出规则诊断凭据( env["maintenance_identity"], run["run_id"], 批准引用="new-export", 有效期=datetime.now(UTC) + timedelta(minutes=30), ) command = 规则状态命令("l002", 1, "enable", credential_id=proof["receipt_id"]) with pytest.raises(审校错误, match="凭据"): env["review"].打开规则状态审阅(env["author"], command) assert env["review"].读取规则状态(env["author"], "l002")["current"] is None