muse-agent-example/tests/契约/test_角色能力与配置.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

314 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""角色策略来自可审配置;不发模型请求。"""
from pathlib import Path
import pytest
import yaml
from muse.任务运行.接口 import 模型协议错误, 角色策略目录
仓库根 = Path(__file__).resolve().parents[2]
源码根 = 仓库根 / "src" / "muse"
评委模型 = ("deepseek-v4.1-flash", "muse-spark-1.3-contributor")
固定模型 = "claude-opus-4-8[1M]"
治理模型 = (
固定模型,
"claude-opus-4-8",
"MiniMax-M3",
"MiniMax-M2.7",
"glm-5.2",
"deepseek-v4-flash",
"qwen3.8-flash",
)
模型别名 = (
"deepseekv4.1flash",
"deepseek-v4-flash",
"deepseek-flash",
"muse spark1.3",
"muse-spark-1.2-contributor",
"zen-go/deepseek-v4.1-flash",
"catproxy-openai/muse-spark-1.3-contributor",
"DEEPSEEK-V4.1-FLASH",
"muse-spark-1.3-contributor ",
)
def 读取策略() -> dict:
return yaml.safe_load((仓库根 / "配置/角色策略.yaml").read_text())
@pytest.mark.case_id(
"NC-role-fixed-model",
environment="离线规则;无真实模型调用",
given="已审角色配置种子",
when="冻结角色及模型工具策略",
then=[
(
"writer/planner 保持原固定策略;judge 按当前版本的独立精确白名单冻结,拒绝未登记模型"
"和别名;单次实际响应身份不得替换"
)
],
contract="docs/系统架构/新版设计/编排与能力资源.md",
)
@pytest.mark.parametrize("角色", ["writer", "planner", "judge"], ids=["writer", "planner", "judge"])
def test_固定角色拒绝降低模型能力__a63001(角色) -> None:
"""judge 接受现行版本登记的初评模型,写手与规划保持原单模型及响应映射。"""
目录 = 角色策略目录(读取策略())
允许 = 评委模型 if 角色 == "judge" else (固定模型,)
assert tuple(目录.定义["models"][目录.定义["roles"][角色]["model_policy"]]) == 允许
for model in 允许:
固定 = 目录.冻结(角色, provider="configured", model=model, thinking="high", 阶段="生成")
assert 固定.model == model and 固定.策略版本 == "role-policy-r2-v2"
assert 固定.允许实际模型 == ((model,) if 角色 == "judge" else (固定模型, "claude-opus-4-8"))
for model in {*治理模型, *评委模型, *模型别名, "claude-opus-4-8", "unknown-model"} - set(允许):
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
目录.冻结(角色, provider="configured", model=model, 阶段="生成")
@pytest.mark.case_id(
"NC-role-tools-phase",
environment="离线规则;无真实模型调用",
given="已审角色配置种子",
when="冻结角色及模型工具策略",
then=["生成写手零工具、盲评冻结输入、角色别名归一"],
contract="docs/系统架构/新版设计/编排与能力资源.md",
)
def test_写手生成与盲评拒绝工具__a63002() -> None:
目录 = 角色策略目录(读取策略())
with pytest.raises(模型协议错误, match="该角色阶段不允许使用工具"):
目录.冻结("writer", provider="configured", model=固定模型, 工具=("read",), 阶段="生成")
for model in 评委模型:
assert 目录.冻结("blind_judge", provider="configured", model=model).角色 == "judge"
with pytest.raises(模型协议错误, match="盲评只使用冻结匿名输入"):
目录.冻结("blind_judge", provider="configured", model=model, 工具=("read",))
for model in (*治理模型, *模型别名, "unknown-model"):
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
目录.冻结("blind_judge", provider="configured", model=model)
for 角色, 阶段, model in [
("writer", "探索", 固定模型),
("planner", "执行", 固定模型),
("judge", "执行", 评委模型[0]),
]:
assert 目录.冻结(
角色, provider="configured", model=model, 工具=("read",), 阶段=阶段
).工具 == ("read",)
assert (
目录.冻结("semantic_detector", provider="configured", model="MiniMax-M3").角色 == "detector"
)
@pytest.mark.case_id(
"NC-judge-other-roles-a63003",
environment="离线契约;不调用真实模型或数据库",
given="可审的角色策略配置与合成模型身份",
when="冻结 detector/extractor 的原模型与新 judge 模型",
then=["原治理集合逐项可用,新 judge 模型不扩散到其他角色"],
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
)
@pytest.mark.parametrize("角色", ["detector", "extractor"], ids=["detector", "extractor"])
def test_其他角色保持原治理模型集合__a63003(角色) -> None:
目录 = 角色策略目录(读取策略())
assert 目录.定义["roles"][角色]["model_policy"] == "governed"
assert tuple(目录.定义["models"]["governed"]) == 治理模型
for model in 治理模型:
assert 目录.冻结(角色, provider="configured", model=model).model == model
for model in 评委模型:
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
目录.冻结(角色, provider="configured", model=model)
@pytest.mark.case_id(
"NC-judge-policy-boundary-a63004",
environment="离线契约;不调用真实模型或数据库",
given="可审的角色策略配置与合成模型身份",
when="将角色策略改成其他角色的模型策略",
then=["writer/planner 与 judge 的策略串用均拒绝"],
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
)
@pytest.mark.parametrize(
("角色", "策略"),
[
("writer", "governed"),
("planner", "governed"),
("writer", "judge-fixed"),
("planner", "judge-fixed"),
("judge", "fixed"),
("judge", "governed"),
],
ids=[
"writer-governed",
"planner-governed",
"writer-judge-fixed",
"planner-judge-fixed",
"judge-fixed",
"judge-governed",
],
)
def test_固定角色拒绝策略串用__a63004(角色, 策略) -> None:
定义 = 读取策略()
定义["roles"][角色]["model_policy"] = 策略
with pytest.raises(模型协议错误):
角色策略目录(定义)
@pytest.mark.case_id(
"NC-judge-response-identity-a63005",
environment="离线契约;不调用真实模型或数据库",
given="可审的角色策略配置与合成模型身份",
when="校验正确模型及其他合法评委、别名、缺失模型的合成响应",
then=["仅本次冻结准确模型可交付,白名单其他成员不能替换响应身份"],
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
)
@pytest.mark.parametrize(
"model", 评委模型, ids=["deepseek-v4.1-flash", "muse-spark-1.3-contributor"]
)
def test_评委合成响应仅接受本次冻结模型__a63005(model) -> None:
"""合成结果直接经过现有 S02 输出校验,不调用宿主或模型。"""
from muse.任务运行.接口 import 模型结果, 模型请求
from muse.任务运行.模型调用 import 校验模型输出
策略 = 角色策略目录(读取策略()).冻结("judge", provider="configured", model=model)
请求 = 模型请求(
"synthetic-judge-call",
策略.provider,
策略.model,
"合成系统说明",
"合成匿名候选",
{"type": "object"},
1024,
30,
允许实际模型=策略.允许实际模型,
)
assert 校验模型输出(请求, 模型结果("completed", "{}", model, None)) == {}
for 实际模型 in (*评委模型, *治理模型, *模型别名, "unknown-model", None):
if 实际模型 == model:
continue
with pytest.raises(模型协议错误, match="实际模型与冻结模型策略不符"):
校验模型输出(请求, 模型结果("completed", "{}", 实际模型, None))
@pytest.mark.case_id(
"NC-judge-version-extension-a63006",
environment="离线契约;不调用真实模型或数据库",
given="可审的角色策略配置与合成模型身份",
when="在独立新策略版本登记合成第三评委",
then=[
"新版本允许准确第三模型及现有初评",
"旧版本与冻结结果不变",
"别名与通配请求拒绝,writer/planner 仍保持原策略",
],
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
)
def test_评委新策略可扩充准确模型且不放宽其他角色__a63006() -> None:
"""合成第三模型仅验证策略扩充合同,不代表真实模型获准或可用。"""
第三模型 = "synthetic-judge-third-v1"
旧目录 = 角色策略目录(读取策略())
旧冻结 = 旧目录.冻结("judge", provider="synthetic", model=评委模型[0])
新定义 = 读取策略()
新定义["version"] = "synthetic-role-policy-v3"
新定义["models"]["judge-fixed"].append(第三模型)
新目录 = 角色策略目录(新定义)
for 角色 in ("judge", "blind_judge"):
for model in (*评委模型, 第三模型):
冻结 = 新目录.冻结(角色, provider="synthetic", model=model)
assert 冻结.model == model and 冻结.允许实际模型 == (model,)
assert 冻结.策略版本 == "synthetic-role-policy-v3"
for model in ("synthetic-judge-third", 第三模型.upper(), "synthetic-judge-*", "*"):
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
新目录.冻结(角色, provider="synthetic", model=model)
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
旧目录.冻结(角色, provider="synthetic", model=第三模型)
for 角色 in ("writer", "planner"):
assert (
新目录.冻结(角色, provider="synthetic", model=固定模型, 阶段="生成").model == 固定模型
)
for model in (*评委模型, 第三模型):
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
新目录.冻结(角色, provider="synthetic", model=model, 阶段="生成")
assert 新定义["models"]["fixed"] == 旧目录.定义["models"]["fixed"]
assert 新定义["models"]["governed"] == 旧目录.定义["models"]["governed"]
assert 旧冻结.策略版本 == 旧目录.定义["version"] == "role-policy-r2-v2"
assert tuple(旧目录.定义["models"]["judge-fixed"]) == 评委模型
@pytest.mark.case_id(
"TC-5274c6c53f9c",
environment="离线协议替身;真实调用单独验收",
when="校验角色能力、配置版本与探针绑定。",
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
def test_active_runtime_has_no_embedded_api_token__5274c6() -> None:
"""活跃源码和递归配置格式按同一凭据形状规则检查,错误不回显值。"""
违规 = _凭据形状问题(仓库根)
assert 违规 == [], "发现内嵌凭据形状:\n" + "\n".join(违规)
def _凭据形状问题(根: Path) -> list[str]:
import re
import tomllib
秘密键 = {"apikey", "token", "secret", "password"}
模式 = (
re.compile(r"sk-[A-Za-z0-9]{20,}"),
re.compile(
r"(?:api[-_]?key|token|secret|password)\s*[:=]\s*[\"'][^\"{}$\n]{8,}[\"']", re.I
),
)
违规 = []
for 文件 in (根 / "src/muse").rglob("*.py"):
if "__pycache__" not in 文件.parts and any(规则.search(文件.read_text()) for 规则 in 模式):
违规.append(f"{文件.relative_to(根)}: 内嵌凭据形状")
def 含秘密(值):
if isinstance(值, dict):
for 键, 子值 in 值.items():
if re.sub(r"[-_]", "", str(键).lower()) in 秘密键 and isinstance(子值, str):
if 子值 and not re.fullmatch(r"\$\{[A-Z_][A-Z0-9_]*\}|<[^<>]+>", 子值):
return True
if 含秘密(子值):
return True
if isinstance(值, list):
return any(含秘密(子值) for 子值 in 值)
return isinstance(值, str) and bool(模式[0].search(值))
for 文件 in sorted((根 / "配置").rglob("*")):
if not 文件.is_file() or 文件.suffix not in {".toml", ".yaml", ".yml"}:
continue
try:
内容 = 文件.read_text(encoding="utf-8")
配置 = tomllib.loads(内容) if 文件.suffix == ".toml" else yaml.safe_load(内容)
except (ValueError, yaml.YAMLError):
违规.append(f"{文件.relative_to(根)}: 配置格式无效")
continue
if 含秘密(配置):
违规.append(f"{文件.relative_to(根)}: 内嵌凭据形状")
return 违规
@pytest.mark.case_id(
"NC-o04-config-secret-shapes",
environment="离线,临时配置文件和合成秘密字面量",
given="嵌套yaml/yml/toml中的受控存储引用与合成明文",
when="递归扫描配置的凭据形状",
then=["引用允许,明文拒绝", "报告只给文件路径和错误类别,不泄露命中值"],
)
@pytest.mark.parametrize("格式", ["yaml", "yml", "toml"], ids=["yaml", "yml", "toml"])
def test_凭据扫描递归涵盖配置格式且不泄露值(tmp_path, 格式):
目录 = tmp_path / "配置/nested"
目录.mkdir(parents=True)
允许 = 目录 / f"引用.{格式}"
拒绝 = 目录 / f"明文.{格式}"
if 格式 == "toml":
允许.write_text('[secret]\n"取值方式"="受控存储"\n"位置"="private.key"\n')
拒绝.write_text('api_key="synthetic-do-not-print-secret"\n')
else:
允许.write_text("secret:\n 取值方式: 受控存储\n 位置: private.key\n")
拒绝.write_text("nested:\n token: synthetic-do-not-print-secret\n")
结果 = _凭据形状问题(tmp_path)
assert 结果 == [f"配置/nested/明文.{格式}: 内嵌凭据形状"]
assert "synthetic-do-not-print-secret" not in str(结果)