实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
314 lines
14 KiB
Python
314 lines
14 KiB
Python
"""角色策略来自可审配置;不发模型请求。"""
|
||
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
import yaml
|
||
|
||
from muse.任务运行.接口 import 模型协议错误, 角色策略目录
|
||
|
||
仓库根 = Path(__file__).resolve().parents[2]
|
||
源码根 = 仓库根 / "src" / "muse"
|
||
|
||
|
||
评委模型 = ("deepseek-v4.1-flash", "muse-spark-1.3-contributor")
|
||
固定模型 = "claude-opus-4-8[1M]"
|
||
治理模型 = (
|
||
固定模型,
|
||
"claude-opus-4-8",
|
||
"MiniMax-M3",
|
||
"MiniMax-M2.7",
|
||
"glm-5.2",
|
||
"deepseek-v4-flash",
|
||
"qwen3.8-flash",
|
||
)
|
||
模型别名 = (
|
||
"deepseekv4.1flash",
|
||
"deepseek-v4-flash",
|
||
"deepseek-flash",
|
||
"muse spark1.3",
|
||
"muse-spark-1.2-contributor",
|
||
"zen-go/deepseek-v4.1-flash",
|
||
"catproxy-openai/muse-spark-1.3-contributor",
|
||
"DEEPSEEK-V4.1-FLASH",
|
||
"muse-spark-1.3-contributor ",
|
||
)
|
||
|
||
|
||
def 读取策略() -> dict:
|
||
return yaml.safe_load((仓库根 / "配置/角色策略.yaml").read_text())
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-role-fixed-model",
|
||
environment="离线规则;无真实模型调用",
|
||
given="已审角色配置种子",
|
||
when="冻结角色及模型工具策略",
|
||
then=[
|
||
(
|
||
"writer/planner 保持原固定策略;judge 按当前版本的独立精确白名单冻结,拒绝未登记模型"
|
||
"和别名;单次实际响应身份不得替换"
|
||
)
|
||
],
|
||
contract="docs/系统架构/新版设计/编排与能力资源.md",
|
||
)
|
||
@pytest.mark.parametrize("角色", ["writer", "planner", "judge"], ids=["writer", "planner", "judge"])
|
||
def test_固定角色拒绝降低模型能力__a63001(角色) -> None:
|
||
"""judge 接受现行版本登记的初评模型,写手与规划保持原单模型及响应映射。"""
|
||
目录 = 角色策略目录(读取策略())
|
||
允许 = 评委模型 if 角色 == "judge" else (固定模型,)
|
||
assert tuple(目录.定义["models"][目录.定义["roles"][角色]["model_policy"]]) == 允许
|
||
for model in 允许:
|
||
固定 = 目录.冻结(角色, provider="configured", model=model, thinking="high", 阶段="生成")
|
||
assert 固定.model == model and 固定.策略版本 == "role-policy-r2-v2"
|
||
assert 固定.允许实际模型 == ((model,) if 角色 == "judge" else (固定模型, "claude-opus-4-8"))
|
||
for model in {*治理模型, *评委模型, *模型别名, "claude-opus-4-8", "unknown-model"} - set(允许):
|
||
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
|
||
目录.冻结(角色, provider="configured", model=model, 阶段="生成")
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-role-tools-phase",
|
||
environment="离线规则;无真实模型调用",
|
||
given="已审角色配置种子",
|
||
when="冻结角色及模型工具策略",
|
||
then=["生成写手零工具、盲评冻结输入、角色别名归一"],
|
||
contract="docs/系统架构/新版设计/编排与能力资源.md",
|
||
)
|
||
def test_写手生成与盲评拒绝工具__a63002() -> None:
|
||
目录 = 角色策略目录(读取策略())
|
||
with pytest.raises(模型协议错误, match="该角色阶段不允许使用工具"):
|
||
目录.冻结("writer", provider="configured", model=固定模型, 工具=("read",), 阶段="生成")
|
||
for model in 评委模型:
|
||
assert 目录.冻结("blind_judge", provider="configured", model=model).角色 == "judge"
|
||
with pytest.raises(模型协议错误, match="盲评只使用冻结匿名输入"):
|
||
目录.冻结("blind_judge", provider="configured", model=model, 工具=("read",))
|
||
for model in (*治理模型, *模型别名, "unknown-model"):
|
||
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
|
||
目录.冻结("blind_judge", provider="configured", model=model)
|
||
for 角色, 阶段, model in [
|
||
("writer", "探索", 固定模型),
|
||
("planner", "执行", 固定模型),
|
||
("judge", "执行", 评委模型[0]),
|
||
]:
|
||
assert 目录.冻结(
|
||
角色, provider="configured", model=model, 工具=("read",), 阶段=阶段
|
||
).工具 == ("read",)
|
||
assert (
|
||
目录.冻结("semantic_detector", provider="configured", model="MiniMax-M3").角色 == "detector"
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-judge-other-roles-a63003",
|
||
environment="离线契约;不调用真实模型或数据库",
|
||
given="可审的角色策略配置与合成模型身份",
|
||
when="冻结 detector/extractor 的原模型与新 judge 模型",
|
||
then=["原治理集合逐项可用,新 judge 模型不扩散到其他角色"],
|
||
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
|
||
)
|
||
@pytest.mark.parametrize("角色", ["detector", "extractor"], ids=["detector", "extractor"])
|
||
def test_其他角色保持原治理模型集合__a63003(角色) -> None:
|
||
目录 = 角色策略目录(读取策略())
|
||
assert 目录.定义["roles"][角色]["model_policy"] == "governed"
|
||
assert tuple(目录.定义["models"]["governed"]) == 治理模型
|
||
for model in 治理模型:
|
||
assert 目录.冻结(角色, provider="configured", model=model).model == model
|
||
for model in 评委模型:
|
||
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
|
||
目录.冻结(角色, provider="configured", model=model)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-judge-policy-boundary-a63004",
|
||
environment="离线契约;不调用真实模型或数据库",
|
||
given="可审的角色策略配置与合成模型身份",
|
||
when="将角色策略改成其他角色的模型策略",
|
||
then=["writer/planner 与 judge 的策略串用均拒绝"],
|
||
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
|
||
)
|
||
@pytest.mark.parametrize(
|
||
("角色", "策略"),
|
||
[
|
||
("writer", "governed"),
|
||
("planner", "governed"),
|
||
("writer", "judge-fixed"),
|
||
("planner", "judge-fixed"),
|
||
("judge", "fixed"),
|
||
("judge", "governed"),
|
||
],
|
||
ids=[
|
||
"writer-governed",
|
||
"planner-governed",
|
||
"writer-judge-fixed",
|
||
"planner-judge-fixed",
|
||
"judge-fixed",
|
||
"judge-governed",
|
||
],
|
||
)
|
||
def test_固定角色拒绝策略串用__a63004(角色, 策略) -> None:
|
||
定义 = 读取策略()
|
||
定义["roles"][角色]["model_policy"] = 策略
|
||
with pytest.raises(模型协议错误):
|
||
角色策略目录(定义)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-judge-response-identity-a63005",
|
||
environment="离线契约;不调用真实模型或数据库",
|
||
given="可审的角色策略配置与合成模型身份",
|
||
when="校验正确模型及其他合法评委、别名、缺失模型的合成响应",
|
||
then=["仅本次冻结准确模型可交付,白名单其他成员不能替换响应身份"],
|
||
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
|
||
)
|
||
@pytest.mark.parametrize(
|
||
"model", 评委模型, ids=["deepseek-v4.1-flash", "muse-spark-1.3-contributor"]
|
||
)
|
||
def test_评委合成响应仅接受本次冻结模型__a63005(model) -> None:
|
||
"""合成结果直接经过现有 S02 输出校验,不调用宿主或模型。"""
|
||
from muse.任务运行.接口 import 模型结果, 模型请求
|
||
from muse.任务运行.模型调用 import 校验模型输出
|
||
|
||
策略 = 角色策略目录(读取策略()).冻结("judge", provider="configured", model=model)
|
||
请求 = 模型请求(
|
||
"synthetic-judge-call",
|
||
策略.provider,
|
||
策略.model,
|
||
"合成系统说明",
|
||
"合成匿名候选",
|
||
{"type": "object"},
|
||
1024,
|
||
30,
|
||
允许实际模型=策略.允许实际模型,
|
||
)
|
||
assert 校验模型输出(请求, 模型结果("completed", "{}", model, None)) == {}
|
||
for 实际模型 in (*评委模型, *治理模型, *模型别名, "unknown-model", None):
|
||
if 实际模型 == model:
|
||
continue
|
||
with pytest.raises(模型协议错误, match="实际模型与冻结模型策略不符"):
|
||
校验模型输出(请求, 模型结果("completed", "{}", 实际模型, None))
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-judge-version-extension-a63006",
|
||
environment="离线契约;不调用真实模型或数据库",
|
||
given="可审的角色策略配置与合成模型身份",
|
||
when="在独立新策略版本登记合成第三评委",
|
||
then=[
|
||
"新版本允许准确第三模型及现有初评",
|
||
"旧版本与冻结结果不变",
|
||
"别名与通配请求拒绝,writer/planner 仍保持原策略",
|
||
],
|
||
contract="docs/系统架构/新版设计/模块设计/S02-任务运行.md",
|
||
)
|
||
def test_评委新策略可扩充准确模型且不放宽其他角色__a63006() -> None:
|
||
"""合成第三模型仅验证策略扩充合同,不代表真实模型获准或可用。"""
|
||
第三模型 = "synthetic-judge-third-v1"
|
||
旧目录 = 角色策略目录(读取策略())
|
||
旧冻结 = 旧目录.冻结("judge", provider="synthetic", model=评委模型[0])
|
||
新定义 = 读取策略()
|
||
新定义["version"] = "synthetic-role-policy-v3"
|
||
新定义["models"]["judge-fixed"].append(第三模型)
|
||
新目录 = 角色策略目录(新定义)
|
||
|
||
for 角色 in ("judge", "blind_judge"):
|
||
for model in (*评委模型, 第三模型):
|
||
冻结 = 新目录.冻结(角色, provider="synthetic", model=model)
|
||
assert 冻结.model == model and 冻结.允许实际模型 == (model,)
|
||
assert 冻结.策略版本 == "synthetic-role-policy-v3"
|
||
for model in ("synthetic-judge-third", 第三模型.upper(), "synthetic-judge-*", "*"):
|
||
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
|
||
新目录.冻结(角色, provider="synthetic", model=model)
|
||
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
|
||
旧目录.冻结(角色, provider="synthetic", model=第三模型)
|
||
|
||
for 角色 in ("writer", "planner"):
|
||
assert (
|
||
新目录.冻结(角色, provider="synthetic", model=固定模型, 阶段="生成").model == 固定模型
|
||
)
|
||
for model in (*评委模型, 第三模型):
|
||
with pytest.raises(模型协议错误, match="请求模型不符合角色能力策略"):
|
||
新目录.冻结(角色, provider="synthetic", model=model, 阶段="生成")
|
||
assert 新定义["models"]["fixed"] == 旧目录.定义["models"]["fixed"]
|
||
assert 新定义["models"]["governed"] == 旧目录.定义["models"]["governed"]
|
||
assert 旧冻结.策略版本 == 旧目录.定义["version"] == "role-policy-r2-v2"
|
||
assert tuple(旧目录.定义["models"]["judge-fixed"]) == 评委模型
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-5274c6c53f9c",
|
||
environment="离线协议替身;真实调用单独验收",
|
||
when="校验角色能力、配置版本与探针绑定。",
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
def test_active_runtime_has_no_embedded_api_token__5274c6() -> None:
|
||
"""活跃源码和递归配置格式按同一凭据形状规则检查,错误不回显值。"""
|
||
违规 = _凭据形状问题(仓库根)
|
||
assert 违规 == [], "发现内嵌凭据形状:\n" + "\n".join(违规)
|
||
|
||
|
||
def _凭据形状问题(根: Path) -> list[str]:
|
||
import re
|
||
import tomllib
|
||
|
||
秘密键 = {"apikey", "token", "secret", "password"}
|
||
模式 = (
|
||
re.compile(r"sk-[A-Za-z0-9]{20,}"),
|
||
re.compile(
|
||
r"(?:api[-_]?key|token|secret|password)\s*[:=]\s*[\"'][^\"{}$\n]{8,}[\"']", re.I
|
||
),
|
||
)
|
||
违规 = []
|
||
for 文件 in (根 / "src/muse").rglob("*.py"):
|
||
if "__pycache__" not in 文件.parts and any(规则.search(文件.read_text()) for 规则 in 模式):
|
||
违规.append(f"{文件.relative_to(根)}: 内嵌凭据形状")
|
||
|
||
def 含秘密(值):
|
||
if isinstance(值, dict):
|
||
for 键, 子值 in 值.items():
|
||
if re.sub(r"[-_]", "", str(键).lower()) in 秘密键 and isinstance(子值, str):
|
||
if 子值 and not re.fullmatch(r"\$\{[A-Z_][A-Z0-9_]*\}|<[^<>]+>", 子值):
|
||
return True
|
||
if 含秘密(子值):
|
||
return True
|
||
if isinstance(值, list):
|
||
return any(含秘密(子值) for 子值 in 值)
|
||
return isinstance(值, str) and bool(模式[0].search(值))
|
||
|
||
for 文件 in sorted((根 / "配置").rglob("*")):
|
||
if not 文件.is_file() or 文件.suffix not in {".toml", ".yaml", ".yml"}:
|
||
continue
|
||
try:
|
||
内容 = 文件.read_text(encoding="utf-8")
|
||
配置 = tomllib.loads(内容) if 文件.suffix == ".toml" else yaml.safe_load(内容)
|
||
except (ValueError, yaml.YAMLError):
|
||
违规.append(f"{文件.relative_to(根)}: 配置格式无效")
|
||
continue
|
||
if 含秘密(配置):
|
||
违规.append(f"{文件.relative_to(根)}: 内嵌凭据形状")
|
||
return 违规
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-o04-config-secret-shapes",
|
||
environment="离线,临时配置文件和合成秘密字面量",
|
||
given="嵌套yaml/yml/toml中的受控存储引用与合成明文",
|
||
when="递归扫描配置的凭据形状",
|
||
then=["引用允许,明文拒绝", "报告只给文件路径和错误类别,不泄露命中值"],
|
||
)
|
||
@pytest.mark.parametrize("格式", ["yaml", "yml", "toml"], ids=["yaml", "yml", "toml"])
|
||
def test_凭据扫描递归涵盖配置格式且不泄露值(tmp_path, 格式):
|
||
目录 = tmp_path / "配置/nested"
|
||
目录.mkdir(parents=True)
|
||
允许 = 目录 / f"引用.{格式}"
|
||
拒绝 = 目录 / f"明文.{格式}"
|
||
if 格式 == "toml":
|
||
允许.write_text('[secret]\n"取值方式"="受控存储"\n"位置"="private.key"\n')
|
||
拒绝.write_text('api_key="synthetic-do-not-print-secret"\n')
|
||
else:
|
||
允许.write_text("secret:\n 取值方式: 受控存储\n 位置: private.key\n")
|
||
拒绝.write_text("nested:\n token: synthetic-do-not-print-secret\n")
|
||
结果 = _凭据形状问题(tmp_path)
|
||
assert 结果 == [f"配置/nested/明文.{格式}: 内嵌凭据形状"]
|
||
assert "synthetic-do-not-print-secret" not in str(结果)
|