实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
210 lines
9.6 KiB
Python
210 lines
9.6 KiB
Python
"""固定实验条件:使用S02真实配置版本,登记不等于实际派发或启用。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from decimal import Decimal
|
||
from typing import Literal
|
||
from uuid import UUID
|
||
|
||
from pydantic import BaseModel, ConfigDict, Field, StrictInt
|
||
|
||
from muse.任务运行.接口 import 角色策略目录, 配置版本管理
|
||
from muse.效果评测.文学评分 import 维度
|
||
from muse.效果评测.模型 import 标定策略, 评测错误
|
||
from muse.效果评测.目标政策 import 方法适用范围, 核对目标范围, 目标标准版本
|
||
from muse.效果评测.配置预检 import 要求文学配置
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
|
||
class _合同(BaseModel):
|
||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||
|
||
|
||
class 目标版本(_合同):
|
||
kind: Literal["method", "rule", "skill", "prompt", "code"]
|
||
target_ref: str = Field(min_length=1)
|
||
version: str = Field(min_length=1)
|
||
content_hash: str = Field(pattern=r"^[0-9a-f]{64}$")
|
||
|
||
|
||
class 配置选择(_合同):
|
||
config_id: str = Field(min_length=1)
|
||
version: str = Field(min_length=1)
|
||
|
||
|
||
class 标定引用(_合同):
|
||
experiment_id: UUID
|
||
receipt_hash: str = Field(pattern=r"^[0-9a-f]{64}$")
|
||
|
||
|
||
class 标定使用(_合同):
|
||
policy: 标定策略
|
||
references: tuple[标定引用, ...] = Field(min_length=1, max_length=8)
|
||
|
||
|
||
class 实验请求(_合同):
|
||
dataset_version_id: UUID
|
||
dataset_hash: str = Field(pattern=r"^[0-9a-f]{64}$")
|
||
split: Literal["discovery", "calibration", "holdout"]
|
||
target: 目标版本
|
||
generator_role: Literal["writer", "planner"]
|
||
generator: 配置选择
|
||
judges: tuple[配置选择, ...] = Field(min_length=1, max_length=8)
|
||
comparison_profile: Literal["preference", "writer_rubric"] = "preference"
|
||
arbitrator: 配置选择 | None = None
|
||
detector: 配置选择 | None = None
|
||
arms: tuple[str, ...] = Field(min_length=2, max_length=8)
|
||
dimensions: tuple[str, ...] = Field(min_length=1, max_length=20)
|
||
max_cost_usd: Decimal = Field(gt=0, allow_inf_nan=False)
|
||
max_calls_per_sample: StrictInt = Field(ge=1, le=100)
|
||
max_judge_corrections: StrictInt = Field(default=1, ge=0, le=2)
|
||
calibration_policy: 标定策略 | None = None
|
||
evaluation_goal: Literal["diagnostic", "qualification"] = "diagnostic"
|
||
calibration_use: 标定使用 | None = None
|
||
effect_policy: Literal["writer-effect-v1"] | 目标标准版本 | None = None
|
||
application_scope: 方法适用范围 | None = None
|
||
effect_dependency_version: Literal["effect-dependency-v2"] | None = None
|
||
reuse_experiments: tuple[UUID, ...] = Field(default=(), max_length=16)
|
||
|
||
def 命令内容(self) -> dict:
|
||
payload = self.model_dump(mode="json")
|
||
if self.effect_policy is None:
|
||
payload.pop("effect_policy")
|
||
if self.application_scope is None:
|
||
payload.pop("application_scope")
|
||
if self.effect_dependency_version is None:
|
||
payload.pop("effect_dependency_version")
|
||
if not self.reuse_experiments:
|
||
payload.pop("reuse_experiments")
|
||
if self.evaluation_goal == "diagnostic":
|
||
payload.pop("evaluation_goal")
|
||
if self.calibration_use is None:
|
||
payload.pop("calibration_use")
|
||
if self.calibration_policy is None:
|
||
payload.pop("calibration_policy")
|
||
# 普通偏好请求沿用原命令身份;新增的默认字段不是条件变化。
|
||
if self.comparison_profile == "preference" and self.arbitrator is None:
|
||
payload.pop("comparison_profile")
|
||
payload.pop("arbitrator")
|
||
return payload
|
||
|
||
|
||
def 固定实验条件(数据库, req: 实验请求) -> dict:
|
||
for rows in (req.arms, req.dimensions):
|
||
if len(set(rows)) != len(rows) or any(not r.strip() for r in rows):
|
||
raise 评测错误("实验臂与维度必须唯一且非空")
|
||
profiles = 配置版本管理(数据库)
|
||
dependency_v2 = bool(
|
||
req.effect_dependency_version
|
||
or req.reuse_experiments
|
||
or (req.effect_policy and req.effect_policy.startswith("writer-goal-"))
|
||
)
|
||
|
||
def resolve(ref, role):
|
||
version = profiles.读取版本(ref.config_id, ref.version)
|
||
declared = version.内容.角色配置.get(role)
|
||
if declared is None:
|
||
raise 评测错误("实际配置版本缺少所需角色")
|
||
# 冻结准确配置哈希和模型;凭据路径/地址只留在S02配置,不复制到实验或模型输入。
|
||
result = {
|
||
"config_id": ref.config_id,
|
||
"version": ref.version,
|
||
"config_hash": version.内容哈希,
|
||
"role": role,
|
||
"provider": declared["provider"],
|
||
"model": declared["model"],
|
||
"thinking": declared["thinking"],
|
||
"resource_release": version.内容.资源发布身份,
|
||
"role_policy": version.内容.角色策略版本,
|
||
"host": version.内容.宿主,
|
||
"host_version": version.内容.宿主版本,
|
||
"pricing_version": version.内容.计价版本,
|
||
"budget_account": version.内容.预算策略引用,
|
||
}
|
||
if dependency_v2:
|
||
endpoints = [p for p in version.内容.提供方 if p.身份 == declared["provider"]]
|
||
if len(endpoints) != 1:
|
||
raise 评测错误("效果依赖需要唯一的实际提供方配置")
|
||
endpoint = endpoints[0]
|
||
result["provider_fingerprint"] = 固定哈希([endpoint.身份, endpoint.协议, endpoint.地址])
|
||
return result
|
||
|
||
generator = resolve(req.generator, req.generator_role)
|
||
literary = req.comparison_profile == "writer_rubric"
|
||
policy = 角色策略目录.从发布包() if literary else None
|
||
if policy is not None:
|
||
要求文学配置(policy, generator=generator)
|
||
judges = [resolve(r, "judge") for r in req.judges]
|
||
if req.evaluation_goal == "qualification" and (
|
||
not literary or req.split != "holdout" or req.calibration_use is None
|
||
):
|
||
raise 评测错误("资格评测需要独立保留集、文学模式及明确的标定引用")
|
||
if req.calibration_use is not None and (
|
||
req.evaluation_goal != "qualification" or req.calibration_policy is not None
|
||
):
|
||
raise 评测错误("标定使用只用于资格评测,标定实验不得递归引用标定")
|
||
if req.calibration_policy is not None and (not literary or req.split != "calibration"):
|
||
raise 评测错误("标定策略只允许用于calibration分割的文学实验")
|
||
if literary and (
|
||
req.generator_role != "writer"
|
||
or req.dimensions != 维度
|
||
or len(judges) != 2
|
||
or req.arbitrator is None
|
||
):
|
||
raise 评测错误("文学评分需要固定五维、两名初评及一名预注册候补评委")
|
||
if not literary and req.arbitrator is not None:
|
||
raise 评测错误("普通偏好实验不能附带未定义条件的第三评委")
|
||
arbitrator = resolve(req.arbitrator, "judge") if req.arbitrator else None
|
||
all_judges = [*judges, *([arbitrator] if arbitrator else [])]
|
||
if policy is not None:
|
||
要求文学配置(policy, generator=generator, judges=all_judges)
|
||
if len({r["model"] for r in all_judges}) != len(all_judges) or any(
|
||
r["model"] == generator["model"] for r in all_judges
|
||
):
|
||
raise 评测错误("独立评委配置不能重复或使用同一生成模型")
|
||
detector = resolve(req.detector, "detector") if req.detector else None
|
||
if detector is not None and detector["model"] == generator["model"]:
|
||
raise 评测错误("语义检测配置不能使用本次生成模型")
|
||
if req.effect_policy is not None and (
|
||
not literary or detector is None or req.calibration_policy is not None
|
||
):
|
||
raise 评测错误("正文效果判据需要文学评分和独立检测,不用于标定实验")
|
||
conditions = req.model_dump(mode="json")
|
||
if req.application_scope is None:
|
||
conditions.pop("application_scope")
|
||
if req.effect_dependency_version is None:
|
||
conditions.pop("effect_dependency_version")
|
||
if req.reuse_experiments:
|
||
if len(set(req.reuse_experiments)) != len(req.reuse_experiments) or req.calibration_policy:
|
||
raise 评测错误("复用来源不得重复,独立标定实验不得复用已有评审")
|
||
conditions["effect_dependency_version"] = "effect-dependency-v2"
|
||
else:
|
||
conditions.pop("reuse_experiments")
|
||
conditions.update(
|
||
generator=generator,
|
||
judges=judges,
|
||
arbitrator=arbitrator,
|
||
detector=detector,
|
||
schema_version="experiment-v1",
|
||
)
|
||
if req.effect_policy is not None:
|
||
from muse.效果评测.启用判据 import 读取效果标准
|
||
|
||
policy = 读取效果标准(req.effect_policy)
|
||
if "goal_policy" in policy:
|
||
if req.application_scope is None or req.target.kind != "method":
|
||
raise 评测错误("按用途目标需要确切方法版本和预注册适用范围")
|
||
核对目标范围(
|
||
policy["goal_policy"],
|
||
req.application_scope.model_dump(mode="json"),
|
||
qualification=req.evaluation_goal == "qualification",
|
||
)
|
||
conditions["effect_dependency_version"] = "effect-dependency-v2"
|
||
elif req.application_scope is not None:
|
||
raise 评测错误("旧效果标准不能附加新的用途资格")
|
||
conditions["effect_policy_snapshot"] = policy
|
||
conditions["effect_policy_hash"] = 固定哈希(policy)
|
||
elif req.application_scope is not None:
|
||
raise 评测错误("适用范围必须绑定具名目标政策")
|
||
return {"conditions": conditions, "conditions_hash": 固定哈希(conditions)}
|