zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

210 lines
9.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""固定实验条件:使用S02真实配置版本,登记不等于实际派发或启用。"""
from __future__ import annotations
from decimal import Decimal
from typing import Literal
from uuid import UUID
from pydantic import BaseModel, ConfigDict, Field, StrictInt
from muse.任务运行.接口 import 角色策略目录, 配置版本管理
from muse.效果评测.文学评分 import 维度
from muse.效果评测.模型 import 标定策略, 评测错误
from muse.效果评测.目标政策 import 方法适用范围, 核对目标范围, 目标标准版本
from muse.效果评测.配置预检 import 要求文学配置
from muse.正式变更.接口 import 固定哈希
class _合同(BaseModel):
model_config = ConfigDict(extra="forbid", frozen=True)
class 目标版本(_合同):
kind: Literal["method", "rule", "skill", "prompt", "code"]
target_ref: str = Field(min_length=1)
version: str = Field(min_length=1)
content_hash: str = Field(pattern=r"^[0-9a-f]{64}$")
class 配置选择(_合同):
config_id: str = Field(min_length=1)
version: str = Field(min_length=1)
class 标定引用(_合同):
experiment_id: UUID
receipt_hash: str = Field(pattern=r"^[0-9a-f]{64}$")
class 标定使用(_合同):
policy: 标定策略
references: tuple[标定引用, ...] = Field(min_length=1, max_length=8)
class 实验请求(_合同):
dataset_version_id: UUID
dataset_hash: str = Field(pattern=r"^[0-9a-f]{64}$")
split: Literal["discovery", "calibration", "holdout"]
target: 目标版本
generator_role: Literal["writer", "planner"]
generator: 配置选择
judges: tuple[配置选择, ...] = Field(min_length=1, max_length=8)
comparison_profile: Literal["preference", "writer_rubric"] = "preference"
arbitrator: 配置选择 | None = None
detector: 配置选择 | None = None
arms: tuple[str, ...] = Field(min_length=2, max_length=8)
dimensions: tuple[str, ...] = Field(min_length=1, max_length=20)
max_cost_usd: Decimal = Field(gt=0, allow_inf_nan=False)
max_calls_per_sample: StrictInt = Field(ge=1, le=100)
max_judge_corrections: StrictInt = Field(default=1, ge=0, le=2)
calibration_policy: 标定策略 | None = None
evaluation_goal: Literal["diagnostic", "qualification"] = "diagnostic"
calibration_use: 标定使用 | None = None
effect_policy: Literal["writer-effect-v1"] | 目标标准版本 | None = None
application_scope: 方法适用范围 | None = None
effect_dependency_version: Literal["effect-dependency-v2"] | None = None
reuse_experiments: tuple[UUID, ...] = Field(default=(), max_length=16)
def 命令内容(self) -> dict:
payload = self.model_dump(mode="json")
if self.effect_policy is None:
payload.pop("effect_policy")
if self.application_scope is None:
payload.pop("application_scope")
if self.effect_dependency_version is None:
payload.pop("effect_dependency_version")
if not self.reuse_experiments:
payload.pop("reuse_experiments")
if self.evaluation_goal == "diagnostic":
payload.pop("evaluation_goal")
if self.calibration_use is None:
payload.pop("calibration_use")
if self.calibration_policy is None:
payload.pop("calibration_policy")
# 普通偏好请求沿用原命令身份;新增的默认字段不是条件变化。
if self.comparison_profile == "preference" and self.arbitrator is None:
payload.pop("comparison_profile")
payload.pop("arbitrator")
return payload
def 固定实验条件(数据库, req: 实验请求) -> dict:
for rows in (req.arms, req.dimensions):
if len(set(rows)) != len(rows) or any(not r.strip() for r in rows):
raise 评测错误("实验臂与维度必须唯一且非空")
profiles = 配置版本管理(数据库)
dependency_v2 = bool(
req.effect_dependency_version
or req.reuse_experiments
or (req.effect_policy and req.effect_policy.startswith("writer-goal-"))
)
def resolve(ref, role):
version = profiles.读取版本(ref.config_id, ref.version)
declared = version.内容.角色配置.get(role)
if declared is None:
raise 评测错误("实际配置版本缺少所需角色")
# 冻结准确配置哈希和模型;凭据路径/地址只留在S02配置,不复制到实验或模型输入。
result = {
"config_id": ref.config_id,
"version": ref.version,
"config_hash": version.内容哈希,
"role": role,
"provider": declared["provider"],
"model": declared["model"],
"thinking": declared["thinking"],
"resource_release": version.内容.资源发布身份,
"role_policy": version.内容.角色策略版本,
"host": version.内容.宿主,
"host_version": version.内容.宿主版本,
"pricing_version": version.内容.计价版本,
"budget_account": version.内容.预算策略引用,
}
if dependency_v2:
endpoints = [p for p in version.内容.提供方 if p.身份 == declared["provider"]]
if len(endpoints) != 1:
raise 评测错误("效果依赖需要唯一的实际提供方配置")
endpoint = endpoints[0]
result["provider_fingerprint"] = 固定哈希([endpoint.身份, endpoint.协议, endpoint.地址])
return result
generator = resolve(req.generator, req.generator_role)
literary = req.comparison_profile == "writer_rubric"
policy = 角色策略目录.从发布包() if literary else None
if policy is not None:
要求文学配置(policy, generator=generator)
judges = [resolve(r, "judge") for r in req.judges]
if req.evaluation_goal == "qualification" and (
not literary or req.split != "holdout" or req.calibration_use is None
):
raise 评测错误("资格评测需要独立保留集、文学模式及明确的标定引用")
if req.calibration_use is not None and (
req.evaluation_goal != "qualification" or req.calibration_policy is not None
):
raise 评测错误("标定使用只用于资格评测,标定实验不得递归引用标定")
if req.calibration_policy is not None and (not literary or req.split != "calibration"):
raise 评测错误("标定策略只允许用于calibration分割的文学实验")
if literary and (
req.generator_role != "writer"
or req.dimensions != 维度
or len(judges) != 2
or req.arbitrator is None
):
raise 评测错误("文学评分需要固定五维、两名初评及一名预注册候补评委")
if not literary and req.arbitrator is not None:
raise 评测错误("普通偏好实验不能附带未定义条件的第三评委")
arbitrator = resolve(req.arbitrator, "judge") if req.arbitrator else None
all_judges = [*judges, *([arbitrator] if arbitrator else [])]
if policy is not None:
要求文学配置(policy, generator=generator, judges=all_judges)
if len({r["model"] for r in all_judges}) != len(all_judges) or any(
r["model"] == generator["model"] for r in all_judges
):
raise 评测错误("独立评委配置不能重复或使用同一生成模型")
detector = resolve(req.detector, "detector") if req.detector else None
if detector is not None and detector["model"] == generator["model"]:
raise 评测错误("语义检测配置不能使用本次生成模型")
if req.effect_policy is not None and (
not literary or detector is None or req.calibration_policy is not None
):
raise 评测错误("正文效果判据需要文学评分和独立检测,不用于标定实验")
conditions = req.model_dump(mode="json")
if req.application_scope is None:
conditions.pop("application_scope")
if req.effect_dependency_version is None:
conditions.pop("effect_dependency_version")
if req.reuse_experiments:
if len(set(req.reuse_experiments)) != len(req.reuse_experiments) or req.calibration_policy:
raise 评测错误("复用来源不得重复,独立标定实验不得复用已有评审")
conditions["effect_dependency_version"] = "effect-dependency-v2"
else:
conditions.pop("reuse_experiments")
conditions.update(
generator=generator,
judges=judges,
arbitrator=arbitrator,
detector=detector,
schema_version="experiment-v1",
)
if req.effect_policy is not None:
from muse.效果评测.启用判据 import 读取效果标准
policy = 读取效果标准(req.effect_policy)
if "goal_policy" in policy:
if req.application_scope is None or req.target.kind != "method":
raise 评测错误("按用途目标需要确切方法版本和预注册适用范围")
核对目标范围(
policy["goal_policy"],
req.application_scope.model_dump(mode="json"),
qualification=req.evaluation_goal == "qualification",
)
conditions["effect_dependency_version"] = "effect-dependency-v2"
elif req.application_scope is not None:
raise 评测错误("旧效果标准不能附加新的用途资格")
conditions["effect_policy_snapshot"] = policy
conditions["effect_policy_hash"] = 固定哈希(policy)
elif req.application_scope is not None:
raise 评测错误("适用范围必须绑定具名目标政策")
return {"conditions": conditions, "conditions_hash": 固定哈希(conditions)}