- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
171 lines
6.7 KiB
Python
171 lines
6.7 KiB
Python
"""数据集公共输入、来源分组与答案分割;不把维护者声明当运行证明。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from datetime import datetime
|
||
from typing import Any, Literal
|
||
from uuid import NAMESPACE_URL, UUID, uuid5
|
||
|
||
from pydantic import BaseModel, ConfigDict, Field, StrictInt
|
||
|
||
from muse.上下文.接口 import 回放材料选择
|
||
from muse.效果评测.文学评分 import 场景, 评判命题
|
||
from muse.效果评测.模型 import 评测错误
|
||
from muse.效果评测.混淆项 import 写手审计声明, 核对审计声明
|
||
from muse.正式变更.接口 import 固定哈希
|
||
|
||
|
||
class _合同(BaseModel):
|
||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||
|
||
|
||
class 公开评测输入(_合同):
|
||
instruction: str = Field(min_length=1, max_length=24000)
|
||
original: str = Field(default="", max_length=200000)
|
||
context: dict[str, Any] = Field(default_factory=dict)
|
||
|
||
|
||
class 样本分层(_合同):
|
||
"""维护者的版本化分层声明;不进入模型输入,不冒充模型或事实验证。"""
|
||
|
||
model_config = ConfigDict(extra="forbid", frozen=True, strict=True)
|
||
work_ref: str = Field(min_length=1, max_length=200)
|
||
annotation_ref: str = Field(min_length=1, max_length=1000)
|
||
new_character_ratio: float | None = Field(default=None, ge=0, le=1, allow_inf_nan=False)
|
||
|
||
|
||
class 数据样本(_合同):
|
||
sample_id: str = Field(min_length=1, max_length=200)
|
||
source_ref: str = Field(min_length=1, max_length=1000)
|
||
license_ref: str = Field(min_length=1, max_length=1000)
|
||
source_groups: tuple[str, ...] = Field(min_length=1)
|
||
split: Literal["discovery", "calibration", "holdout"]
|
||
input: 公开评测输入
|
||
answer: dict[str, Any]
|
||
stratification: 样本分层 | None = None
|
||
|
||
|
||
class 数据集发布(_合同):
|
||
dataset_id: str = Field(min_length=1, max_length=200)
|
||
revision: StrictInt = Field(ge=1)
|
||
samples: tuple[数据样本, ...] = Field(min_length=1, max_length=10000)
|
||
|
||
|
||
class 方法数据集发布(数据集发布):
|
||
method_version_id: UUID
|
||
approval_ref: str = Field(min_length=1)
|
||
expires_at: datetime
|
||
|
||
|
||
class 回放评判声明(_合同):
|
||
scenario: 场景
|
||
assertions: list[评判命题] = Field(max_length=100)
|
||
constraints: list[评判命题] = Field(max_length=100)
|
||
|
||
|
||
class 回放样本发布(_合同):
|
||
sample_id: str = Field(min_length=1, max_length=200)
|
||
source_groups: tuple[str, ...] = Field(min_length=1)
|
||
split: Literal["discovery", "calibration", "holdout"]
|
||
instruction: str = Field(min_length=1, max_length=24000)
|
||
license_ref: str = Field(min_length=1, max_length=1000)
|
||
selection: 回放材料选择
|
||
judging: 回放评判声明 | None = None
|
||
stratification: 样本分层 | None = None
|
||
audit: 写手审计声明 | None = None
|
||
|
||
|
||
class 回放数据集发布(_合同):
|
||
dataset_id: str = Field(min_length=1, max_length=200)
|
||
revision: StrictInt = Field(ge=1)
|
||
approval_ref: str = Field(min_length=1)
|
||
expires_at: datetime
|
||
samples: tuple[回放样本发布, ...] = Field(min_length=1, max_length=10000)
|
||
|
||
|
||
def 来源内容哈希(value: dict) -> str:
|
||
content = {k: value[k] for k in ("original", "context")}
|
||
if not content["original"] and not content["context"]:
|
||
content["instruction"] = value["instruction"]
|
||
return 固定哈希(content)
|
||
|
||
|
||
def 冻结数据集(req: 数据集发布) -> dict:
|
||
# model_dump复制容器;后续写入使用此副本,不持有调用方可变context/answer。
|
||
if not req.dataset_id.strip():
|
||
raise 评测错误("数据集身份不能为空白")
|
||
rows = req.model_dump(mode="json")["samples"]
|
||
ids = [r["sample_id"] for r in rows]
|
||
if len(set(ids)) != len(ids) or any(not s.strip() for s in ids):
|
||
raise 评测错误("数据集样本身份必须非空且唯一")
|
||
group_splits, content_splits = {}, {}
|
||
public, answers = [], []
|
||
for r in sorted(rows, key=lambda r: r["sample_id"]):
|
||
if r["answer"].get("writer_audit") is not None:
|
||
r["answer"]["writer_audit"] = 核对审计声明(r["answer"]["writer_audit"])
|
||
if r.get("stratification") is None:
|
||
# 未提供新声明时保留原数据集字节合同和同版本重放身份。
|
||
r.pop("stratification", None)
|
||
elif any(not r["stratification"][k].strip() for k in ("work_ref", "annotation_ref")):
|
||
raise 评测错误("分层声明需要明确作品与标注来源")
|
||
groups = r["source_groups"]
|
||
if (
|
||
len(groups) != len(set(groups))
|
||
or any(not g.strip() for g in groups)
|
||
or not r["source_ref"].strip()
|
||
or not r["license_ref"].strip()
|
||
or not r["input"]["instruction"].strip()
|
||
):
|
||
raise 评测错误("样本需要明确来源、许可、来源组与任务")
|
||
for g in groups:
|
||
if g in group_splits and group_splits[g] != r["split"]:
|
||
raise 评测错误("同源、变体或相邻样本不能跨分割")
|
||
group_splits[g] = r["split"]
|
||
# 有底稿时忽略任务措辞;纯指令任务则按指令区分,避免全部合并为空底稿。
|
||
content_key = 来源内容哈希(r["input"])
|
||
if content_key in content_splits and content_splits[content_key] != r["split"]:
|
||
raise 评测错误("相同正文或背景不能跨分割")
|
||
content_splits[content_key] = r["split"]
|
||
public.append({k: v for k, v in r.items() if k != "answer"})
|
||
answers.append({"sample_id": r["sample_id"], "answer": r["answer"]})
|
||
payload = {
|
||
"schema_version": "dataset-v1",
|
||
"dataset_id": req.dataset_id,
|
||
"revision": req.revision,
|
||
"samples": public,
|
||
}
|
||
return {
|
||
"version_id": str(
|
||
uuid5(NAMESPACE_URL, "muse:dataset:" + 固定哈希([req.dataset_id, req.revision]))
|
||
),
|
||
"public": payload,
|
||
"public_hash": 固定哈希(payload),
|
||
"answers": answers,
|
||
"answer_hash": 固定哈希(answers),
|
||
}
|
||
|
||
|
||
def 核对数据快照(payload: dict, public_hash: str) -> None:
|
||
if (
|
||
payload.get("schema_version")
|
||
not in {
|
||
"dataset-v1",
|
||
"writer-replay-dataset-v1",
|
||
"method-evaluation-dataset-v1",
|
||
"fixed-revision-pair-dataset-v1",
|
||
"fixed-revision-pair-dataset-v2",
|
||
"rule-diagnostic-dataset-v1",
|
||
}
|
||
or 固定哈希(payload) != public_hash
|
||
):
|
||
raise 评测错误("数据集版本或公开输入哈希不符")
|
||
|
||
|
||
def 选择样本(payload: dict, split: str) -> list[dict]:
|
||
if split not in {"discovery", "calibration", "holdout"}:
|
||
raise 评测错误("实验需要明确分割")
|
||
samples = [r for r in payload["samples"] if r["split"] == split]
|
||
if not samples:
|
||
raise 评测错误("所选分割没有样本,不能创建空实验")
|
||
return samples
|