实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
235 lines
9.2 KiB
Python
235 lines
9.2 KiB
Python
"""从正式owner只读封存ABC资料;不使评测身份获得正式库权限。"""
|
||
|
||
import hashlib
|
||
import json
|
||
from dataclasses import asdict
|
||
from typing import NoReturn
|
||
|
||
from pydantic import BaseModel, ConfigDict, Field, StrictInt
|
||
|
||
from muse.上下文.模型 import 上下文错误
|
||
from muse.作品规划.接口 import 读取作品范围, 读取细纲投影
|
||
from muse.元数据.接口 import 元数据服务
|
||
from muse.基础设施.检索.关键词 import 关键词得分
|
||
from muse.故事世界.接口 import 解析本书规划引用, 读取时点事实投影
|
||
from muse.正式变更.接口 import 固定哈希
|
||
from muse.正文写作.接口 import 可见文本, 读取当前正文依据, 读取正文依据
|
||
|
||
|
||
class 历史版本选择(BaseModel):
|
||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||
chapter_id: str = Field(min_length=1)
|
||
revision: StrictInt = Field(ge=1)
|
||
branch_id: str = Field(default="main", min_length=1)
|
||
|
||
|
||
class 回放材料选择(BaseModel):
|
||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||
work_id: str = Field(min_length=1)
|
||
target_chapter_id: str = Field(min_length=1)
|
||
history: tuple[历史版本选择, ...] = Field(min_length=1, max_length=1000)
|
||
recent_count: StrictInt = Field(default=4, ge=1, le=20)
|
||
supplemental_codepoints: StrictInt = Field(ge=0, le=200000)
|
||
context_bytes: StrictInt = Field(ge=1, le=4000000)
|
||
card_limit: StrictInt = Field(default=20, ge=1, le=200)
|
||
include_target_answer: bool = True
|
||
|
||
|
||
def _拒绝(reason) -> NoReturn:
|
||
raise 上下文错误("REPLAY_SOURCE_INVALID", reason)
|
||
|
||
|
||
def _json(value):
|
||
return json.dumps(value, ensure_ascii=False, sort_keys=True)
|
||
|
||
|
||
def 核对回放字段策略(conn, policies: list[dict], *, 保护到事务结束=False) -> None:
|
||
"""封存投影用于新派发时复检当前字段策略;不读取正式业务实例。"""
|
||
metadata = 元数据服务(conn)
|
||
for policy in sorted(policies, key=lambda p: p["type_id"]):
|
||
if 保护到事务结束:
|
||
metadata.锁定消费策略(policy["type_id"])
|
||
if 固定哈希(asdict(metadata.当前策略(policy["type_id"]))) != policy["hash"]:
|
||
raise 上下文错误("PROJECTION_STALE", "回放资料的字段用途策略已变化,须重新封存并批准")
|
||
|
||
|
||
def _选取足额补充(rows, codepoints):
|
||
remaining = codepoints
|
||
selected = []
|
||
for row in rows:
|
||
if remaining == 0:
|
||
break
|
||
text = row["text"][:remaining]
|
||
if not text:
|
||
continue
|
||
selected.append({**row, "text": text, "start": 0, "end": len(text)})
|
||
remaining -= len(text)
|
||
if remaining:
|
||
_拒绝("A/C等量补充原文不足,不能补造或扩读未声明来源")
|
||
return selected
|
||
|
||
|
||
def _组装实验臂(outline, baseline, a, c, index, context_bytes):
|
||
arms = {}
|
||
for arm, prose, hints in [
|
||
("A", baseline + a, []),
|
||
("B", [], index),
|
||
("C", baseline + c, index),
|
||
]:
|
||
content = {
|
||
"目标章细纲": outline["content"],
|
||
"历史正文": [{"text": r["text"]} for r in prose],
|
||
"卡片索引": hints,
|
||
}
|
||
if len(_json(content).encode()) > context_bytes:
|
||
_拒绝("固定资料超过上下文字节预算,不截断必读材料")
|
||
arms[arm] = content
|
||
return arms
|
||
|
||
|
||
def 读取回放材料(conn, author: str, selection: 回放材料选择) -> dict:
|
||
"""调用方维护事务持有实际读取身份;只用公开owner查询和S04用途投影。"""
|
||
work = 读取作品范围(conn, author, selection.work_id)
|
||
chapters = {c["chapter_id"]: c for c in work["chapters"]}
|
||
target = chapters.get(selection.target_chapter_id)
|
||
if target is None:
|
||
_拒绝("目标章不属于当前作者作品")
|
||
cutoff = target["position"] - 1
|
||
outline = 读取细纲投影(
|
||
conn,
|
||
author,
|
||
selection.work_id,
|
||
selection.target_chapter_id,
|
||
内容用途="generation",
|
||
运行用途="evaluation",
|
||
引用解析=解析本书规划引用,
|
||
)
|
||
facts = 读取时点事实投影(
|
||
conn,
|
||
author,
|
||
selection.work_id,
|
||
cutoff,
|
||
视角="author",
|
||
内容用途="generation",
|
||
运行用途="evaluation",
|
||
)
|
||
histories = []
|
||
seen = set()
|
||
for requested in selection.history:
|
||
row = chapters.get(requested.chapter_id)
|
||
if row is None or row["position"] > cutoff or requested.chapter_id in seen:
|
||
_拒绝("历史版本重复、跨书或包含目标及未来章")
|
||
seen.add(requested.chapter_id)
|
||
body = 读取正文依据(
|
||
conn, author, requested.chapter_id, requested.revision, 分支=requested.branch_id
|
||
)
|
||
if body["current_revision"] != requested.revision:
|
||
_拒绝("封存来源已经改变,请重新选择确切版本")
|
||
histories.append(
|
||
{
|
||
"chapter_id": requested.chapter_id,
|
||
"branch_id": requested.branch_id,
|
||
"revision": requested.revision,
|
||
"document_id": body["document_id"],
|
||
"document_hash": body["document_hash"],
|
||
"text_hash": hashlib.sha256(可见文本(body["document"]).encode()).hexdigest(),
|
||
"position": row["position"],
|
||
"text": 可见文本(body["document"]),
|
||
}
|
||
)
|
||
histories.sort(key=lambda r: r["position"])
|
||
baseline = [r for r in histories if cutoff - selection.recent_count < r["position"] <= cutoff]
|
||
if [r["position"] for r in baseline] != list(
|
||
range(cutoff - selection.recent_count + 1, cutoff + 1)
|
||
):
|
||
_拒绝("声明范围缺少连续历史全文基线")
|
||
cards = []
|
||
omitted = list(facts["omitted"])
|
||
for card in facts["facts"]:
|
||
if any(
|
||
chapters.get(s["chapter_id"], {}).get("position", cutoff + 1) > cutoff
|
||
for s in card["sources"]
|
||
):
|
||
omitted.append({"object_id": card["object_id"], "reason": "FUTURE_EVIDENCE"})
|
||
elif card["content"]:
|
||
cards.append(card)
|
||
query = _json(outline["content"])
|
||
scores = 关键词得分(query, {c["object_id"]: _json(c["content"]) for c in cards})
|
||
cards.sort(key=lambda c: (-scores.get(c["object_id"], 0), c["object_id"]))
|
||
omitted.extend(
|
||
{"object_id": c["object_id"], "reason": "CARD_LIMIT"} for c in cards[selection.card_limit :]
|
||
)
|
||
cards = cards[: selection.card_limit]
|
||
if not cards:
|
||
_拒绝("历史时点没有可用于诊断对照的卡片索引")
|
||
supplemental = [r for r in histories if r not in baseline]
|
||
prose_scores = 关键词得分(query, {r["document_id"]: r["text"] for r in supplemental})
|
||
supplemental.sort(key=lambda r: (-prose_scores.get(r["document_id"], 0), r["position"]))
|
||
references = {
|
||
(s["chapter_id"], s["branch_id"], s["revision"], s["document_hash"])
|
||
for c in cards
|
||
for s in c["sources"]
|
||
}
|
||
|
||
a = _选取足额补充(supplemental, selection.supplemental_codepoints)
|
||
c = _选取足额补充(
|
||
[
|
||
r
|
||
for r in supplemental
|
||
if (r["chapter_id"], r["branch_id"], r["revision"], r["document_hash"]) in references
|
||
],
|
||
selection.supplemental_codepoints,
|
||
)
|
||
index = [{"content": r["content"], "knowledge_mode": r["knowledge_mode"]} for r in cards]
|
||
arms = _组装实验臂(outline, baseline, a, c, index, selection.context_bytes)
|
||
answer = None
|
||
if selection.include_target_answer:
|
||
body = 读取当前正文依据(conn, author, selection.target_chapter_id)
|
||
if body is None:
|
||
_拒绝("没有目标章原文可封存为答案;不能伪造已标定样本")
|
||
answer = {
|
||
"target_text": 可见文本(body["document"]),
|
||
"document_hash": body["document_hash"],
|
||
"revision": body["revision"],
|
||
}
|
||
metadata = 元数据服务(conn)
|
||
binding = outline["schema_binding"]
|
||
plan_type = metadata.读取结构(binding["schema_id"], binding["base_version"]).type_id
|
||
policies = [
|
||
{"type_id": tid, "hash": 固定哈希(asdict(metadata.当前策略(tid)))}
|
||
for tid in sorted({plan_type, *(r["type_id"] for r in cards)})
|
||
]
|
||
return {
|
||
"schema_version": "writer-replay-material-v1",
|
||
"arms": arms,
|
||
"answer": answer,
|
||
"policies": policies,
|
||
"selection": selection.model_dump(mode="json"),
|
||
"basis": {
|
||
"directory_revision": work["revision"],
|
||
"facts_revision": facts["system_revision"],
|
||
"plan": {
|
||
k: outline[k] for k in ("plan_id", "revision", "content_hash", "projection_version")
|
||
},
|
||
"documents": [{k: v for k, v in r.items() if k != "text"} for r in histories],
|
||
"cards": [
|
||
{
|
||
k: r[k]
|
||
for k in (
|
||
"object_id",
|
||
"revision",
|
||
"content_hash",
|
||
"projection_version",
|
||
"sources",
|
||
)
|
||
}
|
||
for r in cards
|
||
],
|
||
"retrieval": {
|
||
"A": [{k: v for k, v in r.items() if k != "text"} for r in a],
|
||
"C": [{k: v for k, v in r.items() if k != "text"} for r in c],
|
||
},
|
||
"omitted": omitted,
|
||
},
|
||
}
|