zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

235 lines
9.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""从正式owner只读封存ABC资料;不使评测身份获得正式库权限。"""
import hashlib
import json
from dataclasses import asdict
from typing import NoReturn
from pydantic import BaseModel, ConfigDict, Field, StrictInt
from muse.上下文.模型 import 上下文错误
from muse.作品规划.接口 import 读取作品范围, 读取细纲投影
from muse.元数据.接口 import 元数据服务
from muse.基础设施.检索.关键词 import 关键词得分
from muse.故事世界.接口 import 解析本书规划引用, 读取时点事实投影
from muse.正式变更.接口 import 固定哈希
from muse.正文写作.接口 import 可见文本, 读取当前正文依据, 读取正文依据
class 历史版本选择(BaseModel):
model_config = ConfigDict(extra="forbid", frozen=True)
chapter_id: str = Field(min_length=1)
revision: StrictInt = Field(ge=1)
branch_id: str = Field(default="main", min_length=1)
class 回放材料选择(BaseModel):
model_config = ConfigDict(extra="forbid", frozen=True)
work_id: str = Field(min_length=1)
target_chapter_id: str = Field(min_length=1)
history: tuple[历史版本选择, ...] = Field(min_length=1, max_length=1000)
recent_count: StrictInt = Field(default=4, ge=1, le=20)
supplemental_codepoints: StrictInt = Field(ge=0, le=200000)
context_bytes: StrictInt = Field(ge=1, le=4000000)
card_limit: StrictInt = Field(default=20, ge=1, le=200)
include_target_answer: bool = True
def _拒绝(reason) -> NoReturn:
raise 上下文错误("REPLAY_SOURCE_INVALID", reason)
def _json(value):
return json.dumps(value, ensure_ascii=False, sort_keys=True)
def 核对回放字段策略(conn, policies: list[dict], *, 保护到事务结束=False) -> None:
"""封存投影用于新派发时复检当前字段策略;不读取正式业务实例。"""
metadata = 元数据服务(conn)
for policy in sorted(policies, key=lambda p: p["type_id"]):
if 保护到事务结束:
metadata.锁定消费策略(policy["type_id"])
if 固定哈希(asdict(metadata.当前策略(policy["type_id"]))) != policy["hash"]:
raise 上下文错误("PROJECTION_STALE", "回放资料的字段用途策略已变化,须重新封存并批准")
def _选取足额补充(rows, codepoints):
remaining = codepoints
selected = []
for row in rows:
if remaining == 0:
break
text = row["text"][:remaining]
if not text:
continue
selected.append({**row, "text": text, "start": 0, "end": len(text)})
remaining -= len(text)
if remaining:
_拒绝("A/C等量补充原文不足,不能补造或扩读未声明来源")
return selected
def _组装实验臂(outline, baseline, a, c, index, context_bytes):
arms = {}
for arm, prose, hints in [
("A", baseline + a, []),
("B", [], index),
("C", baseline + c, index),
]:
content = {
"目标章细纲": outline["content"],
"历史正文": [{"text": r["text"]} for r in prose],
"卡片索引": hints,
}
if len(_json(content).encode()) > context_bytes:
_拒绝("固定资料超过上下文字节预算,不截断必读材料")
arms[arm] = content
return arms
def 读取回放材料(conn, author: str, selection: 回放材料选择) -> dict:
"""调用方维护事务持有实际读取身份;只用公开owner查询和S04用途投影。"""
work = 读取作品范围(conn, author, selection.work_id)
chapters = {c["chapter_id"]: c for c in work["chapters"]}
target = chapters.get(selection.target_chapter_id)
if target is None:
_拒绝("目标章不属于当前作者作品")
cutoff = target["position"] - 1
outline = 读取细纲投影(
conn,
author,
selection.work_id,
selection.target_chapter_id,
内容用途="generation",
运行用途="evaluation",
引用解析=解析本书规划引用,
)
facts = 读取时点事实投影(
conn,
author,
selection.work_id,
cutoff,
视角="author",
内容用途="generation",
运行用途="evaluation",
)
histories = []
seen = set()
for requested in selection.history:
row = chapters.get(requested.chapter_id)
if row is None or row["position"] > cutoff or requested.chapter_id in seen:
_拒绝("历史版本重复、跨书或包含目标及未来章")
seen.add(requested.chapter_id)
body = 读取正文依据(
conn, author, requested.chapter_id, requested.revision, 分支=requested.branch_id
)
if body["current_revision"] != requested.revision:
_拒绝("封存来源已经改变,请重新选择确切版本")
histories.append(
{
"chapter_id": requested.chapter_id,
"branch_id": requested.branch_id,
"revision": requested.revision,
"document_id": body["document_id"],
"document_hash": body["document_hash"],
"text_hash": hashlib.sha256(可见文本(body["document"]).encode()).hexdigest(),
"position": row["position"],
"text": 可见文本(body["document"]),
}
)
histories.sort(key=lambda r: r["position"])
baseline = [r for r in histories if cutoff - selection.recent_count < r["position"] <= cutoff]
if [r["position"] for r in baseline] != list(
range(cutoff - selection.recent_count + 1, cutoff + 1)
):
_拒绝("声明范围缺少连续历史全文基线")
cards = []
omitted = list(facts["omitted"])
for card in facts["facts"]:
if any(
chapters.get(s["chapter_id"], {}).get("position", cutoff + 1) > cutoff
for s in card["sources"]
):
omitted.append({"object_id": card["object_id"], "reason": "FUTURE_EVIDENCE"})
elif card["content"]:
cards.append(card)
query = _json(outline["content"])
scores = 关键词得分(query, {c["object_id"]: _json(c["content"]) for c in cards})
cards.sort(key=lambda c: (-scores.get(c["object_id"], 0), c["object_id"]))
omitted.extend(
{"object_id": c["object_id"], "reason": "CARD_LIMIT"} for c in cards[selection.card_limit :]
)
cards = cards[: selection.card_limit]
if not cards:
_拒绝("历史时点没有可用于诊断对照的卡片索引")
supplemental = [r for r in histories if r not in baseline]
prose_scores = 关键词得分(query, {r["document_id"]: r["text"] for r in supplemental})
supplemental.sort(key=lambda r: (-prose_scores.get(r["document_id"], 0), r["position"]))
references = {
(s["chapter_id"], s["branch_id"], s["revision"], s["document_hash"])
for c in cards
for s in c["sources"]
}
a = _选取足额补充(supplemental, selection.supplemental_codepoints)
c = _选取足额补充(
[
r
for r in supplemental
if (r["chapter_id"], r["branch_id"], r["revision"], r["document_hash"]) in references
],
selection.supplemental_codepoints,
)
index = [{"content": r["content"], "knowledge_mode": r["knowledge_mode"]} for r in cards]
arms = _组装实验臂(outline, baseline, a, c, index, selection.context_bytes)
answer = None
if selection.include_target_answer:
body = 读取当前正文依据(conn, author, selection.target_chapter_id)
if body is None:
_拒绝("没有目标章原文可封存为答案;不能伪造已标定样本")
answer = {
"target_text": 可见文本(body["document"]),
"document_hash": body["document_hash"],
"revision": body["revision"],
}
metadata = 元数据服务(conn)
binding = outline["schema_binding"]
plan_type = metadata.读取结构(binding["schema_id"], binding["base_version"]).type_id
policies = [
{"type_id": tid, "hash": 固定哈希(asdict(metadata.当前策略(tid)))}
for tid in sorted({plan_type, *(r["type_id"] for r in cards)})
]
return {
"schema_version": "writer-replay-material-v1",
"arms": arms,
"answer": answer,
"policies": policies,
"selection": selection.model_dump(mode="json"),
"basis": {
"directory_revision": work["revision"],
"facts_revision": facts["system_revision"],
"plan": {
k: outline[k] for k in ("plan_id", "revision", "content_hash", "projection_version")
},
"documents": [{k: v for k, v in r.items() if k != "text"} for r in histories],
"cards": [
{
k: r[k]
for k in (
"object_id",
"revision",
"content_hash",
"projection_version",
"sources",
)
}
for r in cards
],
"retrieval": {
"A": [{k: v for k, v in r.items() if k != "text"} for r in a],
"C": [{k: v for k, v in r.items() if k != "text"} for r in c],
},
"omitted": omitted,
},
}