实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
346 lines
12 KiB
Python
346 lines
12 KiB
Python
"""合成流验证模型协议;不发送网络请求,不作为真实模型证据。"""
|
||
|
||
import asyncio
|
||
import json
|
||
|
||
import pytest
|
||
|
||
from muse.任务运行.接口 import 模型协议错误
|
||
from muse.基础设施.模型.HTTP传输 import 解码SSE
|
||
from muse.基础设施.模型.Responses import 解析响应
|
||
|
||
|
||
def 完成事件(*, status="completed", usage=True):
|
||
return {
|
||
"type": "response." + status,
|
||
"response": {
|
||
"id": "response-1",
|
||
"status": status,
|
||
"model": "test-model",
|
||
"output": [
|
||
{
|
||
"type": "message",
|
||
"content": [{"type": "output_text", "text": "中文\u0085保持。"}],
|
||
}
|
||
],
|
||
"usage": {"input_tokens": 10, "output_tokens": 4} if usage else None,
|
||
},
|
||
}
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-responses-utf8",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["明确完成对象保留中文与C1字符,实际模型和用量可读"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
def test_Responses明确终态与UTF8字节边界__a61001() -> None:
|
||
帧 = ("data: " + json.dumps(完成事件(), ensure_ascii=False) + "\r\n\r\n").encode()
|
||
|
||
async def 字节流():
|
||
for b in 帧:
|
||
yield bytes([b])
|
||
|
||
async def 收集():
|
||
return [项 async for 项 in 解码SSE(字节流())]
|
||
|
||
结果 = 解析响应(asyncio.run(收集()))
|
||
assert 结果.状态 == "completed" and 结果.文本 == "中文\u0085保持。"
|
||
assert 结果.实际模型 == "test-model"
|
||
assert 结果.用量.输入token == 10 and 结果.用量.输出token == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-failure-terminal",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["缺终态、截断、失败、错误分别拒绝完成"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
@pytest.mark.parametrize(
|
||
"事件,期望",
|
||
[
|
||
([{"type": "response.output_text.delta", "delta": "只到一半"}], "incomplete"),
|
||
([完成事件(status="incomplete")], "incomplete"),
|
||
([完成事件(status="failed")], "failed"),
|
||
([{"type": "error", "code": "server_error"}], "failed"),
|
||
],
|
||
ids=["no-terminal", "incomplete", "failed", "error"],
|
||
)
|
||
def test_断流与失败不计模型完成__a61002(事件, 期望) -> None:
|
||
assert 解析响应(事件).状态 == 期望
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-unknown-usage",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["完成与未知用量分别表示,不填零"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
def test_完成但用量未知独立保留__a61003() -> None:
|
||
结果 = 解析响应([完成事件(usage=False)])
|
||
assert 结果.状态 == "completed" and 结果.用量 is None
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-invalid-frame",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["损坏UTF8拒绝"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
def test_错误SSE帧和UTF8拒绝__a61004() -> None:
|
||
async def 字节流():
|
||
yield b'data: {"text":"\xff"}\n\n'
|
||
|
||
async def 收集():
|
||
return [项 async for 项 in 解码SSE(字节流())]
|
||
|
||
with pytest.raises(模型协议错误):
|
||
asyncio.run(收集())
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-messages-terminal",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["完成须同时有message_stop和合法停止原因"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
@pytest.mark.parametrize(
|
||
"结束,原因,状态",
|
||
[
|
||
(True, "end_turn", "completed"),
|
||
(False, "end_turn", "incomplete"),
|
||
(True, "max_tokens", "incomplete"),
|
||
],
|
||
ids=["complete", "missing-stop", "token-limit"],
|
||
)
|
||
def test_Anthropic结束消息与停止原因共同判定__a61005(结束, 原因, 状态) -> None:
|
||
from muse.基础设施.模型.Messages import 解析响应 as 解析
|
||
|
||
事件 = [
|
||
{
|
||
"type": "message_start",
|
||
"message": {"id": "a1", "model": "opus", "usage": {"input_tokens": 3}},
|
||
},
|
||
{"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}},
|
||
{
|
||
"type": "content_block_delta",
|
||
"index": 0,
|
||
"delta": {"type": "text_delta", "text": "甲乙。"},
|
||
},
|
||
{"type": "message_delta", "delta": {"stop_reason": 原因}, "usage": {"output_tokens": 2}},
|
||
]
|
||
if 结束:
|
||
事件.append({"type": "message_stop"})
|
||
结果 = 解析(事件)
|
||
assert 结果.状态 == 状态 and 结果.文本 == "甲乙。"
|
||
assert 结果.用量.输入token == 3
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-chat-tool-round",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["文本完成、工具回合和未知用量分开"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
def test_ChatCompletions完成与工具回合分开__a61006() -> None:
|
||
from muse.基础设施.模型.ChatCompletions import 解析响应 as 解析
|
||
|
||
结果 = 解析(
|
||
[
|
||
{
|
||
"id": "c1",
|
||
"model": "test",
|
||
"choices": [{"index": 0, "delta": {"content": "完成。"}, "finish_reason": "stop"}],
|
||
}
|
||
]
|
||
)
|
||
assert 结果.状态 == "completed" and 结果.用量 is None
|
||
工具 = 解析(
|
||
[
|
||
{
|
||
"id": "c2",
|
||
"model": "test",
|
||
"choices": [
|
||
{
|
||
"index": 0,
|
||
"delta": {
|
||
"tool_calls": [
|
||
{
|
||
"index": 0,
|
||
"id": "tool1",
|
||
"function": {"name": "read", "arguments": "{}"},
|
||
}
|
||
]
|
||
},
|
||
"finish_reason": "tool_calls",
|
||
}
|
||
],
|
||
}
|
||
]
|
||
)
|
||
assert 工具.状态 == "tool_calls" and 工具.工具调用[0].名称 == "read"
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-r2-a61008",
|
||
environment="离线契约",
|
||
given="chat-completions 的 reasoning_content 与 Messages 的 thinking 块",
|
||
when="解析流式事件",
|
||
then=["推理只计量为推理字符数", "不进入可见文本与回放内容"],
|
||
contract="docs/系统架构/新版设计/文件设计/后端-基础设施.md",
|
||
)
|
||
def test_推理内容只计量不进入可见文本__a61008() -> None:
|
||
from muse.基础设施.模型.ChatCompletions import 解析响应 as 解析对话
|
||
from muse.基础设施.模型.Messages import 解析响应 as 解析消息
|
||
|
||
对话 = 解析对话(
|
||
[
|
||
{
|
||
"id": "c3",
|
||
"model": "test",
|
||
"choices": [
|
||
{
|
||
"index": 0,
|
||
"delta": {"reasoning_content": "先想两字"},
|
||
"finish_reason": None,
|
||
}
|
||
],
|
||
},
|
||
{
|
||
"id": "c3",
|
||
"model": "test",
|
||
"choices": [{"index": 0, "delta": {"content": "答案。"}, "finish_reason": "stop"}],
|
||
},
|
||
]
|
||
)
|
||
assert 对话.文本 == "答案。" and 对话.推理字符数 == 4
|
||
|
||
消息 = 解析消息(
|
||
[
|
||
{
|
||
"type": "message_start",
|
||
"message": {"id": "a3", "model": "opus", "usage": {"input_tokens": 3}},
|
||
},
|
||
{
|
||
"type": "content_block_start",
|
||
"index": 0,
|
||
"content_block": {"type": "thinking", "thinking": ""},
|
||
},
|
||
{
|
||
"type": "content_block_delta",
|
||
"index": 0,
|
||
"delta": {"type": "thinking_delta", "thinking": "推理三字"},
|
||
},
|
||
{
|
||
"type": "content_block_start",
|
||
"index": 1,
|
||
"content_block": {"type": "text", "text": ""},
|
||
},
|
||
{
|
||
"type": "content_block_delta",
|
||
"index": 1,
|
||
"delta": {"type": "text_delta", "text": "答案。"},
|
||
},
|
||
{
|
||
"type": "message_delta",
|
||
"delta": {"stop_reason": "end_turn"},
|
||
"usage": {"output_tokens": 9},
|
||
},
|
||
{"type": "message_stop"},
|
||
]
|
||
)
|
||
assert 消息.文本 == "答案。" and 消息.推理字符数 == 4
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-protocol-output-model",
|
||
environment="离线协议,合成事件;无真实模型调用",
|
||
given="独立合成协议事件或模型结果",
|
||
when="调用新版实际协议解码与终态/结构校验",
|
||
then=["实际模型一致且完整输出符合固定Schema才返回候选"],
|
||
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
|
||
)
|
||
@pytest.mark.parametrize(
|
||
"模型,文本,合法",
|
||
[
|
||
("test", '{"正文":"中文"}', True),
|
||
("other", '{"正文":"中文"}', False),
|
||
("test", '{"正文":1}', False),
|
||
("test", '{"正文":"中文","state":"confirmed"}', False),
|
||
],
|
||
ids=["valid", "model-drift", "wrong-type", "extra-authority"],
|
||
)
|
||
def test_实际模型与输出合同同时满足才产候选__a61007(模型, 文本, 合法) -> None:
|
||
from muse.任务运行.接口 import 校验模型输出, 模型结果, 模型请求
|
||
|
||
请求 = 模型请求(
|
||
"call",
|
||
"synthetic",
|
||
"test",
|
||
"合成系统说明",
|
||
"合成输入",
|
||
{
|
||
"type": "object",
|
||
"properties": {"正文": {"type": "string"}},
|
||
"required": ["正文"],
|
||
"additionalProperties": False,
|
||
},
|
||
100,
|
||
30,
|
||
)
|
||
结果 = 模型结果("completed", 文本, 模型, None)
|
||
if 合法:
|
||
assert 校验模型输出(请求, 结果) == {"正文": "中文"}
|
||
else:
|
||
with pytest.raises(模型协议错误):
|
||
校验模型输出(请求, 结果)
|
||
|
||
|
||
@pytest.mark.case_id("NC-O02-SSE-UTF8-SPLITS")
|
||
def test_中文与表情每个跨块位置都还原同一事件():
|
||
expected = 完成事件()
|
||
expected["response"]["output"][0]["content"][0]["text"] = "中文🙂\u0085保持。"
|
||
frame = ("data: " + json.dumps(expected, ensure_ascii=False) + "\r\n\r\n").encode()
|
||
|
||
async def collect(position):
|
||
async def stream():
|
||
yield frame[:position]
|
||
yield frame[position:]
|
||
|
||
return [item async for item in 解码SSE(stream())]
|
||
|
||
for position in range(1, len(frame)):
|
||
events = asyncio.run(collect(position))
|
||
assert events == [expected]
|
||
assert 解析响应(events).文本 == "中文🙂\u0085保持。"
|
||
|
||
|
||
@pytest.mark.case_id("NC-O02-MALFORMED-PROTOCOL")
|
||
def test_提供方缺字段错索引和错误帧类型均归一领域错误():
|
||
from muse.基础设施.模型.ChatCompletions import 解析响应 as chat
|
||
from muse.基础设施.模型.Messages import 解析响应 as messages
|
||
|
||
for parser, frames in (
|
||
(chat, [{"usage": {"prompt_tokens": 1}}]),
|
||
(chat, [{"choices": [{"delta": {"tool_calls": [{"index": -1}]}}]}]),
|
||
(chat, [{"choices": [None]}]),
|
||
(messages, [{"type": "message_start"}]),
|
||
(messages, [{"type": "content_block_delta", "index": 2, "delta": {}}]),
|
||
(messages, [{"type": "made_up_event"}]),
|
||
):
|
||
with pytest.raises(模型协议错误) as caught:
|
||
parser(frames)
|
||
assert "Chat Completions" in str(caught.value) or "Anthropic Messages" in str(caught.value)
|