muse-agent-example/tests/契约/test_流式终态.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

346 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""合成流验证模型协议;不发送网络请求,不作为真实模型证据。"""
import asyncio
import json
import pytest
from muse.任务运行.接口 import 模型协议错误
from muse.基础设施.模型.HTTP传输 import 解码SSE
from muse.基础设施.模型.Responses import 解析响应
def 完成事件(*, status="completed", usage=True):
return {
"type": "response." + status,
"response": {
"id": "response-1",
"status": status,
"model": "test-model",
"output": [
{
"type": "message",
"content": [{"type": "output_text", "text": "中文\u0085保持。"}],
}
],
"usage": {"input_tokens": 10, "output_tokens": 4} if usage else None,
},
}
@pytest.mark.case_id(
"NC-protocol-responses-utf8",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["明确完成对象保留中文与C1字符,实际模型和用量可读"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
def test_Responses明确终态与UTF8字节边界__a61001() -> None:
帧 = ("data: " + json.dumps(完成事件(), ensure_ascii=False) + "\r\n\r\n").encode()
async def 字节流():
for b in 帧:
yield bytes([b])
async def 收集():
return [项 async for 项 in 解码SSE(字节流())]
结果 = 解析响应(asyncio.run(收集()))
assert 结果.状态 == "completed" and 结果.文本 == "中文\u0085保持。"
assert 结果.实际模型 == "test-model"
assert 结果.用量.输入token == 10 and 结果.用量.输出token == 4
@pytest.mark.case_id(
"NC-protocol-failure-terminal",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["缺终态、截断、失败、错误分别拒绝完成"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
@pytest.mark.parametrize(
"事件,期望",
[
([{"type": "response.output_text.delta", "delta": "只到一半"}], "incomplete"),
([完成事件(status="incomplete")], "incomplete"),
([完成事件(status="failed")], "failed"),
([{"type": "error", "code": "server_error"}], "failed"),
],
ids=["no-terminal", "incomplete", "failed", "error"],
)
def test_断流与失败不计模型完成__a61002(事件, 期望) -> None:
assert 解析响应(事件).状态 == 期望
@pytest.mark.case_id(
"NC-protocol-unknown-usage",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["完成与未知用量分别表示,不填零"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
def test_完成但用量未知独立保留__a61003() -> None:
结果 = 解析响应([完成事件(usage=False)])
assert 结果.状态 == "completed" and 结果.用量 is None
@pytest.mark.case_id(
"NC-protocol-invalid-frame",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["损坏UTF8拒绝"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
def test_错误SSE帧和UTF8拒绝__a61004() -> None:
async def 字节流():
yield b'data: {"text":"\xff"}\n\n'
async def 收集():
return [项 async for 项 in 解码SSE(字节流())]
with pytest.raises(模型协议错误):
asyncio.run(收集())
@pytest.mark.case_id(
"NC-protocol-messages-terminal",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["完成须同时有message_stop和合法停止原因"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
@pytest.mark.parametrize(
"结束,原因,状态",
[
(True, "end_turn", "completed"),
(False, "end_turn", "incomplete"),
(True, "max_tokens", "incomplete"),
],
ids=["complete", "missing-stop", "token-limit"],
)
def test_Anthropic结束消息与停止原因共同判定__a61005(结束, 原因, 状态) -> None:
from muse.基础设施.模型.Messages import 解析响应 as 解析
事件 = [
{
"type": "message_start",
"message": {"id": "a1", "model": "opus", "usage": {"input_tokens": 3}},
},
{"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}},
{
"type": "content_block_delta",
"index": 0,
"delta": {"type": "text_delta", "text": "甲乙。"},
},
{"type": "message_delta", "delta": {"stop_reason": 原因}, "usage": {"output_tokens": 2}},
]
if 结束:
事件.append({"type": "message_stop"})
结果 = 解析(事件)
assert 结果.状态 == 状态 and 结果.文本 == "甲乙。"
assert 结果.用量.输入token == 3
@pytest.mark.case_id(
"NC-protocol-chat-tool-round",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["文本完成、工具回合和未知用量分开"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
def test_ChatCompletions完成与工具回合分开__a61006() -> None:
from muse.基础设施.模型.ChatCompletions import 解析响应 as 解析
结果 = 解析(
[
{
"id": "c1",
"model": "test",
"choices": [{"index": 0, "delta": {"content": "完成。"}, "finish_reason": "stop"}],
}
]
)
assert 结果.状态 == "completed" and 结果.用量 is None
工具 = 解析(
[
{
"id": "c2",
"model": "test",
"choices": [
{
"index": 0,
"delta": {
"tool_calls": [
{
"index": 0,
"id": "tool1",
"function": {"name": "read", "arguments": "{}"},
}
]
},
"finish_reason": "tool_calls",
}
],
}
]
)
assert 工具.状态 == "tool_calls" and 工具.工具调用[0].名称 == "read"
@pytest.mark.case_id(
"NC-r2-a61008",
environment="离线契约",
given="chat-completions 的 reasoning_content 与 Messages 的 thinking 块",
when="解析流式事件",
then=["推理只计量为推理字符数", "不进入可见文本与回放内容"],
contract="docs/系统架构/新版设计/文件设计/后端-基础设施.md",
)
def test_推理内容只计量不进入可见文本__a61008() -> None:
from muse.基础设施.模型.ChatCompletions import 解析响应 as 解析对话
from muse.基础设施.模型.Messages import 解析响应 as 解析消息
对话 = 解析对话(
[
{
"id": "c3",
"model": "test",
"choices": [
{
"index": 0,
"delta": {"reasoning_content": "先想两字"},
"finish_reason": None,
}
],
},
{
"id": "c3",
"model": "test",
"choices": [{"index": 0, "delta": {"content": "答案。"}, "finish_reason": "stop"}],
},
]
)
assert 对话.文本 == "答案。" and 对话.推理字符数 == 4
消息 = 解析消息(
[
{
"type": "message_start",
"message": {"id": "a3", "model": "opus", "usage": {"input_tokens": 3}},
},
{
"type": "content_block_start",
"index": 0,
"content_block": {"type": "thinking", "thinking": ""},
},
{
"type": "content_block_delta",
"index": 0,
"delta": {"type": "thinking_delta", "thinking": "推理三字"},
},
{
"type": "content_block_start",
"index": 1,
"content_block": {"type": "text", "text": ""},
},
{
"type": "content_block_delta",
"index": 1,
"delta": {"type": "text_delta", "text": "答案。"},
},
{
"type": "message_delta",
"delta": {"stop_reason": "end_turn"},
"usage": {"output_tokens": 9},
},
{"type": "message_stop"},
]
)
assert 消息.文本 == "答案。" and 消息.推理字符数 == 4
@pytest.mark.case_id(
"NC-protocol-output-model",
environment="离线协议,合成事件;无真实模型调用",
given="独立合成协议事件或模型结果",
when="调用新版实际协议解码与终态/结构校验",
then=["实际模型一致且完整输出符合固定Schema才返回候选"],
contract="docs/系统架构/新版设计/接口契约/任务工具与事件.md",
)
@pytest.mark.parametrize(
"模型,文本,合法",
[
("test", '{"正文":"中文"}', True),
("other", '{"正文":"中文"}', False),
("test", '{"正文":1}', False),
("test", '{"正文":"中文","state":"confirmed"}', False),
],
ids=["valid", "model-drift", "wrong-type", "extra-authority"],
)
def test_实际模型与输出合同同时满足才产候选__a61007(模型, 文本, 合法) -> None:
from muse.任务运行.接口 import 校验模型输出, 模型结果, 模型请求
请求 = 模型请求(
"call",
"synthetic",
"test",
"合成系统说明",
"合成输入",
{
"type": "object",
"properties": {"正文": {"type": "string"}},
"required": ["正文"],
"additionalProperties": False,
},
100,
30,
)
结果 = 模型结果("completed", 文本, 模型, None)
if 合法:
assert 校验模型输出(请求, 结果) == {"正文": "中文"}
else:
with pytest.raises(模型协议错误):
校验模型输出(请求, 结果)
@pytest.mark.case_id("NC-O02-SSE-UTF8-SPLITS")
def test_中文与表情每个跨块位置都还原同一事件():
expected = 完成事件()
expected["response"]["output"][0]["content"][0]["text"] = "中文🙂\u0085保持。"
frame = ("data: " + json.dumps(expected, ensure_ascii=False) + "\r\n\r\n").encode()
async def collect(position):
async def stream():
yield frame[:position]
yield frame[position:]
return [item async for item in 解码SSE(stream())]
for position in range(1, len(frame)):
events = asyncio.run(collect(position))
assert events == [expected]
assert 解析响应(events).文本 == "中文🙂\u0085保持。"
@pytest.mark.case_id("NC-O02-MALFORMED-PROTOCOL")
def test_提供方缺字段错索引和错误帧类型均归一领域错误():
from muse.基础设施.模型.ChatCompletions import 解析响应 as chat
from muse.基础设施.模型.Messages import 解析响应 as messages
for parser, frames in (
(chat, [{"usage": {"prompt_tokens": 1}}]),
(chat, [{"choices": [{"delta": {"tool_calls": [{"index": -1}]}}]}]),
(chat, [{"choices": [None]}]),
(messages, [{"type": "message_start"}]),
(messages, [{"type": "content_block_delta", "index": 2, "delta": {}}]),
(messages, [{"type": "made_up_event"}]),
):
with pytest.raises(模型协议错误) as caught:
parser(frames)
assert "Chat Completions" in str(caught.value) or "Anthropic Messages" in str(caught.value)