muse-agent-example/tests/端到端/test_诊断入口与只读边界.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

70 lines
3.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""合成场景在隔离PG上跑真实公开用例;不声称自然语言路由或真实模型效果。"""
from pathlib import Path
import pytest
from muse.作品规划.接口 import 档案保存
from muse.效果评测.接口 import 执行行为场景, 服务观察适配, 载入行为场景
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
pytestmark = pytest.mark.数据库
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
@pytest.mark.case_id(
"NC-w24-241005",
environment="隔离PG真服务",
given="固定六场景及逐例原始观察",
when="显式脚本或真实隔离服务执行,注入缺失、禁止动作及伪报告",
then=["场景逐个独立裁决,报告重算和读回、来源前后哈希、失败与模式均可追溯"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize("scene", 场景集, ids=lambda s: s.scenario_id)
def test_真实服务观察报告读回及正文不变__241005(章后环境, scene):
app, author = 章后环境["装配"], 章后环境["作者"]
works, body, cases = app.要求作品(), app.要求正文(), app.要求审校()
schema = works.新档案结构(author, "synthetic:demo", "work_core", 1)
works.保存档案(
author, "create-eval-work", 档案保存("synthetic:demo", 0, {"名称": "合成诊断场景"}, schema)
)
works.添加章节(
author, "create-eval-ch", "synthetic:demo", "synthetic-ch", "场景", 预期目录版本=0
)
body.保存人工(
author,
"seed-eval-text",
"synthetic-ch",
0,
正文草稿((段落("p1", (文本节点(scene.text),)),)),
)
target = {"work_id": "synthetic:demo", "chapter_id": "synthetic-ch", "branch_id": "main"}
before = body.读取正文(author, "synthetic-ch")
with app.要求数据库().连接(只读=True) as conn:
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
result = 执行行为场景(scene, 服务观察适配(body, cases, author, target))
assert result["passed"], result
assert result["mode"] == "service" and not result["model_verified"]
assert body.读取正文(author, "synthetic-ch") == before
with app.要求数据库().连接(只读=True) as conn:
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
assert (
conn.execute(
"SELECT count(*) FROM muse_attempt WHERE call_reference IS NOT NULL"
).fetchone()[0]
== 0
)
assert conn.execute("SELECT count(*) FROM muse_diagnosis").fetchone()[0] == (
1 if scene.expected == "report" else 0
)
if scene.expected == "report":
observation = result["observation"]
assert observation["report"] == observation["report_readback"]
# 覆盖区间只统计真正执行的确定性规则;本场景没有激活规则,故为空。
coverage = observation["report"]["coverage"]
assert coverage["text_length"] == len(scene.text)
assert coverage["rules_executed"] == 0 and observation["report"]["empty_library"] is True
assert coverage["deterministic"] == (
[[0, len(scene.text)]] if coverage["rules_executed"] else []
)