实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
70 lines
3.3 KiB
Python
70 lines
3.3 KiB
Python
"""合成场景在隔离PG上跑真实公开用例;不声称自然语言路由或真实模型效果。"""
|
||
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
|
||
from muse.作品规划.接口 import 档案保存
|
||
from muse.效果评测.接口 import 执行行为场景, 服务观察适配, 载入行为场景
|
||
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
场景集 = 载入行为场景((Path(__file__).parents[1] / "夹具/行为评测/诊断机器味场景.json").read_text())
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w24-241005",
|
||
environment="隔离PG真服务",
|
||
given="固定六场景及逐例原始观察",
|
||
when="显式脚本或真实隔离服务执行,注入缺失、禁止动作及伪报告",
|
||
then=["场景逐个独立裁决,报告重算和读回、来源前后哈希、失败与模式均可追溯"],
|
||
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
|
||
)
|
||
@pytest.mark.parametrize("scene", 场景集, ids=lambda s: s.scenario_id)
|
||
def test_真实服务观察报告读回及正文不变__241005(章后环境, scene):
|
||
app, author = 章后环境["装配"], 章后环境["作者"]
|
||
works, body, cases = app.要求作品(), app.要求正文(), app.要求审校()
|
||
schema = works.新档案结构(author, "synthetic:demo", "work_core", 1)
|
||
works.保存档案(
|
||
author, "create-eval-work", 档案保存("synthetic:demo", 0, {"名称": "合成诊断场景"}, schema)
|
||
)
|
||
works.添加章节(
|
||
author, "create-eval-ch", "synthetic:demo", "synthetic-ch", "场景", 预期目录版本=0
|
||
)
|
||
body.保存人工(
|
||
author,
|
||
"seed-eval-text",
|
||
"synthetic-ch",
|
||
0,
|
||
正文草稿((段落("p1", (文本节点(scene.text),)),)),
|
||
)
|
||
target = {"work_id": "synthetic:demo", "chapter_id": "synthetic-ch", "branch_id": "main"}
|
||
before = body.读取正文(author, "synthetic-ch")
|
||
with app.要求数据库().连接(只读=True) as conn:
|
||
commands = conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0]
|
||
result = 执行行为场景(scene, 服务观察适配(body, cases, author, target))
|
||
assert result["passed"], result
|
||
assert result["mode"] == "service" and not result["model_verified"]
|
||
assert body.读取正文(author, "synthetic-ch") == before
|
||
with app.要求数据库().连接(只读=True) as conn:
|
||
assert conn.execute("SELECT count(*) FROM muse_change_command").fetchone()[0] == commands
|
||
assert (
|
||
conn.execute(
|
||
"SELECT count(*) FROM muse_attempt WHERE call_reference IS NOT NULL"
|
||
).fetchone()[0]
|
||
== 0
|
||
)
|
||
assert conn.execute("SELECT count(*) FROM muse_diagnosis").fetchone()[0] == (
|
||
1 if scene.expected == "report" else 0
|
||
)
|
||
if scene.expected == "report":
|
||
observation = result["observation"]
|
||
assert observation["report"] == observation["report_readback"]
|
||
# 覆盖区间只统计真正执行的确定性规则;本场景没有激活规则,故为空。
|
||
coverage = observation["report"]["coverage"]
|
||
assert coverage["text_length"] == len(scene.text)
|
||
assert coverage["rules_executed"] == 0 and observation["report"]["empty_library"] is True
|
||
assert coverage["deterministic"] == (
|
||
[[0, len(scene.text)]] if coverage["rules_executed"] else []
|
||
)
|