muse-agent-example/tests/集成/test_运行观察与来源.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

738 lines
33 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""真实隔离PG来源与公开owner读回;模型场景使用合成HTTP,不声称真实文学效果。"""
from concurrent.futures import ThreadPoolExecutor
from dataclasses import replace
from threading import Barrier
from uuid import UUID, uuid4
import psycopg
import pytest
import test_两阶段写手 as 写手测试
import test_受控规划方法消费 as 规划测试
import test_模型修订任务 as 修订测试
from psycopg.types.json import Jsonb
from muse.任务运行.接口 import (
任务服务,
任务状态,
任务请求,
步骤处理器,
步骤结果,
步骤计划,
证据服务,
读取运行观察依据于,
)
from muse.作者经验.接口 import 作者经验服务, 改进定位, 运行观察来源, 运行观察请求
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
from muse.共享.错误 import Muse错误
from muse.正式变更.接口 import 参与者目录, 固定哈希, 正式变更服务
from muse.编排.接口 import 流程定义, 流程服务, 流程登记
pytestmark = pytest.mark.数据库
生成环境 = 写手测试.生成环境
规划环境 = 规划测试.规划环境
修订环境 = 修订测试.修订环境
@pytest.fixture
def 观察环境(应用测试库):
pool = 应用测试库[用途.生产]
author = 调用身份("observation-author", None, 用途.生产, 内容用途.生成)
registry = 流程登记()
def 处理(context):
if context.任务.冻结输入["输入"].get("fail"):
raise RuntimeError("合成处理器失败")
return 步骤结果({"recorded": True})
registry.登记处理器(步骤处理器("observation.test", "1", 处理, "v1", "v1"))
registry.登记类型("observation-test", 必需保护=())
registry.登记恢复检查("observation-test", "1", lambda _: None)
runtime = 任务服务(pool, registry)
流程服务(runtime, registry).发布(
流程定义("observation-test", "1", (步骤计划("测试步骤", "observation.test", "1"),))
)
formal = 正式变更服务(pool, 参与者目录())
return {
"pool": pool,
"pools": 应用测试库,
"author": author,
"run": runtime,
"service": 作者经验服务(pool, formal),
}
def _开始(env, *, command=None, author=None, fail=False):
actor = author or env["author"]
request = 任务请求(
"observation-test",
command or str(uuid4()),
actor.作者,
用途.生产,
actor.内容用途,
{"fail": fail, "private_input": "不应进入运行观察的正文"},
"synthetic-policy-1",
"synthetic-release-1",
{
"source_scope": {},
"schema_versions": {},
"authorization": "synthetic-grant",
"budget": {},
"stop_conditions": {},
},
)
tid = env["run"].创建任务(request, "observation-test", "1")
claim = env["run"].领取步骤("worker", ["observation.test"], 任务ID=tid, 作者=actor.作者)
assert claim is not None
return tid, claim
def _来源(env, tid, aid, **kwargs):
return env["service"].读取运行观察来源(env["author"], tid, aid, **kwargs)
def _观察(source, *, title="具名运行现象", issue="同类问题"):
return 运行观察请求(
(运行观察来源(**source["locator"]),),
title,
"本次需要保留的作者原话。",
"隔离合成运行的技术观察。",
issue,
)
def _完成(env, **kwargs):
tid, claim = _开始(env, **kwargs)
env["run"].执行一步(claim)
return tid, claim, _来源(env, tid, claim.尝试ID)
@pytest.mark.case_id(
"TC-15c187553e39",
environment="隔离PostgreSQL与真实S02/业务owner,模型为明确合成HTTP",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_运行观察保留身份原话背景并拒绝非法输入__15c187(观察环境):
env = 观察环境
tid, claim, source = _完成(env)
before = env["run"].读取任务(tid)
saved = env["service"].记录运行观察(env["author"], "record", _观察(source))
assert UUID(str(saved["observation_id"])) and saved["state"] == "observation"
read = env["service"].读取运行观察(env["author"], str(saved["observation_id"]))
assert read["payload"]["quote"] == "本次需要保留的作者原话。"
runtime = read["payload"]["sources"][0]["runtime"]
assert runtime["task_id"] == tid and runtime["attempt_id"] == claim.尝试ID
assert runtime["resource_release_id"] == "synthetic-release-1"
assert runtime["role_policy_version"] == "synthetic-policy-1"
assert runtime["processor_version"] == "1" and runtime["runtime_config"] is None
assert runtime["result_hash"] == 固定哈希({"recorded": True})
assert "不应进入运行观察的正文" not in str(read) and "synthetic-grant" not in str(read)
assert env["run"].读取任务(tid) == before
for change in (
{"phenomenon": ""},
{"phenomenon": "x" * 201},
{"background": ""},
{"quote": " "},
):
with pytest.raises((Muse错误, ValueError)):
replace(_观察(source), **change)
@pytest.mark.case_id(
"TC-1fdbe46e2219",
environment="隔离PostgreSQL与真实S02/业务owner,模型为明确合成HTTP",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_运行观察命令与相同来源幂等并拒绝内容覆盖__1fdbe4(观察环境):
env = 观察环境
_, _, source = _完成(env)
request = _观察(source)
first = env["service"].记录运行观察(env["author"], "one", request)
second = env["service"].记录运行观察(env["author"], "two", request)
replay = env["service"].记录运行观察(env["author"], "one", request)
assert not first["deduped"] and second["deduped"] and replay["deduped"]
assert first["observation_id"] == second["observation_id"] == replay["observation_id"]
assert len(env["service"].列出运行观察(env["author"])) == 1
for command in ("one", "different"):
with pytest.raises(Muse错误, match="内容|原话"):
env["service"].记录运行观察(env["author"], command, replace(request, quote="不同原话"))
@pytest.mark.case_id(
"TC-8baea001ddf5",
environment="隔离PostgreSQL与真实S02/业务owner,模型为明确合成HTTP",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_成功运行只保存观察且不能在数据库升格__8baea0(观察环境):
env = 观察环境
_, _, source = _完成(env)
saved = env["service"].记录运行观察(env["author"], "completed", _观察(source))
assert saved["payload"]["state"] == "observation"
assert saved["payload"]["causality"] == "not_established"
assert env["service"].列出偏好(env["author"]) == []
assert env["service"].读取观察(env["author"])["observations"] == []
with pytest.raises(Muse错误):
env["service"].请求启停(
env["author"],
"no-promotion",
改进定位(str(saved["observation_id"]), 1, saved["content_hash"]),
"enable",
)
with env["pools"][用途.维护].连接() as conn:
with pytest.raises(psycopg.Error), conn.transaction():
conn.execute(
"UPDATE muse_run_observation SET state='promoted' WHERE observation_id=%s",
(saved["observation_id"],),
)
assert (
env["service"].读取运行观察(env["author"], saved["observation_id"])["state"]
== "observation"
)
@pytest.mark.case_id(
"TC-6099d104a8a3",
environment="隔离PostgreSQL与真实S02/业务owner,模型为明确合成HTTP",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_空来源在Python与PG约束都被拒绝__6099d1(观察环境):
env = 观察环境
with pytest.raises((Muse错误, ValueError)):
运行观察请求((), "空来源", "作者原话", "无来源不记录")
# 仅验证数据库约束,故意尝试非法合成行;不为运行能力手造成功证据。
for source_ids, payload, constraint in [
(
[],
{
"sources": [
{
"runtime": {"task_id": str(uuid4()), "attempt_id": str(uuid4())},
"locator": {"content_hash": "a" * 64},
}
]
},
"muse_run_observation_source_ids_nonempty",
),
([str(uuid4())], {"sources": []}, "muse_run_observation_sources_present"),
([str(uuid4())], {}, "muse_run_observation_sources_present"),
([str(uuid4())], {"sources": [{}]}, "muse_run_observation_sources_present"),
([str(uuid4())], {"sources": None}, "muse_run_observation_sources_present"),
]:
with env["pool"].连接() as conn:
with pytest.raises(psycopg.errors.CheckViolation) as caught, conn.transaction():
conn.execute(
"INSERT INTO muse_run_observation(observation_id,author_id,source_ids,"
"source_set_hash,phenomenon,request_hash,payload,content_hash) "
"VALUES (%s,%s,%s,%s,'空来源',%s,%s,%s)",
(
str(uuid4()),
env["author"].作者,
source_ids,
"a" * 64,
"b" * 64,
Jsonb(payload),
"c" * 64,
),
)
assert caught.value.diag.constraint_name == constraint
assert env["service"].列出运行观察(env["author"]) == []
@pytest.mark.case_id(
"TC-7e9bc1e4e12a",
environment="隔离PostgreSQL与真实S02/业务owner,模型为明确合成HTTP",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_缺真实运行或尝试明确拒绝且不留观察__7e9bc1(观察环境):
env = 观察环境
for tid, aid in [("", ""), (str(uuid4()), str(uuid4()))]:
with pytest.raises((Muse错误, ValueError)):
env["service"].读取运行观察来源(env["author"], tid, aid)
fake = 运行观察来源(str(uuid4()), str(uuid4()), 1, "a" * 64)
with pytest.raises(Muse错误):
env["service"].记录运行观察(
env["author"], "missing", 运行观察请求((fake,), "缺源", "原话", "背景")
)
assert env["service"].列出运行观察(env["author"]) == []
@pytest.mark.case_id(
"NC-w26-265101",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["跨作者用途尝试关联和错版本哈希逐项拒绝"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_跨作者用途尝试关联和错版本哈希逐项拒绝__265101(观察环境):
env = 观察环境
tid, claim, source = _完成(env)
_, other_claim, _ = _完成(env)
other = replace(env["author"], 作者="other-author")
for actor, task_id, attempt_id in [
(other, tid, claim.尝试ID),
(replace(env["author"], 内容用途=内容用途.规划), tid, claim.尝试ID),
(replace(env["author"], 用途=用途.评测), tid, claim.尝试ID),
(env["author"], tid, other_claim.尝试ID),
]:
with pytest.raises(Muse错误):
env["service"].读取运行观察来源(actor, task_id, attempt_id)
for change in ({"revision": source["locator"]["revision"] + 1}, {"content_hash": "0" * 64}):
point = replace(运行观察来源(**source["locator"]), **change)
with pytest.raises(Muse错误, match="版本或哈希"):
env["service"].记录运行观察(
env["author"], str(uuid4()), 运行观察请求((point,), "错版本", "原话", "背景")
)
with env["pool"].连接(只读=True) as conn, pytest.raises(Muse错误):
读取运行观察依据于(conn, 用途.评测, env["author"].作者, tid, claim.尝试ID)
assert env["service"].列出运行观察(env["author"]) == []
@pytest.mark.case_id(
"NC-w26-265102",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["历史失败和未知不随后来成功被改写"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_历史失败和未知不随后来成功被改写__265102(观察环境):
env = 观察环境
tid, failed = _开始(env, fail=True)
with pytest.raises(RuntimeError):
env["run"].执行一步(failed)
failure = _来源(env, tid, failed.尝试ID)
assert failure["runtime"]["attempt_state"] == "failed"
assert failure["runtime"]["failure_code"] == "RuntimeError"
failed_row = env["service"].记录运行观察(env["author"], "failure", _观察(failure))
uid, unknown = _开始(env)
# 复现登记外部调用后中断的未知边界;没有响应,不伪造model_delivery。
env["run"].登记外部调用(unknown, "synthetic-unresolved-call")
env["run"].失败步骤(unknown, "INTERRUPTED")
pending = _来源(env, uid, unknown.尝试ID)
assert pending["runtime"]["call_state"] == "unknown"
unknown_request = _观察(pending)
unknown_row = env["service"].记录运行观察(env["author"], "unknown", unknown_request)
env["run"].对账调用(
uid,
unknown.尝试ID,
env["author"].作者,
已保存输出={"recovered": True},
对账回执="synthetic-external-reconciliation",
)
env["run"].控制任务(uid, env["author"].作者, 任务状态.待对账, "恢复", 命令ID="recover")
new = env["run"].领取步骤("recovery", ["observation.test"], 任务ID=uid, 作者=env["author"].作者)
assert new is not None and new.尝试ID != unknown.尝试ID
env["run"].执行一步(new)
current = _来源(env, uid, unknown.尝试ID)
assert current["runtime"]["task_state"] == "completed"
assert current["runtime"]["attempt_state"] != "completed"
replay = env["service"].记录运行观察(env["author"], "unknown", unknown_request)
assert replay["payload"] == unknown_row["payload"]
assert replay["payload"]["sources"][0]["runtime"]["call_state"] == "unknown"
assert (
env["service"].读取运行观察(env["author"], failed_row["observation_id"])["payload"]
== failed_row["payload"]
)
@pytest.mark.case_id(
"NC-w26-265103",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["真实失败证据保留类型版本与哈希且不装作模型消费"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_真实失败证据保留类型版本与哈希且不装作模型消费__265103(观察环境):
env = 观察环境
tid, claim = _开始(env, fail=True)
with pytest.raises(RuntimeError):
env["run"].执行一步(claim)
receipt = 证据服务(env["pool"]).保存(
tid,
claim.尝试ID,
"failure",
"a" * 64,
"failed",
{"error_code": "SYNTHETIC_FAILURE", "policy_version": "synthetic-policy-1"},
)
source = _来源(env, tid, claim.尝试ID)
item = source["runtime"]["evidence"][0]
assert item["evidence_id"] == receipt["evidence_id"] and item["revision"] == 1
assert item["outcome"] == "failed" and item["kind"] == "failure"
assert not item["delivery_confirmed"] and item["retention"] == "hash_only"
assert item["metadata"]["error_code"] == "SYNTHETIC_FAILURE"
saved = env["service"].记录运行观察(env["author"], "error", _观察(source))
assert saved["payload"]["sources"][0]["runtime"]["evidence"] == [item]
@pytest.mark.case_id(
"NC-w26-265104",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["重复问题按实际任务检索且跨作者隔离"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_重复问题按实际任务检索且跨作者隔离__265104(观察环境):
env = 观察环境
tid, _, source = _完成(env)
env["service"].记录运行观察(env["author"], "first", _观察(source))
env["service"].记录运行观察(
env["author"], "second-label", _观察(source, title="同次运行另一条说明")
)
assert env["service"].检索重复问题(env["author"]) == []
other_tid, _, other = _完成(env)
env["service"].记录运行观察(env["author"], "other-run", _观察(other))
groups = env["service"].检索重复问题(env["author"], 问题键="同类问题")
assert len(groups) == 1 and groups[0]["task_count"] == 2
assert groups[0]["observation_count"] == 3 and groups[0]["task_ids"] == sorted([tid, other_tid])
assert len(env["service"].列出运行观察(env["author"], 任务ID=tid)) == 2
assert env["service"].检索重复问题(replace(env["author"], 作者="another")) == []
@pytest.mark.case_id(
"NC-w26-265105",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["新来源状态改变拒绝旧定位而历史命令可原样恢复"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_新来源状态改变拒绝旧定位而历史命令可原样恢复__265105(观察环境):
env = 观察环境
tid, claim = _开始(env)
source = _来源(env, tid, claim.尝试ID)
old_request = _观察(source)
original = env["service"].记录运行观察(env["author"], "original", old_request)
env["run"].执行一步(claim)
# 相同历史观察恢复原记录;以旧快照提出另一现象必须重验并拒绝。
assert (
env["service"].记录运行观察(env["author"], "recover", old_request)["payload"]
== original["payload"]
)
with pytest.raises(Muse错误, match="版本或哈希"):
env["service"].记录运行观察(
env["author"], "stale", replace(old_request, phenomenon="后来完成")
)
for command in ("original", "cross-purpose"):
with pytest.raises(Muse错误, match="用途"):
env["service"].记录运行观察(
replace(env["author"], 内容用途=内容用途.规划), command, old_request
)
assert len(env["service"].列出运行观察(env["author"])) == 1
@pytest.mark.case_id(
"NC-w26-265106",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["来源集合重排及并发新命令只生成同一条观察"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_来源集合重排及并发新命令只生成同一条观察__265106(观察环境):
env = 观察环境
_, _, first = _完成(env)
_, _, second = _完成(env)
request = replace(
_观察(first), sources=(运行观察来源(**first["locator"]), 运行观察来源(**second["locator"]))
)
start = Barrier(2)
def record(index):
start.wait()
item = request if index == 0 else replace(request, sources=tuple(reversed(request.sources)))
return env["service"].记录运行观察(env["author"], f"concurrent-{index}", item)
with ThreadPoolExecutor(max_workers=2) as executor:
rows = list(executor.map(record, (0, 1)))
assert rows[0]["observation_id"] == rows[1]["observation_id"]
assert sorted(row["deduped"] for row in rows) == [False, True]
assert len(env["service"].列出运行观察(env["author"])) == 1
@pytest.mark.case_id(
"NC-w26-265107",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["已存证据错版本明确拒绝且不覆盖S02失败记录"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
@pytest.mark.parametrize("key", ["policy_version", "resource_release_id"])
def test_已存证据错版本明确拒绝且不覆盖S02失败记录__265107(观察环境, key):
env = 观察环境
tid, claim = _开始(env, fail=True)
with pytest.raises(RuntimeError):
env["run"].执行一步(claim)
receipt = 证据服务(env["pool"]).保存(
tid,
claim.尝试ID,
"failure",
"f" * 64,
"failed",
{key: "wrong-version", "error_code": "SYNTHETIC_FAILURE"},
)
with pytest.raises(Muse错误, match="版本"):
_来源(env, tid, claim.尝试ID)
assert 证据服务(env["pool"]).读取回执(tid, receipt["evidence_id"]) == receipt
assert env["service"].列出运行观察(env["author"]) == []
@pytest.mark.case_id(
"NC-w26-265108",
environment="隔离PG/合成HTTP及浏览器",
given="固定请求、来源与作者;按用例环境准备独立测试输入",
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
then=["评测数据库权限不能读写运行观察"],
contract="docs/系统架构/新版设计/模块设计/B07-作者经验.md#有来源的运行观察",
)
def test_评测数据库权限不能读写运行观察__265108(观察环境):
env = 观察环境
for sql_text in (
"SELECT * FROM public.muse_run_observation",
"SELECT * FROM public.muse_run_observation_command",
):
with env["pools"][用途.评测].连接() as conn:
with pytest.raises(psycopg.errors.InsufficientPrivilege), conn.transaction():
conn.execute(sql_text)
def _业务环境(env, purpose=内容用途.生成):
return {
"service": 作者经验服务(env["库"], env["正文"].正式),
"author": replace(env["作者"], 内容用途=purpose),
}
@pytest.mark.case_id(
"TC-2fbacb191cbb",
environment="隔离PostgreSQL与真实S02/业务owner,模型为明确合成HTTP",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
@pytest.mark.parametrize("failed", [False, True], ids=["passed", "failed"])
def test_正文机械门结果来自真实报告并保留原尝试版本__2fbacb(生成环境, failed):
paragraphs = [
"雨点敲着铁皮屋顶,林深把退回的信压在掌心。",
"他沿跳板走向仓库,门后传来一声应答。",
]
if failed:
paragraphs[0] = "上一章结尾雨还没停,林深攥着信封走进雨幕。"
tid, task = 写手测试.发起并跑完(生成环境, 剧本=写手测试.正常剧本(paragraphs))
env = _业务环境(生成环境)
writer = next(s for s in task.步骤 if s["step_id"] == "无工具写作")
check_step = next(s for s in task.步骤 if s["step_id"] == "检查与候选")
check = check_step["result"]
source = _来源(
env,
tid,
str(check_step["current_attempt"]),
业务类型="review_report",
业务引用=check["candidate_id"],
业务版本=1,
)
result = source["domain"]["result"]
assert result["verdict"] == ("failed" if failed else "passed")
report = 生成环境["装配"].要求审校().读取报告(check["candidate_id"], 1)
assert str(result["report_id"]) == str(report["report_id"])
assert source["runtime"]["runtime_config"]["config_id"] == "gen-config"
writer_source = _来源(env, tid, str(writer["current_attempt"]))
assert any(e["delivery_confirmed"] for e in writer_source["runtime"]["evidence"])
with pytest.raises(Muse错误, match="检查事件"):
_来源(
env,
tid,
str(writer["current_attempt"]),
业务类型="review_report",
业务引用=check["candidate_id"],
业务版本=1,
)
if failed:
assert any(r["outcome"] == "conflict" for r in result["findings"])
saved = env["service"].记录运行观察(
env["author"],
"writer-gate",
replace(
_观察(source, title="正文生产机械门"),
sources=(运行观察来源(**source["locator"]), 运行观察来源(**writer_source["locator"])),
),
)
assert saved["state"] == "observation"
assert (
生成环境["正文"].读取候选(生成环境["作者"], check["candidate_id"])["decision"]
== "undecided"
)
@pytest.mark.case_id(
"TC-ee1c05270fa2",
environment="真实业务入口与隔离 PostgreSQL,评测替身须明示",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_实际冻结步骤留具名观察不冒充模型消费__ee1c05(生成环境):
tid, task = 写手测试.发起并跑完(
生成环境, 剧本=写手测试.正常剧本(["林深敲了门。", "门后有人应声。"])
)
step = next(s for s in task.步骤 if s["step_id"] == "冻结回放")
env = _业务环境(生成环境)
source = _来源(env, tid, str(step["current_attempt"]))
assert source["runtime"]["processor_id"] == "ctx.freeze"
assert source["runtime"]["result_hash"] == 固定哈希(step["result"])
assert source["runtime"]["call_state"] == "not_sent"
assert source["runtime"]["evidence"] == []
saved = env["service"].记录运行观察(
env["author"], "freeze", _观察(source, title="写作上下文冻结步骤完成")
)
assert (
saved["payload"]["phenomenon"] == "写作上下文冻结步骤完成"
and saved["state"] == "observation"
)
@pytest.mark.case_id(
"TC-90b874ef9830",
environment="真实业务入口与隔离 PostgreSQL,评测替身须明示",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_规划观察读取实际候选版本并保持未确认__90b874(规划环境):
tid, task = 规划测试._运行(规划环境)
env = _业务环境(规划环境, 内容用途.规划)
saved = next(s for s in task.步骤 if s["step_id"] == "保存规划候选")["result"]
model = next(s for s in task.步骤 if s["step_id"] == "模型规划")
source = _来源(
env,
tid,
str(model["current_attempt"]),
业务类型="planning_candidate",
业务引用=saved["candidate_id"],
业务版本=1,
)
candidate = 规划环境["规划"].读取候选(规划环境["作者"], saved["candidate_id"])
assert source["domain"]["content_hash"] == candidate["candidate_hash"]
assert source["domain"]["result"]["schema_binding"] == candidate["schema_binding"]
assert source["domain"]["result"]["state"] == "candidate"
freezing = next(s for s in task.步骤 if s["step_id"] == "冻结规划")
with pytest.raises(Muse错误, match="实际交付"):
_来源(
env,
tid,
str(freezing["current_attempt"]),
业务类型="planning_candidate",
业务引用=saved["candidate_id"],
业务版本=1,
)
row = env["service"].记录运行观察(
env["author"], "plan", _观察(source, title="规划候选已持久化")
)
assert row["state"] == "observation" and candidate["decision"] == "undecided"
assert 规划环境["规划"].读取规划(规划环境["作者"], candidate["plan_id"])["revision"] == 1
@pytest.mark.case_id(
"TC-2623c0036a0a",
environment="真实业务入口与隔离 PostgreSQL,评测替身须明示",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
@pytest.mark.parametrize("choice", ["replace", "retain", "bad_number"])
def test_修订观察分别保留候选原文与阻断结果__2623c0(修订环境, choice):
env = 修订环境
tid = 修订测试.发起(env)
env["剧本"].append(
{
"类型": "文本",
"文本": {
"edits": [
{
"finding_id": env["发现"],
"action": "retain" if choice == "retain" else "replace",
"replacement": "4枚铜钱," if choice == "bad_number" else "",
"reason": "合成修改建议",
}
]
},
}
)
task = 修订测试.跑(env, tid)
model = next(s for s in task.步骤 if s["step_id"] == "模型修订")
observation = _业务环境(env)
source = _来源(
observation,
tid,
str(model["current_attempt"]),
业务类型="revision_result",
业务引用=tid,
业务版本=1,
)
result = env["装配"].要求审校().读取模型修订(env["作者"], tid)
assert source["domain"]["content_hash"] == 固定哈希(result)
assert (
source["domain"]["result"]["outcome"]
== {"replace": "candidate", "retain": "original", "bad_number": "rejected"}[choice]
)
row = observation["service"].记录运行观察(
observation["author"], "revision", _观察(source, title="本次修订结果")
)
assert row["state"] == "observation"
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
@pytest.mark.case_id(
"TC-ee5f4ff42c54",
environment="真实业务入口与隔离 PostgreSQL,评测替身须明示",
when="记录观察、提交改进、验证固定目标并由作者批准。",
contract="docs/系统架构/新版设计/功能规格/P06-经验验证与启用.md",
)
def test_拆书具名观察读取B03真实分析而非自造run标签__ee5f4f(研究环境):
tool = 研究环境
source = tool.导入来源("运行观察合成来源", "第1回 机舱\n林深照看机舱。\n")
started = tool.发起(source["source_id"])
task = tool.装配.任务运行.读取任务(started["task_id"])
for _ in task.冻结输入["输入"]["窗计划"]:
tool.剧本.append(
tool.分析输出([{"name": "林深", "type": "人物", "evidence": ["林深照看机舱。"]}])
)
done = tool.跑任务(started["task_id"])
step = next(s for s in done.步骤 if s["step_id"] == "分析版本登记")
actor = 调用身份("post-author", None, 用途.生产, 内容用途.抽取)
service = 作者经验服务(tool.库, tool.装配.要求正文().正式)
observed = service.读取运行观察来源(
actor,
started["task_id"],
str(step["current_attempt"]),
业务类型="reference_analysis",
业务引用=started["task_id"],
业务版本=1,
)
with tool.库.连接(只读=True) as conn:
actual = tool.服务.读取分析(conn, actor.作者, started["task_id"])
assert observed["domain"]["content_hash"] == actual["content_hash"]
assert observed["domain"]["result"]["source_revision"] == actual["source_revision"]
assert observed["domain"]["result"]["version"]["完成窗"] == [
w["窗ID"] for w in task.冻结输入["输入"]["窗计划"]
]
with pytest.raises(Muse错误, match="确切结果"):
service.读取运行观察来源(
actor,
started["task_id"],
str(step["current_attempt"]),
业务类型="reference_analysis",
业务引用=started["task_id"],
业务版本=2,
)
saved = service.记录运行观察(actor, "analysis", _观察(observed, title="拆书分析已登记"))
assert (
saved["state"] == "observation"
and saved["payload"]["sources"][0]["domain"] == observed["domain"]
)