muse-agent-example/tests/集成/test_原观察返修比较.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

325 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""两条旧观察以新版真实持久返修、规则复扫和固定双盲比较重放。"""
from __future__ import annotations
import hashlib
import json
from dataclasses import asdict
from pathlib import Path
import pytest
import test_持久返修会话 as 返修测试
import test_模型修订任务 as 修订测试
import test_返修独立比较 as 比较测试
from muse.审校修订.接口 import 允许选区, 核对规则库, 模型修订授权, 诊断文本
from muse.正式变更.接口 import 固定哈希
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落, 选区坐标, 选段写作授权
from muse.编排.审校修订 import 发起有限返修, 读取有限返修
pytestmark = pytest.mark.数据库
生成环境 = 比较测试.生成环境
修订环境 = 比较测试.修订环境
比较环境 = 比较测试.比较环境
def _规则副本(env, 诊断报告: dict) -> dict:
服务 = env["装配"].要求审校()
当前 = 服务.读取规则库(env["作者"], 预览候选=True)
规则, 例证 = [], {}
for rule in 当前["rules"].values():
version = 服务.读取规则版本(env["作者"], rule["id"], rule["version"])
assert version["payload"]["rule"] == rule
规则.append(version["payload"]["rule"])
for sample_id, sample in version["payload"]["samples"].items():
assert sample_id not in 例证 or 例证[sample_id] == sample
例证[sample_id] = sample
frozen = 核对规则库(规则, list(例证.values()))
assert frozen["fingerprint"] == 诊断报告["rule_library_version"]
return frozen
def _完成并读取证明(env, experiment: dict) -> dict:
face = 比较测试._启动并运行(env, experiment)
assert len(face["units"]) == 2
proof = (
env["eval_app"]
.要求评测()
.导出返修比较证明(
env["eval_actor"],
experiment["experiment_id"],
env["pools"][比较测试.用途.维护],
env["maint_actor"],
)
)
return (
env["装配"]
.要求审校()
.读取返修比较证明(env["作者"], env["装配"].任务运行, proof["receipt_id"])
)
@pytest.mark.case_id(
"TC-446746f1ed42",
environment="隔离PostgreSQL与实际S02合成HTTP双评委;不认证外部模型文学效果",
when="执行具名公开保真、离线回放或实际CLI与隔离PG入口",
then=[
"保留原CLEAN_TEXT及确切候选门外已经下起了雨;真实S02修订的机械保真与规则复扫通过,同一候选经两名独立评委实际任务选择并留可读证明;当前正文不变。"
],
contract="docs/系统架构/新版设计/功能规格/P04-声音诊断与受控修订.md",
)
def test_full_patch_chain_passes_gates__446746(比较环境):
env = 比较环境
# 保留旧 RevisionContractTest 的 CLEAN_TEXT 与确切候选观察。
original = "值得注意的是,门外已经下起了雨。"
env["正文"].保存人工(
env["作者"],
"simple-original",
"ch-2",
0,
正文草稿((段落("simple", (文本节点(original),)),)),
)
saved = env["正文"].读取正文(env["作者"], "ch-2")
target = {
k: saved[k] for k in ("work_id", "chapter_id", "revision", "document_hash", "branch_id")
}
diagnosis = env["装配"].要求审校().生成诊断(env["作者"], target, 预览候选=True)
finding = next(f for f in diagnosis["report"]["findings"] if f["rule_id"] == "l002")
env["授权"] = 模型修订授权(
diagnosis["diagnosis_id"],
固定哈希(diagnosis["report"]),
(允许选区("simple", 0, 7),),
(finding["id"],),
"删除无功能套语",
)
env["发现"] = finding["id"]
session, candidate_id, _, _, experiment = 比较测试._准备实验(env, "tc-446746f1ed42")
result = session["rounds"][0]["result"]
assert result["checks"]["mechanical_pass"] and result["report"]["verdict"] == "passed"
candidate = env["正文"].读取候选(env["作者"], candidate_id)
assert candidate["visible_text"] == "门外已经下起了雨。"
rescanned = 诊断文本(
candidate["visible_text"],
work_id="gen-work",
规则库=_规则副本(env, diagnosis["report"]),
预览候选=True,
)
assert rescanned["findings"] == []
read = _完成并读取证明(env, experiment)
assert read["proof"]["version"] == "revision-comparison-proof-v1"
assert read["proof"]["outcome"] == "candidate"
assert read["proof"]["comparison_complete"] and len(read["proof"]["decisions"]) == 2
assert read["current"]["selection"] == "candidate"
assert not read["current"]["automatic_adoption"]
assert env["正文"].读取正文(env["作者"], "ch-2") == saved
def _唯一坐标(lines: list[str], original: str) -> 选区坐标:
matches = [
(index, line.index(original))
for index, line in enumerate(lines)
if line.count(original) == 1
]
assert len(matches) == 1, original
index, start = matches[0]
return 选区坐标(f"u0-{index}", start, start + len(original), original)
@pytest.mark.case_id(
"TC-98ee99c7cf24",
environment="隔离PostgreSQL与实际S02合成HTTP双评委;不认证外部模型文学效果",
when="应用最小修改、重放或触发有限返修。",
contract="docs/系统架构/新版设计/功能规格/P04-声音诊断与受控修订.md",
)
def test_full_pipeline_replay__98ee99(比较环境):
env = 比较环境
fixture = json.loads((Path(__file__).parents[1] / "夹具/返修回放/U0原观察.json").read_text())
original = fixture["text"]
assert len(original) == 408
assert hashlib.sha256(original.encode()).hexdigest() == fixture["text_sha256"]
assert fixture["patches"][1]["original"] not in original
after_first = original.replace(
fixture["patches"][0]["original"], fixture["patches"][0]["replacement"], 1
)
assert fixture["patches"][1]["original"] in after_first
lines = original.split("\n")
document = 正文草稿(
tuple(
段落(f"u0-{index}", (文本节点(line),) if line else ())
for index, line in enumerate(lines)
)
)
env["正文"].保存人工(env["作者"], "u0-original", "ch-2", 0, document)
saved = env["正文"].读取正文(env["作者"], "ch-2")
assert saved["visible_text"] == original
target = {
key: saved[key]
for key in ("work_id", "chapter_id", "revision", "document_hash", "branch_id")
}
diagnosis = env["装配"].要求审校().生成诊断(env["作者"], target, 预览候选=True)
findings = diagnosis["report"]["findings"]
assert len(findings) == 7
assert "也许这就是命运吧" not in {
original[finding["start"] : finding["end"]] for finding in findings
}
combined = "**那伙计上下打量了她一眼,**嘴角微微上扬"
intents = [(combined, "那伙计上下打量了她一眼")]
intents.extend((patch["original"], patch["replacement"]) for patch in fixture["patches"][2:])
selections = tuple(
env["正文"].读取选段定位(
env["作者"],
"gen-work",
"ch-2",
"main",
saved["revision"],
saved["document_hash"],
tuple(_唯一坐标(lines, text) for text, _ in intents),
)
)
protected = tuple(
env["正文"].读取选段定位(
env["作者"],
"gen-work",
"ch-2",
"main",
saved["revision"],
saved["document_hash"],
tuple(_唯一坐标(lines, text) for text in fixture["protected"]),
)
)
authorization = 选段写作授权(
"gen-work",
"ch-2",
"main",
saved["revision"],
saved["document_hash"],
"rewrite",
"按作者明确的八处编辑意图改写;前两项合并为同一原文上的不重叠选区",
selections,
protected,
)
started = 发起有限返修(
env["装配"],
env["作者"],
"u0-selection-session",
authorization,
配置ID="gen-config",
最大轮数=3,
)
task_id = started["rounds"][0]["task_id"]
返修测试.批准轮预算(env, task_id)
output = {
"edits": [
{
"selection_id": selection.selection_id,
"action": "replace",
"replacement": replacement,
"reason": "作者明确选段编辑意图",
}
for selection, (_, replacement) in zip(selections, intents, strict=True)
]
}
env["剧本"].append({"类型": "文本", "文本": output})
assert 修订测试.跑(env, task_id).状态.value == "completed"
session = 读取有限返修(env["装配"], env["作者"], started["session_id"])
result = session["rounds"][0]["result"]
candidate_id = session["rounds"][0]["candidate_id"]
assert result["source_kind"] == "selection" and result["diagnosis_id"] is None
assert result["checks"]["mechanical_pass"] and result["report"]["verdict"] == "passed"
candidate = env["正文"].读取候选(env["作者"], candidate_id)
candidate_text = candidate["visible_text"]
assert all(patch["marker"] not in candidate_text for patch in fixture["patches"])
assert all(text in candidate_text for text in fixture["protected"])
assert all(
text in candidate_text
for text in (
"「姑娘,这东西值三百两银子,不是您一枚铜钱就能当的。」",
"「三百年?我说小子,你这铺子也就立了三百年,可我手里这枚铜钱——」",
"「它传了三千年。」",
"三天后",
)
)
assert env["正文"].读取正文(env["作者"], "ch-2") == saved
rescanned = 诊断文本(
candidate_text,
work_id="gen-work",
规则库=_规则副本(env, diagnosis["report"]),
预览候选=True,
保护表达=tuple(fixture["protected"]),
)
assert rescanned["findings"] == []
session, same_candidate, publication, data, experiment = 比较测试._从既有返修会话准备实验(
env, "tc-98ee99c7cf24", session
)
assert same_candidate == candidate_id
svc = env["eval_app"].要求评测()
loaded_data = svc.读取数据集(env["eval_actor"], data["version_id"])
assert loaded_data["public_manifest"]["schema_version"] == "fixed-revision-pair-dataset-v2"
with pytest.raises(比较测试.Muse错误, match="没有生成输入"):
svc.读取生成输入(
env["eval_actor"], experiment["experiment_id"], experiment["sample_ids"][0]
)
assert data["source_hash"]
duplicate = 比较测试.评测服务(env["pools"][比较测试.用途.维护]).发布返修比较数据集(
env["maint_actor"], publication, env["库"], env["装配"].任务运行
)
assert duplicate["duplicate"] and duplicate["source_hash"] == data["source_hash"]
read = _完成并读取证明(env, experiment)
source = read["proof"]["source"]
assert read["proof"]["version"] == "revision-comparison-proof-v2"
assert source["version"] == "revision-comparison-source-v2"
assert "diagnosis" not in source
assert source["selection_authorization"] == {
"authorization_hash": 固定哈希(asdict(authorization)),
"basis_hash": session["source"]["basis_hash"],
"mode": "rewrite",
"purpose_hash": 固定哈希(authorization.purpose),
"selection_ids": [selection.selection_id for selection in selections],
"selections_hash": 固定哈希(asdict(authorization)["selections"]),
"protected_range_ids": [selection.selection_id for selection in protected],
"protected_ranges_hash": 固定哈希(asdict(authorization)["protected_ranges"]),
}
assert source["original"]["text_hash"] == fixture["text_sha256"]
assert source["candidate"]["text_hash"] == hashlib.sha256(candidate_text.encode()).hexdigest()
assert source["target"] == {
"work_id": "gen-work",
"chapter_id": "ch-2",
"branch_id": "main",
"original_revision": saved["revision"],
"original_document_hash": saved["document_hash"],
}
assert source["delivery"]["task_id"] == task_id
assert source["delivery"]["config_id"] == "gen-config"
assert source["delivery"]["resource_release"] == env["装配"].配置.资源发布身份
assert source["invariance"]["report_id"] == result["report"]["report_id"]
assert read["proof"]["outcome"] == "candidate"
assert read["proof"]["comparison_complete"] and len(read["proof"]["decisions"]) == 2
assert read["current"]["selection"] == "candidate"
assert not read["current"]["automatic_adoption"]
calls = len(env["received_judges"])
replay = (
env["eval_app"]
.要求评测()
.导出返修比较证明(
env["eval_actor"],
experiment["experiment_id"],
env["pools"][比较测试.用途.维护],
env["maint_actor"],
)
)
assert replay["receipt_id"] == read["receipt_id"]
assert len(env["received_judges"]) == calls
assert (
env["装配"]
.要求审校()
.读取返修比较证明(env["作者"], env["装配"].任务运行, read["receipt_id"])
== read
)
assert len(env["received_judges"]) == calls
assert env["正文"].读取正文(env["作者"], "ch-2") == saved