实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
325 lines
13 KiB
Python
325 lines
13 KiB
Python
"""两条旧观察以新版真实持久返修、规则复扫和固定双盲比较重放。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import json
|
||
from dataclasses import asdict
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
import test_持久返修会话 as 返修测试
|
||
import test_模型修订任务 as 修订测试
|
||
import test_返修独立比较 as 比较测试
|
||
|
||
from muse.审校修订.接口 import 允许选区, 核对规则库, 模型修订授权, 诊断文本
|
||
from muse.正式变更.接口 import 固定哈希
|
||
from muse.正文写作.接口 import 文本节点, 正文草稿, 段落, 选区坐标, 选段写作授权
|
||
from muse.编排.审校修订 import 发起有限返修, 读取有限返修
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
生成环境 = 比较测试.生成环境
|
||
修订环境 = 比较测试.修订环境
|
||
比较环境 = 比较测试.比较环境
|
||
|
||
|
||
def _规则副本(env, 诊断报告: dict) -> dict:
|
||
服务 = env["装配"].要求审校()
|
||
当前 = 服务.读取规则库(env["作者"], 预览候选=True)
|
||
规则, 例证 = [], {}
|
||
for rule in 当前["rules"].values():
|
||
version = 服务.读取规则版本(env["作者"], rule["id"], rule["version"])
|
||
assert version["payload"]["rule"] == rule
|
||
规则.append(version["payload"]["rule"])
|
||
for sample_id, sample in version["payload"]["samples"].items():
|
||
assert sample_id not in 例证 or 例证[sample_id] == sample
|
||
例证[sample_id] = sample
|
||
frozen = 核对规则库(规则, list(例证.values()))
|
||
assert frozen["fingerprint"] == 诊断报告["rule_library_version"]
|
||
return frozen
|
||
|
||
|
||
def _完成并读取证明(env, experiment: dict) -> dict:
|
||
face = 比较测试._启动并运行(env, experiment)
|
||
assert len(face["units"]) == 2
|
||
proof = (
|
||
env["eval_app"]
|
||
.要求评测()
|
||
.导出返修比较证明(
|
||
env["eval_actor"],
|
||
experiment["experiment_id"],
|
||
env["pools"][比较测试.用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
)
|
||
return (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修比较证明(env["作者"], env["装配"].任务运行, proof["receipt_id"])
|
||
)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-446746f1ed42",
|
||
environment="隔离PostgreSQL与实际S02合成HTTP双评委;不认证外部模型文学效果",
|
||
when="执行具名公开保真、离线回放或实际CLI与隔离PG入口",
|
||
then=[
|
||
"保留原CLEAN_TEXT及确切候选门外已经下起了雨;真实S02修订的机械保真与规则复扫通过,同一候选经两名独立评委实际任务选择并留可读证明;当前正文不变。"
|
||
],
|
||
contract="docs/系统架构/新版设计/功能规格/P04-声音诊断与受控修订.md",
|
||
)
|
||
def test_full_patch_chain_passes_gates__446746(比较环境):
|
||
env = 比较环境
|
||
# 保留旧 RevisionContractTest 的 CLEAN_TEXT 与确切候选观察。
|
||
original = "值得注意的是,门外已经下起了雨。"
|
||
env["正文"].保存人工(
|
||
env["作者"],
|
||
"simple-original",
|
||
"ch-2",
|
||
0,
|
||
正文草稿((段落("simple", (文本节点(original),)),)),
|
||
)
|
||
saved = env["正文"].读取正文(env["作者"], "ch-2")
|
||
target = {
|
||
k: saved[k] for k in ("work_id", "chapter_id", "revision", "document_hash", "branch_id")
|
||
}
|
||
diagnosis = env["装配"].要求审校().生成诊断(env["作者"], target, 预览候选=True)
|
||
finding = next(f for f in diagnosis["report"]["findings"] if f["rule_id"] == "l002")
|
||
env["授权"] = 模型修订授权(
|
||
diagnosis["diagnosis_id"],
|
||
固定哈希(diagnosis["report"]),
|
||
(允许选区("simple", 0, 7),),
|
||
(finding["id"],),
|
||
"删除无功能套语",
|
||
)
|
||
env["发现"] = finding["id"]
|
||
session, candidate_id, _, _, experiment = 比较测试._准备实验(env, "tc-446746f1ed42")
|
||
result = session["rounds"][0]["result"]
|
||
assert result["checks"]["mechanical_pass"] and result["report"]["verdict"] == "passed"
|
||
candidate = env["正文"].读取候选(env["作者"], candidate_id)
|
||
assert candidate["visible_text"] == "门外已经下起了雨。"
|
||
rescanned = 诊断文本(
|
||
candidate["visible_text"],
|
||
work_id="gen-work",
|
||
规则库=_规则副本(env, diagnosis["report"]),
|
||
预览候选=True,
|
||
)
|
||
assert rescanned["findings"] == []
|
||
|
||
read = _完成并读取证明(env, experiment)
|
||
assert read["proof"]["version"] == "revision-comparison-proof-v1"
|
||
assert read["proof"]["outcome"] == "candidate"
|
||
assert read["proof"]["comparison_complete"] and len(read["proof"]["decisions"]) == 2
|
||
assert read["current"]["selection"] == "candidate"
|
||
assert not read["current"]["automatic_adoption"]
|
||
assert env["正文"].读取正文(env["作者"], "ch-2") == saved
|
||
|
||
|
||
def _唯一坐标(lines: list[str], original: str) -> 选区坐标:
|
||
matches = [
|
||
(index, line.index(original))
|
||
for index, line in enumerate(lines)
|
||
if line.count(original) == 1
|
||
]
|
||
assert len(matches) == 1, original
|
||
index, start = matches[0]
|
||
return 选区坐标(f"u0-{index}", start, start + len(original), original)
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"TC-98ee99c7cf24",
|
||
environment="隔离PostgreSQL与实际S02合成HTTP双评委;不认证外部模型文学效果",
|
||
when="应用最小修改、重放或触发有限返修。",
|
||
contract="docs/系统架构/新版设计/功能规格/P04-声音诊断与受控修订.md",
|
||
)
|
||
def test_full_pipeline_replay__98ee99(比较环境):
|
||
env = 比较环境
|
||
fixture = json.loads((Path(__file__).parents[1] / "夹具/返修回放/U0原观察.json").read_text())
|
||
original = fixture["text"]
|
||
assert len(original) == 408
|
||
assert hashlib.sha256(original.encode()).hexdigest() == fixture["text_sha256"]
|
||
assert fixture["patches"][1]["original"] not in original
|
||
after_first = original.replace(
|
||
fixture["patches"][0]["original"], fixture["patches"][0]["replacement"], 1
|
||
)
|
||
assert fixture["patches"][1]["original"] in after_first
|
||
lines = original.split("\n")
|
||
document = 正文草稿(
|
||
tuple(
|
||
段落(f"u0-{index}", (文本节点(line),) if line else ())
|
||
for index, line in enumerate(lines)
|
||
)
|
||
)
|
||
env["正文"].保存人工(env["作者"], "u0-original", "ch-2", 0, document)
|
||
saved = env["正文"].读取正文(env["作者"], "ch-2")
|
||
assert saved["visible_text"] == original
|
||
|
||
target = {
|
||
key: saved[key]
|
||
for key in ("work_id", "chapter_id", "revision", "document_hash", "branch_id")
|
||
}
|
||
diagnosis = env["装配"].要求审校().生成诊断(env["作者"], target, 预览候选=True)
|
||
findings = diagnosis["report"]["findings"]
|
||
assert len(findings) == 7
|
||
assert "也许这就是命运吧" not in {
|
||
original[finding["start"] : finding["end"]] for finding in findings
|
||
}
|
||
|
||
combined = "**那伙计上下打量了她一眼,**嘴角微微上扬"
|
||
intents = [(combined, "那伙计上下打量了她一眼")]
|
||
intents.extend((patch["original"], patch["replacement"]) for patch in fixture["patches"][2:])
|
||
selections = tuple(
|
||
env["正文"].读取选段定位(
|
||
env["作者"],
|
||
"gen-work",
|
||
"ch-2",
|
||
"main",
|
||
saved["revision"],
|
||
saved["document_hash"],
|
||
tuple(_唯一坐标(lines, text) for text, _ in intents),
|
||
)
|
||
)
|
||
protected = tuple(
|
||
env["正文"].读取选段定位(
|
||
env["作者"],
|
||
"gen-work",
|
||
"ch-2",
|
||
"main",
|
||
saved["revision"],
|
||
saved["document_hash"],
|
||
tuple(_唯一坐标(lines, text) for text in fixture["protected"]),
|
||
)
|
||
)
|
||
authorization = 选段写作授权(
|
||
"gen-work",
|
||
"ch-2",
|
||
"main",
|
||
saved["revision"],
|
||
saved["document_hash"],
|
||
"rewrite",
|
||
"按作者明确的八处编辑意图改写;前两项合并为同一原文上的不重叠选区",
|
||
selections,
|
||
protected,
|
||
)
|
||
started = 发起有限返修(
|
||
env["装配"],
|
||
env["作者"],
|
||
"u0-selection-session",
|
||
authorization,
|
||
配置ID="gen-config",
|
||
最大轮数=3,
|
||
)
|
||
task_id = started["rounds"][0]["task_id"]
|
||
返修测试.批准轮预算(env, task_id)
|
||
output = {
|
||
"edits": [
|
||
{
|
||
"selection_id": selection.selection_id,
|
||
"action": "replace",
|
||
"replacement": replacement,
|
||
"reason": "作者明确选段编辑意图",
|
||
}
|
||
for selection, (_, replacement) in zip(selections, intents, strict=True)
|
||
]
|
||
}
|
||
env["剧本"].append({"类型": "文本", "文本": output})
|
||
assert 修订测试.跑(env, task_id).状态.value == "completed"
|
||
session = 读取有限返修(env["装配"], env["作者"], started["session_id"])
|
||
result = session["rounds"][0]["result"]
|
||
candidate_id = session["rounds"][0]["candidate_id"]
|
||
assert result["source_kind"] == "selection" and result["diagnosis_id"] is None
|
||
assert result["checks"]["mechanical_pass"] and result["report"]["verdict"] == "passed"
|
||
candidate = env["正文"].读取候选(env["作者"], candidate_id)
|
||
candidate_text = candidate["visible_text"]
|
||
assert all(patch["marker"] not in candidate_text for patch in fixture["patches"])
|
||
assert all(text in candidate_text for text in fixture["protected"])
|
||
assert all(
|
||
text in candidate_text
|
||
for text in (
|
||
"「姑娘,这东西值三百两银子,不是您一枚铜钱就能当的。」",
|
||
"「三百年?我说小子,你这铺子也就立了三百年,可我手里这枚铜钱——」",
|
||
"「它传了三千年。」",
|
||
"三天后",
|
||
)
|
||
)
|
||
assert env["正文"].读取正文(env["作者"], "ch-2") == saved
|
||
|
||
rescanned = 诊断文本(
|
||
candidate_text,
|
||
work_id="gen-work",
|
||
规则库=_规则副本(env, diagnosis["report"]),
|
||
预览候选=True,
|
||
保护表达=tuple(fixture["protected"]),
|
||
)
|
||
assert rescanned["findings"] == []
|
||
|
||
session, same_candidate, publication, data, experiment = 比较测试._从既有返修会话准备实验(
|
||
env, "tc-98ee99c7cf24", session
|
||
)
|
||
assert same_candidate == candidate_id
|
||
svc = env["eval_app"].要求评测()
|
||
loaded_data = svc.读取数据集(env["eval_actor"], data["version_id"])
|
||
assert loaded_data["public_manifest"]["schema_version"] == "fixed-revision-pair-dataset-v2"
|
||
with pytest.raises(比较测试.Muse错误, match="没有生成输入"):
|
||
svc.读取生成输入(
|
||
env["eval_actor"], experiment["experiment_id"], experiment["sample_ids"][0]
|
||
)
|
||
assert data["source_hash"]
|
||
duplicate = 比较测试.评测服务(env["pools"][比较测试.用途.维护]).发布返修比较数据集(
|
||
env["maint_actor"], publication, env["库"], env["装配"].任务运行
|
||
)
|
||
assert duplicate["duplicate"] and duplicate["source_hash"] == data["source_hash"]
|
||
read = _完成并读取证明(env, experiment)
|
||
source = read["proof"]["source"]
|
||
assert read["proof"]["version"] == "revision-comparison-proof-v2"
|
||
assert source["version"] == "revision-comparison-source-v2"
|
||
assert "diagnosis" not in source
|
||
assert source["selection_authorization"] == {
|
||
"authorization_hash": 固定哈希(asdict(authorization)),
|
||
"basis_hash": session["source"]["basis_hash"],
|
||
"mode": "rewrite",
|
||
"purpose_hash": 固定哈希(authorization.purpose),
|
||
"selection_ids": [selection.selection_id for selection in selections],
|
||
"selections_hash": 固定哈希(asdict(authorization)["selections"]),
|
||
"protected_range_ids": [selection.selection_id for selection in protected],
|
||
"protected_ranges_hash": 固定哈希(asdict(authorization)["protected_ranges"]),
|
||
}
|
||
assert source["original"]["text_hash"] == fixture["text_sha256"]
|
||
assert source["candidate"]["text_hash"] == hashlib.sha256(candidate_text.encode()).hexdigest()
|
||
assert source["target"] == {
|
||
"work_id": "gen-work",
|
||
"chapter_id": "ch-2",
|
||
"branch_id": "main",
|
||
"original_revision": saved["revision"],
|
||
"original_document_hash": saved["document_hash"],
|
||
}
|
||
assert source["delivery"]["task_id"] == task_id
|
||
assert source["delivery"]["config_id"] == "gen-config"
|
||
assert source["delivery"]["resource_release"] == env["装配"].配置.资源发布身份
|
||
assert source["invariance"]["report_id"] == result["report"]["report_id"]
|
||
assert read["proof"]["outcome"] == "candidate"
|
||
assert read["proof"]["comparison_complete"] and len(read["proof"]["decisions"]) == 2
|
||
assert read["current"]["selection"] == "candidate"
|
||
assert not read["current"]["automatic_adoption"]
|
||
calls = len(env["received_judges"])
|
||
replay = (
|
||
env["eval_app"]
|
||
.要求评测()
|
||
.导出返修比较证明(
|
||
env["eval_actor"],
|
||
experiment["experiment_id"],
|
||
env["pools"][比较测试.用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
)
|
||
assert replay["receipt_id"] == read["receipt_id"]
|
||
assert len(env["received_judges"]) == calls
|
||
assert (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修比较证明(env["作者"], env["装配"].任务运行, read["receipt_id"])
|
||
== read
|
||
)
|
||
assert len(env["received_judges"]) == calls
|
||
assert env["正文"].读取正文(env["作者"], "ch-2") == saved
|