zizi cf0f4fb985 W20–W24:保存知识方法、迁移框架、审校修订与案例行为的集成成果
接续 88dd570,保存 W20–W24 已实现的共享接口、业务入口、迁移、工作台、测试与文档。
W20/W22/W23 保持 in_progress,W21/W24 保持 verified;此提交不宣称方法或规则正式启用、多轮返修、真实角色评测完成。

W25 新增实验、标定、逐调用交付与角色执行及其迁移/测试/索引留在实施工作树,原有私人和旧实现保留项不纳入。

验证:离线 571、前端 33 通过;PG 469 项通过、2 项浏览器未启用,2 项误带入的 W25 用例已移出本提交;最终任务与交付边界 37 项通过。make 检查、最终类型、84 项资源及 diff 检查通过。未重跑浏览器或 Pi 宿主,不以合成调用认证外部模型效果。
独立整体审查四维通过;证据保存在 R2-20260909/提交W20-W24。
2026-09-13 16:24:42 +08:00

176 lines
6.9 KiB
Python

"""显式离线修订回放;只验证输入副本和机械行为,不认证数据库当前状态。"""
from collections import Counter
from dataclasses import asdict
from typing import Literal
from pydantic import ConfigDict, StrictBool
from pydantic.dataclasses import dataclass
from muse.审校修订.受控修订 import 允许选区, 核对诊断选段
from muse.审校修订.机器味规则 import 读取内置规则
from muse.审校修订.机器味诊断 import 诊断文本
from muse.审校修订.模型 import 审校错误
from muse.审校修订.返修策略 import 选择返修
from muse.正式变更.接口 import 固定哈希
from muse.正文写作.接口 import 可见文本, 正文结构哈希, 正文草稿, 选段修改
@dataclass(frozen=True, config=ConfigDict(extra="forbid"))
class 离线事实副本:
payload: dict
content_hash: str
def __post_init__(self):
if 固定哈希(self.payload) != self.content_hash:
raise 审校错误("离线事实副本哈希不符")
@dataclass(frozen=True, config=ConfigDict(extra="forbid"))
class 离线修订请求:
work_id: str
document: 正文草稿
diagnosis: dict
patches: tuple[选段修改, ...]
allowed_ranges: tuple[允许选区, ...]
approved_findings: tuple[str, ...]
purpose: str
author_approved: StrictBool
fact_snapshot: 离线事实副本 | None = None
no_snapshot_authorization: StrictBool = False
author_choice: Literal["original", "candidate", "retry"] | None = None
protected_expressions: tuple[str, ...] = ()
def __post_init__(self):
if not self.work_id or not self.purpose.strip() or not self.author_approved:
raise 审校错误("离线回放需要作品、目的及明确的本地预览授权")
if not self.diagnosis or not self.patches:
raise 审校错误("离线回放需要完整诊断产物和修改,不能无事生成报告")
def 离线修订回放(请求: 离线修订请求):
doc, artifact = 请求.document, 请求.diagnosis
text = 可见文本(doc)
library = 读取内置规则()
if artifact.get("mode") not in {"active", "candidate_preview"}:
raise 审校错误("诊断头缺失或模式无效")
voice = artifact.get("voice")
if voice is not None and not isinstance(voice, dict):
raise 审校错误("诊断声音副本必须为对象或空值")
voice_protection = (voice or {}).get("protected_expressions", [])
if not isinstance(voice_protection, (list, tuple)) or any(
not isinstance(p, str) for p in voice_protection
):
raise 审校错误("诊断保护表达副本必须为字符串列表")
protected = tuple(
dict.fromkeys(
请求.protected_expressions + tuple((voice or {}).get("protected_expressions", []))
)
)
expected = 诊断文本(
text,
work_id=请求.work_id,
规则库=library,
预览候选=artifact["mode"] == "candidate_preview",
保护表达=protected,
)
expected.pop("persisted")
actual = {k: v for k, v in artifact.items() if k not in {"persisted", "target", "voice"}}
if actual != expected:
raise 审校错误("诊断与原文、当前规则副本或确定性发现不一致")
target = artifact.get("target")
if target is not None and not isinstance(target, dict):
raise 审校错误("诊断目标副本必须为对象或空值")
if target and (
target.get("work_id") != 请求.work_id or target.get("document_hash") != 正文结构哈希(doc)
):
raise 审校错误("诊断的文档依据与回放原文不一致")
result = {
"mode": "offline_preview",
"persisted": False,
"model_verified": False,
"production_validity": "not_checked",
"input_hash": 固定哈希(asdict(请求)),
"source_hash": 正文结构哈希(doc),
"diagnosis_hash": 固定哈希(artifact),
"fact_snapshot_hash": 请求.fact_snapshot.content_hash if 请求.fact_snapshot else None,
"fact_snapshot_verification": "local_hash_only" if 请求.fact_snapshot else "missing",
"original_text": text,
"candidate_text": text,
"proposed_document": None,
"hard_gate": None,
"regression_gate": None,
"author_choice": 请求.author_choice,
"summary": {
"work_id": 请求.work_id,
"requested_changes": len(请求.patches),
"author_preview_only": True,
},
}
if 请求.fact_snapshot is None and not 请求.no_snapshot_authorization:
result["final"] = 选择返修(
已用轮数=0, 最大轮数=1, 硬门通过=False, 有修改=False, 有完整依据=False
)
return result
try:
candidate, checks = 核对诊断选段(
doc,
artifact,
请求.patches,
请求.allowed_ranges,
请求.approved_findings,
保护表达=protected,
)
except 审校错误 as exc:
# 只有合法范围上的保真失败形成失败回放;伪发现与越范围不得伪装成执行过。
if not exc.说明.startswith("修订保真失败:"):
raise
result["hard_gate"] = {
"mechanical_pass": False,
"errors": [exc.说明],
"semantic_status": "unverified",
}
result["final"] = 选择返修(已用轮数=1, 最大轮数=1, 硬门通过=False, 有修改=True)
return result
after = 可见文本(candidate)
# 保护表达本身已由硬门检验;复扫用于发现目标规则是否仍有未消除命中。
rescanned = (
诊断文本(
after,
work_id=请求.work_id,
规则库=library,
预览候选=artifact["mode"] == "candidate_preview",
保护表达=protected,
)
if after.strip()
else None
)
before_count = Counter(f["rule_id"] for f in artifact["findings"])
after_count = Counter(f["rule_id"] for f in (rescanned or {}).get("findings", []))
by_id = {f["id"]: f for f in artifact["findings"]}
removed = Counter(
by_id[p.finding_id]["rule_id"] for p in 请求.patches if p.original != p.replacement
)
residual = [r for r, n in removed.items() if after_count[r] > before_count[r] - n]
result.update(
proposed_document=asdict(candidate),
hard_gate=checks,
regression_gate={
"pass": not residual,
"residual_rules": residual,
"semantic_status": "unreviewed",
"report": rescanned,
},
)
final = 选择返修(
已用轮数=1,
最大轮数=1,
硬门通过=checks["mechanical_pass"] and not residual,
有修改=after != text,
作者选择=请求.author_choice,
)
result["final"] = final
if final["action"] == "candidate":
result["candidate_text"] = after
return result