421 lines
22 KiB
Python
421 lines
22 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""合同负路测试 + U0 端到端回放。
|
||
|
||
覆盖专题-09 §12 一级验收中可自动化的条目:
|
||
- 验收 4:诊断产物头缺项时修订拒绝执行
|
||
- 验收 6:事实快照缺失时 Patch 自动降为 Audit(无显式授权不得继续)
|
||
- 验收 7:成对选择模型与改写模型相同时视为阶段未执行
|
||
其余为合同负路(无发现禁改、唯一匹配、样例不齐拒绝激活、结构校验、事实增量、口癖保护)。
|
||
"""
|
||
import sys
|
||
import unittest
|
||
from pathlib import Path
|
||
|
||
# 内网环境不 pip install:直接把 src/ 加入导入路径
|
||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))
|
||
|
||
from deai import cards, diagnose, gates, load, pairwise, patch, report
|
||
from deai.pipeline import DowngradedToAudit, resolve_mode, run_audit, run_patch
|
||
|
||
|
||
class TestAssetCompleteness(unittest.TestCase):
|
||
"""规则库/样例库装载:active 规则必须配齐四类样例(专题-09 §4.1)。"""
|
||
|
||
def test_all_shipped_rules_load_with_complete_samples(self):
|
||
samples = load.load_samples()
|
||
rules = load.load_rules(samples=samples)
|
||
actives = load.active_rules(rules)
|
||
# 一级目标:至少 10 条 active 规则,每条四类样例齐全(装载器已强制)
|
||
self.assertGreaterEqual(len(actives), 10)
|
||
for rule in actives:
|
||
for stype in load.SAMPLE_TYPES:
|
||
self.assertTrue(rule["samples"][stype], f"{rule['id']} 缺 {stype}")
|
||
for sample_id in rule["samples"][stype]:
|
||
self.assertIn(rule["id"], samples[sample_id].get("rules", []))
|
||
|
||
def test_rule_without_four_sample_types_rejected(self):
|
||
samples = load.load_samples()
|
||
rules = load.load_rules(samples=samples)
|
||
broken = dict(rules["l002"])
|
||
broken["samples"] = dict(broken["samples"])
|
||
broken["samples"]["regression"] = [] # 缺回归陷阱样例
|
||
with self.assertRaises(load.LoadError):
|
||
load.check_activation(broken, samples)
|
||
|
||
def test_rule_referencing_missing_sample_rejected(self):
|
||
samples = load.load_samples()
|
||
rules = load.load_rules(samples=samples)
|
||
broken = dict(rules["l002"])
|
||
broken["samples"] = dict(broken["samples"])
|
||
broken["samples"]["regression"] = ["不存在的样例id"]
|
||
with self.assertRaises(load.LoadError):
|
||
load.check_activation(broken, samples)
|
||
|
||
|
||
class TestCaseCards(unittest.TestCase):
|
||
"""反向回填/创作反馈的案例卡:快采集,慢确认,禁止越过资产门。"""
|
||
|
||
def _capture(self, license_name="owned", label="sf", mode="backfill"):
|
||
return cards.capture_case_card(
|
||
card_id=f"card-{license_name}-{label}-{mode}",
|
||
label=label,
|
||
layer="lexical",
|
||
carrier="narration",
|
||
source_kind="existing_work" if mode == "backfill" else "creation_feedback",
|
||
source_license=license_name,
|
||
source_text="来源全文:值得注意的是,门外下雨了。",
|
||
excerpt="值得注意的是,门外下雨了。",
|
||
context="她抬头。值得注意的是,门外下雨了。",
|
||
location="chapter-1:paragraph-2",
|
||
pattern="无功能元话语",
|
||
rationale="待复核的表面模式观察",
|
||
function_check=["是否承担转折"],
|
||
risk_if_changed="可能误删节奏或信息",
|
||
capture_mode=mode,
|
||
)
|
||
|
||
def test_shipped_case_card_fixture_uses_current_schema(self):
|
||
loaded = cards.load_case_cards()
|
||
self.assertIn("card-backfill-l002-001", loaded)
|
||
self.assertEqual(loaded["card-backfill-l002-001"]["schema_version"], "ai-flavor-case-v1")
|
||
|
||
def test_backfill_starts_as_shadow_and_confirmation_is_explicit(self):
|
||
shadow = self._capture()
|
||
self.assertEqual(shadow["state"], "shadow")
|
||
with self.assertRaises(cards.CardError):
|
||
cards.promote_card_to_sample(shadow)
|
||
canonical = cards.confirm_case_card(
|
||
shadow, label="sf", review_note="作者确认:此处无功能"
|
||
)
|
||
sample = cards.promote_card_to_sample(canonical)
|
||
self.assertEqual(sample["case_card_id"], canonical["id"])
|
||
self.assertEqual(sample["type"], "sf")
|
||
|
||
def test_unlicensed_source_is_hash_only_and_cannot_be_confirmed(self):
|
||
card = self._capture(license_name="unauthorized")
|
||
self.assertEqual(card["excerpt"], "")
|
||
self.assertEqual(card["context"], "")
|
||
self.assertIn("excerpt_hash", card["source"])
|
||
with self.assertRaises(cards.CardError):
|
||
cards.confirm_case_card(card, label="sf", review_note="不应确认")
|
||
|
||
def test_hash_only_card_rejects_text_injection(self):
|
||
card = self._capture(license_name="research_only")
|
||
broken = dict(card)
|
||
broken["excerpt"] = "偷偷保存的第三方原文"
|
||
with self.assertRaises(cards.CardError):
|
||
cards.validate_card(broken)
|
||
|
||
def test_rule_candidate_is_always_candidate_and_needs_opposing_evidence(self):
|
||
sf = self._capture(label="sf")
|
||
snf = self._capture(label="snf")
|
||
candidate = cards.propose_rule_from_cards(
|
||
[sf, snf],
|
||
rule_id="candidate-l999",
|
||
name="待评估元话语",
|
||
layer="lexical",
|
||
carrier_scope="narration",
|
||
trigger={"type": "model_judgment", "criteria": "上下文功能判断"},
|
||
fix_hint="先仲裁再决定",
|
||
function_check=["是否承担叙事功能"],
|
||
)
|
||
self.assertEqual(candidate["status"], "candidate")
|
||
self.assertEqual(candidate["default_disposition"], "candidate")
|
||
# 候选规则可以先引用 shadow 卡,供后续人工确认;不得因此变 active。
|
||
load.check_case_card_refs(candidate, {sf["id"]: sf, snf["id"]: snf})
|
||
active = dict(candidate)
|
||
active["status"] = "active"
|
||
with self.assertRaises(load.LoadError):
|
||
load.check_case_card_refs(active, {sf["id"]: sf, snf["id"]: snf})
|
||
with self.assertRaises(cards.CardError):
|
||
cards.propose_rule_from_cards(
|
||
[sf],
|
||
rule_id="candidate-l998",
|
||
name="单样本禁令",
|
||
layer="lexical",
|
||
carrier_scope="narration",
|
||
trigger={"type": "model_judgment", "criteria": "读感"},
|
||
fix_hint="删除",
|
||
function_check=["是否有功能"],
|
||
)
|
||
|
||
def test_live_feedback_requires_feedback_source(self):
|
||
card = self._capture(mode="live_feedback")
|
||
self.assertEqual(card["source"]["kind"], "creation_feedback")
|
||
broken = dict(card)
|
||
broken["source"] = dict(card["source"])
|
||
broken["source"]["kind"] = "existing_work"
|
||
with self.assertRaises(cards.CardError):
|
||
cards.validate_card(broken)
|
||
|
||
|
||
class TestNegativePaths(unittest.TestCase):
|
||
"""一级验收 4/6/7 的负路(必须可测、必须红)。"""
|
||
|
||
def test_artifact_header_missing_blocks_patch(self):
|
||
# 验收 4:产物头缺项 → 视为诊断未发生
|
||
artifact = {"text_hash": "sha256:x", "findings": []} # 缺 mode/规则库版本
|
||
with self.assertRaises(diagnose.ArtifactIncomplete):
|
||
diagnose.validate_artifact(artifact)
|
||
|
||
def test_snapshot_missing_downgrades_to_audit(self):
|
||
# 验收 6:快照缺失且无授权 → Patch 强制降为 Audit
|
||
with self.assertRaises(DowngradedToAudit):
|
||
resolve_mode({"mode": "Patch", "author_approved_revision": True, "allowed_scope": "全文"}, fact_snapshot=None)
|
||
|
||
def test_snapshot_missing_with_explicit_authorization_proceeds(self):
|
||
mode = resolve_mode(
|
||
{"mode": "Patch", "author_approved_revision": True, "allowed_scope": "全文", "no_snapshot_authorization": True}, fact_snapshot=None)
|
||
self.assertEqual(mode, "Patch")
|
||
|
||
def test_same_model_pairwise_treated_as_not_executed(self):
|
||
# 验收 7:选择模型 == 改写模型 → 阶段 10 视为未执行
|
||
record = {"choice": "candidate", "rationale": "理由", "selection_model": "claude"}
|
||
with self.assertRaises(pairwise.PairwiseNotExecuted):
|
||
pairwise.validate_choice(record, rewrite_model="claude")
|
||
|
||
def test_pairwise_without_rationale_rejected(self):
|
||
record = {"choice": "candidate", "rationale": "", "selection_model": "gpt-5.6-sol"}
|
||
with self.assertRaises(pairwise.PairwiseNotExecuted):
|
||
pairwise.validate_choice(record, rewrite_model="claude")
|
||
|
||
|
||
class TestPatchContracts(unittest.TestCase):
|
||
def _finding(self, span):
|
||
return {
|
||
"id": "f1", "text_hash": "sha256:0000000000000000",
|
||
"rule_id": "l002", "rule_version": 1, "spans": [span],
|
||
"context_window": span, "layer": "lexical", "evidence": "测试",
|
||
"possible_function": "none", "confidence": "high",
|
||
"decision_proposal": "repair",
|
||
}
|
||
|
||
def _patch(self, finding_id="f1", original="值得注意的是,", replacement="", action="delete"):
|
||
return {"finding_id": finding_id, "action": action,
|
||
"original_exact": original, "replacement": replacement,
|
||
"rationale": "测试删除", "protected_invariants": []}
|
||
|
||
def test_patch_without_finding_rejected(self):
|
||
with self.assertRaises(patch.PatchError):
|
||
patch.apply_patches("值得注意的是,他来了。", [self._patch()], {})
|
||
|
||
def test_patch_nonunique_match_rejected(self):
|
||
text = "值得注意的是,值得注意的是,他来了。"
|
||
findings = {"f1": self._finding("值得注意的是")}
|
||
with self.assertRaises(patch.PatchError):
|
||
patch.apply_patches(text, [self._patch()], findings)
|
||
|
||
def test_patch_not_covering_span_rejected(self):
|
||
# original_exact 与发现 span 无关 → 禁止(防止借诊断之名改别处)
|
||
findings = {"f1": self._finding("他来了")}
|
||
with self.assertRaises(patch.PatchError):
|
||
patch.apply_patches("值得注意的是,他来了。", [self._patch()], findings)
|
||
|
||
def test_skip_patch_cannot_smuggle_replacement(self):
|
||
finding = self._finding("值得注意的是")
|
||
malicious = self._patch(action="skip", replacement="他突然笑了")
|
||
with self.assertRaises(patch.PatchError):
|
||
patch.apply_patches("值得注意的是,他来了。", [malicious], {"f1": finding})
|
||
|
||
def test_patch_does_not_normalize_unrelated_blank_lines(self):
|
||
finding = self._finding("值得注意的是")
|
||
text = "前文。\n\n\n值得注意的是,他来了。"
|
||
candidate, _ = patch.apply_patches(
|
||
text, [self._patch(original="值得注意的是,", replacement="")], {"f1": finding}
|
||
)
|
||
self.assertTrue(candidate.startswith("前文。\n\n\n"))
|
||
|
||
def test_adjacent_punctuation_extension_allowed(self):
|
||
# U0 修正 #6:span 允许紧邻标点的最小扩展(避免删除后标点粘连)
|
||
text = "他打量她,嘴角微微上扬。「姑娘」"
|
||
finding = self._finding("嘴角微微上扬")
|
||
p = self._patch(original=",嘴角微微上扬", replacement="")
|
||
candidate, _ = patch.apply_patches(text, [p], {"f1": finding})
|
||
self.assertEqual(candidate, "他打量她。「姑娘」")
|
||
|
||
|
||
class TestHardGate(unittest.TestCase):
|
||
def test_structural_verify_detects_outside_change(self):
|
||
original = "她关上门。值得注意的是,天黑了。"
|
||
patches = [{"finding_id": "f1", "action": "delete",
|
||
"original_exact": "值得注意的是,", "replacement": "",
|
||
"rationale": "测试", "protected_invariants": []}]
|
||
findings = {"f1": {"id": "f1", "spans": ["值得注意的是,"]}}
|
||
# 候选稿在 patch 之外还被偷偷改了一处 → 结构校验必须红
|
||
tampered = "她关上门。天黑了。另外多出一句。"
|
||
result = gates.structural_verify(original, tampered, patches, findings)
|
||
self.assertFalse(result["pass"])
|
||
|
||
def test_fact_delta_blocks_new_number(self):
|
||
patches = [{"finding_id": "f1", "action": "local_rewrite",
|
||
"original_exact": "值三百两银子", "replacement": "值三百五十两银子",
|
||
"rationale": "测试", "protected_invariants": []}]
|
||
result = gates.fact_delta(patches)
|
||
self.assertFalse(result["pass"])
|
||
|
||
def test_number_attribution_allows_approved_deletion(self):
|
||
# 数字随批准删除消失是合法的(U0 修正 #1:不做计数守恒)
|
||
original = "值三百两银子。值得一提的是,三天后开门。"
|
||
candidate = "值三百两银子。开门。"
|
||
patches = [{"finding_id": "f1", "action": "delete",
|
||
"original_exact": "值得一提的是,三天后", "replacement": "",
|
||
"rationale": "测试", "protected_invariants": []}]
|
||
result = gates.number_attribution(original, candidate, patches)
|
||
self.assertTrue(result["pass"], result["failures"])
|
||
|
||
def test_number_attribution_blocks_unexplained_disappearance(self):
|
||
original = "值三百两银子。"
|
||
candidate = "值银子。" # 数字凭空消失,无对应 patch
|
||
result = gates.number_attribution(original, candidate, [])
|
||
self.assertFalse(result["pass"])
|
||
|
||
def test_quote_tamper_fails(self):
|
||
# 引文/场内文本被篡改(未在任何批准删除片段内)→ 必须红
|
||
result = gates.quote_attribution(
|
||
"面板显示:【等级:待定】。", "面板显示:【等级:最高】。", [])
|
||
self.assertFalse(result["pass"])
|
||
|
||
def test_quote_duplicate_loss_cannot_hide_behind_one_deleted_quote(self):
|
||
patches = [{"finding_id": "f1", "action": "delete",
|
||
"original_exact": "前文【同一条】", "replacement": "",
|
||
"rationale": "测试", "protected_invariants": []}]
|
||
result = gates.quote_attribution(
|
||
"前文【同一条】。后文【同一条】。",
|
||
"前文。后文。",
|
||
patches,
|
||
)
|
||
self.assertFalse(result["pass"])
|
||
|
||
def test_quote_approved_deletion_passes(self):
|
||
# 引文整段位于批准删除片段内 → 合法消失
|
||
patches = [{"finding_id": "f1", "action": "delete",
|
||
"original_exact": "面板显示:【等级:待定】。", "replacement": "",
|
||
"rationale": "测试", "protected_invariants": []}]
|
||
result = gates.quote_attribution(
|
||
"前文。面板显示:【等级:待定】。后文。", "前文。后文。", patches)
|
||
self.assertTrue(result["pass"], result["failures"])
|
||
|
||
def test_protected_tic_deletion_fails(self):
|
||
voice_thin = {"untouchable_verbal_tics": {"老周": ["我说小子"]}}
|
||
result = gates.protected_checks(
|
||
"「我说小子,你来了。」", "「你来了。」", voice_thin)
|
||
self.assertFalse(result["pass"])
|
||
|
||
|
||
class TestU0Replay(unittest.TestCase):
|
||
"""端到端回放 U0 演练:真实规则库 + 真实目标文本 + 8 条 patch + 跨模型选择记录。"""
|
||
|
||
TEXT = """林晚儿站在万宝阁的柜台前,手指无意识地摩挲着那枚铜钱的边缘。
|
||
|
||
**那伙计上下打量了她一眼,**嘴角微微上扬。「姑娘,这东西值三百两银子,不是您一枚铜钱就能当的。」
|
||
|
||
值得注意的是,万宝阁在青云城立了三百年,从没人敢在这里讨价还价。研究表明,能进这种地方的修士,多半背景不凡。伙计眼中闪过一丝精光,显然已经把林晚儿的来历掂量了一遍。
|
||
|
||
「三百年?我说小子,你这铺子也就立了三百年,可我手里这枚铜钱——」老周把铜钱往柜台上一推,声音忽然压低,「它传了三千年。」
|
||
|
||
[占位:此处揭示铜钱来历]
|
||
|
||
万宝阁的大堂里安静了一瞬。系统面板的流光在半空停住,鉴定结果的金字明灭不定:
|
||
|
||
【物品:无法识别|等级:待定|建议操作:上报上峰】
|
||
|
||
无论是伙计的讥笑,还是看客的沉默,似乎都在这一刻定格了。三天后,这件事被摆进了执法堂的晨会。那时谁也不知道,这枚铜钱会掀翻整座青云城。也许这就是命运吧,总在人不经意的时候,悄悄安排好了一切。"""
|
||
|
||
PATCH_SPANS = [
|
||
# (original_exact, replacement, 对应发现 span 的识别片段)
|
||
("**那伙计上下打量了她一眼,**", "那伙计上下打量了她一眼,", "**那伙计"),
|
||
(",嘴角微微上扬", "", "嘴角微微上扬"),
|
||
("眼中闪过一丝精光,", "", "眼中闪过一丝精光"),
|
||
("研究表明,", "", "研究表明"),
|
||
("值得注意的是,", "", "值得注意的是"),
|
||
("无论是伙计的讥笑,还是看客的沉默,似乎都在这一刻定格了。", "", "无论是伙计的讥笑"),
|
||
("也许这就是命运吧,总在人不经意的时候,悄悄安排好了一切。", "", "也许这就是命运吧"),
|
||
("[占位:此处揭示铜钱来历]", "", "[占位:此处揭示铜钱来历]"),
|
||
]
|
||
|
||
def _build_artifact(self, rules):
|
||
artifact = run_audit(self.TEXT, rules, load.rule_library_version(rules))
|
||
artifact["mode"] = "Patch"
|
||
# 语义层发现由外部判定注入(U0 阶段 4 人工判定的代码化形态)
|
||
sem_finding = {
|
||
"id": "f-sem", "text_hash": artifact["text_hash"],
|
||
"rule_id": "sem001", "rule_version": 1,
|
||
"spans": ["也许这就是命运吧,总在人不经意的时候,悄悄安排好了一切。"],
|
||
"context_window": "那时谁也不知道…悄悄安排好了一切。",
|
||
"layer": "semantic",
|
||
"evidence": "段尾脱离场景的命运评注,钩子由前句承担",
|
||
"possible_function": "none", "confidence": "high",
|
||
"decision_proposal": "repair",
|
||
"arbitration_note": "五问皆否;前句为伏笔钩子,升华句稀释钩子",
|
||
}
|
||
diagnose.merge_model_findings(artifact, [sem_finding])
|
||
# 模拟仲裁完成:全部 repair(U0 阶段 4 结论)
|
||
for f in artifact["findings"]:
|
||
f["decision_proposal"] = "repair"
|
||
f["possible_function"] = f.get("possible_function", "none")
|
||
f.setdefault("arbitration_note", "演练回放:五问仲裁通过")
|
||
return artifact
|
||
|
||
def test_full_pipeline_replay(self):
|
||
samples = load.load_samples()
|
||
rules = load.load_rules(samples=samples)
|
||
artifact = self._build_artifact(rules)
|
||
diagnose.validate_artifact(artifact)
|
||
|
||
# patch 的 finding_id 按 span 覆盖关系解析(回放时不硬编码顺序)
|
||
patches = []
|
||
for original, replacement, marker in self.PATCH_SPANS:
|
||
fid = next(f["id"] for f in artifact["findings"] if marker in f["spans"][0]
|
||
or f["spans"][0] in original)
|
||
patches.append({"finding_id": fid, "action": "delete" if not replacement else "patch",
|
||
"original_exact": original, "replacement": replacement,
|
||
"rationale": "U0 回放", "protected_invariants": []})
|
||
|
||
result = run_patch(
|
||
self.TEXT, rules, artifact, patches,
|
||
task_contract={"mode": "Patch", "allowed_scope": "全文", "author_approved_revision": True},
|
||
fact_snapshot={"entities": ["林晚儿", "老周", "伙计", "万宝阁", "青云城", "执法堂"]},
|
||
voice_thin={"untouchable_verbal_tics": {"老周": ["我说小子"]},
|
||
"protected_spans": ["【物品:无法识别|等级:待定|建议操作:上报上峰】"]},
|
||
rewrite_model="claude",
|
||
pairwise_record={"choice": "candidate",
|
||
"rationale": "保真完整,钩子有力(U0 gpt-5.6-sol 判定)",
|
||
"selection_model": "gpt-5.6-sol"},
|
||
rule_library_version=load.rule_library_version(rules))
|
||
|
||
# 硬门全绿(结构/事实增量/数字归因/保护项)
|
||
self.assertTrue(result["hard_gate"]["pass"], result["hard_gate"])
|
||
# 复扫零残留
|
||
self.assertTrue(result["regression_gate"]["pass"],
|
||
result["regression_gate"]["residual_hits"])
|
||
# 跨模型选择成立
|
||
self.assertEqual(result["pairwise_choice"]["choice"], "candidate")
|
||
# 候选稿不再含任何确定性命中点
|
||
for marker in ("**", "值得注意的是", "研究表明", "嘴角微微上扬", "[占位", "也许这就是命运"):
|
||
self.assertNotIn(marker, result["candidate_text"])
|
||
# 保护项仍在
|
||
self.assertIn("我说小子", result["candidate_text"])
|
||
self.assertIn("【物品:无法识别|等级:待定|建议操作:上报上峰】", result["candidate_text"])
|
||
|
||
# 审计报告可组装且过合同(含禁止总分检查)
|
||
rep = report.assemble(
|
||
artifact, patches, result["candidate_text"], result["hard_gate"],
|
||
result["voice_gate"], result["regression_gate"], result["pairwise_choice"],
|
||
unresolved_risks=result["hard_gate"]["unverified"])
|
||
self.assertIn("pairwise_choice", rep["final"])
|
||
|
||
def test_report_rejects_score_fields(self):
|
||
samples = load.load_samples()
|
||
rules = load.load_rules(samples=samples)
|
||
artifact = run_audit(self.TEXT, rules, load.rule_library_version(rules))
|
||
with self.assertRaises(report.ForbiddenScoreError):
|
||
report.assemble(
|
||
artifact, [], None,
|
||
{"pass": True, "checks": {}, "unverified": [], "human_score": 0.9},
|
||
{"status": "unverified", "pass": None, "note": ""},
|
||
{"pass": True, "residual_hits": []}, None, [])
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|