# -*- coding: utf-8 -*- """U4 评测运行器(专题-09 §9 / §12 一级验收的自动化部分)。 两个探针: 1. 规则判别试跑:每条 regex 规则在自己的 SF 样例上必须命中(召回自查); 在 SNF 样例上的命中是**预期内的表面碰撞**——SNF 的定义就是「相同表面形式 但承担功能」,判别力在仲裁/carve_out(模型与人),不在触发器。 model_judgment 规则无法自动跑,标「待模型判定」。 2. 回归陷阱门禁探针:对每条 regression 样例的「模拟坏改写」跑机械化探针—— 无授权重写探针(结构校验,patches=[])+ 不变量探针(数字/引文/口癖归因)。 分类输出:机械可拦 / 语义级(门禁 unverified,需仲裁与盲评兜底)。 诚实约束:样例量是冷启动级别(n 小),本报告只出试跑数据与工具验证, 不出「规则有效」结论(专题-09 §9.3:n=1 不下结论)。 """ import re import sys from pathlib import Path ROOT = Path(__file__).resolve().parent.parent sys.path.insert(0, str(ROOT / "src")) from deai import evaluation, gates, load # noqa: E402 def parse_regression_sample(text: str): """回归样例题干格式:原文:X / 模拟坏改写:Y。抽象描述类样例返回 None。""" m_orig = re.search(r"原文:(.*?)(?:\n|$)", text) m_bad = re.search(r"模拟坏改写:(.*?)(?:\n|$)", text, re.DOTALL) if not m_orig or not m_bad: return None return m_orig.group(1).strip(), m_bad.group(1).strip() def probe_rule_discrimination(rules: dict, samples: dict) -> list: rows = [] for rule in sorted(rules.values(), key=lambda r: r["id"]): if rule["trigger"]["type"] == "model_judgment": rows.append({"rule": rule["id"], "kind": "model_judgment", "sf_hit": "-", "snf_surface": "-", "note": "待模型判定(一级人工/二级接入)"}) continue report = evaluation.evaluate_rule_contract(rule, samples) sf_rows = [row for row in report["rows"] if row["kind"] == "sf"] snf_rows = [row for row in report["rows"] if row["kind"] == "snf"] sf_hit = sum(1 for row in sf_rows if row["surface_hit"]) snf_surf = sum(1 for row in snf_rows if row["surface_hit"]) rows.append({ "rule": rule["id"], "kind": rule["trigger"]["type"], "sf_hit": f"{sf_hit}/{len(sf_rows)}", "snf_surface": f"{snf_surf}/{len(snf_rows)}", "note": "SF 召回正常" if sf_hit == len(sf_rows) else "SF 召回有缺口,规则或样例需修", }) return rows def probe_regression_traps(samples: dict) -> list: rows = [] voice_tics = {"老周": ["我说小子"]} for sid, s in sorted(samples.items()): if s["type"] != "regression": continue parsed = parse_regression_sample(s["text"]) if parsed is None: rows.append({"sample": sid, "mechanical": "—", "verdict": "抽象描述类,需人工评审"}) continue original, bad = parsed caught = [] # 探针 1:无授权重写——任何不经批准 patch 的整体改写都违反结构校验 if gates.structural_verify(original, bad, [], {})["pass"] is False and bad != original: caught.append("结构校验(无授权重写)") # 探针 2:数字归因(无 patch 可归因 → 任何数字增减都红) if gates.number_attribution(original, bad, [])["pass"] is False: caught.append("数字归因") # 探针 3:引文归因 if gates.quote_attribution(original, bad, [])["pass"] is False: caught.append("引文归因") # 探针 4:口癖保护(样例涉及老周口癖时) if "我说小子" in original and gates.protected_checks(original, bad, {"untouchable_verbal_tics": voice_tics})["pass"] is False: caught.append("口癖保护") rows.append({ "sample": sid, "mechanical": "、".join(caught) if caught else "未拦", "verdict": "机械可拦" if caught else "语义级:门禁 unverified,需仲裁/盲评兜底", }) return rows # patch 形态伤害探针:把伤害放进「看似合法的 patch replacement」里, # 验证硬门在结构校验必然通过的情况下还能不能拦到实质伤害。 # expected: caught = 机械应拦;semantic = 已知机械拦不住(门禁 unverified,仲裁/盲评兜底) PATCH_FORM_PROBES = [ dict(sample="reg-004", desc="数字篡改(三百两→三百五十两)", original_exact="值三百两银子", replacement="值三百五十两银子", expected="caught"), dict(sample="reg-001", desc="新增日期(秋分的夜里)", original_exact="那时谁也不知道", replacement="那年秋分的夜里,谁也不知道", expected="caught"), dict(sample="reg-l001-01", desc="模态篡改(多半→都)", original_exact="研究表明,能进这种地方的修士,多半背景不凡。", replacement="能进这种地方的修士,背景都不凡。", expected="caught"), dict(sample="reg-m002-01", desc="replacement 续写新内容", original_exact="抱歉,我无法继续这个故事。", replacement="她裹紧衣领,走进了巷子深处。", expected="semantic"), ] def probe_patch_form_damage() -> list: rows = [] for probe in PATCH_FORM_PROBES: patches = [{"finding_id": "f1", "action": "local_rewrite", "original_exact": probe["original_exact"], "replacement": probe["replacement"], "rationale": "探针", "protected_invariants": []}] caught = [] if gates.fact_delta(patches)["pass"] is False: caught.append("事实增量") # 数字归因在单 patch 场景等价于 fact_delta 的数字部分,补充跑一遍防漏 original = "前文。" + probe["original_exact"] + "后文。" candidate = "前文。" + probe["replacement"] + "后文。" if gates.number_attribution(original, candidate, patches)["pass"] is False: caught.append("数字归因") if gates.quote_attribution(original, candidate, patches)["pass"] is False: caught.append("引文归因") if gates.temporal_fact_delta(patches)["pass"] is False: caught.append("时间锚点") if gates.modality_attribution(patches)["pass"] is False: caught.append("模态守恒") got = "caught" if caught else "semantic" rows.append({ "sample": probe["sample"], "desc": probe["desc"], "caught": "、".join(caught) if caught else "未拦(语义级)", "match_expectation": got == probe["expected"], }) return rows def main(): samples = load.load_samples() rules = load.load_rules(samples=samples) print("== 规则判别试跑 ==") for row in probe_rule_discrimination(rules, samples): print(f"{row['rule']:<8} {row['kind']:<15} SF命中 {row['sf_hit']:<6} " f"SNF表面碰撞 {row['snf_surface']:<6} {row['note']}") print() print("== 回归陷阱门禁探针(无授权重写形态) ==") for row in probe_regression_traps(samples): print(f"{row['sample']:<14} {row['verdict']:<28} 探针: {row['mechanical']}") print() print("== 回归陷阱门禁探针(patch 内伤害形态) ==") for row in probe_patch_form_damage(): flag = "符合预期" if row["match_expectation"] else "!! 与预期不符" print(f"{row['sample']:<14} {row['desc']:<26} {row['caught']:<22} {flag}") if __name__ == "__main__": main()