157 lines
7.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# -*- coding: utf-8 -*-
"""U4 评测运行器(专题-09 §9 / §12 一级验收的自动化部分)。
两个探针:
1. 规则判别试跑:每条 regex 规则在自己的 SF 样例上必须命中(召回自查);
在 SNF 样例上的命中是**预期内的表面碰撞**——SNF 的定义就是「相同表面形式
但承担功能」,判别力在仲裁/carve_out(模型与人),不在触发器。
model_judgment 规则无法自动跑,标「待模型判定」。
2. 回归陷阱门禁探针:对每条 regression 样例的「模拟坏改写」跑机械化探针——
无授权重写探针(结构校验,patches=[])+ 不变量探针(数字/引文/口癖归因)。
分类输出:机械可拦 / 语义级(门禁 unverified,需仲裁与盲评兜底)。
诚实约束:样例量是冷启动级别(n 小),本报告只出试跑数据与工具验证,
不出「规则有效」结论(专题-09 §9.3:n=1 不下结论)。
"""
import re
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(ROOT / "src"))
from deai import evaluation, gates, load # noqa: E402
def parse_regression_sample(text: str):
"""回归样例题干格式:原文:X / 模拟坏改写:Y。抽象描述类样例返回 None。"""
m_orig = re.search(r"原文:(.*?)(?:\n|$)", text)
m_bad = re.search(r"模拟坏改写:(.*?)(?:\n|$)", text, re.DOTALL)
if not m_orig or not m_bad:
return None
return m_orig.group(1).strip(), m_bad.group(1).strip()
def probe_rule_discrimination(rules: dict, samples: dict) -> list:
rows = []
for rule in sorted(rules.values(), key=lambda r: r["id"]):
if rule["trigger"]["type"] == "model_judgment":
rows.append({"rule": rule["id"], "kind": "model_judgment",
"sf_hit": "-", "snf_surface": "-", "note": "待模型判定(一级人工/二级接入)"})
continue
report = evaluation.evaluate_rule_contract(rule, samples)
sf_rows = [row for row in report["rows"] if row["kind"] == "sf"]
snf_rows = [row for row in report["rows"] if row["kind"] == "snf"]
sf_hit = sum(1 for row in sf_rows if row["surface_hit"])
snf_surf = sum(1 for row in snf_rows if row["surface_hit"])
rows.append({
"rule": rule["id"], "kind": rule["trigger"]["type"],
"sf_hit": f"{sf_hit}/{len(sf_rows)}",
"snf_surface": f"{snf_surf}/{len(snf_rows)}",
"note": "SF 召回正常" if sf_hit == len(sf_rows) else "SF 召回有缺口,规则或样例需修",
})
return rows
def probe_regression_traps(samples: dict) -> list:
rows = []
voice_tics = {"老周": ["我说小子"]}
for sid, s in sorted(samples.items()):
if s["type"] != "regression":
continue
parsed = parse_regression_sample(s["text"])
if parsed is None:
rows.append({"sample": sid, "mechanical": "—", "verdict": "抽象描述类,需人工评审"})
continue
original, bad = parsed
caught = []
# 探针 1:无授权重写——任何不经批准 patch 的整体改写都违反结构校验
if gates.structural_verify(original, bad, [], {})["pass"] is False and bad != original:
caught.append("结构校验(无授权重写)")
# 探针 2:数字归因(无 patch 可归因 → 任何数字增减都红)
if gates.number_attribution(original, bad, [])["pass"] is False:
caught.append("数字归因")
# 探针 3:引文归因
if gates.quote_attribution(original, bad, [])["pass"] is False:
caught.append("引文归因")
# 探针 4:口癖保护(样例涉及老周口癖时)
if "我说小子" in original and gates.protected_checks(original, bad, {"untouchable_verbal_tics": voice_tics})["pass"] is False:
caught.append("口癖保护")
rows.append({
"sample": sid,
"mechanical": "、".join(caught) if caught else "未拦",
"verdict": "机械可拦" if caught else "语义级:门禁 unverified,需仲裁/盲评兜底",
})
return rows
# patch 形态伤害探针:把伤害放进「看似合法的 patch replacement」里,
# 验证硬门在结构校验必然通过的情况下还能不能拦到实质伤害。
# expected: caught = 机械应拦;semantic = 已知机械拦不住(门禁 unverified,仲裁/盲评兜底)
PATCH_FORM_PROBES = [
dict(sample="reg-004", desc="数字篡改(三百两→三百五十两)",
original_exact="值三百两银子", replacement="值三百五十两银子", expected="caught"),
dict(sample="reg-001", desc="新增日期(秋分的夜里)",
original_exact="那时谁也不知道", replacement="那年秋分的夜里,谁也不知道", expected="caught"),
dict(sample="reg-l001-01", desc="模态篡改(多半→都)",
original_exact="研究表明,能进这种地方的修士,多半背景不凡。",
replacement="能进这种地方的修士,背景都不凡。", expected="caught"),
dict(sample="reg-m002-01", desc="replacement 续写新内容",
original_exact="抱歉,我无法继续这个故事。",
replacement="她裹紧衣领,走进了巷子深处。", expected="semantic"),
]
def probe_patch_form_damage() -> list:
rows = []
for probe in PATCH_FORM_PROBES:
patches = [{"finding_id": "f1", "action": "local_rewrite",
"original_exact": probe["original_exact"],
"replacement": probe["replacement"],
"rationale": "探针", "protected_invariants": []}]
caught = []
if gates.fact_delta(patches)["pass"] is False:
caught.append("事实增量")
# 数字归因在单 patch 场景等价于 fact_delta 的数字部分,补充跑一遍防漏
original = "前文。" + probe["original_exact"] + "后文。"
candidate = "前文。" + probe["replacement"] + "后文。"
if gates.number_attribution(original, candidate, patches)["pass"] is False:
caught.append("数字归因")
if gates.quote_attribution(original, candidate, patches)["pass"] is False:
caught.append("引文归因")
if gates.temporal_fact_delta(patches)["pass"] is False:
caught.append("时间锚点")
if gates.modality_attribution(patches)["pass"] is False:
caught.append("模态守恒")
got = "caught" if caught else "semantic"
rows.append({
"sample": probe["sample"], "desc": probe["desc"],
"caught": "、".join(caught) if caught else "未拦(语义级)",
"match_expectation": got == probe["expected"],
})
return rows
def main():
samples = load.load_samples()
rules = load.load_rules(samples=samples)
print("== 规则判别试跑 ==")
for row in probe_rule_discrimination(rules, samples):
print(f"{row['rule']:<8} {row['kind']:<15} SF命中 {row['sf_hit']:<6} "
f"SNF表面碰撞 {row['snf_surface']:<6} {row['note']}")
print()
print("== 回归陷阱门禁探针(无授权重写形态) ==")
for row in probe_regression_traps(samples):
print(f"{row['sample']:<14} {row['verdict']:<28} 探针: {row['mechanical']}")
print()
print("== 回归陷阱门禁探针(patch 内伤害形态) ==")
for row in probe_patch_form_damage():
flag = "符合预期" if row["match_expectation"] else "!! 与预期不符"
print(f"{row['sample']:<14} {row['desc']:<26} {row['caught']:<22} {flag}")
if __name__ == "__main__":
main()