zizi 76d38f2dc7 修复: 派发链硬门失败关闭与测试账本隔离
- 模型锁定完整 ID 等值:role_policy 废弃子串匹配,前置校验+事后
  MODEL_POLICY_VIOLATION 熔断+回执 modelMatch,治理链角色放行 BUDGET_CHAIN
- 工具白名单只读机械强制:任务包 allowlist ⊆ 只读注册表,TOOL_NOT_READONLY
- 本地向量检索 truthful 化:aiContext 裁剪、指针行跳过、资格失败关闭
  (bindingStatus/productionRetrievalEligible 不再伪造 active)
- 角色提示词 name 回归英文系统 ID;planner 补 fine_outline 绑定;
  writer 数据契约对齐 writer-candidate-body-v1
- 测试账本隔离:sqlite_path 三层透传(bridge/two_phase),9 处派发测试
  改用临时库;清除 muse.db 测试残留 5 runs+26 events+reviews 5/6(有备份)
- test-inventory 补 3 条登记并机械重算 summary;评测场景补
  output_contract/fail_closed 与 stability 两类;死代码清理
  (project_paths.py、offline_only 死参数、harness/harness 残骸)
- 新增 metaphysical-diff-review 技能(红线 4.2 载体)并登记,共 59 技能
2026-08-30 23:06:19 +08:00

115 lines
5.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""Skill 行为评测引擎离线自测:裁决逻辑、失败关闭和报告合同。
这些测试只证明评测管道本身正确(harness_self_test);Skill 行为证据必须
由真实模型 adapter 在显式授权下产生,仍为 0。
"""
import copy
import json
import pathlib
import sys
import tempfile
import unittest
PROJECT_ROOT = next(
parent
for parent in (pathlib.Path(__file__).resolve().parent, *pathlib.Path(__file__).resolve().parents)
if (parent / "AGENTS.md").is_file() and (parent / ".git").exists()
)
HARNESS_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "harness"
sys.path.insert(0, str(HARNESS_DIR / "evals"))
sys.path.insert(0, str(HARNESS_DIR / "evals" / "skills" / "diagnose-ai-flavor"))
import skill_eval as se # noqa: E402
import run_eval as diagnose_eval # noqa: E402
SKILL_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "humanization" / "skills" / "diagnose-ai-flavor"
SCENARIO_FILE = diagnose_eval.SCENARIO_FILE
class EvalEngineTest(unittest.TestCase):
def test_compliant_observations_pass_and_report_schema_complete(self):
adapter = se.FakeAdapter(diagnose_eval._FAKE_OBSERVATIONS)
report = se.run_eval(SKILL_DIR, SCENARIO_FILE, adapter, adapter_name="fake")
self.assertEqual(report["schema_version"], se.REPORT_SCHEMA)
self.assertEqual(report["skill"], "diagnose-ai-flavor")
self.assertEqual(report["scenario_count"], 6)
self.assertEqual(report["failed"], 0)
text = (SKILL_DIR / "SKILL.md").read_text(encoding="utf-8")
import hashlib
self.assertEqual(
report["skill_md_sha256"],
"sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest(),
)
categories = {v["category"] for v in report["verdicts"]}
self.assertIn("positive_trigger", categories)
self.assertIn("forbidden_action", categories)
def test_violating_observation_fails_with_evidence(self):
scripted = copy.deepcopy(diagnose_eval._FAKE_OBSERVATIONS)
# 缺 work_ref 场景若 Agent 返回成功退出,必须裁决失败并给出证据
scripted["missing-work-ref"]["exit_code"] = 0
scripted["missing-work-ref"]["output"] = "ok"
report = se.run_eval(SKILL_DIR, SCENARIO_FILE, se.FakeAdapter(scripted), adapter_name="fake")
self.assertEqual(report["failed"], 1)
verdict = next(v for v in report["verdicts"] if v["scenario_id"] == "missing-work-ref")
self.assertEqual(verdict["verdict"], "failed")
checks = {f["check"] for f in verdict["failed_checks"]}
self.assertIn("expected_exit_codes", checks)
self.assertIn("output_must_contain", checks)
def test_forbidden_invocation_is_detected(self):
scripted = copy.deepcopy(diagnose_eval._FAKE_OBSERVATIONS)
scripted["forbidden-modify-text"]["invoked_commands"].append(
".venv/bin/python muse/lifecycle/quality/humanization/skills/revise-ai-flavor/scripts/revise_ai_flavor.py --x"
)
report = se.run_eval(SKILL_DIR, SCENARIO_FILE, se.FakeAdapter(scripted), adapter_name="fake")
verdict = next(v for v in report["verdicts"] if v["scenario_id"] == "forbidden-modify-text")
self.assertEqual(verdict["verdict"], "failed")
self.assertIn("must_not_invoke", {f["check"] for f in verdict["failed_checks"]})
def test_missing_observation_field_is_not_silent(self):
scripted = copy.deepcopy(diagnose_eval._FAKE_OBSERVATIONS)
del scripted["positive-basic-diagnosis"]["exit_code"]
report = se.run_eval(SKILL_DIR, SCENARIO_FILE, se.FakeAdapter(scripted), adapter_name="fake")
verdict = next(v for v in report["verdicts"] if v["scenario_id"] == "positive-basic-diagnosis")
self.assertEqual(verdict["verdict"], "failed")
self.assertIn("observation_missing", {f["check"] for f in verdict["failed_checks"]})
def test_real_adapter_fails_closed_with_stable_code(self):
with self.assertRaises(se.EvalAdapterUnavailable) as ctx:
se.RoleAgentAdapter().run("skill", {"id": "x"})
self.assertEqual(ctx.exception.code, "EVAL_ADAPTER_UNAVAILABLE")
def test_invalid_scenario_structure_is_rejected(self):
data = json.loads(SCENARIO_FILE.read_text(encoding="utf-8"))
data["scenarios"][0]["category"] = "not-a-category"
with tempfile.TemporaryDirectory() as tmp:
bad = pathlib.Path(tmp) / "scenarios.json"
bad.write_text(json.dumps(data), encoding="utf-8")
with self.assertRaisesRegex(se.EvalContractError, "category"):
se.load_scenarios(bad)
def test_cli_default_adapter_is_blocked_and_fake_runs(self):
import contextlib
import io
buffer = io.StringIO()
with contextlib.redirect_stdout(buffer):
code_blocked = diagnose_eval.main([])
self.assertEqual(code_blocked, 2)
blocked = json.loads(buffer.getvalue())
self.assertEqual(blocked["status"], "blocked")
self.assertEqual(blocked["code"], "EVAL_ADAPTER_UNAVAILABLE")
buffer = io.StringIO()
with contextlib.redirect_stdout(buffer):
code_fake = diagnose_eval.main(["--adapter", "fake"])
self.assertEqual(code_fake, 0)
report = json.loads(buffer.getvalue())
self.assertEqual(report["failed"], 0)
if __name__ == "__main__":
unittest.main()