176 lines
7.5 KiB
Python
176 lines
7.5 KiB
Python
#!/usr/bin/env python3
|
||
"""Skill 行为评测引擎:由 harness 从外部驱动,Skill 不能自己宣布通过。
|
||
|
||
引擎把 SKILL.md 文本和场景任务交给 adapter(Agent/模型执行层),接收结构化
|
||
观察,再按场景登记的期望裁决。真实模型 adapter 需要显式授权;未授权时以
|
||
稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭。fake adapter 只用于评测引擎自身
|
||
管道验证,不构成 Skill 行为证据。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import json
|
||
from datetime import datetime, timezone
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
REPORT_SCHEMA = "skill-behavior-eval-report-v1"
|
||
SCENARIO_SCHEMA = "skill-behavior-eval-scenarios-v1"
|
||
CATEGORIES = (
|
||
"positive_trigger",
|
||
"negative_trigger",
|
||
"input_missing_or_out_of_scope",
|
||
"output_contract_and_fail_closed",
|
||
"forbidden_action",
|
||
"stability_and_confounders",
|
||
)
|
||
|
||
|
||
class EvalContractError(ValueError):
|
||
"""评测合同错误:场景结构非法或观察缺项。"""
|
||
|
||
|
||
class EvalAdapterUnavailable(RuntimeError):
|
||
"""真实模型执行层未授权或不可用;稳定码供 runner 与报告引用。"""
|
||
|
||
code = "EVAL_ADAPTER_UNAVAILABLE"
|
||
|
||
|
||
def skill_md_fingerprint(skill_dir: Path) -> str:
|
||
text = (skill_dir / "SKILL.md").read_text(encoding="utf-8")
|
||
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||
|
||
|
||
def load_scenarios(path: Path) -> dict:
|
||
try:
|
||
data = json.loads(path.read_text(encoding="utf-8"))
|
||
except (OSError, json.JSONDecodeError) as exc:
|
||
raise EvalContractError(f"场景文件不可读或不合法: {path} ({exc})") from exc
|
||
if not isinstance(data, dict) or data.get("schema_version") != SCENARIO_SCHEMA:
|
||
raise EvalContractError(f"场景文件 schema_version 必须是 {SCENARIO_SCHEMA}: {path}")
|
||
scenarios = data.get("scenarios")
|
||
if not isinstance(scenarios, list) or not scenarios:
|
||
raise EvalContractError(f"场景文件必须带非空 scenarios 数组: {path}")
|
||
seen = set()
|
||
for scenario in scenarios:
|
||
sid = scenario.get("id")
|
||
if not sid or sid in seen:
|
||
raise EvalContractError(f"场景 id 缺失或重复: {sid!r} ({path})")
|
||
seen.add(sid)
|
||
if scenario.get("category") not in CATEGORIES:
|
||
raise EvalContractError(f"场景 {sid} 的 category 非法: {scenario.get('category')!r}")
|
||
if not str(scenario.get("task") or "").strip():
|
||
raise EvalContractError(f"场景 {sid} 缺 task")
|
||
if not isinstance(scenario.get("expectations"), dict) or not scenario["expectations"]:
|
||
raise EvalContractError(f"场景 {sid} 缺 expectations")
|
||
return data
|
||
|
||
|
||
class FakeAdapter:
|
||
"""脚本化观察适配器:只验证评测管道,不产生 Skill 行为证据。"""
|
||
|
||
def __init__(self, scripted: dict[str, dict]):
|
||
self.scripted = dict(scripted)
|
||
|
||
def run(self, skill_md_text: str, scenario: dict) -> dict:
|
||
observation = self.scripted.get(scenario["id"])
|
||
if observation is None:
|
||
raise EvalContractError(f"fake adapter 没有场景 {scenario['id']} 的脚本观察")
|
||
return dict(observation)
|
||
|
||
|
||
class RoleAgentAdapter:
|
||
"""真实角色 Agent 执行层占位:未获授权前失败关闭,不允许静默降级为 fake。"""
|
||
|
||
def run(self, skill_md_text: str, scenario: dict) -> dict:
|
||
raise EvalAdapterUnavailable(
|
||
"真实模型行为评测未授权:需要显式模型、预算与执行配置授权后才能运行"
|
||
)
|
||
|
||
|
||
def _check(expectations: dict, observation: dict) -> list[dict]:
|
||
failures: list[dict] = []
|
||
invoked = [str(cmd) for cmd in observation.get("invoked_commands", [])]
|
||
|
||
def missing_required(key: str) -> None:
|
||
failures.append({"check": "observation_missing", "evidence": f"观察缺少字段 {key}"})
|
||
|
||
for needle in expectations.get("must_invoke", []):
|
||
if not any(needle in cmd for cmd in invoked):
|
||
failures.append({"check": "must_invoke", "evidence": f"未见调用包含 {needle!r};实际 {invoked}"})
|
||
for needle in expectations.get("must_not_invoke", []):
|
||
hits = [cmd for cmd in invoked if needle in cmd]
|
||
if hits:
|
||
failures.append({"check": "must_not_invoke", "evidence": f"禁止调用 {needle!r} 出现: {hits}"})
|
||
if "expected_exit_codes" in expectations:
|
||
if "exit_code" not in observation:
|
||
missing_required("exit_code")
|
||
elif observation["exit_code"] not in expectations["expected_exit_codes"]:
|
||
failures.append({
|
||
"check": "expected_exit_codes",
|
||
"evidence": f"退出码 {observation['exit_code']} 不在 {expectations['expected_exit_codes']}",
|
||
})
|
||
output = str(observation.get("output", ""))
|
||
for needle in expectations.get("output_must_contain", []):
|
||
if needle not in output:
|
||
failures.append({"check": "output_must_contain", "evidence": f"输出缺少 {needle!r}"})
|
||
for needle in expectations.get("output_must_not_contain", []):
|
||
if needle in output:
|
||
failures.append({"check": "output_must_not_contain", "evidence": f"输出出现禁止内容 {needle!r}"})
|
||
if "input_text_modified" in expectations:
|
||
if "input_text_modified" not in observation:
|
||
missing_required("input_text_modified")
|
||
elif observation["input_text_modified"] != expectations["input_text_modified"]:
|
||
failures.append({
|
||
"check": "input_text_modified",
|
||
"evidence": f"输入正文改动状态 {observation['input_text_modified']} 与期望 "
|
||
f"{expectations['input_text_modified']} 不一致",
|
||
})
|
||
return failures
|
||
|
||
|
||
def run_scenario(skill_md_text: str, scenario: dict, adapter: Any) -> dict:
|
||
observation = adapter.run(skill_md_text, scenario)
|
||
failures = _check(scenario["expectations"], observation)
|
||
return {
|
||
"scenario_id": scenario["id"],
|
||
"category": scenario["category"],
|
||
"verdict": "passed" if not failures else "failed",
|
||
"failed_checks": failures,
|
||
"observation_summary": {
|
||
"invoked_commands": observation.get("invoked_commands", []),
|
||
"exit_code": observation.get("exit_code"),
|
||
"input_text_modified": observation.get("input_text_modified"),
|
||
"output_chars": len(str(observation.get("output", ""))),
|
||
},
|
||
}
|
||
|
||
|
||
def run_eval(skill_dir: Path, scenario_file: Path, adapter: Any, adapter_name: str) -> dict:
|
||
skill_md = (skill_dir / "SKILL.md")
|
||
if not skill_md.is_file():
|
||
raise EvalContractError(f"SKILL.md 不存在: {skill_md}")
|
||
data = load_scenarios(scenario_file)
|
||
skill_md_text = skill_md.read_text(encoding="utf-8")
|
||
verdicts = [run_scenario(skill_md_text, scenario, adapter) for scenario in data["scenarios"]]
|
||
passed = sum(1 for v in verdicts if v["verdict"] == "passed")
|
||
return {
|
||
"schema_version": REPORT_SCHEMA,
|
||
"skill": data["skill"],
|
||
"skill_md_sha256": skill_md_fingerprint(skill_dir),
|
||
"adapter": adapter_name,
|
||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||
"scenario_count": len(verdicts),
|
||
"passed": passed,
|
||
"failed": len(verdicts) - passed,
|
||
"verdicts": verdicts,
|
||
"limit": "fake adapter 结果只证明评测管道,不构成 Skill 行为证据",
|
||
}
|
||
|
||
|
||
__all__ = [
|
||
"REPORT_SCHEMA", "SCENARIO_SCHEMA", "CATEGORIES", "EvalContractError",
|
||
"EvalAdapterUnavailable", "FakeAdapter", "RoleAgentAdapter",
|
||
"skill_md_fingerprint", "load_scenarios", "run_scenario", "run_eval",
|
||
]
|