范围(不含 design-story-foundation、docs/design、docs/write-chapter、
craft/、humanization/README.md 等进行中改动):
1. humanization 规则/样例运行时数据库权威
- db/ddl/111:example_ai_flavor_rule / example_ai_flavor_sample /
example_ai_flavor_rule_event(append-only 生命周期留痕),已应用到 muse-example
- deai/load_db.py:数据库装载器,激活门/重复检测/指纹与文件装载器同源;
数据库失败关闭,不静默回退 Git 文件资产
- humanization/tools/seed_rules_db.py:YAML 种子单事务同步,幂等、
变化留痕、--strict 对 db-only 行失败关闭;真实库已种入 26 规则/107 样例
- prevent/diagnose/revise 生产路径切到数据库读取(--offline 显式读文件),
落库前新鲜度检查与合同声明来源一致;四个 SKILL.md 数据库合同同步
- 真实库验证:规则库指纹与文件种子一致(v-609bc40e21d0b5db),
三个生产脚本端到端从库装载通过
2. PostgreSQL 集成:显式授权后 9/9 通过
- 此前被依赖门阻断的 6 个 _db/smoke 测试全部通过
- extract rollback 冒烟改为回滚事务内自给夹具(pending 窗/草稿缺失时自建),
不再依赖瞬时生产状态;夹具残留核验为 0
3. Skill 行为评测脚手架(真实执行数量仍为 0)
- harness/evals/skill_eval.py:场景合同、六类评测范畴、适配器和结构化裁决报告
- diagnose-ai-flavor 参考场景 4 条 + 管道自测 7 项通过
- 真实模型适配器未授权时以稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭;
清单登记 skill_behavior_eval 条目,默认被依赖门阻断
4. evaluate-frozen-replay raw 存储边界冲突
- docs/2026-08-19 备忘录:平台 DB-first 合同(创始人批准)与回放链
仓外 vault 强制的冲突事实、两个选项和裁决前约束;运行时合同未单方面改写
5. harness 自身修复
- runner 对账语义:行为评测入口不参与测试资产双向等值,但登记文件必须存在;
manifest 保留 skill_behavior_eval 布尔字段并校验类型
- 新增 2 条对账回归用例
验证证据: harness 三组自测 15+15+7 通过;静态审计 32 Skill / 0 问题;
76 个非数据库条目通过;9 个 PostgreSQL 集成条目显式授权后通过;
行为评测条目默认阻断;py_compile 与 git diff --check 通过。
未调用真实模型、embedding 或额度;真实行为评测执行数量仍为 0。
176 lines
7.5 KiB
Python
176 lines
7.5 KiB
Python
#!/usr/bin/env python3
|
||
"""Skill 行为评测引擎:由 harness 从外部驱动,Skill 不能自己宣布通过。
|
||
|
||
引擎把 SKILL.md 文本和场景任务交给 adapter(Agent/模型执行层),接收结构化
|
||
观察,再按场景登记的期望裁决。真实模型 adapter 需要显式授权;未授权时以
|
||
稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭。fake adapter 只用于评测引擎自身
|
||
管道验证,不构成 Skill 行为证据。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import json
|
||
from datetime import datetime, timezone
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
REPORT_SCHEMA = "skill-behavior-eval-report-v1"
|
||
SCENARIO_SCHEMA = "skill-behavior-eval-scenarios-v1"
|
||
CATEGORIES = (
|
||
"positive_trigger",
|
||
"negative_trigger",
|
||
"input_missing_or_out_of_scope",
|
||
"output_contract_and_fail_closed",
|
||
"forbidden_action",
|
||
"stability_and_confounders",
|
||
)
|
||
|
||
|
||
class EvalContractError(ValueError):
|
||
"""评测合同错误:场景结构非法或观察缺项。"""
|
||
|
||
|
||
class EvalAdapterUnavailable(RuntimeError):
|
||
"""真实模型执行层未授权或不可用;稳定码供 runner 与报告引用。"""
|
||
|
||
code = "EVAL_ADAPTER_UNAVAILABLE"
|
||
|
||
|
||
def skill_md_fingerprint(skill_dir: Path) -> str:
|
||
text = (skill_dir / "SKILL.md").read_text(encoding="utf-8")
|
||
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||
|
||
|
||
def load_scenarios(path: Path) -> dict:
|
||
try:
|
||
data = json.loads(path.read_text(encoding="utf-8"))
|
||
except (OSError, json.JSONDecodeError) as exc:
|
||
raise EvalContractError(f"场景文件不可读或不合法: {path} ({exc})") from exc
|
||
if not isinstance(data, dict) or data.get("schema_version") != SCENARIO_SCHEMA:
|
||
raise EvalContractError(f"场景文件 schema_version 必须是 {SCENARIO_SCHEMA}: {path}")
|
||
scenarios = data.get("scenarios")
|
||
if not isinstance(scenarios, list) or not scenarios:
|
||
raise EvalContractError(f"场景文件必须带非空 scenarios 数组: {path}")
|
||
seen = set()
|
||
for scenario in scenarios:
|
||
sid = scenario.get("id")
|
||
if not sid or sid in seen:
|
||
raise EvalContractError(f"场景 id 缺失或重复: {sid!r} ({path})")
|
||
seen.add(sid)
|
||
if scenario.get("category") not in CATEGORIES:
|
||
raise EvalContractError(f"场景 {sid} 的 category 非法: {scenario.get('category')!r}")
|
||
if not str(scenario.get("task") or "").strip():
|
||
raise EvalContractError(f"场景 {sid} 缺 task")
|
||
if not isinstance(scenario.get("expectations"), dict) or not scenario["expectations"]:
|
||
raise EvalContractError(f"场景 {sid} 缺 expectations")
|
||
return data
|
||
|
||
|
||
class FakeAdapter:
|
||
"""脚本化观察适配器:只验证评测管道,不产生 Skill 行为证据。"""
|
||
|
||
def __init__(self, scripted: dict[str, dict]):
|
||
self.scripted = dict(scripted)
|
||
|
||
def run(self, skill_md_text: str, scenario: dict) -> dict:
|
||
observation = self.scripted.get(scenario["id"])
|
||
if observation is None:
|
||
raise EvalContractError(f"fake adapter 没有场景 {scenario['id']} 的脚本观察")
|
||
return dict(observation)
|
||
|
||
|
||
class ClaudeCliAdapter:
|
||
"""真实模型执行层占位:未获授权前失败关闭,不允许静默降级为 fake。"""
|
||
|
||
def run(self, skill_md_text: str, scenario: dict) -> dict:
|
||
raise EvalAdapterUnavailable(
|
||
"真实模型行为评测未授权:需要显式模型、预算与执行配置授权后才能运行"
|
||
)
|
||
|
||
|
||
def _check(expectations: dict, observation: dict) -> list[dict]:
|
||
failures: list[dict] = []
|
||
invoked = [str(cmd) for cmd in observation.get("invoked_commands", [])]
|
||
|
||
def missing_required(key: str) -> None:
|
||
failures.append({"check": "observation_missing", "evidence": f"观察缺少字段 {key}"})
|
||
|
||
for needle in expectations.get("must_invoke", []):
|
||
if not any(needle in cmd for cmd in invoked):
|
||
failures.append({"check": "must_invoke", "evidence": f"未见调用包含 {needle!r};实际 {invoked}"})
|
||
for needle in expectations.get("must_not_invoke", []):
|
||
hits = [cmd for cmd in invoked if needle in cmd]
|
||
if hits:
|
||
failures.append({"check": "must_not_invoke", "evidence": f"禁止调用 {needle!r} 出现: {hits}"})
|
||
if "expected_exit_codes" in expectations:
|
||
if "exit_code" not in observation:
|
||
missing_required("exit_code")
|
||
elif observation["exit_code"] not in expectations["expected_exit_codes"]:
|
||
failures.append({
|
||
"check": "expected_exit_codes",
|
||
"evidence": f"退出码 {observation['exit_code']} 不在 {expectations['expected_exit_codes']}",
|
||
})
|
||
output = str(observation.get("output", ""))
|
||
for needle in expectations.get("output_must_contain", []):
|
||
if needle not in output:
|
||
failures.append({"check": "output_must_contain", "evidence": f"输出缺少 {needle!r}"})
|
||
for needle in expectations.get("output_must_not_contain", []):
|
||
if needle in output:
|
||
failures.append({"check": "output_must_not_contain", "evidence": f"输出出现禁止内容 {needle!r}"})
|
||
if "input_text_modified" in expectations:
|
||
if "input_text_modified" not in observation:
|
||
missing_required("input_text_modified")
|
||
elif observation["input_text_modified"] != expectations["input_text_modified"]:
|
||
failures.append({
|
||
"check": "input_text_modified",
|
||
"evidence": f"输入正文改动状态 {observation['input_text_modified']} 与期望 "
|
||
f"{expectations['input_text_modified']} 不一致",
|
||
})
|
||
return failures
|
||
|
||
|
||
def run_scenario(skill_md_text: str, scenario: dict, adapter: Any) -> dict:
|
||
observation = adapter.run(skill_md_text, scenario)
|
||
failures = _check(scenario["expectations"], observation)
|
||
return {
|
||
"scenario_id": scenario["id"],
|
||
"category": scenario["category"],
|
||
"verdict": "passed" if not failures else "failed",
|
||
"failed_checks": failures,
|
||
"observation_summary": {
|
||
"invoked_commands": observation.get("invoked_commands", []),
|
||
"exit_code": observation.get("exit_code"),
|
||
"input_text_modified": observation.get("input_text_modified"),
|
||
"output_chars": len(str(observation.get("output", ""))),
|
||
},
|
||
}
|
||
|
||
|
||
def run_eval(skill_dir: Path, scenario_file: Path, adapter: Any, adapter_name: str) -> dict:
|
||
skill_md = (skill_dir / "SKILL.md")
|
||
if not skill_md.is_file():
|
||
raise EvalContractError(f"SKILL.md 不存在: {skill_md}")
|
||
data = load_scenarios(scenario_file)
|
||
skill_md_text = skill_md.read_text(encoding="utf-8")
|
||
verdicts = [run_scenario(skill_md_text, scenario, adapter) for scenario in data["scenarios"]]
|
||
passed = sum(1 for v in verdicts if v["verdict"] == "passed")
|
||
return {
|
||
"schema_version": REPORT_SCHEMA,
|
||
"skill": data["skill"],
|
||
"skill_md_sha256": skill_md_fingerprint(skill_dir),
|
||
"adapter": adapter_name,
|
||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||
"scenario_count": len(verdicts),
|
||
"passed": passed,
|
||
"failed": len(verdicts) - passed,
|
||
"verdicts": verdicts,
|
||
"limit": "fake adapter 结果只证明评测管道,不构成 Skill 行为证据",
|
||
}
|
||
|
||
|
||
__all__ = [
|
||
"REPORT_SCHEMA", "SCENARIO_SCHEMA", "CATEGORIES", "EvalContractError",
|
||
"EvalAdapterUnavailable", "FakeAdapter", "ClaudeCliAdapter",
|
||
"skill_md_fingerprint", "load_scenarios", "run_scenario", "run_eval",
|
||
]
|