116 lines
4.8 KiB
Python
116 lines
4.8 KiB
Python
#!/usr/bin/env python3
|
||
"""诊断机器味 行为评测入口。
|
||
|
||
--adapter fake:脚本化观察,只验证评测管道(不产生 Skill 行为证据)。
|
||
--adapter role-agent:真实 prompt 角色执行层;未获显式授权时以稳定码失败关闭。
|
||
真实行为评测需要模型、预算与执行配置授权;在此之前本入口被 harness
|
||
依赖门阻断,不伪装通过。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
EVAL_DIR = Path(__file__).resolve().parent
|
||
AGENT_ROOT = next(
|
||
parent
|
||
for parent in (EVAL_DIR, *EVAL_DIR.parents)
|
||
if (parent / "AGENTS.md").is_file() and (parent / ".git").exists()
|
||
)
|
||
HARNESS_DIR = AGENT_ROOT / "muse" / "lifecycle" / "quality" / "harness"
|
||
sys.path.insert(0, str(HARNESS_DIR / "evals"))
|
||
|
||
from skill_eval import ( # noqa: E402
|
||
EvalAdapterUnavailable, EvalContractError, FakeAdapter, RoleAgentAdapter, run_eval,
|
||
)
|
||
|
||
SKILL_DIR = AGENT_ROOT / "muse" / "lifecycle" / "quality" / "humanization" / "skills" / "诊断机器味"
|
||
SCENARIO_FILE = EVAL_DIR / "scenarios.json"
|
||
|
||
# 管道验证用脚本观察:与 scenarios.json 一一对应;只证明引擎裁决链路可用。
|
||
_FAKE_OBSERVATIONS = {
|
||
"positive-basic-diagnosis": {
|
||
"invoked_commands": [
|
||
".venv/bin/python muse/lifecycle/quality/humanization/skills/诊断机器味/scripts/diagnose_ai_flavor.py run "
|
||
"--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json"
|
||
],
|
||
"exit_code": 0,
|
||
"output": '{"findings": 2, "rule_library_version": "v-demo"}',
|
||
"input_text_modified": False,
|
||
},
|
||
"negative-direct-rewrite": {
|
||
"invoked_commands": [],
|
||
"exit_code": 0,
|
||
"output": "按写作任务处理,未调用诊断技能。",
|
||
"input_text_modified": True,
|
||
},
|
||
"missing-work-ref": {
|
||
"invoked_commands": [
|
||
".venv/bin/python muse/lifecycle/quality/humanization/skills/诊断机器味/scripts/diagnose_ai_flavor.py run "
|
||
"--text-file /tmp/text.txt --output /tmp/artifact.json"
|
||
],
|
||
"exit_code": 2,
|
||
"output": "DIAGNOSE_CONTRACT_FAILED: 诊断必须带 work_ref",
|
||
"input_text_modified": False,
|
||
},
|
||
"forbidden-modify-text": {
|
||
"invoked_commands": [
|
||
".venv/bin/python muse/lifecycle/quality/humanization/skills/诊断机器味/scripts/diagnose_ai_flavor.py run "
|
||
"--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json"
|
||
],
|
||
"exit_code": 0,
|
||
"output": '{"findings": 1, "rule_library_version": "v-demo"}',
|
||
"input_text_modified": False,
|
||
},
|
||
"clean-text-still-emits-contract-report": {
|
||
"invoked_commands": [
|
||
".venv/bin/python muse/lifecycle/quality/humanization/skills/诊断机器味/scripts/diagnose_ai_flavor.py run "
|
||
"--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json"
|
||
],
|
||
"exit_code": 0,
|
||
"output": '{"findings": 0, "rule_library_version": "v-demo"}',
|
||
"input_text_modified": False,
|
||
},
|
||
"resists-skip-rules-suggestion": {
|
||
"invoked_commands": [
|
||
".venv/bin/python muse/lifecycle/quality/humanization/skills/诊断机器味/scripts/diagnose_ai_flavor.py run "
|
||
"--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json"
|
||
],
|
||
"exit_code": 0,
|
||
"output": '{"findings": 1, "rule_library_version": "v-demo"}',
|
||
"input_text_modified": False,
|
||
},
|
||
}
|
||
|
||
|
||
def main(argv: list[str] | None = None) -> int:
|
||
parser = argparse.ArgumentParser(description="诊断机器味 Skill 行为评测")
|
||
parser.add_argument("--adapter", choices=["fake", "role-agent"], default="role-agent",
|
||
help="默认真实执行层(未授权即失败关闭);fake 仅限管道验证")
|
||
parser.add_argument("--output", type=Path, help="评测报告输出路径(缺省只打印)")
|
||
args = parser.parse_args(argv)
|
||
|
||
adapter = FakeAdapter(_FAKE_OBSERVATIONS) if args.adapter == "fake" else RoleAgentAdapter()
|
||
try:
|
||
report = run_eval(SKILL_DIR, SCENARIO_FILE, adapter, adapter_name=args.adapter)
|
||
except EvalAdapterUnavailable as exc:
|
||
print(json.dumps({"status": "blocked", "code": exc.code, "message": str(exc)},
|
||
ensure_ascii=False))
|
||
return 2
|
||
except (EvalContractError, OSError, ValueError) as exc:
|
||
print(f"SKILL_EVAL_CONTRACT_FAILED: {exc}")
|
||
return 2
|
||
|
||
payload = json.dumps(report, ensure_ascii=False, indent=2)
|
||
if args.output:
|
||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||
args.output.write_text(payload + "\n", encoding="utf-8")
|
||
print(payload)
|
||
return 0 if report["failed"] == 0 else 1
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|