#!/usr/bin/env python3 """对项目运行时 Skill 文档执行只读静态审计。 本模块只读取 `.agent/skills/*/SKILL.md`,不连接数据库、网络或模型,也不修改 工作树。机器调用使用 :func:`audit_skills`,命令行入口同时支持人读摘要和 JSON。 """ from __future__ import annotations import argparse import json import re import sys from collections import Counter from pathlib import Path from typing import Any, Iterable, Optional, Sequence SCHEMA_VERSION = 4 SKILLS_ROOT = Path(".agent") / "skills" DEFAULT_MANIFEST_PATH = Path("harness") / "manifests" / "skills.json" # 分类取值与质量预算的定义 Owner 是 harness/specs/skill-quality-rubric.md;本模块只做机械判定。 VALID_LIFECYCLES = ( "platform", "ingest", "knowledge", "concept", "planning", "writing", "review", "humanization", "sovereignty", ) VALID_INVOCATIONS = ("orchestrated", "model_routed") VALID_SIDE_EFFECTS = ("none", "db_write", "external_call") VALID_COMPOUNDING = ("closed_loop", "partial", "none") LINE_BUDGET = 120 DESCRIPTION_BUDGET = 200 # 创作方法密集的周期段需要列同义触发说法,描述预算放宽。 WIDE_DESCRIPTION_LIFECYCLES = ("concept", "planning", "writing", "review") WIDE_DESCRIPTION_BUDGET = 1000 # 必备节的机械判定只覆盖标题词汇稳定的一部分;唯一目的、消费者仍由人工审查。 _SECTION_PATTERNS = { "输入": re.compile(r"输入|用法|入口|参数"), "输出": re.compile(r"输出|产出|返回|报告"), "红线": re.compile(r"红线|禁止|不得|边界|纪律|铁律"), "数据边界": re.compile(r"数据边界|数据库读写|落库|副作用|存储|数据与"), "复利合同": re.compile(r"复利|升格|回写|沉淀|经验"), } _BASE_SECTIONS = ("输入", "输出", "红线") _SIDE_EFFECT_SECTIONS = ("数据边界",) _COMPOUNDING_SECTIONS = ("复利合同",) _FRONTMATTER_KEY = re.compile(r"^(?P\s*)(?P[A-Za-z0-9_-]+)\s*:\s*(?P.*?)\s*$") # 单个 references 文件的体量上限:references/ 本就是 D6 的外移目的地,只对 # “一次读进来就吃掉整窗”的超大单文件设限,不对 references/ 总量设预算。 REFERENCE_FILE_BUDGET_BYTES = 20 * 1024 # 跨 Skill 检查用的文档集合与忽略目录。 _DOC_SUFFIXES = (".md",) _SCAN_SKIP_DIRS = frozenset({".git", ".venv", "__pycache__", "node_modules", ".pytest_cache"}) # 内部代号引用:形如 `MERGE_PLAN §4`。全大写加节号是很窄的签名, # 正常业务词与 DDL-111 这类带连字符的编号都不会命中。 _CODENAME_CITATION = re.compile(r"(?=|>|:|:)?\s*" r"\d+(?:\.\d+)?\s*[%%]", re.IGNORECASE, ) _BUSINESS_COVERAGE_CONTEXT_PATTERN = re.compile( r"(?:细纲|事件|硬约束|软约束|伏笔|实体|设定|情节|场景|知识|事实|正文|" r"作品|业务|候选|规划|大纲|章)[^。\n]{0,24}覆盖率" r"|覆盖率[^。\n]{0,24}(?:细纲|事件|硬约束|软约束|伏笔|实体|设定|情节|" r"场景|知识|事实|正文|作品|业务|候选|规划|大纲|章)", re.IGNORECASE, ) # 这些规则只针对运行时 Skill 文档中的明显开发验证残留;业务正文中的普通“检查” # 等词不在扫描范围内。规则保持确定性,方便报告中的 code 作为机械门禁输入。 _DEVELOPMENT_HEADING_PATTERN = re.compile(r"^\s*(#{1,6})\s*(.*?)\s*$") _DEVELOPMENT_SECTION_TITLE_PATTERN = re.compile( r"(?:自测|测试|开发验证|离线验证|开发测试|离线测试)", re.IGNORECASE, ) _DEVELOPMENT_COMMAND_PATTERN = re.compile( r"(? Path: """把审计根目录解析为绝对路径,但不要求它已经存在。""" path = Path(root).expanduser() if not path.is_absolute(): path = Path.cwd() / path return path.resolve() def _relative_path(path: Path, root: Path) -> str: """返回报告中的稳定 POSIX 相对路径。""" try: return path.resolve().relative_to(root).as_posix() except ValueError: return path.resolve().as_posix() def _issue( code: str, message: str, *, path: Optional[str] = None, line: Optional[int] = None, severity: str = "blocking", **details: Any, ) -> dict[str, Any]: result: dict[str, Any] = { "code": code, "severity": severity, "message": message, } if path is not None: result["path"] = path if line is not None: result["line"] = line result.update(details) return result def _unquote_scalar(value: str) -> str: """解析 Skill frontmatter 中足够支持 name 的简单标量。""" value = value.strip() if len(value) >= 2 and value[0] == value[-1] == '"': try: parsed = json.loads(value) except json.JSONDecodeError: return value[1:-1] return parsed if isinstance(parsed, str) else value if len(value) >= 2 and value[0] == value[-1] == "'": return value[1:-1].replace("''", "'") # 允许常见的 YAML 行尾注释,不引入 YAML 依赖。 return re.sub(r"\s+#.*$", "", value).strip() def _parse_frontmatter(text: str) -> tuple[Optional[dict[str, str]], Optional[int], Optional[str]]: """读取 frontmatter 的简单键值视图。 返回值为 ``(fields, closing_line, error_code)``。该解析器识别顶层标量键与 ``|``/``>`` 块标量(description 需要),不试图替代完整 YAML 解析器;不合法或 缺失的边界会明确进入失败报告。 """ lines = text.splitlines() if not lines or lines[0].lstrip("\ufeff").strip() != "---": return None, None, "frontmatter_missing" closing_line: Optional[int] = None for index in range(1, len(lines)): if lines[index].strip() in {"---", "..."}: closing_line = index + 1 break if closing_line is None: return None, None, "frontmatter_unclosed" fields: dict[str, str] = {} body = lines[1 : closing_line - 1] index = 0 while index < len(body): match = _FRONTMATTER_KEY.match(body[index]) if not match or match.group("indent"): index += 1 continue key = match.group("key") value = match.group("value") if value in {"|", ">", "|-", ">-", "|+", ">+"}: block: list[str] = [] index += 1 while index < len(body) and (not body[index].strip() or body[index][:1] in " \t"): block.append(body[index].strip()) index += 1 fields[key] = "\n".join(block).strip() continue fields[key] = _unquote_scalar(value) index += 1 return fields, closing_line, None def _snippet(line: str) -> str: compact = line.strip() return compact if len(compact) <= 200 else compact[:197] + "..." def _is_development_coverage_reference(line: str) -> bool: """只把明确落在开发测试语境中的 coverage 文字判为污染。""" if _COVERAGE_TOOL_PATTERN.search(line) or _EXPLICIT_TEST_COVERAGE_PATTERN.search(line): return True if not _COVERAGE_PERCENT_PATTERN.search(line): return False return not _BUSINESS_COVERAGE_CONTEXT_PATTERN.search(line) def _inspect_skill_file(path: Path, root: Path) -> tuple[dict[str, Any], list[dict[str, Any]]]: relative = _relative_path(path, root) entry: dict[str, Any] = { "directory": path.parent.name, "skill_path": relative, "name": None, "frontmatter_present": False, "invocation": None, "description": "", "description_length": 0, "line_count": 0, "headings": [], } issues: list[dict[str, Any]] = [] try: text = path.read_text(encoding="utf-8") except (OSError, UnicodeError) as exc: issues.append( _issue( "skill_file_unreadable", "无法读取 SKILL.md", path=relative, error=str(exc), ) ) return entry, issues if not text.strip(): issues.append( _issue( "empty_skill_file", "SKILL.md 为空", path=relative, line=1, ) ) return entry, issues fields, closing_line, frontmatter_error = _parse_frontmatter(text) if frontmatter_error is not None: line = 1 if frontmatter_error == "frontmatter_missing" else max(1, len(text.splitlines())) messages = { "frontmatter_missing": "缺少以 --- 开始的 frontmatter", "frontmatter_unclosed": "frontmatter 未闭合", } issues.append( _issue( frontmatter_error, messages[frontmatter_error], path=relative, line=line, ) ) else: entry["frontmatter_present"] = True fields = fields or {} entry["invocation"] = ( "orchestrated" if "disable-model-invocation" in fields else "model_routed" ) entry["description"] = fields.get("description", "") entry["description_length"] = len(entry["description"]) name = fields.get("name") entry["name"] = name or None if not name: issues.append( _issue( "frontmatter_name_missing", "frontmatter 缺少非空 name", path=relative, line=1, ) ) elif name != path.parent.name: issues.append( _issue( "name_directory_mismatch", "frontmatter name 与 Skill 目录名不一致", path=relative, line=2, expected=path.parent.name, actual=name, ) ) development_section_level: Optional[int] = None entry["line_count"] = len(text.splitlines()) for line_number, line in enumerate(text.splitlines(), start=1): heading_match = _DEVELOPMENT_HEADING_PATTERN.match(line) if heading_match: heading_level = len(heading_match.group(1)) heading_title = re.sub(r"\s+#+\s*$", "", heading_match.group(2)).strip() entry["headings"].append(heading_title) if ( development_section_level is not None and heading_level <= development_section_level ): development_section_level = None if ( 2 <= heading_level <= 6 and _DEVELOPMENT_SECTION_TITLE_PATTERN.search(heading_title) ): development_section_level = heading_level issues.append( _issue( "development_test_heading", "发现开发测试章节标题", path=relative, line=line_number, snippet=_snippet(line), ) ) for code, pattern, message in _POLLUTION_RULES: if pattern.search(line): issues.append( _issue( code, message, path=relative, line=line_number, snippet=_snippet(line), ) ) if ( development_section_level is not None and _DEVELOPMENT_COMMAND_PATTERN.search(line) ): issues.append( _issue( "development_test_command_reference", "发现开发测试章节中的合同测试命令", path=relative, line=line_number, snippet=_snippet(line), ) ) if _is_development_coverage_reference(line): issues.append( _issue( "coverage_reference", "发现开发测试语境中的覆盖率声明或工具名", path=relative, line=line_number, snippet=_snippet(line), ) ) return entry, issues def _scan_skill_directories(root: Path) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: skills_root = root / SKILLS_ROOT relative_skills_root = _relative_path(skills_root, root) entries: list[dict[str, Any]] = [] issues: list[dict[str, Any]] = [] if not skills_root.exists(): issues.append( _issue( "skills_directory_missing", "运行时 Skill 目录不存在", path=relative_skills_root, ) ) return entries, issues if not skills_root.is_dir(): issues.append( _issue( "skills_directory_not_directory", "运行时 Skill 路径不是目录", path=relative_skills_root, ) ) return entries, issues try: children = sorted(skills_root.iterdir(), key=lambda item: item.name) except OSError as exc: issues.append( _issue( "skills_directory_unreadable", "无法列出运行时 Skill 目录", path=relative_skills_root, error=str(exc), ) ) return entries, issues directories = [child for child in children if child.is_dir()] if not directories: issues.append( _issue( "no_skills_found", "运行时 Skill 目录中未发现 Skill 子目录", path=relative_skills_root, ) ) return entries, issues for directory in directories: skill_file = directory / "SKILL.md" relative_skill_file = _relative_path(skill_file, root) disk_entry: dict[str, Any] = { "directory": directory.name, "skill_path": relative_skill_file, "name": None, "frontmatter_present": False, } if not skill_file.exists(): entries.append(disk_entry) issues.append( _issue( "missing_skill_file", "Skill 目录缺少 SKILL.md", path=relative_skill_file, ) ) continue if not skill_file.is_file(): entries.append(disk_entry) issues.append( _issue( "skill_file_not_file", "SKILL.md 路径不是普通文件", path=relative_skill_file, ) ) continue entry, file_issues = _inspect_skill_file(skill_file, root) entries.append(entry) issues.extend(file_issues) return entries, issues def _resolve_manifest_path(root: Path, manifest: str | Path | None) -> Path: path = DEFAULT_MANIFEST_PATH if manifest is None else Path(manifest).expanduser() if not path.is_absolute(): path = root / path return path.resolve() def _normalise_manifest_skill_path( root: Path, value: str ) -> tuple[Optional[str], Optional[Path]]: declared = Path(value) if declared.is_absolute(): return None, None candidate = (root / declared).resolve() try: relative = candidate.relative_to(root).as_posix() except ValueError: return None, None return relative, candidate def _validate_manifest( root: Path, entries: Sequence[dict[str, Any]], manifest_path: Path, ) -> tuple[dict[str, Any], list[dict[str, Any]]]: """验证 Skill 登记表,并要求它与磁盘目录一一对应。""" relative_manifest = _relative_path(manifest_path, root) metadata: dict[str, Any] = { "path": relative_manifest, "loaded": False, "skills_declared": 0, } issues: list[dict[str, Any]] = [] if not manifest_path.exists(): issues.append( _issue( "manifest_missing", "Skill manifest 不存在", path=relative_manifest, ) ) return metadata, issues if not manifest_path.is_file(): issues.append( _issue( "manifest_not_file", "Skill manifest 路径不是普通文件", path=relative_manifest, ) ) return metadata, issues try: manifest_text = manifest_path.read_text(encoding="utf-8") manifest = json.loads(manifest_text) except UnicodeError as exc: issues.append( _issue( "manifest_unreadable", "无法按 UTF-8 读取 Skill manifest", path=relative_manifest, error=str(exc), ) ) return metadata, issues except OSError as exc: issues.append( _issue( "manifest_unreadable", "无法读取 Skill manifest", path=relative_manifest, error=str(exc), ) ) return metadata, issues except json.JSONDecodeError as exc: issues.append( _issue( "manifest_invalid_json", "Skill manifest 不是合法 JSON", path=relative_manifest, line=exc.lineno, column=exc.colno, error=exc.msg, ) ) return metadata, issues metadata["loaded"] = True if not isinstance(manifest, dict): issues.append( _issue( "manifest_invalid_structure", "Skill manifest 顶层必须是对象", path=relative_manifest, ) ) return metadata, issues raw_skills = manifest.get("skills") if not isinstance(raw_skills, list): issues.append( _issue( "manifest_skills_invalid", "Skill manifest.skills 必须是数组", path=relative_manifest, ) ) return metadata, issues metadata["skills_declared"] = len(raw_skills) disk_paths = { str(entry["skill_path"]) for entry in entries if isinstance(entry.get("skill_path"), str) } disk_invocation = { str(entry["skill_path"]): entry.get("invocation") for entry in entries if isinstance(entry.get("skill_path"), str) } declared_paths: set[str] = set() path_occurrences: dict[str, list[int]] = {} name_occurrences: dict[str, list[int]] = {} skills_root = (root / SKILLS_ROOT).resolve() for index, item in enumerate(raw_skills): if not isinstance(item, dict): issues.append( _issue( "manifest_entry_invalid", "Skill manifest 条目必须是对象", path=relative_manifest, entry_index=index, ) ) continue name = item.get("name") if ( not isinstance(name, str) or not name.strip() or "/" in name or "\\" in name or name in {".", ".."} ): issues.append( _issue( "manifest_name_invalid", "manifest 条目的 name 必须是非空目录名", path=relative_manifest, entry_index=index, actual=name, ) ) else: name_occurrences.setdefault(name, []).append(index) lifecycle = item.get("lifecycle") if lifecycle not in VALID_LIFECYCLES: issues.append( _issue( "manifest_lifecycle_invalid", "manifest 条目的 lifecycle 必须是作品创建周期九段之一", path=relative_manifest, entry_index=index, actual=lifecycle, allowed=list(VALID_LIFECYCLES), ) ) compounding = item.get("compounding") if compounding not in VALID_COMPOUNDING: issues.append( _issue( "manifest_compounding_invalid", "manifest 条目的 compounding 必须是 closed_loop、partial 或 none", path=relative_manifest, entry_index=index, actual=compounding, ) ) invocation = item.get("invocation") if invocation not in VALID_INVOCATIONS: issues.append( _issue( "manifest_invocation_invalid", "manifest 条目的 invocation 必须是 orchestrated 或 model_routed", path=relative_manifest, entry_index=index, actual=invocation, ) ) invocation = None side_effects = item.get("side_effects") if ( not isinstance(side_effects, list) or not side_effects or any(value not in VALID_SIDE_EFFECTS for value in side_effects) or ("none" in side_effects and len(side_effects) > 1) ): issues.append( _issue( "manifest_side_effects_invalid", "manifest 条目的 side_effects 必须是 none/db_write/external_call 的非空数组,且 none 不与其它值并列", path=relative_manifest, entry_index=index, actual=side_effects, ) ) # runtime 由领域表登记合同责任方;methodology 不进领域表,改以 domain 说明能力域。 # 两类 Skill 都在 AGENTS.md 第 3 节的领域表登记合同责任方。 contract_owner = item.get("contract_owner") if not isinstance(contract_owner, str) or not contract_owner.strip(): issues.append( _issue( "manifest_contract_owner_invalid", "manifest 条目的 contract_owner 必须非空", path=relative_manifest, entry_index=index, ) ) collaborations = item.get("collaborates_with") if ( not isinstance(collaborations, list) or any(not isinstance(value, str) or not value.strip() for value in collaborations) ): issues.append( _issue( "manifest_collaborates_with_invalid", "manifest 条目的 collaborates_with 必须是非空字符串数组", path=relative_manifest, entry_index=index, ) ) declared_skill_path = item.get("skill_path") if not isinstance(declared_skill_path, str) or not declared_skill_path.strip(): issues.append( _issue( "manifest_skill_path_invalid", "manifest 条目的 skill_path 必须是非空相对路径", path=relative_manifest, entry_index=index, actual=declared_skill_path, ) ) continue normalised_path, candidate = _normalise_manifest_skill_path( root, declared_skill_path ) if normalised_path is None or candidate is None: issues.append( _issue( "manifest_skill_path_invalid", "manifest 条目的 skill_path 必须位于项目根目录内", path=relative_manifest, entry_index=index, actual=declared_skill_path, ) ) continue declared_paths.add(normalised_path) path_occurrences.setdefault(normalised_path, []).append(index) # "不得由 Agent 自行触发"只有 orchestrated 才算生效;散文声明不能代替 frontmatter 开关。 actual_invocation = disk_invocation.get(normalised_path) if invocation is not None and actual_invocation is not None and invocation != actual_invocation: issues.append( _issue( "invocation_frontmatter_mismatch", "manifest invocation 与 frontmatter 的 disable-model-invocation 不一致", path=normalised_path, entry_index=index, expected=actual_invocation, actual=invocation, ) ) if candidate != candidate.parent / "SKILL.md" or not candidate.is_relative_to(skills_root): issues.append( _issue( "manifest_skill_path_invalid", "manifest 条目的 skill_path 必须指向 .agent/skills/*/SKILL.md", path=normalised_path, entry_index=index, ) ) if not candidate.exists(): issues.append( _issue( "manifest_skill_path_missing", "manifest 登记的 Skill 路径不存在", path=normalised_path, entry_index=index, ) ) elif not candidate.is_file(): issues.append( _issue( "manifest_skill_path_not_file", "manifest 登记的 Skill 路径不是普通文件", path=normalised_path, entry_index=index, ) ) directory_name = candidate.parent.name if isinstance(name, str) and name.strip() and name != directory_name: issues.append( _issue( "manifest_name_directory_mismatch", "manifest name 与 skill_path 的目录名不一致", path=normalised_path, entry_index=index, expected=directory_name, actual=name, ) ) if isinstance(name, str) and name.strip(): expected_path = (root / SKILLS_ROOT / name / "SKILL.md").resolve() expected_relative = _relative_path(expected_path, root) if normalised_path != expected_relative: issues.append( _issue( "manifest_skill_path_mismatch", "manifest skill_path 与 name 对应的 Skill 路径不一致", path=normalised_path, entry_index=index, expected=expected_relative, actual=normalised_path, ) ) for path, indexes in sorted(path_occurrences.items()): if len(indexes) > 1: issues.append( _issue( "manifest_duplicate_skill_path", "manifest 重复登记同一个 Skill 路径", path=path, entry_indexes=indexes, ) ) for name, indexes in sorted(name_occurrences.items()): if len(indexes) > 1: issues.append( _issue( "manifest_duplicate_name", "manifest 重复登记同一个 Skill name", path=relative_manifest, name=name, entry_indexes=indexes, ) ) for path in sorted(disk_paths - declared_paths): issues.append( _issue( "manifest_skill_missing", "磁盘 Skill 未在 manifest 中登记", path=path, ) ) for path in sorted(declared_paths - disk_paths): issues.append( _issue( "manifest_skill_extra", "manifest 登记了磁盘中不存在的额外 Skill", path=path, ) ) return metadata, issues def _load_manifest_classes(manifest_path: Path, root: Path) -> dict[str, dict[str, Any]]: """只取评分需要的分类视图;manifest 本身的合法性由 :func:`_validate_manifest` 判定。""" try: manifest = json.loads(manifest_path.read_text(encoding="utf-8")) except (OSError, UnicodeError, json.JSONDecodeError): return {} if not isinstance(manifest, dict) or not isinstance(manifest.get("skills"), list): return {} view: dict[str, dict[str, Any]] = {} for item in manifest["skills"]: if not isinstance(item, dict): continue declared = item.get("skill_path") if not isinstance(declared, str): continue normalised, _ = _normalise_manifest_skill_path(root, declared) if normalised is None: continue view[normalised] = { "lifecycle": item.get("lifecycle"), "side_effects": item.get("side_effects") or [], "compounding": item.get("compounding"), } return view def _collect_markdown_stems(root: Path) -> set[str]: """收集审计根目录下全部 Markdown 文件的裸名,用于判定代号引用是否落空。""" stems: set[str] = set() for path in root.rglob("*.md"): if any(part in _SCAN_SKIP_DIRS for part in path.parts): continue stems.add(path.stem) return stems def _skill_documents(skill_dir: Path) -> list[Path]: """一个 Skill 包内会被人或模型读到的 Markdown 文档。""" documents: list[Path] = [] skill_file = skill_dir / "SKILL.md" if skill_file.is_file(): documents.append(skill_file) for sub in ("references", "scripts"): directory = skill_dir / sub if not directory.is_dir(): continue for suffix in _DOC_SUFFIXES: documents.extend(sorted(directory.rglob(f"*{suffix}"))) return documents def _document_advisories( root: Path, skill_dir: Path, stems: set[str] ) -> list[dict[str, Any]]: """逐文档检查落空的内部代号引用与不可解析的文档路径。""" advisories: list[dict[str, Any]] = [] for document in _skill_documents(skill_dir): try: text = document.read_text(encoding="utf-8") except (OSError, UnicodeError): continue relative = _relative_path(document, root) for line_number, line in enumerate(text.splitlines(), start=1): for codename in _CODENAME_CITATION.findall(line): if codename in _CODENAME_ALLOWLIST or codename in stems: continue advisories.append( _issue( "dangling_internal_codename", f"引用了不存在的内部代号文档:{codename}", path=relative, line=line_number, severity="major", codename=codename, snippet=_snippet(line), ) ) targets = list(_MARKDOWN_LINK.findall(line)) + list( _INLINE_DOC_PATH.findall(line) ) for target in targets: resolved = _resolve_doc_target(document, target) if resolved is None or resolved.exists(): continue advisories.append( _issue( "reference_target_missing", f"引用的文档路径不存在:{target}", path=relative, line=line_number, severity="major", target=target, ) ) return advisories def _resolve_doc_target(document: Path, target: str) -> Optional[Path]: """把文档内引用解析为磁盘路径;外链与纯锚点返回 None 表示不判定。""" candidate = target.strip().strip("<>") if not candidate or candidate.startswith(("#", "http://", "https://", "mailto:")): return None candidate = candidate.split("#", 1)[0].strip() if not candidate or candidate.startswith("/"): return None return (document.parent / candidate).resolve() def _reference_hygiene_advisories( root: Path, skill_dir: Path, skill_path: str ) -> list[dict[str, Any]]: """references/ 的孤儿文件与超大单文件。""" advisories: list[dict[str, Any]] = [] references = skill_dir / "references" if not references.is_dir(): return advisories skill_file = skill_dir / "SKILL.md" try: skill_text = skill_file.read_text(encoding="utf-8") except (OSError, UnicodeError): skill_text = "" for path in sorted(references.rglob("*.md")): relative = _relative_path(path, root) try: size = path.stat().st_size except OSError: size = 0 if size > REFERENCE_FILE_BUDGET_BYTES: advisories.append( _issue( "reference_file_too_large", f"单个 references 文件 {size // 1024} KB 超出 " f"{REFERENCE_FILE_BUDGET_BYTES // 1024} KB 上限,应按主题拆分", path=relative, severity="minor", budget=REFERENCE_FILE_BUDGET_BYTES, actual=size, ) ) # 下划线开头是包内记账文件(如 _coverage.md),不是可加载参考。 if path.parent != references or path.name.startswith("_"): continue if path.name not in skill_text: advisories.append( _issue( "orphan_reference_file", f"references/{path.name} 未被 SKILL.md 引用,属包内死重", path=relative, severity="major", skill_path=skill_path, ) ) return advisories def _trigger_tokens(description: str) -> set[str]: """取 description 中“何时用”那半段的触发词;排除项之后不计入。""" head = description.split(_DESCRIPTION_EXCLUSION_MARKER, 1)[0].replace("’", "'") tokens = set() for raw in _TRIGGER_SPLIT.split(head): token = raw.strip().strip(".-_") if not token or token in _TRIGGER_STOPWORDS: continue if _TRIGGER_MIN_LEN <= len(token) <= _TRIGGER_MAX_LEN: tokens.add(token) return tokens def _shared_triggers(left: set[str], right: set[str]) -> list[str]: """重合的触发说法:完全相同,或一方把另一方整句包进更长的说法里。""" shared = set(left & right) for token in left: if len(token) < _TRIGGER_CONTAINMENT_MIN_LEN: continue if any(token in other for other in right if other != token): shared.add(token) for token in right: if len(token) < _TRIGGER_CONTAINMENT_MIN_LEN: continue if any(token in other for other in left if other != token): shared.add(token) return sorted(shared) def _cross_skill_advisories( root: Path, entries: Sequence[dict[str, Any]], classes: dict[str, dict[str, Any]], ) -> list[dict[str, Any]]: """集合级检查:点名不存在的 Skill、同周期段触发词碰撞、案例范本跨 Skill 重复。""" advisories: list[dict[str, Any]] = [] known = { str(entry.get("directory")) for entry in entries if isinstance(entry.get("directory"), str) } routed: list[tuple[str, str, str, str, set[str]]] = [] for entry in entries: skill_path = str(entry.get("skill_path", "")) declared = classes.get(skill_path) if declared is None: continue directory = str(entry.get("directory", "")) description = str(entry.get("description", "")) package_docs = { path.stem for path in (root / SKILLS_ROOT / directory).rglob("*.md") } for token in sorted(set(_SKILL_NAME_TOKEN.findall(description))): # 指向自己包内 references/ 文档的连字符名不是路由目标,不算点名。 if token in known or token in _KNOWN_NON_SKILL_TOKENS or token in package_docs: continue advisories.append( _issue( "description_names_unknown_skill", f"description 点名的 {token} 不是现存 Skill,模型无法按它路由", path=skill_path, severity="major", token=token, ) ) routed.append( ( skill_path, directory, str(declared.get("lifecycle") or ""), description, _trigger_tokens(description), ) ) for index, left in enumerate(routed): for right in routed[index + 1 :]: if not left[2] or left[2] != right[2]: continue shared = _shared_triggers(left[4], right[4]) if not shared: continue if right[1] in left[3] or left[1] in right[3]: continue advisories.append( _issue( "trigger_collision_without_handoff", f"与 {right[1]} 同处 {left[2]} 周期段且触发词重合" f"({'、'.join(shared)}),双方都没点名对方", path=left[0], severity="major", peer=right[1], shared_triggers=shared, ) ) return advisories def _exemplar_advisories(root: Path, entries: Sequence[dict[str, Any]]) -> list[dict[str, Any]]: """同一案例范本被两个 Skill 各写一遍:出范式卡时会重复登记同一条方法。""" occurrences: dict[str, list[tuple[str, str, int]]] = {} for entry in entries: directory = str(entry.get("directory", "")) skill_dir = root / SKILLS_ROOT / directory for document in _skill_documents(skill_dir): try: text = document.read_text(encoding="utf-8") except (OSError, UnicodeError): continue relative = _relative_path(document, root) for line_number, line in enumerate(text.splitlines(), start=1): match = _EXEMPLAR_HEADING.match(line) if not match: continue occurrences.setdefault(match.group(1), []).append( (directory, relative, line_number) ) advisories: list[dict[str, Any]] = [] for exemplar, hits in sorted(occurrences.items()): owners = sorted({directory for directory, _, _ in hits}) if len(owners) < 2: continue for directory, relative, line_number in hits: peers = [owner for owner in owners if owner != directory] advisories.append( _issue( "shared_exemplar_heading", f"案例范本「{exemplar}」同时出现在 {'、'.join(peers)}," "应只留一份、其余改链接", path=relative, line=line_number, severity="minor", exemplar=exemplar, owners=owners, ) ) return advisories def _quality_advisories( root: Path, entries: Sequence[dict[str, Any]], classes: dict[str, dict[str, Any]], ) -> list[dict[str, Any]]: """产出不阻断退出码的质量发现,评分口径见 specs/skill-quality-rubric.md。""" advisories: list[dict[str, Any]] = [] stems = _collect_markdown_stems(root) for entry in entries: skill_path = str(entry.get("skill_path", "")) declared = classes.get(skill_path) if declared is None or declared["lifecycle"] not in VALID_LIFECYCLES: continue skill_dir = root / SKILLS_ROOT / str(entry.get("directory", "")) advisories.extend(_document_advisories(root, skill_dir, stems)) advisories.extend(_reference_hygiene_advisories(root, skill_dir, skill_path)) lifecycle = str(declared["lifecycle"]) side_effects = declared["side_effects"] compounding = declared["compounding"] required = list(_BASE_SECTIONS) if side_effects != ["none"]: required.extend(_SIDE_EFFECT_SECTIONS) if compounding in ("closed_loop", "partial"): required.extend(_COMPOUNDING_SECTIONS) headings = [str(value) for value in entry.get("headings", [])] for section in required: pattern = _SECTION_PATTERNS[section] if not any(pattern.search(heading) for heading in headings): advisories.append( _issue( "required_section_missing", f"缺少必备节:{section}", path=skill_path, severity="major", section=section, ) ) line_count = int(entry.get("line_count", 0)) if line_count > LINE_BUDGET: advisories.append( _issue( "line_budget_exceeded", f"正文 {line_count} 行超出预算 {LINE_BUDGET} 行", path=skill_path, severity="major" if line_count > LINE_BUDGET * 1.5 else "minor", budget=LINE_BUDGET, actual=line_count, ) ) if entry.get("invocation") == "model_routed": desc_budget = ( WIDE_DESCRIPTION_BUDGET if lifecycle in WIDE_DESCRIPTION_LIFECYCLES else DESCRIPTION_BUDGET ) desc_length = int(entry.get("description_length", 0)) if desc_length > desc_budget: advisories.append( _issue( "description_budget_exceeded", f"description {desc_length} 字符超出预算 {desc_budget} 字符", path=skill_path, severity="major" if desc_length > desc_budget * 1.5 else "minor", budget=desc_budget, actual=desc_length, ) ) # 复利未接入是待补齐的债,不是合法终态;平台底座不承载创作经验,豁免。 if compounding == "none" and lifecycle != "platform": advisories.append( _issue( "compounding_not_wired", "未接入经验复利:方法与判断只活在 SKILL.md 与 references/,用过之后不留下可被后续消费的资产", path=skill_path, severity="major", lifecycle=lifecycle, ) ) scripts_dir = root / SKILLS_ROOT / str(entry.get("directory", "")) / "scripts" if scripts_dir.is_dir(): try: names = sorted(item.name for item in scripts_dir.iterdir() if item.is_file()) except OSError: names = [] tests = [name for name in names if _SCRIPT_TEST_PATTERN.match(name)] implementations = [ name for name in names if name.endswith(".py") and name not in tests ] for name in tests: advisories.append( _issue( "scripts_test_file", "scripts/ 内存放测试文件,应移入 tests/skills//", path=f"{skill_path.rsplit('/', 1)[0]}/scripts/{name}", severity="major", ) ) if not implementations: others = [name for name in names if name not in tests] message = ( f"scripts/ 只有非可执行文件({'、'.join(others)}),清单与模板类内容应放 references/" if others else "scripts/ 目录为空,应补运行时实现或删除该目录" ) advisories.append( _issue( "scripts_without_implementation", message, path=f"{skill_path.rsplit('/', 1)[0]}/scripts", severity="major", files=others, ) ) advisories.extend(_cross_skill_advisories(root, entries, classes)) advisories.extend(_exemplar_advisories(root, entries)) return advisories def _duplicate_name_issues(entries: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: by_name: dict[str, list[str]] = {} for entry in entries: name = entry.get("name") if isinstance(name, str) and name: by_name.setdefault(name, []).append(str(entry["skill_path"])) issues: list[dict[str, Any]] = [] for name in sorted(by_name): paths = sorted(by_name[name]) if len(paths) > 1: issues.append( _issue( "duplicate_name", "多个 Skill 使用相同的 frontmatter name", path=paths[0], name=name, paths=paths, ) ) return issues def _summarize( issues: Sequence[dict[str, Any]], advisories: Sequence[dict[str, Any]], skill_count: int, *, ok: bool, ) -> dict[str, Any]: counts = Counter(str(issue["code"]) for issue in issues) advisory_counts = Counter(str(item["code"]) for item in advisories) severities = Counter(str(item["severity"]) for item in advisories) return { "status": "passed" if ok else "failed", "skills_scanned": skill_count, "issue_count": len(issues), "issue_codes": {code: counts[code] for code in sorted(counts)}, "advisory_count": len(advisories), "advisory_codes": {code: advisory_counts[code] for code in sorted(advisory_counts)}, "advisory_severities": {name: severities[name] for name in sorted(severities)}, } def audit_skills( root: str | Path = ".", manifest: str | Path | None = None, *, strict: bool = False, ) -> dict[str, Any]: """审计 Skill 文档与 manifest,并返回 JSON 可序列化报告。 阻断级问题进 ``issues`` 并决定退出码;严重与一般级发现进 ``advisories``, 只有 ``strict=True`` 时才一并阻断。 """ root_path = _resolve_root(root) manifest_path = _resolve_manifest_path(root_path, manifest) report: dict[str, Any] = { "schema_version": SCHEMA_VERSION, "root": str(root_path), "skills_root": SKILLS_ROOT.as_posix(), "manifest": { "path": _relative_path(manifest_path, root_path), "loaded": False, }, "skills_scanned": 0, "skills": [], "issues": [], "advisories": [], } if not root_path.exists(): report["issues"] = [ _issue("root_missing", "审计根目录不存在", path=".") ] elif not root_path.is_dir(): report["issues"] = [ _issue("root_not_directory", "审计根路径不是目录", path=".") ] else: entries, issues = _scan_skill_directories(root_path) issues.extend(_duplicate_name_issues(entries)) manifest_info, manifest_issues = _validate_manifest( root_path, entries, manifest_path ) issues.extend(manifest_issues) report["manifest"] = manifest_info report["skills"] = entries report["skills_scanned"] = len(entries) report["issues"] = issues findings = _quality_advisories( root_path, entries, _load_manifest_classes(manifest_path, root_path) ) issues.extend( item for item in findings if item["code"] in _BLOCKING_ADVISORY_CODES ) report["advisories"] = [ item for item in findings if item["code"] not in _BLOCKING_ADVISORY_CODES ] report["issue_count"] = len(report["issues"]) report["advisory_count"] = len(report["advisories"]) report["ok"] = not report["issues"] and not (strict and report["advisories"]) report["status"] = "passed" if report["ok"] else "failed" report["strict"] = strict report["summary"] = _summarize( report["issues"], report["advisories"], report["skills_scanned"], ok=report["ok"] ) return report def human_summary(report: dict[str, Any]) -> str: """把结构化报告渲染为简洁的人读摘要。""" status = "通过" if report["ok"] else "失败" lines = [ f"Skill 静态审计:{status}", f"扫描 Skill:{report['skills_scanned']} 个;" f"阻断问题:{report['issue_count']} 个;" f"质量发现:{report.get('advisory_count', 0)} 个", ] def render(items: Sequence[dict[str, Any]]) -> None: for item in items: location = str(item.get("path", "")) if item.get("line") is not None: location += f":{item['line']}" lines.append(f"- [{item['severity']}] {location} [{item['code']}] {item['message']}") if report["issues"]: lines.append("阻断问题:") render(report["issues"]) else: lines.append("未发现分类、frontmatter、目录命名、空文件或开发测试污染的阻断问题。") if report.get("advisories"): lines.append("质量发现(默认不阻断,评分口径见 specs/skill-quality-rubric.md):") render(report["advisories"]) return "\n".join(lines) def _build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( description="只读扫描项目运行时 .agent/skills/*/SKILL.md 的静态卫生审计器。" ) parser.add_argument( "--root", default=".", help="项目根目录,默认为当前目录", ) parser.add_argument( "--manifest", default=None, help="Skill manifest 路径;相对路径按项目根目录解析,默认 harness/manifests/skills.json", ) parser.add_argument( "--json", action="store_true", help="输出结构化 JSON 报告", ) parser.add_argument( "--quiet", action="store_true", help="不输出人读摘要;与 --json 同用时仍输出 JSON", ) parser.add_argument( "--strict", action="store_true", help="把严重与一般级质量发现一并计入退出码;存量清理完成后应长期开启", ) return parser def main(argv: Optional[Sequence[str]] = None) -> int: args = _build_parser().parse_args(argv) report = audit_skills(args.root, args.manifest, strict=args.strict) if args.json: print(json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True)) elif not args.quiet: print(human_summary(report)) return 0 if report["ok"] else 1 if __name__ == "__main__": sys.exit(main())