#!/usr/bin/env python3 """对项目运行时 Skill 文档执行只读静态审计。 本模块只读取 `.claude/skills/*/SKILL.md`,不连接数据库、网络或模型,也不修改 工作树。机器调用使用 :func:`audit_skills`,命令行入口同时支持人读摘要和 JSON。 """ from __future__ import annotations import argparse import json import re import sys from collections import Counter from pathlib import Path from typing import Any, Iterable, Optional, Sequence SCHEMA_VERSION = 1 SKILLS_ROOT = Path(".claude") / "skills" DEFAULT_MANIFEST_PATH = Path("harness") / "manifests" / "skills.json" _FRONTMATTER_NAME = re.compile(r"^\s*name\s*:\s*(.*?)\s*$") _COVERAGE_TOOL_PATTERN = re.compile( r"(?:\bcoverage\.py\b|\bpytest-cov\b|" r"(?=|>|:|:)?\s*" r"\d+(?:\.\d+)?\s*[%%]", re.IGNORECASE, ) _BUSINESS_COVERAGE_CONTEXT_PATTERN = re.compile( r"(?:细纲|事件|硬约束|软约束|伏笔|实体|设定|情节|场景|知识|事实|正文|" r"作品|业务|候选|规划|大纲|章)[^。\n]{0,24}覆盖率" r"|覆盖率[^。\n]{0,24}(?:细纲|事件|硬约束|软约束|伏笔|实体|设定|情节|" r"场景|知识|事实|正文|作品|业务|候选|规划|大纲|章)", re.IGNORECASE, ) # 这些规则只针对运行时 Skill 文档中的明显开发验证残留;业务正文中的普通“检查” # 等词不在扫描范围内。规则保持确定性,方便报告中的 code 作为机械门禁输入。 _DEVELOPMENT_HEADING_PATTERN = re.compile(r"^\s*(#{1,6})\s*(.*?)\s*$") _DEVELOPMENT_SECTION_TITLE_PATTERN = re.compile( r"(?:自测|测试|开发验证|离线验证|开发测试|离线测试)", re.IGNORECASE, ) _DEVELOPMENT_COMMAND_PATTERN = re.compile( r"(? Path: """把审计根目录解析为绝对路径,但不要求它已经存在。""" path = Path(root).expanduser() if not path.is_absolute(): path = Path.cwd() / path return path.resolve() def _relative_path(path: Path, root: Path) -> str: """返回报告中的稳定 POSIX 相对路径。""" try: return path.resolve().relative_to(root).as_posix() except ValueError: return path.resolve().as_posix() def _issue( code: str, message: str, *, path: Optional[str] = None, line: Optional[int] = None, **details: Any, ) -> dict[str, Any]: result: dict[str, Any] = { "code": code, "severity": "error", "message": message, } if path is not None: result["path"] = path if line is not None: result["line"] = line result.update(details) return result def _unquote_scalar(value: str) -> str: """解析 Skill frontmatter 中足够支持 name 的简单标量。""" value = value.strip() if len(value) >= 2 and value[0] == value[-1] == '"': try: parsed = json.loads(value) except json.JSONDecodeError: return value[1:-1] return parsed if isinstance(parsed, str) else value if len(value) >= 2 and value[0] == value[-1] == "'": return value[1:-1].replace("''", "'") # 允许常见的 YAML 行尾注释,不引入 YAML 依赖。 return re.sub(r"\s+#.*$", "", value).strip() def _parse_frontmatter(text: str) -> tuple[Optional[dict[str, str]], Optional[int], Optional[str]]: """读取 frontmatter 的简单键值视图。 返回值为 ``(fields, closing_line, error_code)``。该解析器只需要识别 name, 不试图替代完整 YAML 解析器;不合法或缺失的边界会明确进入失败报告。 """ lines = text.splitlines() if not lines or lines[0].lstrip("\ufeff").strip() != "---": return None, None, "frontmatter_missing" closing_line: Optional[int] = None for index in range(1, len(lines)): if lines[index].strip() in {"---", "..."}: closing_line = index + 1 break if closing_line is None: return None, None, "frontmatter_unclosed" fields: dict[str, str] = {} for line in lines[1 : closing_line - 1]: match = _FRONTMATTER_NAME.match(line) if match: fields["name"] = _unquote_scalar(match.group(1)) return fields, closing_line, None def _snippet(line: str) -> str: compact = line.strip() return compact if len(compact) <= 200 else compact[:197] + "..." def _is_development_coverage_reference(line: str) -> bool: """只把明确落在开发测试语境中的 coverage 文字判为污染。""" if _COVERAGE_TOOL_PATTERN.search(line) or _EXPLICIT_TEST_COVERAGE_PATTERN.search(line): return True if not _COVERAGE_PERCENT_PATTERN.search(line): return False return not _BUSINESS_COVERAGE_CONTEXT_PATTERN.search(line) def _inspect_skill_file(path: Path, root: Path) -> tuple[dict[str, Any], list[dict[str, Any]]]: relative = _relative_path(path, root) entry: dict[str, Any] = { "directory": path.parent.name, "skill_path": relative, "name": None, "frontmatter_present": False, } issues: list[dict[str, Any]] = [] try: text = path.read_text(encoding="utf-8") except (OSError, UnicodeError) as exc: issues.append( _issue( "skill_file_unreadable", "无法读取 SKILL.md", path=relative, error=str(exc), ) ) return entry, issues if not text.strip(): issues.append( _issue( "empty_skill_file", "SKILL.md 为空", path=relative, line=1, ) ) return entry, issues fields, closing_line, frontmatter_error = _parse_frontmatter(text) if frontmatter_error is not None: line = 1 if frontmatter_error == "frontmatter_missing" else max(1, len(text.splitlines())) messages = { "frontmatter_missing": "缺少以 --- 开始的 frontmatter", "frontmatter_unclosed": "frontmatter 未闭合", } issues.append( _issue( frontmatter_error, messages[frontmatter_error], path=relative, line=line, ) ) else: entry["frontmatter_present"] = True name = fields.get("name") if fields is not None else None entry["name"] = name or None if not name: issues.append( _issue( "frontmatter_name_missing", "frontmatter 缺少非空 name", path=relative, line=1, ) ) elif name != path.parent.name: issues.append( _issue( "name_directory_mismatch", "frontmatter name 与 Skill 目录名不一致", path=relative, line=2, expected=path.parent.name, actual=name, ) ) development_section_level: Optional[int] = None for line_number, line in enumerate(text.splitlines(), start=1): heading_match = _DEVELOPMENT_HEADING_PATTERN.match(line) if heading_match: heading_level = len(heading_match.group(1)) heading_title = re.sub(r"\s+#+\s*$", "", heading_match.group(2)).strip() if ( development_section_level is not None and heading_level <= development_section_level ): development_section_level = None if ( 2 <= heading_level <= 6 and _DEVELOPMENT_SECTION_TITLE_PATTERN.search(heading_title) ): development_section_level = heading_level issues.append( _issue( "development_test_heading", "发现开发测试章节标题", path=relative, line=line_number, snippet=_snippet(line), ) ) for code, pattern, message in _POLLUTION_RULES: if pattern.search(line): issues.append( _issue( code, message, path=relative, line=line_number, snippet=_snippet(line), ) ) if ( development_section_level is not None and _DEVELOPMENT_COMMAND_PATTERN.search(line) ): issues.append( _issue( "development_test_command_reference", "发现开发测试章节中的合同测试命令", path=relative, line=line_number, snippet=_snippet(line), ) ) if _is_development_coverage_reference(line): issues.append( _issue( "coverage_reference", "发现开发测试语境中的覆盖率声明或工具名", path=relative, line=line_number, snippet=_snippet(line), ) ) return entry, issues def _scan_skill_directories(root: Path) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: skills_root = root / SKILLS_ROOT relative_skills_root = _relative_path(skills_root, root) entries: list[dict[str, Any]] = [] issues: list[dict[str, Any]] = [] if not skills_root.exists(): issues.append( _issue( "skills_directory_missing", "运行时 Skill 目录不存在", path=relative_skills_root, ) ) return entries, issues if not skills_root.is_dir(): issues.append( _issue( "skills_directory_not_directory", "运行时 Skill 路径不是目录", path=relative_skills_root, ) ) return entries, issues try: children = sorted(skills_root.iterdir(), key=lambda item: item.name) except OSError as exc: issues.append( _issue( "skills_directory_unreadable", "无法列出运行时 Skill 目录", path=relative_skills_root, error=str(exc), ) ) return entries, issues directories = [child for child in children if child.is_dir()] if not directories: issues.append( _issue( "no_skills_found", "运行时 Skill 目录中未发现 Skill 子目录", path=relative_skills_root, ) ) return entries, issues for directory in directories: skill_file = directory / "SKILL.md" relative_skill_file = _relative_path(skill_file, root) disk_entry: dict[str, Any] = { "directory": directory.name, "skill_path": relative_skill_file, "name": None, "frontmatter_present": False, } if not skill_file.exists(): entries.append(disk_entry) issues.append( _issue( "missing_skill_file", "Skill 目录缺少 SKILL.md", path=relative_skill_file, ) ) continue if not skill_file.is_file(): entries.append(disk_entry) issues.append( _issue( "skill_file_not_file", "SKILL.md 路径不是普通文件", path=relative_skill_file, ) ) continue entry, file_issues = _inspect_skill_file(skill_file, root) entries.append(entry) issues.extend(file_issues) return entries, issues def _resolve_manifest_path(root: Path, manifest: str | Path | None) -> Path: path = DEFAULT_MANIFEST_PATH if manifest is None else Path(manifest).expanduser() if not path.is_absolute(): path = root / path return path.resolve() def _normalise_manifest_skill_path( root: Path, value: str ) -> tuple[Optional[str], Optional[Path]]: declared = Path(value) if declared.is_absolute(): return None, None candidate = (root / declared).resolve() try: relative = candidate.relative_to(root).as_posix() except ValueError: return None, None return relative, candidate def _validate_manifest( root: Path, entries: Sequence[dict[str, Any]], manifest_path: Path, ) -> tuple[dict[str, Any], list[dict[str, Any]]]: """验证 Skill 登记表,并要求它与磁盘目录一一对应。""" relative_manifest = _relative_path(manifest_path, root) metadata: dict[str, Any] = { "path": relative_manifest, "loaded": False, "skills_declared": 0, } issues: list[dict[str, Any]] = [] if not manifest_path.exists(): issues.append( _issue( "manifest_missing", "Skill manifest 不存在", path=relative_manifest, ) ) return metadata, issues if not manifest_path.is_file(): issues.append( _issue( "manifest_not_file", "Skill manifest 路径不是普通文件", path=relative_manifest, ) ) return metadata, issues try: manifest_text = manifest_path.read_text(encoding="utf-8") manifest = json.loads(manifest_text) except UnicodeError as exc: issues.append( _issue( "manifest_unreadable", "无法按 UTF-8 读取 Skill manifest", path=relative_manifest, error=str(exc), ) ) return metadata, issues except OSError as exc: issues.append( _issue( "manifest_unreadable", "无法读取 Skill manifest", path=relative_manifest, error=str(exc), ) ) return metadata, issues except json.JSONDecodeError as exc: issues.append( _issue( "manifest_invalid_json", "Skill manifest 不是合法 JSON", path=relative_manifest, line=exc.lineno, column=exc.colno, error=exc.msg, ) ) return metadata, issues metadata["loaded"] = True if not isinstance(manifest, dict): issues.append( _issue( "manifest_invalid_structure", "Skill manifest 顶层必须是对象", path=relative_manifest, ) ) return metadata, issues raw_skills = manifest.get("skills") if not isinstance(raw_skills, list): issues.append( _issue( "manifest_skills_invalid", "Skill manifest.skills 必须是数组", path=relative_manifest, ) ) return metadata, issues metadata["skills_declared"] = len(raw_skills) disk_paths = { str(entry["skill_path"]) for entry in entries if isinstance(entry.get("skill_path"), str) } declared_paths: set[str] = set() path_occurrences: dict[str, list[int]] = {} name_occurrences: dict[str, list[int]] = {} skills_root = (root / SKILLS_ROOT).resolve() for index, item in enumerate(raw_skills): if not isinstance(item, dict): issues.append( _issue( "manifest_entry_invalid", "Skill manifest 条目必须是对象", path=relative_manifest, entry_index=index, ) ) continue name = item.get("name") if ( not isinstance(name, str) or not name.strip() or "/" in name or "\\" in name or name in {".", ".."} ): issues.append( _issue( "manifest_name_invalid", "manifest 条目的 name 必须是非空目录名", path=relative_manifest, entry_index=index, actual=name, ) ) else: name_occurrences.setdefault(name, []).append(index) contract_owner = item.get("contract_owner") if not isinstance(contract_owner, str) or not contract_owner.strip(): issues.append( _issue( "manifest_contract_owner_invalid", "manifest 条目的 contract_owner 必须非空", path=relative_manifest, entry_index=index, ) ) collaborations = item.get("collaborates_with") if ( not isinstance(collaborations, list) or any(not isinstance(value, str) or not value.strip() for value in collaborations) ): issues.append( _issue( "manifest_collaborates_with_invalid", "manifest 条目的 collaborates_with 必须是非空字符串数组", path=relative_manifest, entry_index=index, ) ) declared_skill_path = item.get("skill_path") if not isinstance(declared_skill_path, str) or not declared_skill_path.strip(): issues.append( _issue( "manifest_skill_path_invalid", "manifest 条目的 skill_path 必须是非空相对路径", path=relative_manifest, entry_index=index, actual=declared_skill_path, ) ) continue normalised_path, candidate = _normalise_manifest_skill_path( root, declared_skill_path ) if normalised_path is None or candidate is None: issues.append( _issue( "manifest_skill_path_invalid", "manifest 条目的 skill_path 必须位于项目根目录内", path=relative_manifest, entry_index=index, actual=declared_skill_path, ) ) continue declared_paths.add(normalised_path) path_occurrences.setdefault(normalised_path, []).append(index) if candidate != candidate.parent / "SKILL.md" or not candidate.is_relative_to(skills_root): issues.append( _issue( "manifest_skill_path_invalid", "manifest 条目的 skill_path 必须指向 .claude/skills/*/SKILL.md", path=normalised_path, entry_index=index, ) ) if not candidate.exists(): issues.append( _issue( "manifest_skill_path_missing", "manifest 登记的 Skill 路径不存在", path=normalised_path, entry_index=index, ) ) elif not candidate.is_file(): issues.append( _issue( "manifest_skill_path_not_file", "manifest 登记的 Skill 路径不是普通文件", path=normalised_path, entry_index=index, ) ) directory_name = candidate.parent.name if isinstance(name, str) and name.strip() and name != directory_name: issues.append( _issue( "manifest_name_directory_mismatch", "manifest name 与 skill_path 的目录名不一致", path=normalised_path, entry_index=index, expected=directory_name, actual=name, ) ) if isinstance(name, str) and name.strip(): expected_path = (root / SKILLS_ROOT / name / "SKILL.md").resolve() expected_relative = _relative_path(expected_path, root) if normalised_path != expected_relative: issues.append( _issue( "manifest_skill_path_mismatch", "manifest skill_path 与 name 对应的 Skill 路径不一致", path=normalised_path, entry_index=index, expected=expected_relative, actual=normalised_path, ) ) for path, indexes in sorted(path_occurrences.items()): if len(indexes) > 1: issues.append( _issue( "manifest_duplicate_skill_path", "manifest 重复登记同一个 Skill 路径", path=path, entry_indexes=indexes, ) ) for name, indexes in sorted(name_occurrences.items()): if len(indexes) > 1: issues.append( _issue( "manifest_duplicate_name", "manifest 重复登记同一个 Skill name", path=relative_manifest, name=name, entry_indexes=indexes, ) ) for path in sorted(disk_paths - declared_paths): issues.append( _issue( "manifest_skill_missing", "磁盘 Skill 未在 manifest 中登记", path=path, ) ) for path in sorted(declared_paths - disk_paths): issues.append( _issue( "manifest_skill_extra", "manifest 登记了磁盘中不存在的额外 Skill", path=path, ) ) return metadata, issues def _duplicate_name_issues(entries: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: by_name: dict[str, list[str]] = {} for entry in entries: name = entry.get("name") if isinstance(name, str) and name: by_name.setdefault(name, []).append(str(entry["skill_path"])) issues: list[dict[str, Any]] = [] for name in sorted(by_name): paths = sorted(by_name[name]) if len(paths) > 1: issues.append( _issue( "duplicate_name", "多个 Skill 使用相同的 frontmatter name", path=paths[0], name=name, paths=paths, ) ) return issues def _summarize(issues: Sequence[dict[str, Any]], skill_count: int) -> dict[str, Any]: counts = Counter(str(issue["code"]) for issue in issues) return { "status": "passed" if not issues else "failed", "skills_scanned": skill_count, "issue_count": len(issues), "issue_codes": {code: counts[code] for code in sorted(counts)}, } def audit_skills( root: str | Path = ".", manifest: str | Path | None = None ) -> dict[str, Any]: """审计 Skill 文档与 manifest,并返回 JSON 可序列化报告。""" root_path = _resolve_root(root) manifest_path = _resolve_manifest_path(root_path, manifest) report: dict[str, Any] = { "schema_version": SCHEMA_VERSION, "root": str(root_path), "skills_root": SKILLS_ROOT.as_posix(), "manifest": { "path": _relative_path(manifest_path, root_path), "loaded": False, }, "skills_scanned": 0, "skills": [], "issues": [], } if not root_path.exists(): report["issues"] = [ _issue("root_missing", "审计根目录不存在", path=".") ] elif not root_path.is_dir(): report["issues"] = [ _issue("root_not_directory", "审计根路径不是目录", path=".") ] else: entries, issues = _scan_skill_directories(root_path) issues.extend(_duplicate_name_issues(entries)) manifest_info, manifest_issues = _validate_manifest( root_path, entries, manifest_path ) issues.extend(manifest_issues) report["manifest"] = manifest_info report["skills"] = entries report["skills_scanned"] = len(entries) report["issues"] = issues report["summary"] = _summarize(report["issues"], report["skills_scanned"]) report["issue_count"] = len(report["issues"]) report["ok"] = not report["issues"] report["status"] = "passed" if report["ok"] else "failed" return report def human_summary(report: dict[str, Any]) -> str: """把结构化报告渲染为简洁的人读摘要。""" status = "通过" if report["ok"] else "失败" lines = [ f"Skill 静态审计:{status}", f"扫描 Skill:{report['skills_scanned']} 个;问题:{report['issue_count']} 个", ] issues = report["issues"] if not issues: lines.append("未发现 frontmatter、目录命名、空文件或开发测试污染问题。") return "\n".join(lines) lines.append("问题明细:") for issue in issues: location = str(issue.get("path", "")) if issue.get("line") is not None: location += f":{issue['line']}" lines.append(f"- {location} [{issue['code']}] {issue['message']}") return "\n".join(lines) def _build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( description="只读扫描项目运行时 .claude/skills/*/SKILL.md 的静态卫生审计器。" ) parser.add_argument( "--root", default=".", help="项目根目录,默认为当前目录", ) parser.add_argument( "--manifest", default=None, help="Skill manifest 路径;相对路径按项目根目录解析,默认 harness/manifests/skills.json", ) parser.add_argument( "--json", action="store_true", help="输出结构化 JSON 报告", ) parser.add_argument( "--quiet", action="store_true", help="不输出人读摘要;与 --json 同用时仍输出 JSON", ) return parser def main(argv: Optional[Sequence[str]] = None) -> int: args = _build_parser().parse_args(argv) report = audit_skills(args.root, args.manifest) if args.json: print(json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True)) elif not args.quiet: print(human_summary(report)) return 0 if report["ok"] else 1 if __name__ == "__main__": sys.exit(main())