283 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""生成与机械校验 Skill 发现总索引 `.agent/skills/_index.md`。
发现合同(见 AGENTS.md §3):任何 agent(不限宿主)读 AGENTS.md 后必须读总索引,
按 `skill_file` 读取所需能力的 `SKILL.md`;不依赖任何 coding agent 的 skill 自动发现,
物理目录位置完全取自 manifest 的 `skill_path`,允许在挂载面内嵌套。
索引只登记三字段:
- `skill_name`:目录名,即调用名;
- `skill_file`:合同文件路径(来自 `muse/lifecycle/quality/harness/manifests/skills.json` 的 `skill_path`);
- `skill_description`:适用与边界描述,与 SKILL.md frontmatter 的 `description` 逐字一致
(frontmatter 是 SoT,索引是生成物)。
生命周期分组取自 `muse/lifecycle/quality/harness/manifests/skills.json` 的 `lifecycle` 字段;取值定义见
`muse/lifecycle/quality/harness/specs/skill-quality-rubric.md`。
用法:
.venv/bin/python harness/skills_index.py --check # 校验索引与磁盘、manifest 一致(默认)
.venv/bin/python harness/skills_index.py --write # 按磁盘与 manifest 重新生成索引
"""
from __future__ import annotations
import argparse
import difflib
from pathlib import Path
from typing import Optional, Sequence
import yaml
# 生命周期域 → 索引分节标题;顺序即索引分节顺序,标题沿用既有中文域名。
LIFECYCLE_SECTIONS: tuple[tuple[str, str], ...] = (
("platform", "0 平台底座"),
("ingest", "1 素材导入与拆解"),
("knowledge", "2 知识与上下文供给"),
("concept", "3 概念与前期设计"),
("planning", "4 结构与规划"),
("writing", "5 正文写作与呈现"),
("review", "6 检测、评分与诊断"),
("humanization", "7 去 AI 味与人感"),
("sovereignty", "8 候选主权与落库"),
)
INDEX_RELPATH = Path(".agent") / "skills" / "_index.md"
BUSINESS_INDEX_RELPATH = Path("muse") / "_skills_index.md"
MANIFEST_RELPATH = Path("muse") / "lifecycle" / "quality" / "harness" / "manifests" / "skills.json"
HEADER = """# Skill 方法发现总索引
> **发现合同**:任何 agent(不限宿主)读 [`AGENTS.md`](../../AGENTS.md) 后必须读本索引,\
需要某项能力时按 `skill_file` 读对应 `SKILL.md`。发现只靠 AGENTS.md → 本索引 → SKILL.md \
的渐进披露,不依赖任何 coding agent 的 skill 自动发现;物理目录位置完全取自 manifest 的 `skill_path`,允许嵌套。
本索引只登记三字段:`skill_name`(目录名,即调用名)、`skill_file`(合同文件路径)、\
`skill_description`(适用与边界描述,与 SKILL.md frontmatter 逐字一致,frontmatter 是 SoT)。\
分类字段(`lifecycle` / `invocation` / `side_effects` / `compounding`)逐个登记在 \
[`muse/lifecycle/quality/harness/manifests/skills.json`](../../muse/lifecycle/quality/harness/manifests/skills.json),由 \
`muse/lifecycle/quality/harness/skill_harness.py` 机械校验,不在本索引重复。
本文件由 `muse/lifecycle/quality/harness/skills_index.py --write` 生成,手改会被覆盖;一致性由 `--check` 与 \
`tests/architecture/test_skills_index.py` 机械把关。按生命周期分域,共 {count} 个 skill。
"""
BUSINESS_HEADER = """# Muse 编排 Skill 索引
> **业务发现合同**:主代理与业务编排按本索引的 `skill_file` 读取对应 `SKILL.md`。编排 Skill 不进入 Agent 框架的运行期 catalog;物理位置完全取自 manifest 的 `skill_path`。
本索引只登记三字段:`skill_name`、`skill_file`、`skill_description`。分类字段与责任方仍以 [`lifecycle/quality/harness/manifests/skills.json`](lifecycle/quality/harness/manifests/skills.json) 为准,由 `muse/lifecycle/quality/harness` 机械校验。
本文件由 `muse/lifecycle/quality/harness/skills_index.py --write` 生成,手改会被覆盖;按生命周期分域,共 {count} 个编排 skill。
"""
def _load_manifest(root: Path) -> dict[str, dict[str, str]]:
"""读取 skills.json,返回 name → 条目;仅接受与磁盘一致的规范路径。"""
import json
manifest_path = root / MANIFEST_RELPATH
if not manifest_path.exists():
legacy = root / "harness" / "manifests" / "skills.json"
if legacy.exists():
manifest_path = legacy
data = json.loads(manifest_path.read_text(encoding="utf-8"))
entries = data["skills"] if isinstance(data, dict) and "skills" in data else data
result: dict[str, dict[str, str]] = {}
for entry in entries:
result[entry["name"]] = entry
return result
def _read_description(skill_md: Path) -> str:
"""取 SKILL.md frontmatter 的 description 并折叠换行为单行。"""
text = skill_md.read_text(encoding="utf-8")
if not text.startswith("---\n"):
raise ValueError(f"{skill_md}: 缺少 frontmatter")
end = text.index("\n---\n", 4)
fields = yaml.safe_load(text[4:end])
description = (fields or {}).get("description")
if not isinstance(description, str) or not description.strip():
raise ValueError(f"{skill_md}: frontmatter 缺少非空 description")
# YAML 多行块按行折叠;中文行间接缝不需要空格。
return "".join(line.strip() for line in description.strip().splitlines())
def build_index_text(root: Path) -> tuple[str, int]:
"""按 catalog 磁盘 + skills.json 组装索引全文。"""
skills_root = root / ".agent" / "skills"
all_manifest = _load_manifest(root)
manifest = {
name: entry
for name, entry in all_manifest.items()
if entry.get("invocation") == "model_routed"
}
# 运行期 catalog 只收 model_routed;物理目录允许规划/写作/诊断等中间分组。
disk_by_path: dict[str, tuple[str, str]] = {}
for skill_md in sorted(skills_root.rglob("SKILL.md")):
relative = skill_md.relative_to(root).as_posix()
disk_by_path[relative] = (skill_md.parent.name, _read_description(skill_md))
disk: dict[str, tuple[str, str]] = {}
for name, entry in manifest.items():
declared_path = str(entry.get("skill_path", ""))
disk_entry = disk_by_path.get(declared_path)
if disk_entry is not None:
disk[name] = (declared_path, disk_entry[1])
only_manifest = sorted(set(manifest) - set(disk))
if only_manifest:
raise ValueError(
"skills.json 登记的 model_routed Skill 不在挂载面磁盘上:"
+ ", ".join(only_manifest)
)
catalog_paths = {entry.get("skill_path") for entry in manifest.values()}
unregistered_catalog = sorted(
path
for path, (name, _description) in disk_by_path.items()
if path.startswith(".agent/skills/") and path not in catalog_paths
and name not in all_manifest
)
if unregistered_catalog:
raise ValueError(
"挂载面存在未登记 Skill:" + ", ".join(unregistered_catalog)
)
for name, entry in manifest.items():
expected = disk[name][0]
if entry.get("skill_path") != expected:
raise ValueError(
f"{name}: skills.json skill_path={entry.get('skill_path')!r},"
f"期望 {expected!r}"
)
lifecycle = entry.get("lifecycle")
if lifecycle not in dict(LIFECYCLE_SECTIONS):
raise ValueError(f"{name}: 未知 lifecycle={lifecycle!r}")
lines: list[str] = [HEADER.format(count=len(disk)).rstrip()]
for lifecycle, title in LIFECYCLE_SECTIONS:
names = sorted(
name for name, entry in manifest.items() if entry["lifecycle"] == lifecycle
)
if not names:
continue
lines.append("")
lines.append(f"## {title}")
lines.append("")
lines.append("| skill_name | skill_file | skill_description |")
lines.append("|---|---|---|")
for name in names:
lines.append(
f"| {name} | `{disk[name][0]}` | {disk[name][1]} |"
)
lines.append("")
return "\n".join(lines), len(disk)
def build_business_index_text(root: Path) -> tuple[str, int]:
"""按 manifest 生成 Muse 编排 Skill 索引。"""
manifest = {
name: entry
for name, entry in _load_manifest(root).items()
if entry.get("invocation") == "orchestrated"
}
disk: dict[str, tuple[str, str]] = {}
for name, entry in manifest.items():
declared_path = str(entry.get("skill_path", ""))
if not declared_path.startswith("muse/"):
raise ValueError(f"orchestrated Skill 不在 muse/ 业务树:{name}")
skill_md = root / declared_path
if not skill_md.is_file():
raise ValueError(f"编排 Skill 路径不存在:{declared_path}")
disk[name] = (declared_path, _read_description(skill_md))
if skill_md.parent.name != name:
raise ValueError(f"Skill 目录名与 name 不一致:{name} -> {declared_path}")
lines: list[str] = [BUSINESS_HEADER.format(count=len(disk)).rstrip()]
for lifecycle, title in LIFECYCLE_SECTIONS:
names = sorted(
name for name, entry in manifest.items() if entry["lifecycle"] == lifecycle
)
if not names:
continue
lines.extend(["", f"## {title}", "", "| skill_name | skill_file | skill_description |", "|---|---|---|"])
for name in names:
lines.append(f"| {name} | `{disk[name][0]}` | {disk[name][1]} |")
lines.append("")
return "\n".join(lines), len(disk)
def check(root: Path) -> Optional[str]:
"""校验两份索引与 manifest、磁盘一致;一致返回 None,否则返回差异文本。"""
expected_catalog, _ = build_index_text(root)
expected_business, _ = build_business_index_text(root)
results: list[str] = []
for relative_path, expected in (
(INDEX_RELPATH, expected_catalog),
(BUSINESS_INDEX_RELPATH, expected_business),
):
index_path = root / relative_path
actual = index_path.read_text(encoding="utf-8") if index_path.exists() else ""
if actual == expected:
continue
results.append(
"".join(
difflib.unified_diff(
actual.splitlines(keepends=True),
expected.splitlines(keepends=True),
fromfile=str(relative_path) + "(现状)",
tofile=str(relative_path) + "(按磁盘与 skills.json 重建)",
n=2,
)
)
)
return "\n".join(results) or None
def main(argv: Optional[Sequence[str]] = None) -> int:
parser = argparse.ArgumentParser(
description="生成与机械校验两份 Skill 发现索引(三字段)。"
)
parser.add_argument("--root", default=".", help="项目根目录,默认为当前目录")
group = parser.add_mutually_exclusive_group()
group.add_argument(
"--check",
action="store_true",
help="校验索引与磁盘、skills.json 一致(默认模式)",
)
group.add_argument(
"--write", action="store_true", help="按磁盘与 skills.json 重新生成索引"
)
args = parser.parse_args(argv)
root = Path(args.root).resolve()
try:
if args.write:
catalog_text, catalog_count = build_index_text(root)
business_text, business_count = build_business_index_text(root)
(root / INDEX_RELPATH).write_text(catalog_text, encoding="utf-8")
(root / BUSINESS_INDEX_RELPATH).parent.mkdir(parents=True, exist_ok=True)
(root / BUSINESS_INDEX_RELPATH).write_text(business_text, encoding="utf-8")
print(
f"已重新生成 {INDEX_RELPATH}({catalog_count} 个方法 skill)与 "
f"{BUSINESS_INDEX_RELPATH}({business_count} 个编排 skill)"
)
return 0
diff = check(root)
except (ValueError, KeyError, FileNotFoundError) as error:
print(f"skills_index 失败:{error}")
return 2
if diff is None:
print("skills_index 一致:索引与磁盘、skills.json 相符")
return 0
print("skills_index 漂移:请用 muse/lifecycle/quality/harness/skills_index.py --write 重新生成")
print(diff)
return 1
if __name__ == "__main__":
raise SystemExit(main())