muse-agent-example/工具/维护索引.py
zizi 8a44364539 工程(索引): 索引只维护会提交内容;修复独占锁内收集自锁
- 维护索引按 git 视角枚举(已跟踪 + 未跟踪且未忽略):被 .gitignore 忽略的运行痕迹与
  过程性检查报告不再进入目录派生或影响索引门禁;非 git 场景退回原磁盘枚举
- conftest 验证闸门在 `验证锁 --执行` 启动的会话内不再取共享锁,修复 `make 索引生成`
  在独占锁内收集用例时的自锁(原 180s 超时失败 → 现 6s 通过)
- .gitignore 登记过程性检查报告与工具会话缓存(.zcode/):可复跑,不入历史
- 新增两条测试:忽略内容不参与目录派生、独占锁内收集不自锁
2026-09-18 15:07:32 +08:00

602 lines
24 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""由源文件派生导航,并复用 pytest 收集的稳定用例身份。
显式 --写入 生成目录表;--检查 不写磁盘。技能说明来自 SKILL.md
frontmatter,作者导航语义随源文件保存,手写导读保留。历史逐文件设计
不再参与每日源码定位门禁。所有入口可用显式根目录在临时项目复现。
"""
from __future__ import annotations
import argparse
import ast
import json
import os
import re
import subprocess
import sys
import unicodedata
from fnmatch import fnmatch
from pathlib import Path
from typing import Any
import yaml
仓库根 = Path(__file__).resolve().parent.parent
_忽略目录片段 = frozenset(
{
".venv",
".venv-旧",
".git",
"node_modules",
"__pycache__",
".agents.local",
"研究依据",
"文件设计",
}
)
# 新版用例目录:旧测试树(tests/skills、tests/adapters、tests/architecture 英文目录等)
# 由旧环境运行,不进入新版用例身份检查
_新版用例目录 = ("单元", "契约", "集成", "架构", "迁移", "端到端", "真实调用")
# 按 git 工作树顶层缓存「会提交内容」,同一轮多次遍历不重复询问 git。
_仓库缓存: dict[Path, frozenset[Path] | None] = {}
def _在忽略目录(路径: Path, 根: Path) -> bool:
try:
相对 = 路径.relative_to(根)
except ValueError:
return True
return any(片段 in 相对.parts for 片段 in _忽略目录片段)
def _仓库顶层(根: Path) -> Path | None:
"""根所在 git 工作树的顶层目录;不在工作树内时返回 None。"""
try:
输出 = subprocess.run(
["git", "-C", str(根), "rev-parse", "--show-toplevel"],
capture_output=True,
text=True,
timeout=30,
check=True,
).stdout.strip()
except (OSError, subprocess.SubprocessError):
return None
return Path(输出).resolve() if 输出 else None
def _会提交内容(根: Path) -> frozenset[Path] | None:
"""根下会提交的路径(文件与其祖先目录);非 git 场景返回 None,调用方退回磁盘枚举。
索引只维护会提交的内容:被 .gitignore 忽略的运行痕迹、检查报告与缓存既不
进入派生,也不承担扫描成本。
"""
顶层 = _仓库顶层(根)
if 顶层 is None:
return None
if 顶层 not in _仓库缓存:
try:
输出 = subprocess.run(
[
"git",
"-C",
str(顶层),
"ls-files",
"-z",
"--cached",
"--others",
"--exclude-standard",
],
capture_output=True,
timeout=60,
check=True,
).stdout
except (OSError, subprocess.SubprocessError):
_仓库缓存[顶层] = None
return None
文件集: set[Path] = set()
for 名 in 输出.decode("utf-8", "surrogateescape").split("\0"):
路径 = 顶层 / 名
if 名 and 路径.is_file():
文件集.add(路径)
目录集 = {
祖先 for p in 文件集 for 祖先 in p.parents if 祖先 == 顶层 or 顶层 in 祖先.parents
}
_仓库缓存[顶层] = frozenset(文件集 | 目录集)
内容 = _仓库缓存[顶层]
if 内容 is None:
return None
本根 = 根.resolve()
return frozenset(p for p in 内容 if p == 本根 or 本根 in p.parents)
def _会提交条目(根: Path, 目录: Path) -> list[Path] | None:
"""目录下会提交的一级条目(文件与子目录);非 git 场景返回 None。"""
内容 = _会提交内容(根)
if 内容 is None:
return None
return sorted(p for p in 内容 if p.parent == 目录)
def _会提交子目录(根: Path, 目录: Path) -> list[Path]:
"""目录下会提交的一级子目录;非 git 场景退回磁盘枚举。"""
条目 = _会提交条目(根, 目录)
if 条目 is None:
return sorted(p for p in 目录.iterdir() if p.is_dir())
return [p for p in 条目 if p.is_dir()]
def 遍历文件(根: Path, 模式: str):
"""只遍历会提交的内容;忽略的运行痕迹不参与派生,也不承担扫描成本。"""
内容 = _会提交内容(根)
if 内容 is not None:
for 文件 in sorted(内容):
if 文件.is_file() and fnmatch(文件.name, 模式) and not _在忽略目录(文件, 根):
yield 文件
return
for 当前, 目录名, 文件名 in os.walk(根, followlinks=False):
目录名[:] = sorted(名 for 名 in 目录名 if 名 not in _忽略目录片段)
for 名 in sorted(文件名):
if fnmatch(名, 模式):
yield Path(当前) / 名
def 检查目录链接(根: Path) -> list[str]:
"""核对仓库内 目录.md 的表格相对链接;返回问题列表。"""
问题: list[str] = []
for 目录文件 in 遍历文件(根, "目录.md"):
if _在忽略目录(目录文件, 根):
continue
for 序, _, 原地址 in _表格行(目录文件.read_text(encoding="utf-8")):
地址 = 原地址.split("#")[0].strip()
if not 地址 or 地址 == "暂无" or 地址.startswith(("http://", "https://", "mailto:")):
continue
目标 = (目录文件.parent / 地址).resolve()
if not 目标.exists():
问题.append(f"{目录文件.relative_to(根)}:{序 + 1} 链接不存在:{地址}")
return 问题
def _读frontmatter(文件: Path) -> dict:
"""与源文件一起移动的声明;支持 YAML 折叠及引号,不解析正文。"""
文本 = 文件.read_text(encoding="utf-8")
匹配 = re.match(r"\A---\r?\n(.*?)\r?\n---(?:\r?\n|$)", 文本, re.S)
if 匹配 is None:
return {}
try:
声明 = yaml.safe_load(匹配[1])
except yaml.YAMLError as 错误:
raise ValueError(f"frontmatter 无法解析:{文件}") from 错误
return 声明 if isinstance(声明, dict) else {}
def _读frontmatter名(文件: Path) -> str | None:
名称 = _读frontmatter(文件).get("name")
return 名称 if isinstance(名称, str) else None
def _含中文(名称: str) -> bool:
return any("\u4e00" <= ch <= "\u9fff" for ch in 名称)
def 检查技能目录(根: Path) -> list[str]:
"""核对 .agent/skills 分组:每个技能有非空 SKILL.md、名称不重复、
frontmatter name 与目录一致、目录名全中文且规范化(NFC、无空白)。"""
技能根 = 根 / ".agent" / "skills"
if not 技能根.is_dir():
return ["技能目录不存在:.agent/skills(缺失即失败,不静默跳过)"]
问题: list[str] = []
已见: dict[str, str] = {}
for 分组 in _会提交子目录(根, 技能根):
for 技能目录 in _会提交子目录(根, 分组):
名称 = 技能目录.name
相对 = str(技能目录.relative_to(根))
技能文件 = 技能目录 / "SKILL.md"
if not 技能文件.is_file():
问题.append(f"技能缺少 SKILL.md:{相对}")
continue
正文 = 技能文件.read_text(encoding="utf-8").strip()
if not 正文 or 正文 == "---":
问题.append(f"SKILL.md 为空:{相对}")
continue
try:
front名 = _读frontmatter名(技能文件)
except ValueError as 错误:
问题.append(str(错误))
continue
if front名 is None:
问题.append(f"SKILL.md 缺少 frontmatter name:{相对}")
elif front名 != 名称:
问题.append(f"frontmatter name 与目录不一致:{相对}(name: {front名})")
if not _含中文(名称):
问题.append(f"技能名称必须全中文(政策 chinese_only):{相对}")
if 名称 != unicodedata.normalize("NFC", 名称):
问题.append(f"技能名称不是 NFC 规范形式:{相对}")
if any(ch.isspace() for ch in 名称):
问题.append(f"技能名称含空白:{相对}")
if 名称 in 已见:
问题.append(f"技能名称重复:{名称}({已见[名称]} 与 {相对})")
else:
已见[名称] = 相对
return 问题
_导航表头 = "| 名称 | 相对地址 | 内容描述 | 使用场景 | 使用要求 |"
_导航分隔 = "|------|----------|----------|----------|----------|"
_导航元信息 = re.compile(r"<!-- 导航元信息: (.*?) -->")
_表格链接 = re.compile(r"\[([^\]]+)\]\(([^)]+)\)")
def _表格行(文本: str) -> list[tuple[int, list[str], str]]:
结果 = []
for 序, 行 in enumerate(文本.splitlines()):
if not 行.startswith("|"):
continue
列 = [v.strip() for v in re.split(r"(?<!\\)\|", 行)[1:-1]]
if len(列) != 5 or 列[0] == "名称" or re.fullmatch(r"[- :]+", 列[0]):
continue
链接 = _表格链接.fullmatch(列[1])
结果.append((序, 列, 链接[2] if 链接 else 列[1]))
return 结果
def _导航声明(文件: Path) -> dict[str, str]:
if 文件.suffix != ".md" or not 文件.is_file():
return {}
匹配 = _导航元信息.findall(文件.read_text(encoding="utf-8"))
if len(匹配) > 1:
raise ValueError(f"导航元信息重复:{文件}")
if not 匹配:
return {}
数据 = json.loads(匹配[0])
if not isinstance(数据, dict) or any(
k not in {"内容描述", "使用场景", "使用要求"} or not isinstance(v, str)
for k, v in 数据.items()
):
raise ValueError(f"导航元信息应为内容描述、使用场景、使用要求的字符串对象:{文件}")
return 数据
def _标题(文件: Path) -> str:
if 文件.suffix == ".md" and 文件.is_file():
匹配 = re.search(r"^# +(.+?) *$", 文件.read_text(encoding="utf-8"), re.M)
if 匹配:
return 匹配[1]
return 文件.parent.name if 文件.name == "目录.md" else 文件.stem
def _目录文件(根: Path) -> list[Path]:
文件组 = {
p
for 树 in (根 / "docs", 根 / ".agent")
for p in 遍历文件(树, "目录.md")
if not _在忽略目录(p, 根)
}
技能根 = 根 / ".agent/skills"
if 技能根.is_dir():
文件组.update(p / "目录.md" for p in _会提交子目录(根, 技能根))
return sorted(文件组)
def 导航元信息迁移方案(根: Path) -> dict[Path, str]:
"""一次性保全旧索引独有语义,返回源文件修改提案;不写磁盘。
只承接本目录的 Markdown 文件及下一级目录的入口。跨目录导读、
非 Markdown 材料简介与“暂无”是手写导航,仍留原处;不把它们变成
发布资源声明,也不从名称补造场景或要求。
"""
声明组: dict[Path, dict[str, str]] = {}
for 索引 in _目录文件(根):
if not 索引.is_file():
continue
for _, 列, 地址 in _表格行(索引.read_text(encoding="utf-8")):
if 地址.startswith(("../", "http:", "https:")) or "#" in 地址:
continue
目标 = 索引.parent / 地址
if not 目标.is_file() or 目标.suffix != ".md":
continue
_导航声明(目标)
if _导航元信息.search(目标.read_text(encoding="utf-8")):
continue
字段 = (
("使用场景", "使用要求")
if 目标.name == "SKILL.md"
else ("内容描述", "使用场景", "使用要求")
)
新声明 = {
k: 列[("名称", "相对地址", "内容描述", "使用场景", "使用要求").index(k)]
for k in 字段
}
if not any(新声明.values()):
continue
已有 = 声明组.setdefault(目标, {})
for 键, 值 in 新声明.items():
if 值 and 值 not in 已有.get(键, "").split("\n"):
已有[键] = "\n".join(filter(None, (已有.get(键, ""), 值)))
修改 = {}
for 文件, 声明 in 声明组.items():
文本 = 文件.read_text(encoding="utf-8")
注记 = "<!-- 导航元信息: " + json.dumps(声明, ensure_ascii=False) + " -->\n"
# frontmatter 必须仍在文件首部,正文及作者措辞逐字保留。
匹配 = re.match(r"\A---\r?\n.*?\r?\n---(?:\r?\n|$)", 文本, re.S)
位置 = 匹配.end() if 匹配 else 0
修改[文件] = 文本[:位置] + 注记 + 文本[位置:]
return 修改
def _转义单元格(值: str) -> str:
return " ".join(值.splitlines()).replace("|", "\\|")
def 派生目录(根: Path) -> dict[Path, str]:
"""按真实文件派生本级目录表;手写导读、分节与跨目录导航原样保留。"""
产物 = {}
技能根 = 根 / ".agent/skills"
for 索引 in _目录文件(根):
原文 = 索引.read_text(encoding="utf-8") if 索引.is_file() else ""
旧行 = _表格行(原文)
分组 = 索引.parent.parent == 技能根
条目 = _会提交条目(根, 索引.parent)
if 条目 is None:
条目 = sorted(索引.parent.iterdir())
if 分组:
文件组 = [p / "SKILL.md" for p in 条目 if p.is_dir() and (p / "SKILL.md").is_file()]
else:
文件组 = [
(p / "目录.md" if (p / "目录.md").is_file() else p)
for p in 条目
if p != 索引 and not p.name.startswith(".") and p.name != "__pycache__"
]
新行 = {}
for 文件 in 文件组:
地址 = 文件.relative_to(索引.parent).as_posix() + ("/" if 文件.is_dir() else "")
元信息 = _导航声明(文件)
if 分组:
声明 = _读frontmatter(文件)
名称 = str(声明.get("name", 文件.parent.name))
描述 = str(声明.get("description", ""))
else:
名称 = _标题(文件)
描述 = 元信息.get("内容描述", "")
# 非 Markdown 材料没有强塞额外元信息,保留作者已有的唯一简介。
旧 = next((列 for _, 列, 路径 in 旧行 if 路径 == 地址), None)
if 文件.suffix != ".md" and 旧:
名称, 描述 = 旧[0], 旧[2]
元信息 = dict(zip(("内容描述", "使用场景", "使用要求"), 旧[2:], strict=True))
列 = [
名称,
f"[{文件.name if 文件.is_file() else 文件.name + '/'}]({地址})",
描述,
元信息.get("使用场景", ""),
元信息.get("使用要求", ""),
]
新行[地址] = "| " + " | ".join(_转义单元格(v) for v in 列) + " |"
行组 = 原文.splitlines()
旧行映射 = {序: (列, 地址) for 序, 列, 地址 in 旧行}
结果 = []
插入位置 = None
表数量 = 0
for 序, 行 in enumerate(行组):
if 序 in 旧行映射:
_, 地址 = 旧行映射[序]
if 地址 in 新行:
结果.append(新行.pop(地址))
elif 地址.startswith(("../", "http:", "https:")) or 地址 == "暂无" or "#" in 地址:
结果.append(行) # 手写导读,不作为本级文件清单。
continue
结果.append(行)
if re.fullmatch(r"\|[- :|]+\|", 行):
表数量 += 1
if 插入位置 is None:
插入位置 = len(结果)
if 插入位置 is None:
if 结果 and 结果[-1]:
结果.append("")
结果.extend((_导航表头, _导航分隔))
插入位置 = len(结果)
if 表数量 > 1 and 新行:
# 未声明作者分节的新来源不被悄悄归到第一个语义分节。
结果.extend(("", "## 未分节条目", "", _导航表头, _导航分隔))
插入位置 = len(结果)
结果[插入位置:插入位置] = list(新行.values())
产物[索引] = "\n".join(结果).rstrip() + "\n"
return 产物
def 写入目录(根: Path) -> list[Path]:
"""显式生成导航;存量独有语义必须先迁回源文件,不静默丢弃。"""
待迁移 = 导航元信息迁移方案(根)
if 待迁移:
raise ValueError(
"索引独有语义尚未迁回源文件:" + "、".join(str(p.relative_to(根)) for p in 待迁移)
)
已写 = []
for 文件, 文本 in 派生目录(根).items():
if not 文件.is_file() or 文件.read_text(encoding="utf-8") != 文本:
文件.write_text(文本, encoding="utf-8")
已写.append(文件)
return 已写
def 检查派生目录(根: Path) -> list[str]:
"""显式索引检查比较派生结果;不要求历史设计与现行源码逐文件一致。"""
try:
return [
f"目录需要重新生成:{文件.relative_to(根)}"
for 文件, 文本 in 派生目录(根).items()
if not 文件.is_file() or 文件.read_text(encoding="utf-8") != 文本
]
except (ValueError, yaml.YAMLError) as 错误:
return [f"目录源元信息无效:{错误}"]
def 检查技能索引(根: Path) -> list[str]:
"""分组与技能双向核对的唯一实现,供命令和 pytest 共用。"""
技能根 = 根 / ".agent/skills"
if not 技能根.is_dir():
return ["技能目录不存在:.agent/skills"]
问题 = []
for 分组 in _会提交子目录(根, 技能根):
索引 = 分组 / "目录.md"
if not 索引.is_file():
问题.append(f"分组缺目录:{分组.name}")
continue
索引名 = [
地址
for _, _, 地址 in _表格行(索引.read_text(encoding="utf-8"))
if 地址.endswith("/SKILL.md")
]
磁盘名 = {
f"{p.name}/SKILL.md" for p in _会提交子目录(根, 分组) if (p / "SKILL.md").is_file()
}
if set(索引名) != 磁盘名:
问题.append(
f"分组 {分组.name} 索引与磁盘不一致:"
f"索引多出 {sorted(set(索引名) - 磁盘名)},"
f"磁盘多出 {sorted(磁盘名 - set(索引名))}"
)
if len(索引名) != len(set(索引名)):
问题.append(f"分组 {分组.name} 索引重复登记技能")
return 问题
def 解析测试符号(文件: Path) -> list[ast.FunctionDef]:
"""用 AST 提取 test_ 开头的函数与方法;不执行任何代码。"""
树 = ast.parse(文件.read_text(encoding="utf-8"))
return [
节点
for 节点 in ast.walk(树)
if isinstance(节点, ast.FunctionDef) and 节点.name.startswith("test_")
]
def _有判定语句(节点: ast.FunctionDef) -> bool:
"""函数体内是否有 assert、raise 或 pytest.fail/skip/xfail 调用。"""
for 子 in ast.walk(节点):
if isinstance(子, ast.Assert):
return True
if isinstance(子, ast.Raise):
return True
if isinstance(子, ast.Call) and isinstance(子.func, ast.Attribute):
if 子.func.attr in {"fail", "skip", "xfail", "raises"}:
return True
return False
def 收集索引用例(根: Path) -> list[dict[str, Any]]:
"""Python、前端各由本语言的唯一收集实现导出,再验证跨语言身份。"""
from 用例身份 import 收集用例
用例 = 收集用例(根)
if (根 / "web/tests").is_dir():
try:
结果 = subprocess.run(
["node", str(根 / "工具/前端用例身份.mjs"), str(根)],
cwd=根,
capture_output=True,
text=True,
timeout=30,
check=True,
)
前端 = json.loads(结果.stdout)
if not isinstance(前端, list):
raise ValueError("前端用例导出必须为列表")
用例.extend(前端)
except subprocess.CalledProcessError as 错误:
raise ValueError(f"前端用例收集失败(exit {错误.returncode}):{错误.stderr}") from 错误
except (OSError, subprocess.TimeoutExpired) as 错误:
raise ValueError(f"前端用例收集未完成:{错误}") from 错误
已见 = set()
for 条 in 用例:
if not isinstance(条, dict) or not isinstance(条.get("case_id"), str):
raise ValueError("用例导出缺少完整 case_id")
身份 = 条["case_id"]
if 身份 in 已见:
raise ValueError(f"用例身份重复:{身份}")
已见.add(身份)
if not 用例:
raise ValueError("未收集到用例,不能通过身份检查")
return sorted(用例, key=lambda 条: 条["case_id"])
def _检查用例判定(根: Path) -> list[str]:
问题 = []
for 文件 in sorted(遍历文件(根 / "tests", "test_*.py")):
if _在忽略目录(文件, 根) or 文件.relative_to(根 / "tests").parts[0] not in _新版用例目录:
continue
for 节点 in 解析测试符号(文件):
if not _有判定语句(节点):
问题.append(
f"用例无可判定语句(print 不能代替断言):{文件.relative_to(根)}::{节点.name}"
)
return 问题
def 检查用例身份(根: Path) -> list[str]:
"""源码身份由共享收集入口校验;不读取派生清单或旧后缀。"""
问题 = _检查用例判定(根)
try:
收集索引用例(根)
except ValueError as 错误:
问题.append(str(错误))
return 问题
def 写入用例清单(根: Path, 用例: list[dict[str, Any]]) -> bool:
目标 = 根 / "tests/用例清单.json"
文本 = (
json.dumps(
{
"schema_version": 2,
"scope": "源码收集派生;不作为执行前置,不表示执行或通过。",
"cases": 用例,
},
ensure_ascii=False,
indent=2,
)
+ "\n"
)
if 目标.is_file() and 目标.read_text(encoding="utf-8") == 文本:
return False
目标.write_text(文本, encoding="utf-8")
return True
def 主() -> int:
解析器 = argparse.ArgumentParser(description="索引生成、查重及检查")
解析器.add_argument("--写入", action="store_true", help="按源文件声明与实际文件生成导航表")
解析器.add_argument("--检查", action="store_true", help="只检查并报告差异")
参数 = 解析器.parse_args()
try:
用例 = 收集索引用例(仓库根)
except ValueError as 错误:
print(f"索引问题:{错误}", file=sys.stderr)
return 1
if 参数.写入:
try:
源问题 = 检查技能目录(仓库根)
if 源问题:
raise ValueError(";".join(源问题))
print(f"索引写入:更新 {len(写入目录(仓库根))} 个目录")
print(f"用例导出:{len(用例)} 条,更新 {int(写入用例清单(仓库根, 用例))} 个清单")
except ValueError as 错误:
print(f"索引生成失败:{错误}", file=sys.stderr)
return 1
问题 = (
检查目录链接(仓库根)
+ 检查技能目录(仓库根)
+ _检查用例判定(仓库根)
+ 检查技能索引(仓库根)
+ 检查派生目录(仓库根)
)
if 问题:
for 条 in 问题:
print(f"索引问题:{条}", file=sys.stderr)
return 1
print("索引检查通过:目录链接、技能目录与用例身份一致(历史逐文件设计不参与活跃门禁)")
return 0
if __name__ == "__main__":
raise SystemExit(主())