muse-agent-example/工具/维护索引.py
zizi 0658d178ad 工程底座:单一生成入口、验证期写闸门与并行分片
- Makefile:新增 验收数据库(生成→检查→库层全量,前置校验隔离库连接串)、数据库分片、浏览器测试三个入口;
  格式写入 末尾就地重新生成;数据库测试 与 验收数据库 统一排除 浏览器/网络;快检 不再静默跳过类型门;
  前端旅程 预检补齐四个必需变量;pytest 目标改用仓内解释器(uv run 在嵌套检出会解析到外层环境)。
- 工具/验证锁.py:验证会话持共享锁,写入口用 --执行 在独占锁内落盘,生成/格式写入/索引生成走同一闸门。
- 工具/并行数据库测试.py:按文件分片并行,缺连接串在采集前拒绝。
- 工具/构建编排.py、环境预检.py、构建资源包.py、维护索引.py 与上述口径对齐。
- .gitignore / CI / README / AGENTS:忽略构建产物、CI 与 Makefile 单一口径、README 按实际实现陈述、AGENTS 补日常入口与验证纪律。
2026-09-18 01:14:50 +08:00

521 lines
21 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""由源文件派生导航,并复用 pytest 收集的稳定用例身份。
显式 --写入 生成目录表;--检查 不写磁盘。技能说明来自 SKILL.md
frontmatter,作者导航语义随源文件保存,手写导读保留。历史逐文件设计
不再参与每日源码定位门禁。所有入口可用显式根目录在临时项目复现。
"""
from __future__ import annotations
import argparse
import ast
import json
import os
import re
import subprocess
import sys
import unicodedata
from fnmatch import fnmatch
from pathlib import Path
from typing import Any
import yaml
仓库根 = Path(__file__).resolve().parent.parent
_忽略目录片段 = frozenset(
{
".venv",
".venv-旧",
".git",
"node_modules",
"__pycache__",
".agents.local",
"研究依据",
"文件设计",
}
)
# 新版用例目录:旧测试树(tests/skills、tests/adapters、tests/architecture 英文目录等)
# 由旧环境运行,不进入新版用例身份检查
_新版用例目录 = ("单元", "契约", "集成", "架构", "迁移", "端到端", "真实调用")
def _在忽略目录(路径: Path, 根: Path) -> bool:
try:
相对 = 路径.relative_to(根)
except ValueError:
return True
return any(片段 in 相对.parts for 片段 in _忽略目录片段)
def 遍历文件(根: Path, 模式: str):
"""在进入目录前剪枝,忽略缓存和历史树不承担扫描成本。"""
for 当前, 目录名, 文件名 in os.walk(根, followlinks=False):
目录名[:] = sorted(名 for 名 in 目录名 if 名 not in _忽略目录片段)
for 名 in sorted(文件名):
if fnmatch(名, 模式):
yield Path(当前) / 名
def 检查目录链接(根: Path) -> list[str]:
"""核对仓库内 目录.md 的表格相对链接;返回问题列表。"""
问题: list[str] = []
for 目录文件 in 遍历文件(根, "目录.md"):
if _在忽略目录(目录文件, 根):
continue
for 序, _, 原地址 in _表格行(目录文件.read_text(encoding="utf-8")):
地址 = 原地址.split("#")[0].strip()
if not 地址 or 地址 == "暂无" or 地址.startswith(("http://", "https://", "mailto:")):
continue
目标 = (目录文件.parent / 地址).resolve()
if not 目标.exists():
问题.append(f"{目录文件.relative_to(根)}:{序 + 1} 链接不存在:{地址}")
return 问题
def _读frontmatter(文件: Path) -> dict:
"""与源文件一起移动的声明;支持 YAML 折叠及引号,不解析正文。"""
文本 = 文件.read_text(encoding="utf-8")
匹配 = re.match(r"\A---\r?\n(.*?)\r?\n---(?:\r?\n|$)", 文本, re.S)
if 匹配 is None:
return {}
try:
声明 = yaml.safe_load(匹配[1])
except yaml.YAMLError as 错误:
raise ValueError(f"frontmatter 无法解析:{文件}") from 错误
return 声明 if isinstance(声明, dict) else {}
def _读frontmatter名(文件: Path) -> str | None:
名称 = _读frontmatter(文件).get("name")
return 名称 if isinstance(名称, str) else None
def _含中文(名称: str) -> bool:
return any("\u4e00" <= ch <= "\u9fff" for ch in 名称)
def 检查技能目录(根: Path) -> list[str]:
"""核对 .agent/skills 分组:每个技能有非空 SKILL.md、名称不重复、
frontmatter name 与目录一致、目录名全中文且规范化(NFC、无空白)。"""
技能根 = 根 / ".agent" / "skills"
if not 技能根.is_dir():
return ["技能目录不存在:.agent/skills(缺失即失败,不静默跳过)"]
问题: list[str] = []
已见: dict[str, str] = {}
for 分组 in sorted(p for p in 技能根.iterdir() if p.is_dir()):
for 技能目录 in sorted(p for p in 分组.iterdir() if p.is_dir()):
名称 = 技能目录.name
相对 = str(技能目录.relative_to(根))
技能文件 = 技能目录 / "SKILL.md"
if not 技能文件.is_file():
问题.append(f"技能缺少 SKILL.md:{相对}")
continue
正文 = 技能文件.read_text(encoding="utf-8").strip()
if not 正文 or 正文 == "---":
问题.append(f"SKILL.md 为空:{相对}")
continue
try:
front名 = _读frontmatter名(技能文件)
except ValueError as 错误:
问题.append(str(错误))
continue
if front名 is None:
问题.append(f"SKILL.md 缺少 frontmatter name:{相对}")
elif front名 != 名称:
问题.append(f"frontmatter name 与目录不一致:{相对}(name: {front名})")
if not _含中文(名称):
问题.append(f"技能名称必须全中文(政策 chinese_only):{相对}")
if 名称 != unicodedata.normalize("NFC", 名称):
问题.append(f"技能名称不是 NFC 规范形式:{相对}")
if any(ch.isspace() for ch in 名称):
问题.append(f"技能名称含空白:{相对}")
if 名称 in 已见:
问题.append(f"技能名称重复:{名称}({已见[名称]} 与 {相对})")
else:
已见[名称] = 相对
return 问题
_导航表头 = "| 名称 | 相对地址 | 内容描述 | 使用场景 | 使用要求 |"
_导航分隔 = "|------|----------|----------|----------|----------|"
_导航元信息 = re.compile(r"<!-- 导航元信息: (.*?) -->")
_表格链接 = re.compile(r"\[([^\]]+)\]\(([^)]+)\)")
def _表格行(文本: str) -> list[tuple[int, list[str], str]]:
结果 = []
for 序, 行 in enumerate(文本.splitlines()):
if not 行.startswith("|"):
continue
列 = [v.strip() for v in re.split(r"(?<!\\)\|", 行)[1:-1]]
if len(列) != 5 or 列[0] == "名称" or re.fullmatch(r"[- :]+", 列[0]):
continue
链接 = _表格链接.fullmatch(列[1])
结果.append((序, 列, 链接[2] if 链接 else 列[1]))
return 结果
def _导航声明(文件: Path) -> dict[str, str]:
if 文件.suffix != ".md" or not 文件.is_file():
return {}
匹配 = _导航元信息.findall(文件.read_text(encoding="utf-8"))
if len(匹配) > 1:
raise ValueError(f"导航元信息重复:{文件}")
if not 匹配:
return {}
数据 = json.loads(匹配[0])
if not isinstance(数据, dict) or any(
k not in {"内容描述", "使用场景", "使用要求"} or not isinstance(v, str)
for k, v in 数据.items()
):
raise ValueError(f"导航元信息应为内容描述、使用场景、使用要求的字符串对象:{文件}")
return 数据
def _标题(文件: Path) -> str:
if 文件.suffix == ".md" and 文件.is_file():
匹配 = re.search(r"^# +(.+?) *$", 文件.read_text(encoding="utf-8"), re.M)
if 匹配:
return 匹配[1]
return 文件.parent.name if 文件.name == "目录.md" else 文件.stem
def _目录文件(根: Path) -> list[Path]:
文件组 = {
p
for 树 in (根 / "docs", 根 / ".agent")
for p in 遍历文件(树, "目录.md")
if not _在忽略目录(p, 根)
}
技能根 = 根 / ".agent/skills"
if 技能根.is_dir():
文件组.update(p / "目录.md" for p in 技能根.iterdir() if p.is_dir())
return sorted(文件组)
def 导航元信息迁移方案(根: Path) -> dict[Path, str]:
"""一次性保全旧索引独有语义,返回源文件修改提案;不写磁盘。
只承接本目录的 Markdown 文件及下一级目录的入口。跨目录导读、
非 Markdown 材料简介与“暂无”是手写导航,仍留原处;不把它们变成
发布资源声明,也不从名称补造场景或要求。
"""
声明组: dict[Path, dict[str, str]] = {}
for 索引 in _目录文件(根):
if not 索引.is_file():
continue
for _, 列, 地址 in _表格行(索引.read_text(encoding="utf-8")):
if 地址.startswith(("../", "http:", "https:")) or "#" in 地址:
continue
目标 = 索引.parent / 地址
if not 目标.is_file() or 目标.suffix != ".md":
continue
_导航声明(目标)
if _导航元信息.search(目标.read_text(encoding="utf-8")):
continue
字段 = (
("使用场景", "使用要求")
if 目标.name == "SKILL.md"
else ("内容描述", "使用场景", "使用要求")
)
新声明 = {
k: 列[("名称", "相对地址", "内容描述", "使用场景", "使用要求").index(k)]
for k in 字段
}
if not any(新声明.values()):
continue
已有 = 声明组.setdefault(目标, {})
for 键, 值 in 新声明.items():
if 值 and 值 not in 已有.get(键, "").split("\n"):
已有[键] = "\n".join(filter(None, (已有.get(键, ""), 值)))
修改 = {}
for 文件, 声明 in 声明组.items():
文本 = 文件.read_text(encoding="utf-8")
注记 = "<!-- 导航元信息: " + json.dumps(声明, ensure_ascii=False) + " -->\n"
# frontmatter 必须仍在文件首部,正文及作者措辞逐字保留。
匹配 = re.match(r"\A---\r?\n.*?\r?\n---(?:\r?\n|$)", 文本, re.S)
位置 = 匹配.end() if 匹配 else 0
修改[文件] = 文本[:位置] + 注记 + 文本[位置:]
return 修改
def _转义单元格(值: str) -> str:
return " ".join(值.splitlines()).replace("|", "\\|")
def 派生目录(根: Path) -> dict[Path, str]:
"""按真实文件派生本级目录表;手写导读、分节与跨目录导航原样保留。"""
产物 = {}
技能根 = 根 / ".agent/skills"
for 索引 in _目录文件(根):
原文 = 索引.read_text(encoding="utf-8") if 索引.is_file() else ""
旧行 = _表格行(原文)
分组 = 索引.parent.parent == 技能根
if 分组:
文件组 = [
p / "SKILL.md"
for p in sorted(索引.parent.iterdir())
if p.is_dir() and (p / "SKILL.md").is_file()
]
else:
文件组 = [
(p / "目录.md" if (p / "目录.md").is_file() else p)
for p in sorted(索引.parent.iterdir())
if p != 索引 and not p.name.startswith(".") and p.name != "__pycache__"
]
新行 = {}
for 文件 in 文件组:
地址 = 文件.relative_to(索引.parent).as_posix() + ("/" if 文件.is_dir() else "")
元信息 = _导航声明(文件)
if 分组:
声明 = _读frontmatter(文件)
名称 = str(声明.get("name", 文件.parent.name))
描述 = str(声明.get("description", ""))
else:
名称 = _标题(文件)
描述 = 元信息.get("内容描述", "")
# 非 Markdown 材料没有强塞额外元信息,保留作者已有的唯一简介。
旧 = next((列 for _, 列, 路径 in 旧行 if 路径 == 地址), None)
if 文件.suffix != ".md" and 旧:
名称, 描述 = 旧[0], 旧[2]
元信息 = dict(zip(("内容描述", "使用场景", "使用要求"), 旧[2:], strict=True))
列 = [
名称,
f"[{文件.name if 文件.is_file() else 文件.name + '/'}]({地址})",
描述,
元信息.get("使用场景", ""),
元信息.get("使用要求", ""),
]
新行[地址] = "| " + " | ".join(_转义单元格(v) for v in 列) + " |"
行组 = 原文.splitlines()
旧行映射 = {序: (列, 地址) for 序, 列, 地址 in 旧行}
结果 = []
插入位置 = None
表数量 = 0
for 序, 行 in enumerate(行组):
if 序 in 旧行映射:
_, 地址 = 旧行映射[序]
if 地址 in 新行:
结果.append(新行.pop(地址))
elif 地址.startswith(("../", "http:", "https:")) or 地址 == "暂无" or "#" in 地址:
结果.append(行) # 手写导读,不作为本级文件清单。
continue
结果.append(行)
if re.fullmatch(r"\|[- :|]+\|", 行):
表数量 += 1
if 插入位置 is None:
插入位置 = len(结果)
if 插入位置 is None:
if 结果 and 结果[-1]:
结果.append("")
结果.extend((_导航表头, _导航分隔))
插入位置 = len(结果)
if 表数量 > 1 and 新行:
# 未声明作者分节的新来源不被悄悄归到第一个语义分节。
结果.extend(("", "## 未分节条目", "", _导航表头, _导航分隔))
插入位置 = len(结果)
结果[插入位置:插入位置] = list(新行.values())
产物[索引] = "\n".join(结果).rstrip() + "\n"
return 产物
def 写入目录(根: Path) -> list[Path]:
"""显式生成导航;存量独有语义必须先迁回源文件,不静默丢弃。"""
待迁移 = 导航元信息迁移方案(根)
if 待迁移:
raise ValueError(
"索引独有语义尚未迁回源文件:" + "、".join(str(p.relative_to(根)) for p in 待迁移)
)
已写 = []
for 文件, 文本 in 派生目录(根).items():
if not 文件.is_file() or 文件.read_text(encoding="utf-8") != 文本:
文件.write_text(文本, encoding="utf-8")
已写.append(文件)
return 已写
def 检查派生目录(根: Path) -> list[str]:
"""显式索引检查比较派生结果;不要求历史设计与现行源码逐文件一致。"""
try:
return [
f"目录需要重新生成:{文件.relative_to(根)}"
for 文件, 文本 in 派生目录(根).items()
if not 文件.is_file() or 文件.read_text(encoding="utf-8") != 文本
]
except (ValueError, yaml.YAMLError) as 错误:
return [f"目录源元信息无效:{错误}"]
def 检查技能索引(根: Path) -> list[str]:
"""分组与技能双向核对的唯一实现,供命令和 pytest 共用。"""
技能根 = 根 / ".agent/skills"
if not 技能根.is_dir():
return ["技能目录不存在:.agent/skills"]
问题 = []
for 分组 in sorted(p for p in 技能根.iterdir() if p.is_dir()):
索引 = 分组 / "目录.md"
if not 索引.is_file():
问题.append(f"分组缺目录:{分组.name}")
continue
索引名 = [
地址
for _, _, 地址 in _表格行(索引.read_text(encoding="utf-8"))
if 地址.endswith("/SKILL.md")
]
磁盘名 = {
f"{p.name}/SKILL.md"
for p in 分组.iterdir()
if p.is_dir() and (p / "SKILL.md").is_file()
}
if set(索引名) != 磁盘名:
问题.append(
f"分组 {分组.name} 索引与磁盘不一致:"
f"索引多出 {sorted(set(索引名) - 磁盘名)},"
f"磁盘多出 {sorted(磁盘名 - set(索引名))}"
)
if len(索引名) != len(set(索引名)):
问题.append(f"分组 {分组.name} 索引重复登记技能")
return 问题
def 解析测试符号(文件: Path) -> list[ast.FunctionDef]:
"""用 AST 提取 test_ 开头的函数与方法;不执行任何代码。"""
树 = ast.parse(文件.read_text(encoding="utf-8"))
return [
节点
for 节点 in ast.walk(树)
if isinstance(节点, ast.FunctionDef) and 节点.name.startswith("test_")
]
def _有判定语句(节点: ast.FunctionDef) -> bool:
"""函数体内是否有 assert、raise 或 pytest.fail/skip/xfail 调用。"""
for 子 in ast.walk(节点):
if isinstance(子, ast.Assert):
return True
if isinstance(子, ast.Raise):
return True
if isinstance(子, ast.Call) and isinstance(子.func, ast.Attribute):
if 子.func.attr in {"fail", "skip", "xfail", "raises"}:
return True
return False
def 收集索引用例(根: Path) -> list[dict[str, Any]]:
"""Python、前端各由本语言的唯一收集实现导出,再验证跨语言身份。"""
from 用例身份 import 收集用例
用例 = 收集用例(根)
if (根 / "web/tests").is_dir():
try:
结果 = subprocess.run(
["node", str(根 / "工具/前端用例身份.mjs"), str(根)],
cwd=根,
capture_output=True,
text=True,
timeout=30,
check=True,
)
前端 = json.loads(结果.stdout)
if not isinstance(前端, list):
raise ValueError("前端用例导出必须为列表")
用例.extend(前端)
except subprocess.CalledProcessError as 错误:
raise ValueError(f"前端用例收集失败(exit {错误.returncode}):{错误.stderr}") from 错误
except (OSError, subprocess.TimeoutExpired) as 错误:
raise ValueError(f"前端用例收集未完成:{错误}") from 错误
已见 = set()
for 条 in 用例:
if not isinstance(条, dict) or not isinstance(条.get("case_id"), str):
raise ValueError("用例导出缺少完整 case_id")
身份 = 条["case_id"]
if 身份 in 已见:
raise ValueError(f"用例身份重复:{身份}")
已见.add(身份)
if not 用例:
raise ValueError("未收集到用例,不能通过身份检查")
return sorted(用例, key=lambda 条: 条["case_id"])
def _检查用例判定(根: Path) -> list[str]:
问题 = []
for 文件 in sorted(遍历文件(根 / "tests", "test_*.py")):
if _在忽略目录(文件, 根) or 文件.relative_to(根 / "tests").parts[0] not in _新版用例目录:
continue
for 节点 in 解析测试符号(文件):
if not _有判定语句(节点):
问题.append(
f"用例无可判定语句(print 不能代替断言):{文件.relative_to(根)}::{节点.name}"
)
return 问题
def 检查用例身份(根: Path) -> list[str]:
"""源码身份由共享收集入口校验;不读取派生清单或旧后缀。"""
问题 = _检查用例判定(根)
try:
收集索引用例(根)
except ValueError as 错误:
问题.append(str(错误))
return 问题
def 写入用例清单(根: Path, 用例: list[dict[str, Any]]) -> bool:
目标 = 根 / "tests/用例清单.json"
文本 = (
json.dumps(
{
"schema_version": 2,
"scope": "源码收集派生;不作为执行前置,不表示执行或通过。",
"cases": 用例,
},
ensure_ascii=False,
indent=2,
)
+ "\n"
)
if 目标.is_file() and 目标.read_text(encoding="utf-8") == 文本:
return False
目标.write_text(文本, encoding="utf-8")
return True
def 主() -> int:
解析器 = argparse.ArgumentParser(description="索引生成、查重及检查")
解析器.add_argument("--写入", action="store_true", help="按源文件声明与实际文件生成导航表")
解析器.add_argument("--检查", action="store_true", help="只检查并报告差异")
参数 = 解析器.parse_args()
try:
用例 = 收集索引用例(仓库根)
except ValueError as 错误:
print(f"索引问题:{错误}", file=sys.stderr)
return 1
if 参数.写入:
try:
源问题 = 检查技能目录(仓库根)
if 源问题:
raise ValueError(";".join(源问题))
print(f"索引写入:更新 {len(写入目录(仓库根))} 个目录")
print(f"用例导出:{len(用例)} 条,更新 {int(写入用例清单(仓库根, 用例))} 个清单")
except ValueError as 错误:
print(f"索引生成失败:{错误}", file=sys.stderr)
return 1
问题 = (
检查目录链接(仓库根)
+ 检查技能目录(仓库根)
+ _检查用例判定(仓库根)
+ 检查技能索引(仓库根)
+ 检查派生目录(仓库根)
)
if 问题:
for 条 in 问题:
print(f"索引问题:{条}", file=sys.stderr)
return 1
print("索引检查通过:目录链接、技能目录与用例身份一致(历史逐文件设计不参与活跃门禁)")
return 0
if __name__ == "__main__":
raise SystemExit(主())