360 lines
15 KiB
Python
360 lines
15 KiB
Python
#!/usr/bin/env python3
|
||
"""校验作品设定初始化的根设定与同构候选。
|
||
|
||
只检查机械易错且人眼难以抽查的项:三级标题树同构、S 编号唯一递增不越号段、
|
||
占位符残留、根设定格式与句长。内容是否完整由语义复核按完成判定核对,
|
||
本脚本不做数量、字数或内容质量门槛。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import re
|
||
import sys
|
||
from dataclasses import asdict, dataclass
|
||
from pathlib import Path
|
||
from typing import Iterable
|
||
|
||
TREE_PATH = Path(__file__).with_name("candidate_tree.json")
|
||
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$", re.MULTILINE)
|
||
RANGE_RE = re.compile(r"<!--\s*S(\d+)\s*[-—–]\s*S(\d+)\s*-->")
|
||
SETTING_RE = re.compile(r"^\s*[-*]\s+(?:\*\*)?(S\d{3,})(?=[^\d])", re.MULTILINE)
|
||
PLACEHOLDER_RE = re.compile(
|
||
r"(?<![A-Za-z0-9])(?:TODO|TBD|PLACEHOLDER)(?![A-Za-z0-9])|\[待填\]|待补充|待生成|在此填写",
|
||
re.IGNORECASE,
|
||
)
|
||
ROOT_ITEM_RE = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*(.+)$")
|
||
ROOT_H2_RE = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE)
|
||
ROOT_NESTED_RE = re.compile(r"^#{3,6}\s+", re.MULTILINE)
|
||
PROCESS_TERMS = ("Claude", "claude", "子代理", "候选文档", "提示词", "工具调用", "设计过程", "方案来源")
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class ValidationConfig:
|
||
min_candidates: int = 1
|
||
min_item_chars: int = 20
|
||
root_min_chars: int = 30
|
||
root_max_chars: int = 60
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class Problem:
|
||
code: str
|
||
file: str
|
||
message: str
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class SectionReport:
|
||
heading: str
|
||
range_start: int
|
||
range_end: int
|
||
setting_count: int
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class CandidateReport:
|
||
file: str
|
||
setting_count: int
|
||
headings: tuple[str, ...]
|
||
sections: tuple[SectionReport, ...]
|
||
|
||
|
||
def visible_chars(text: str) -> int:
|
||
"""去掉常见 Markdown 标记后统计非空白可见字符。"""
|
||
cleaned = re.sub(r"[*_`#>\[\]()]", "", text)
|
||
return len(re.sub(r"\s+", "", cleaned))
|
||
|
||
|
||
def _problem(code: str, path: Path, message: str) -> Problem:
|
||
return Problem(code=code, file=str(path), message=message)
|
||
|
||
|
||
def load_tree(path: Path) -> dict:
|
||
return json.loads(path.read_text(encoding="utf-8"))
|
||
|
||
|
||
def expected_heading_sequence(tree: dict) -> list[str]:
|
||
"""按冻结树返回 章→节→单元 的扁平标题序列。"""
|
||
sequence: list[str] = []
|
||
for chapter in tree["chapters"]:
|
||
sequence.append(chapter["heading"])
|
||
for section in chapter["sections"]:
|
||
sequence.append(section["heading"])
|
||
sequence.extend(section["units"])
|
||
return sequence
|
||
|
||
|
||
def validate_root(path: Path, config: ValidationConfig) -> list[Problem]:
|
||
problems: list[Problem] = []
|
||
text = path.read_text(encoding="utf-8")
|
||
headings = list(ROOT_H2_RE.finditer(text))
|
||
root_index = next((index for index, match in enumerate(headings) if "根设定" in match.group(1)), None)
|
||
if root_index is None:
|
||
return [_problem("ROOT_SECTION_MISSING", path, "未找到包含“根设定”的二级标题")]
|
||
|
||
start = headings[root_index].end()
|
||
end = headings[root_index + 1].start() if root_index + 1 < len(headings) else len(text)
|
||
body = text[start:end]
|
||
|
||
if ROOT_NESTED_RE.search(body):
|
||
problems.append(_problem("ROOT_STRUCTURE", path, "根设定内部出现三级或更深标题"))
|
||
|
||
for term in PROCESS_TERMS:
|
||
if term in body:
|
||
problems.append(_problem("ROOT_PROCESS_LEAK", path, f"根设定包含设计过程词:{term}"))
|
||
|
||
labels: set[str] = set()
|
||
items = 0
|
||
for line_number, raw_line in enumerate(body.splitlines(), start=1):
|
||
line = raw_line.strip()
|
||
if not line or line == "---":
|
||
continue
|
||
match = ROOT_ITEM_RE.match(line)
|
||
if not match:
|
||
problems.append(
|
||
_problem("ROOT_ITEM_FORMAT", path, f"根设定第 {line_number} 个相对行不是单行加粗标签条目")
|
||
)
|
||
continue
|
||
items += 1
|
||
label = match.group(1).strip()
|
||
if label in labels:
|
||
problems.append(_problem("ROOT_LABEL_DUPLICATE", path, f"根设定标签重复:{label}"))
|
||
labels.add(label)
|
||
length = visible_chars(line.lstrip("- "))
|
||
if not config.root_min_chars <= length <= config.root_max_chars:
|
||
problems.append(
|
||
_problem(
|
||
"ROOT_ITEM_LENGTH",
|
||
path,
|
||
f"根设定“{label}”为 {length} 个可见字符,要求 {config.root_min_chars}—{config.root_max_chars}",
|
||
)
|
||
)
|
||
if items == 0:
|
||
problems.append(_problem("ROOT_EMPTY", path, "根设定没有可校验条目"))
|
||
return problems
|
||
|
||
|
||
def _parse_candidate(path: Path, tree: dict, config: ValidationConfig) -> tuple[CandidateReport, list[Problem]]:
|
||
text = path.read_text(encoding="utf-8")
|
||
problems: list[Problem] = []
|
||
|
||
if PLACEHOLDER_RE.search(text):
|
||
problems.append(_problem("CANDIDATE_PLACEHOLDER", path, "候选仍含待填占位文字"))
|
||
|
||
headings = [(len(match.group(1)), match.group(2).strip(), match.start(), match.end()) for match in HEADING_RE.finditer(text)]
|
||
h1_count = sum(1 for level, _, _, _ in headings if level == 1)
|
||
if h1_count != 1:
|
||
problems.append(_problem("CANDIDATE_STRUCTURE", path, f"候选应有且只有一个一级标题,实际 {h1_count} 个"))
|
||
deep = [heading for level, heading, _, _ in headings if level >= 4]
|
||
if deep:
|
||
problems.append(_problem("CANDIDATE_STRUCTURE", path, f"候选出现四级或更深标题:{deep[0]}"))
|
||
|
||
h2_list = [(heading, start, end) for level, heading, start, end in headings if level == 2]
|
||
toc_seen = False
|
||
chapter_index = 0
|
||
all_ids: list[int] = []
|
||
previous_range_end = 0
|
||
sections: list[SectionReport] = []
|
||
chapter_titles = [chapter["heading"] for chapter in tree["chapters"]]
|
||
|
||
for position, (heading, start, end) in enumerate(h2_list):
|
||
body_end = h2_list[position + 1][1] if position + 1 < len(h2_list) else len(text)
|
||
body = text[end:body_end]
|
||
|
||
if heading == "目录":
|
||
toc_seen = True
|
||
if re.search(r"^###\s+", body, re.MULTILINE):
|
||
problems.append(_problem("CANDIDATE_STRUCTURE", path, "目录下不得出现三级标题"))
|
||
continue
|
||
|
||
if chapter_index >= len(chapter_titles):
|
||
problems.append(_problem("TREE_MISMATCH", path, f"多出章节标题:{heading}"))
|
||
continue
|
||
expected_title = chapter_titles[chapter_index]
|
||
if heading != expected_title:
|
||
problems.append(
|
||
_problem("TREE_MISMATCH", path, f"第 {chapter_index + 1} 章应为“{expected_title}”,实际为“{heading}”")
|
||
)
|
||
chapter_index += 1
|
||
continue
|
||
chapter_index += 1
|
||
|
||
expected_subs: list[str] = []
|
||
for section in tree["chapters"][chapter_index - 1]["sections"]:
|
||
expected_subs.append(section["heading"])
|
||
expected_subs.extend(section["units"])
|
||
actual_subs = [sub_heading for level, sub_heading, sub_start, _ in headings if level == 3 and start < sub_start < body_end]
|
||
if actual_subs != expected_subs:
|
||
expected_set = set(expected_subs)
|
||
missing = [item for item in expected_subs if item not in set(actual_subs)]
|
||
extra = [item for item in actual_subs if item not in expected_set]
|
||
detail = []
|
||
if missing:
|
||
detail.append(f"缺 {len(missing)} 个(首个:{missing[0]})")
|
||
if extra:
|
||
detail.append(f"多 {len(extra)} 个(首个:{extra[0]})")
|
||
if not detail:
|
||
detail.append("顺序与冻结树不一致")
|
||
problems.append(_problem("TREE_MISMATCH", path, f"章节“{heading}”标题树漂移:{';'.join(detail)}"))
|
||
|
||
ranges = RANGE_RE.findall(body)
|
||
if len(ranges) != 1:
|
||
problems.append(_problem("SECTION_RANGE", path, f"章节“{heading}”必须且只能有一个号段标记"))
|
||
continue
|
||
range_start, range_end = (int(value) for value in ranges[0])
|
||
if range_start > range_end:
|
||
problems.append(_problem("SECTION_RANGE", path, f"章节号段倒置:S{range_start:03d}-S{range_end:03d}"))
|
||
if range_start <= previous_range_end:
|
||
problems.append(_problem("SECTION_RANGE", path, "章节号段未按顺序递增或发生重叠"))
|
||
previous_range_end = max(previous_range_end, range_end)
|
||
|
||
setting_matches = list(SETTING_RE.finditer(body))
|
||
ids = [int(match.group(1)[1:]) for match in setting_matches]
|
||
all_ids.extend(ids)
|
||
for setting_id in ids:
|
||
if not range_start <= setting_id <= range_end:
|
||
problems.append(
|
||
_problem(
|
||
"SETTING_ID_OUT_OF_RANGE",
|
||
path,
|
||
f"S{setting_id:03d} 不在章节号段 S{range_start:03d}-S{range_end:03d} 内",
|
||
)
|
||
)
|
||
for item_index, setting_match in enumerate(setting_matches):
|
||
item_end = setting_matches[item_index + 1].start() if item_index + 1 < len(setting_matches) else len(body)
|
||
item_text = body[setting_match.start():item_end]
|
||
if visible_chars(item_text) < config.min_item_chars:
|
||
problems.append(
|
||
_problem("SETTING_ITEM_TOO_SHORT", path, f"{setting_match.group(1)} 内容过短,疑似只有标题或占位句")
|
||
)
|
||
sections.append(
|
||
SectionReport(heading=heading, range_start=range_start, range_end=range_end, setting_count=len(ids))
|
||
)
|
||
|
||
if chapter_index < len(chapter_titles):
|
||
problems.append(
|
||
_problem(
|
||
"TREE_MISMATCH",
|
||
path,
|
||
f"缺少章节:{ '、'.join(chapter_titles[chapter_index:]) }",
|
||
)
|
||
)
|
||
if toc_seen and chapter_index == 0:
|
||
problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选只有目录,没有内容章节"))
|
||
|
||
duplicates = sorted({setting_id for setting_id in all_ids if all_ids.count(setting_id) > 1})
|
||
if duplicates:
|
||
display = ", ".join(f"S{setting_id:03d}" for setting_id in duplicates[:10])
|
||
problems.append(_problem("SETTING_ID_DUPLICATE", path, f"全文编号重复:{display}"))
|
||
if any(later <= earlier for earlier, later in zip(all_ids, all_ids[1:])):
|
||
problems.append(_problem("SETTING_ORDER", path, "全文 S 编号必须严格递增"))
|
||
|
||
report = CandidateReport(
|
||
file=str(path),
|
||
setting_count=len(all_ids),
|
||
headings=tuple(heading for heading, _, _ in h2_list),
|
||
sections=tuple(sections),
|
||
)
|
||
return report, problems
|
||
|
||
|
||
def validate_candidates(
|
||
paths: Iterable[Path],
|
||
config: ValidationConfig,
|
||
root_doc: Path | None = None,
|
||
tree_path: Path = TREE_PATH,
|
||
) -> tuple[list[CandidateReport], list[Problem]]:
|
||
candidate_paths = sorted((Path(path) for path in paths), key=lambda item: str(item))
|
||
problems: list[Problem] = []
|
||
reports: list[CandidateReport] = []
|
||
|
||
if len(candidate_paths) < config.min_candidates:
|
||
problems.append(
|
||
Problem(
|
||
code="CANDIDATE_COUNT",
|
||
file="<group>",
|
||
message=f"只有 {len(candidate_paths)} 份候选,要求至少 {config.min_candidates} 份",
|
||
)
|
||
)
|
||
if root_doc is not None:
|
||
if not root_doc.is_file():
|
||
problems.append(_problem("ROOT_FILE_MISSING", root_doc, "根设定文档不存在"))
|
||
else:
|
||
problems.extend(validate_root(root_doc, config))
|
||
|
||
tree = load_tree(tree_path)
|
||
for path in candidate_paths:
|
||
if not path.is_file():
|
||
problems.append(_problem("CANDIDATE_FILE_MISSING", path, "候选文件不存在"))
|
||
continue
|
||
report, file_problems = _parse_candidate(path, tree, config)
|
||
reports.append(report)
|
||
problems.extend(file_problems)
|
||
|
||
if len(reports) > 1:
|
||
baseline = reports[0]
|
||
baseline_shape = tuple((section.heading, section.range_start, section.range_end) for section in baseline.sections)
|
||
for report in reports[1:]:
|
||
shape = tuple((section.heading, section.range_start, section.range_end) for section in report.sections)
|
||
if report.headings != baseline.headings or shape != baseline_shape:
|
||
problems.append(
|
||
Problem(
|
||
code="CANDIDATE_STRUCTURE_MISMATCH",
|
||
file=report.file,
|
||
message=f"目录或号段与基准候选 {baseline.file} 不一致",
|
||
)
|
||
)
|
||
return reports, problems
|
||
|
||
|
||
def _build_parser() -> argparse.ArgumentParser:
|
||
parser = argparse.ArgumentParser(description="校验作品设定初始化候选")
|
||
parser.add_argument("files", nargs="+", type=Path, help="候选 Markdown 文件")
|
||
parser.add_argument("--root-doc", type=Path, help="含根设定的唯一前期设计文档")
|
||
parser.add_argument("--min-candidates", type=int, default=1)
|
||
parser.add_argument("--min-item-chars", type=int, default=20)
|
||
parser.add_argument("--root-min-chars", type=int, default=30)
|
||
parser.add_argument("--root-max-chars", type=int, default=60)
|
||
parser.add_argument("--tree", type=Path, default=TREE_PATH, help="冻结三级树 JSON")
|
||
parser.add_argument("--json", action="store_true", help="输出稳定 JSON")
|
||
return parser
|
||
|
||
|
||
def main(argv: list[str] | None = None) -> int:
|
||
parser = _build_parser()
|
||
args = parser.parse_args(argv)
|
||
if args.root_min_chars > args.root_max_chars:
|
||
parser.error("--root-min-chars 不能大于 --root-max-chars")
|
||
|
||
config = ValidationConfig(
|
||
min_candidates=args.min_candidates,
|
||
min_item_chars=args.min_item_chars,
|
||
root_min_chars=args.root_min_chars,
|
||
root_max_chars=args.root_max_chars,
|
||
)
|
||
reports, problems = validate_candidates(args.files, config, args.root_doc, args.tree)
|
||
status = "ok" if not problems else "invalid"
|
||
code = "SETTING_INIT_OK" if not problems else "SETTING_INIT_VALIDATION_FAILED"
|
||
payload = {
|
||
"status": status,
|
||
"code": code,
|
||
"reports": [asdict(report) for report in reports],
|
||
"problems": [asdict(problem) for problem in problems],
|
||
}
|
||
|
||
if args.json:
|
||
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
||
else:
|
||
print(f"{code}: {len(reports)} 份候选,{len(problems)} 个问题")
|
||
for report in reports:
|
||
print(f"- {report.file}: {report.setting_count} 项")
|
||
for problem in problems:
|
||
print(f"[{problem.code}] {problem.file}: {problem.message}")
|
||
return 0 if not problems else 1
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|