#!/usr/bin/env python3 """只校验阶段 0 最终交付稿的格式,不参与探索调度或创作判断。""" from __future__ import annotations import argparse from collections import Counter from dataclasses import asdict, dataclass import json from pathlib import Path import re from typing import Iterable TREE_PATH = Path(__file__).with_name("candidate_tree.json") HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*$", re.MULTILINE) SETTING_RE = re.compile(r"^\s*[-*]\s+(?:\*\*)?S(\d{3,})(?=[^\d]|$)", re.MULTILINE) PLACEHOLDER_RE = re.compile( r"^\s*(?:[-*>]\s*)?(?:TODO|TBD|PLACEHOLDER|\[待填\])\s*$", re.MULTILINE | re.IGNORECASE, ) @dataclass(frozen=True) class Problem: code: str file: str message: str @dataclass(frozen=True) class CandidateReport: file: str setting_count: int def _validate(path: Path, tree: dict) -> tuple[CandidateReport, list[Problem]]: text = path.read_text(encoding="utf-8") headings = list(HEADING_RE.finditer(text)) problems: list[Problem] = [] def fail(code: str, message: str) -> None: problems.append(Problem(code, str(path), message)) if sum(len(h.group(1)) == 1 for h in headings) != 1: fail("CANDIDATE_STRUCTURE", "交付稿须有且只有一个一级标题") if headings and len(headings[0].group(1)) != 1: fail("CANDIDATE_STRUCTURE", "首个标题须为作品一级标题") if any(len(h.group(1)) >= 4 for h in headings): fail("CANDIDATE_STRUCTURE", "交付稿不使用四级及更深标题") if PLACEHOLDER_RE.search(text): fail("CANDIDATE_PLACEHOLDER", "请把裸占位符改为具体的待定或未探索说明") chapters = {chapter["heading"]: chapter for chapter in tree["chapters"]} expected = ["根设定", *chapters, "待决与取舍"] h2 = [h for h in headings if len(h.group(1)) == 2] titles = [h.group(2) for h in h2] if titles and titles[0] == "目录": titles = titles[1:] if titles != expected: fail("TREE_MISMATCH", "二级标题须按顺序包含根设定、九章和待决与取舍;目录仅可置于最前") if any( len(h.group(1)) == 3 and (not h2 or h.start() < h2[0].start()) for h in headings ): fail("CANDIDATE_STRUCTURE", "三级标题须位于对应章节内") for index, heading in enumerate(h2): title = heading.group(2) end = h2[index + 1].start() if index + 1 < len(h2) else len(text) body = text[heading.end():end] # 标题和注释不能代替说明;未知内容可以用一句明确说明表达。 prose = HEADING_RE.sub("", body) prose = re.sub(r"", "", prose, flags=re.DOTALL) if not re.sub(r"[\s>*_\x60-]", "", prose): fail("SECTION_EMPTY", f"“{title}”为空,请写已有内容或说明尚未探索的部分") actual = [ h.group(2) for h in headings if len(h.group(1)) == 3 and heading.end() <= h.start() < end ] allowed: list[str] = [] if title in chapters: for section in chapters[title]["sections"]: allowed.append(section["heading"]) allowed.extend(section["units"]) positions = {name: position for position, name in enumerate(allowed)} if any(name not in positions for name in actual): fail("TREE_MISMATCH", f"“{title}”含导航之外的三级标题") elif any(positions[right] <= positions[left] for left, right in zip(actual, actual[1:])): fail("TREE_MISMATCH", f"“{title}”的已展开标题重复或顺序错误") ids = [int(value) for value in SETTING_RE.findall(text)] duplicates = [value for value, count in Counter(ids).items() if count > 1] if duplicates: fail("SETTING_ID_DUPLICATE", "编号重复:" + "、".join(f"S{value:03d}" for value in duplicates)) return CandidateReport(str(path), len(ids)), problems def validate_candidates(paths: Iterable[Path]) -> tuple[list[CandidateReport], list[Problem]]: """各稿独立校验;不要求候选间同构,也不裁决是否已经完成创作。""" tree = json.loads(TREE_PATH.read_text(encoding="utf-8")) reports: list[CandidateReport] = [] problems: list[Problem] = [] for item in paths: path = Path(item) try: report, found = _validate(path, tree) except (OSError, UnicodeError) as exc: problems.append(Problem("CANDIDATE_FILE_UNREADABLE", str(path), str(exc))) continue reports.append(report) problems.extend(found) if not reports and not problems: problems.append(Problem("CANDIDATE_FILE_MISSING", "", "请提供一份最终交付稿")) return reports, problems def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description="校验阶段 0 最终交付稿格式") parser.add_argument("files", nargs="+", type=Path, help="已整理的作品探索稿") parser.add_argument("--json", action="store_true", help="输出稳定 JSON") args = parser.parse_args(argv) reports, problems = validate_candidates(args.files) code = "SETTING_INIT_VALIDATION_FAILED" if problems else "SETTING_INIT_OK" if args.json: print(json.dumps({ "status": "invalid" if problems else "ok", "code": code, "reports": [asdict(report) for report in reports], "problems": [asdict(problem) for problem in problems], }, ensure_ascii=False, indent=2)) else: print(f"{code}: {len(reports)} 份交付稿,{len(problems)} 个格式问题") for problem in problems: print(f"[{problem.code}] {problem.file}: {problem.message}") return 1 if problems else 0 if __name__ == "__main__": raise SystemExit(main())