#!/usr/bin/env python3 """校验作品设定初始化的根设定与同构候选。 只检查机械易错且人眼难以抽查的项:三级标题树同构、S 编号唯一递增不越号段、 占位符残留、根设定格式与句长。内容是否完整由语义复核按完成判定核对, 本脚本不做数量、字数或内容质量门槛。 """ from __future__ import annotations import argparse import json import re import sys from dataclasses import asdict, dataclass from pathlib import Path from typing import Iterable TREE_PATH = Path(__file__).with_name("candidate_tree.json") HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$", re.MULTILINE) RANGE_RE = re.compile(r"") SETTING_RE = re.compile(r"^\s*[-*]\s+(?:\*\*)?(S\d{3,})(?=[^\d])", re.MULTILINE) PLACEHOLDER_RE = re.compile( r"(? int: """去掉常见 Markdown 标记后统计非空白可见字符。""" cleaned = re.sub(r"[*_`#>\[\]()]", "", text) return len(re.sub(r"\s+", "", cleaned)) def _problem(code: str, path: Path, message: str) -> Problem: return Problem(code=code, file=str(path), message=message) def load_tree(path: Path) -> dict: return json.loads(path.read_text(encoding="utf-8")) def expected_heading_sequence(tree: dict) -> list[str]: """按冻结树返回 章→节→单元 的扁平标题序列。""" sequence: list[str] = [] for chapter in tree["chapters"]: sequence.append(chapter["heading"]) for section in chapter["sections"]: sequence.append(section["heading"]) sequence.extend(section["units"]) return sequence def validate_root(path: Path, config: ValidationConfig) -> list[Problem]: problems: list[Problem] = [] text = path.read_text(encoding="utf-8") headings = list(ROOT_H2_RE.finditer(text)) root_index = next((index for index, match in enumerate(headings) if "根设定" in match.group(1)), None) if root_index is None: return [_problem("ROOT_SECTION_MISSING", path, "未找到包含“根设定”的二级标题")] start = headings[root_index].end() end = headings[root_index + 1].start() if root_index + 1 < len(headings) else len(text) body = text[start:end] if ROOT_NESTED_RE.search(body): problems.append(_problem("ROOT_STRUCTURE", path, "根设定内部出现三级或更深标题")) for term in PROCESS_TERMS: if term in body: problems.append(_problem("ROOT_PROCESS_LEAK", path, f"根设定包含设计过程词:{term}")) labels: set[str] = set() items = 0 for line_number, raw_line in enumerate(body.splitlines(), start=1): line = raw_line.strip() if not line or line == "---": continue match = ROOT_ITEM_RE.match(line) if not match: problems.append( _problem("ROOT_ITEM_FORMAT", path, f"根设定第 {line_number} 个相对行不是单行加粗标签条目") ) continue items += 1 label = match.group(1).strip() if label in labels: problems.append(_problem("ROOT_LABEL_DUPLICATE", path, f"根设定标签重复:{label}")) labels.add(label) length = visible_chars(line.lstrip("- ")) if not config.root_min_chars <= length <= config.root_max_chars: problems.append( _problem( "ROOT_ITEM_LENGTH", path, f"根设定“{label}”为 {length} 个可见字符,要求 {config.root_min_chars}—{config.root_max_chars}", ) ) if items == 0: problems.append(_problem("ROOT_EMPTY", path, "根设定没有可校验条目")) return problems def _parse_candidate(path: Path, tree: dict, config: ValidationConfig) -> tuple[CandidateReport, list[Problem]]: text = path.read_text(encoding="utf-8") problems: list[Problem] = [] if PLACEHOLDER_RE.search(text): problems.append(_problem("CANDIDATE_PLACEHOLDER", path, "候选仍含待填占位文字")) headings = [(len(match.group(1)), match.group(2).strip(), match.start(), match.end()) for match in HEADING_RE.finditer(text)] h1_count = sum(1 for level, _, _, _ in headings if level == 1) if h1_count != 1: problems.append(_problem("CANDIDATE_STRUCTURE", path, f"候选应有且只有一个一级标题,实际 {h1_count} 个")) deep = [heading for level, heading, _, _ in headings if level >= 4] if deep: problems.append(_problem("CANDIDATE_STRUCTURE", path, f"候选出现四级或更深标题:{deep[0]}")) h2_list = [(heading, start, end) for level, heading, start, end in headings if level == 2] toc_seen = False chapter_index = 0 all_ids: list[int] = [] previous_range_end = 0 sections: list[SectionReport] = [] chapter_titles = [chapter["heading"] for chapter in tree["chapters"]] for position, (heading, start, end) in enumerate(h2_list): body_end = h2_list[position + 1][1] if position + 1 < len(h2_list) else len(text) body = text[end:body_end] if heading == "目录": toc_seen = True if re.search(r"^###\s+", body, re.MULTILINE): problems.append(_problem("CANDIDATE_STRUCTURE", path, "目录下不得出现三级标题")) continue if chapter_index >= len(chapter_titles): problems.append(_problem("TREE_MISMATCH", path, f"多出章节标题:{heading}")) continue expected_title = chapter_titles[chapter_index] if heading != expected_title: problems.append( _problem("TREE_MISMATCH", path, f"第 {chapter_index + 1} 章应为“{expected_title}”,实际为“{heading}”") ) chapter_index += 1 continue chapter_index += 1 expected_subs: list[str] = [] for section in tree["chapters"][chapter_index - 1]["sections"]: expected_subs.append(section["heading"]) expected_subs.extend(section["units"]) actual_subs = [sub_heading for level, sub_heading, sub_start, _ in headings if level == 3 and start < sub_start < body_end] if actual_subs != expected_subs: expected_set = set(expected_subs) missing = [item for item in expected_subs if item not in set(actual_subs)] extra = [item for item in actual_subs if item not in expected_set] detail = [] if missing: detail.append(f"缺 {len(missing)} 个(首个:{missing[0]})") if extra: detail.append(f"多 {len(extra)} 个(首个:{extra[0]})") if not detail: detail.append("顺序与冻结树不一致") problems.append(_problem("TREE_MISMATCH", path, f"章节“{heading}”标题树漂移:{';'.join(detail)}")) ranges = RANGE_RE.findall(body) if len(ranges) != 1: problems.append(_problem("SECTION_RANGE", path, f"章节“{heading}”必须且只能有一个号段标记")) continue range_start, range_end = (int(value) for value in ranges[0]) if range_start > range_end: problems.append(_problem("SECTION_RANGE", path, f"章节号段倒置:S{range_start:03d}-S{range_end:03d}")) if range_start <= previous_range_end: problems.append(_problem("SECTION_RANGE", path, "章节号段未按顺序递增或发生重叠")) previous_range_end = max(previous_range_end, range_end) setting_matches = list(SETTING_RE.finditer(body)) ids = [int(match.group(1)[1:]) for match in setting_matches] all_ids.extend(ids) for setting_id in ids: if not range_start <= setting_id <= range_end: problems.append( _problem( "SETTING_ID_OUT_OF_RANGE", path, f"S{setting_id:03d} 不在章节号段 S{range_start:03d}-S{range_end:03d} 内", ) ) for item_index, setting_match in enumerate(setting_matches): item_end = setting_matches[item_index + 1].start() if item_index + 1 < len(setting_matches) else len(body) item_text = body[setting_match.start():item_end] if visible_chars(item_text) < config.min_item_chars: problems.append( _problem("SETTING_ITEM_TOO_SHORT", path, f"{setting_match.group(1)} 内容过短,疑似只有标题或占位句") ) sections.append( SectionReport(heading=heading, range_start=range_start, range_end=range_end, setting_count=len(ids)) ) if chapter_index < len(chapter_titles): problems.append( _problem( "TREE_MISMATCH", path, f"缺少章节:{ '、'.join(chapter_titles[chapter_index:]) }", ) ) if toc_seen and chapter_index == 0: problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选只有目录,没有内容章节")) duplicates = sorted({setting_id for setting_id in all_ids if all_ids.count(setting_id) > 1}) if duplicates: display = ", ".join(f"S{setting_id:03d}" for setting_id in duplicates[:10]) problems.append(_problem("SETTING_ID_DUPLICATE", path, f"全文编号重复:{display}")) if any(later <= earlier for earlier, later in zip(all_ids, all_ids[1:])): problems.append(_problem("SETTING_ORDER", path, "全文 S 编号必须严格递增")) report = CandidateReport( file=str(path), setting_count=len(all_ids), headings=tuple(heading for heading, _, _ in h2_list), sections=tuple(sections), ) return report, problems def validate_candidates( paths: Iterable[Path], config: ValidationConfig, root_doc: Path | None = None, tree_path: Path = TREE_PATH, ) -> tuple[list[CandidateReport], list[Problem]]: candidate_paths = sorted((Path(path) for path in paths), key=lambda item: str(item)) problems: list[Problem] = [] reports: list[CandidateReport] = [] if len(candidate_paths) < config.min_candidates: problems.append( Problem( code="CANDIDATE_COUNT", file="", message=f"只有 {len(candidate_paths)} 份候选,要求至少 {config.min_candidates} 份", ) ) if root_doc is not None: if not root_doc.is_file(): problems.append(_problem("ROOT_FILE_MISSING", root_doc, "根设定文档不存在")) else: problems.extend(validate_root(root_doc, config)) tree = load_tree(tree_path) for path in candidate_paths: if not path.is_file(): problems.append(_problem("CANDIDATE_FILE_MISSING", path, "候选文件不存在")) continue report, file_problems = _parse_candidate(path, tree, config) reports.append(report) problems.extend(file_problems) if len(reports) > 1: baseline = reports[0] baseline_shape = tuple((section.heading, section.range_start, section.range_end) for section in baseline.sections) for report in reports[1:]: shape = tuple((section.heading, section.range_start, section.range_end) for section in report.sections) if report.headings != baseline.headings or shape != baseline_shape: problems.append( Problem( code="CANDIDATE_STRUCTURE_MISMATCH", file=report.file, message=f"目录或号段与基准候选 {baseline.file} 不一致", ) ) return reports, problems def _build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description="校验作品设定初始化候选") parser.add_argument("files", nargs="+", type=Path, help="候选 Markdown 文件") parser.add_argument("--root-doc", type=Path, help="含根设定的唯一前期设计文档") parser.add_argument("--min-candidates", type=int, default=1) parser.add_argument("--min-item-chars", type=int, default=20) parser.add_argument("--root-min-chars", type=int, default=30) parser.add_argument("--root-max-chars", type=int, default=60) parser.add_argument("--tree", type=Path, default=TREE_PATH, help="冻结三级树 JSON") parser.add_argument("--json", action="store_true", help="输出稳定 JSON") return parser def main(argv: list[str] | None = None) -> int: parser = _build_parser() args = parser.parse_args(argv) if args.root_min_chars > args.root_max_chars: parser.error("--root-min-chars 不能大于 --root-max-chars") config = ValidationConfig( min_candidates=args.min_candidates, min_item_chars=args.min_item_chars, root_min_chars=args.root_min_chars, root_max_chars=args.root_max_chars, ) reports, problems = validate_candidates(args.files, config, args.root_doc, args.tree) status = "ok" if not problems else "invalid" code = "SETTING_INIT_OK" if not problems else "SETTING_INIT_VALIDATION_FAILED" payload = { "status": status, "code": code, "reports": [asdict(report) for report in reports], "problems": [asdict(problem) for problem in problems], } if args.json: print(json.dumps(payload, ensure_ascii=False, indent=2)) else: print(f"{code}: {len(reports)} 份候选,{len(problems)} 个问题") for report in reports: print(f"- {report.file}: {report.setting_count} 项") for problem in problems: print(f"[{problem.code}] {problem.file}: {problem.message}") return 0 if not problems else 1 if __name__ == "__main__": sys.exit(main())