360 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""校验作品设定初始化的根设定与同构候选。
只检查机械易错且人眼难以抽查的项:三级标题树同构、S 编号唯一递增不越号段、
占位符残留、根设定格式与句长。内容是否完整由语义复核按完成判定核对,
本脚本不做数量、字数或内容质量门槛。
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import Iterable
TREE_PATH = Path(__file__).with_name("candidate_tree.json")
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$", re.MULTILINE)
RANGE_RE = re.compile(r"<!--\s*S(\d+)\s*[-—–]\s*S(\d+)\s*-->")
SETTING_RE = re.compile(r"^\s*[-*]\s+(?:\*\*)?(S\d{3,})(?=[^\d])", re.MULTILINE)
PLACEHOLDER_RE = re.compile(
r"(?<![A-Za-z0-9])(?:TODO|TBD|PLACEHOLDER)(?![A-Za-z0-9])|\[待填\]|待补充|待生成|在此填写",
re.IGNORECASE,
)
ROOT_ITEM_RE = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*(.+)$")
ROOT_H2_RE = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE)
ROOT_NESTED_RE = re.compile(r"^#{3,6}\s+", re.MULTILINE)
PROCESS_TERMS = ("Claude", "claude", "子代理", "候选文档", "提示词", "工具调用", "设计过程", "方案来源")
@dataclass(frozen=True)
class ValidationConfig:
min_candidates: int = 1
min_item_chars: int = 20
root_min_chars: int = 30
root_max_chars: int = 60
@dataclass(frozen=True)
class Problem:
code: str
file: str
message: str
@dataclass(frozen=True)
class SectionReport:
heading: str
range_start: int
range_end: int
setting_count: int
@dataclass(frozen=True)
class CandidateReport:
file: str
setting_count: int
headings: tuple[str, ...]
sections: tuple[SectionReport, ...]
def visible_chars(text: str) -> int:
"""去掉常见 Markdown 标记后统计非空白可见字符。"""
cleaned = re.sub(r"[*_`#>\[\]()]", "", text)
return len(re.sub(r"\s+", "", cleaned))
def _problem(code: str, path: Path, message: str) -> Problem:
return Problem(code=code, file=str(path), message=message)
def load_tree(path: Path) -> dict:
return json.loads(path.read_text(encoding="utf-8"))
def expected_heading_sequence(tree: dict) -> list[str]:
"""按冻结树返回 章→节→单元 的扁平标题序列。"""
sequence: list[str] = []
for chapter in tree["chapters"]:
sequence.append(chapter["heading"])
for section in chapter["sections"]:
sequence.append(section["heading"])
sequence.extend(section["units"])
return sequence
def validate_root(path: Path, config: ValidationConfig) -> list[Problem]:
problems: list[Problem] = []
text = path.read_text(encoding="utf-8")
headings = list(ROOT_H2_RE.finditer(text))
root_index = next((index for index, match in enumerate(headings) if "根设定" in match.group(1)), None)
if root_index is None:
return [_problem("ROOT_SECTION_MISSING", path, "未找到包含“根设定”的二级标题")]
start = headings[root_index].end()
end = headings[root_index + 1].start() if root_index + 1 < len(headings) else len(text)
body = text[start:end]
if ROOT_NESTED_RE.search(body):
problems.append(_problem("ROOT_STRUCTURE", path, "根设定内部出现三级或更深标题"))
for term in PROCESS_TERMS:
if term in body:
problems.append(_problem("ROOT_PROCESS_LEAK", path, f"根设定包含设计过程词:{term}"))
labels: set[str] = set()
items = 0
for line_number, raw_line in enumerate(body.splitlines(), start=1):
line = raw_line.strip()
if not line or line == "---":
continue
match = ROOT_ITEM_RE.match(line)
if not match:
problems.append(
_problem("ROOT_ITEM_FORMAT", path, f"根设定第 {line_number} 个相对行不是单行加粗标签条目")
)
continue
items += 1
label = match.group(1).strip()
if label in labels:
problems.append(_problem("ROOT_LABEL_DUPLICATE", path, f"根设定标签重复:{label}"))
labels.add(label)
length = visible_chars(line.lstrip("- "))
if not config.root_min_chars <= length <= config.root_max_chars:
problems.append(
_problem(
"ROOT_ITEM_LENGTH",
path,
f"根设定“{label}”为 {length} 个可见字符,要求 {config.root_min_chars}—{config.root_max_chars}",
)
)
if items == 0:
problems.append(_problem("ROOT_EMPTY", path, "根设定没有可校验条目"))
return problems
def _parse_candidate(path: Path, tree: dict, config: ValidationConfig) -> tuple[CandidateReport, list[Problem]]:
text = path.read_text(encoding="utf-8")
problems: list[Problem] = []
if PLACEHOLDER_RE.search(text):
problems.append(_problem("CANDIDATE_PLACEHOLDER", path, "候选仍含待填占位文字"))
headings = [(len(match.group(1)), match.group(2).strip(), match.start(), match.end()) for match in HEADING_RE.finditer(text)]
h1_count = sum(1 for level, _, _, _ in headings if level == 1)
if h1_count != 1:
problems.append(_problem("CANDIDATE_STRUCTURE", path, f"候选应有且只有一个一级标题,实际 {h1_count} 个"))
deep = [heading for level, heading, _, _ in headings if level >= 4]
if deep:
problems.append(_problem("CANDIDATE_STRUCTURE", path, f"候选出现四级或更深标题:{deep[0]}"))
h2_list = [(heading, start, end) for level, heading, start, end in headings if level == 2]
toc_seen = False
chapter_index = 0
all_ids: list[int] = []
previous_range_end = 0
sections: list[SectionReport] = []
chapter_titles = [chapter["heading"] for chapter in tree["chapters"]]
for position, (heading, start, end) in enumerate(h2_list):
body_end = h2_list[position + 1][1] if position + 1 < len(h2_list) else len(text)
body = text[end:body_end]
if heading == "目录":
toc_seen = True
if re.search(r"^###\s+", body, re.MULTILINE):
problems.append(_problem("CANDIDATE_STRUCTURE", path, "目录下不得出现三级标题"))
continue
if chapter_index >= len(chapter_titles):
problems.append(_problem("TREE_MISMATCH", path, f"多出章节标题:{heading}"))
continue
expected_title = chapter_titles[chapter_index]
if heading != expected_title:
problems.append(
_problem("TREE_MISMATCH", path, f"第 {chapter_index + 1} 章应为“{expected_title}”,实际为“{heading}”")
)
chapter_index += 1
continue
chapter_index += 1
expected_subs: list[str] = []
for section in tree["chapters"][chapter_index - 1]["sections"]:
expected_subs.append(section["heading"])
expected_subs.extend(section["units"])
actual_subs = [sub_heading for level, sub_heading, sub_start, _ in headings if level == 3 and start < sub_start < body_end]
if actual_subs != expected_subs:
expected_set = set(expected_subs)
missing = [item for item in expected_subs if item not in set(actual_subs)]
extra = [item for item in actual_subs if item not in expected_set]
detail = []
if missing:
detail.append(f"缺 {len(missing)} 个(首个:{missing[0]})")
if extra:
detail.append(f"多 {len(extra)} 个(首个:{extra[0]})")
if not detail:
detail.append("顺序与冻结树不一致")
problems.append(_problem("TREE_MISMATCH", path, f"章节“{heading}”标题树漂移:{';'.join(detail)}"))
ranges = RANGE_RE.findall(body)
if len(ranges) != 1:
problems.append(_problem("SECTION_RANGE", path, f"章节“{heading}”必须且只能有一个号段标记"))
continue
range_start, range_end = (int(value) for value in ranges[0])
if range_start > range_end:
problems.append(_problem("SECTION_RANGE", path, f"章节号段倒置:S{range_start:03d}-S{range_end:03d}"))
if range_start <= previous_range_end:
problems.append(_problem("SECTION_RANGE", path, "章节号段未按顺序递增或发生重叠"))
previous_range_end = max(previous_range_end, range_end)
setting_matches = list(SETTING_RE.finditer(body))
ids = [int(match.group(1)[1:]) for match in setting_matches]
all_ids.extend(ids)
for setting_id in ids:
if not range_start <= setting_id <= range_end:
problems.append(
_problem(
"SETTING_ID_OUT_OF_RANGE",
path,
f"S{setting_id:03d} 不在章节号段 S{range_start:03d}-S{range_end:03d} 内",
)
)
for item_index, setting_match in enumerate(setting_matches):
item_end = setting_matches[item_index + 1].start() if item_index + 1 < len(setting_matches) else len(body)
item_text = body[setting_match.start():item_end]
if visible_chars(item_text) < config.min_item_chars:
problems.append(
_problem("SETTING_ITEM_TOO_SHORT", path, f"{setting_match.group(1)} 内容过短,疑似只有标题或占位句")
)
sections.append(
SectionReport(heading=heading, range_start=range_start, range_end=range_end, setting_count=len(ids))
)
if chapter_index < len(chapter_titles):
problems.append(
_problem(
"TREE_MISMATCH",
path,
f"缺少章节:{ '、'.join(chapter_titles[chapter_index:]) }",
)
)
if toc_seen and chapter_index == 0:
problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选只有目录,没有内容章节"))
duplicates = sorted({setting_id for setting_id in all_ids if all_ids.count(setting_id) > 1})
if duplicates:
display = ", ".join(f"S{setting_id:03d}" for setting_id in duplicates[:10])
problems.append(_problem("SETTING_ID_DUPLICATE", path, f"全文编号重复:{display}"))
if any(later <= earlier for earlier, later in zip(all_ids, all_ids[1:])):
problems.append(_problem("SETTING_ORDER", path, "全文 S 编号必须严格递增"))
report = CandidateReport(
file=str(path),
setting_count=len(all_ids),
headings=tuple(heading for heading, _, _ in h2_list),
sections=tuple(sections),
)
return report, problems
def validate_candidates(
paths: Iterable[Path],
config: ValidationConfig,
root_doc: Path | None = None,
tree_path: Path = TREE_PATH,
) -> tuple[list[CandidateReport], list[Problem]]:
candidate_paths = sorted((Path(path) for path in paths), key=lambda item: str(item))
problems: list[Problem] = []
reports: list[CandidateReport] = []
if len(candidate_paths) < config.min_candidates:
problems.append(
Problem(
code="CANDIDATE_COUNT",
file="<group>",
message=f"只有 {len(candidate_paths)} 份候选,要求至少 {config.min_candidates} 份",
)
)
if root_doc is not None:
if not root_doc.is_file():
problems.append(_problem("ROOT_FILE_MISSING", root_doc, "根设定文档不存在"))
else:
problems.extend(validate_root(root_doc, config))
tree = load_tree(tree_path)
for path in candidate_paths:
if not path.is_file():
problems.append(_problem("CANDIDATE_FILE_MISSING", path, "候选文件不存在"))
continue
report, file_problems = _parse_candidate(path, tree, config)
reports.append(report)
problems.extend(file_problems)
if len(reports) > 1:
baseline = reports[0]
baseline_shape = tuple((section.heading, section.range_start, section.range_end) for section in baseline.sections)
for report in reports[1:]:
shape = tuple((section.heading, section.range_start, section.range_end) for section in report.sections)
if report.headings != baseline.headings or shape != baseline_shape:
problems.append(
Problem(
code="CANDIDATE_STRUCTURE_MISMATCH",
file=report.file,
message=f"目录或号段与基准候选 {baseline.file} 不一致",
)
)
return reports, problems
def _build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description="校验作品设定初始化候选")
parser.add_argument("files", nargs="+", type=Path, help="候选 Markdown 文件")
parser.add_argument("--root-doc", type=Path, help="含根设定的唯一前期设计文档")
parser.add_argument("--min-candidates", type=int, default=1)
parser.add_argument("--min-item-chars", type=int, default=20)
parser.add_argument("--root-min-chars", type=int, default=30)
parser.add_argument("--root-max-chars", type=int, default=60)
parser.add_argument("--tree", type=Path, default=TREE_PATH, help="冻结三级树 JSON")
parser.add_argument("--json", action="store_true", help="输出稳定 JSON")
return parser
def main(argv: list[str] | None = None) -> int:
parser = _build_parser()
args = parser.parse_args(argv)
if args.root_min_chars > args.root_max_chars:
parser.error("--root-min-chars 不能大于 --root-max-chars")
config = ValidationConfig(
min_candidates=args.min_candidates,
min_item_chars=args.min_item_chars,
root_min_chars=args.root_min_chars,
root_max_chars=args.root_max_chars,
)
reports, problems = validate_candidates(args.files, config, args.root_doc, args.tree)
status = "ok" if not problems else "invalid"
code = "SETTING_INIT_OK" if not problems else "SETTING_INIT_VALIDATION_FAILED"
payload = {
"status": status,
"code": code,
"reports": [asdict(report) for report in reports],
"problems": [asdict(problem) for problem in problems],
}
if args.json:
print(json.dumps(payload, ensure_ascii=False, indent=2))
else:
print(f"{code}: {len(reports)} 份候选,{len(problems)} 个问题")
for report in reports:
print(f"- {report.file}: {report.setting_count} 项")
for problem in problems:
print(f"[{problem.code}] {problem.file}: {problem.message}")
return 0 if not problems else 1
if __name__ == "__main__":
sys.exit(main())