一、技能重组(动作-对象命名) - 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为 clean-book-text/decide-candidate/write-next-chapter/access-database/ check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持) - agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名 二、先审后入创作闭环(本次核心) 正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交", DB 级兜底,编排层跳步即被硬拒。 - candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链 - fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量, 模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本 - projection_registry.py + example_projection_run(107):投影登记与恢复 - acceptance_state.py:接受前置实时状态重读 - lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格 - DDL 105:example_candidate 增 semantic_status/semantic_report_sha256 - write_canonical.accept:语义兜底+同事务合并增量+登记投影; run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链 - claude_runtime:兼容新 CLI modelUsage 信息字段 三、审查修复(独立子代理四维审查后) - 事实增量 propose→approve 翻态正道,不撞唯一键 - 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记 - 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理 测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。 创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
339 lines
13 KiB
Python
339 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
"""校验作品设定初始化的根设定与同构候选,不调用模型、不访问数据库。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import re
|
||
import sys
|
||
from dataclasses import asdict, dataclass
|
||
from pathlib import Path
|
||
from typing import Iterable
|
||
|
||
|
||
H2_RE = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE)
|
||
NESTED_HEADING_RE = re.compile(r"^#{3,6}\s+", re.MULTILINE)
|
||
RANGE_RE = re.compile(r"<!--\s*S(\d+)\s*[-—–]\s*S(\d+)\s*-->")
|
||
SETTING_RE = re.compile(r"^\s*[-*]\s+(?:\*\*)?(S\d{3,})(?=[^\d])", re.MULTILINE)
|
||
ROOT_ITEM_RE = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*(.+)$")
|
||
PLACEHOLDER_RE = re.compile(
|
||
r"(?:\bTODO\b|\bTBD\b|\[待填\]|待补充|待生成|在此填写|PLACEHOLDER)",
|
||
re.IGNORECASE,
|
||
)
|
||
PROCESS_TERMS = ("Claude", "claude", "子代理", "候选文档", "提示词", "工具调用", "设计过程", "方案来源")
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class ValidationConfig:
|
||
min_candidates: int = 1
|
||
min_settings: int = 100
|
||
min_section_settings: int = 10
|
||
max_section_settings: int = 50
|
||
min_chars: int = 50000
|
||
min_item_chars: int = 20
|
||
root_min_chars: int = 30
|
||
root_max_chars: int = 60
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class Problem:
|
||
code: str
|
||
file: str
|
||
message: str
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class SectionReport:
|
||
heading: str
|
||
range_start: int
|
||
range_end: int
|
||
setting_count: int
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class CandidateReport:
|
||
file: str
|
||
effective_chars: int
|
||
setting_count: int
|
||
headings: tuple[str, ...]
|
||
sections: tuple[SectionReport, ...]
|
||
|
||
|
||
def effective_chars(text: str) -> int:
|
||
"""按非空白字符计数,避免用空行填充篇幅。"""
|
||
return len(re.sub(r"\s+", "", text))
|
||
|
||
|
||
def visible_chars(text: str) -> int:
|
||
"""去掉常见 Markdown 标记后统计可见字符。"""
|
||
cleaned = re.sub(r"[*_`#>\[\]()]", "", text)
|
||
return effective_chars(cleaned)
|
||
|
||
|
||
def _problem(code: str, path: Path, message: str) -> Problem:
|
||
return Problem(code=code, file=str(path), message=message)
|
||
|
||
|
||
def validate_root(path: Path, config: ValidationConfig) -> list[Problem]:
|
||
problems: list[Problem] = []
|
||
text = path.read_text(encoding="utf-8")
|
||
headings = list(H2_RE.finditer(text))
|
||
root_index = next((index for index, match in enumerate(headings) if "根设定" in match.group(1)), None)
|
||
if root_index is None:
|
||
return [_problem("ROOT_SECTION_MISSING", path, "未找到包含“根设定”的二级标题")]
|
||
|
||
start = headings[root_index].end()
|
||
end = headings[root_index + 1].start() if root_index + 1 < len(headings) else len(text)
|
||
body = text[start:end]
|
||
|
||
if NESTED_HEADING_RE.search(body):
|
||
problems.append(_problem("ROOT_STRUCTURE", path, "根设定内部出现三级或更深标题"))
|
||
|
||
for term in PROCESS_TERMS:
|
||
if term in body:
|
||
problems.append(_problem("ROOT_PROCESS_LEAK", path, f"根设定包含设计过程词:{term}"))
|
||
|
||
labels: set[str] = set()
|
||
items = 0
|
||
for line_number, raw_line in enumerate(body.splitlines(), start=1):
|
||
line = raw_line.strip()
|
||
if not line or line == "---":
|
||
continue
|
||
match = ROOT_ITEM_RE.match(line)
|
||
if not match:
|
||
problems.append(
|
||
_problem("ROOT_ITEM_FORMAT", path, f"根设定第 {line_number} 个相对行不是单行加粗标签条目")
|
||
)
|
||
continue
|
||
items += 1
|
||
label = match.group(1).strip()
|
||
if label in labels:
|
||
problems.append(_problem("ROOT_LABEL_DUPLICATE", path, f"根设定标签重复:{label}"))
|
||
labels.add(label)
|
||
length = visible_chars(line.lstrip("- "))
|
||
if not config.root_min_chars <= length <= config.root_max_chars:
|
||
problems.append(
|
||
_problem(
|
||
"ROOT_ITEM_LENGTH",
|
||
path,
|
||
f"根设定“{label}”为 {length} 个可见字符,要求 {config.root_min_chars}—{config.root_max_chars}",
|
||
)
|
||
)
|
||
if items == 0:
|
||
problems.append(_problem("ROOT_EMPTY", path, "根设定没有可校验条目"))
|
||
return problems
|
||
|
||
|
||
def _parse_candidate(path: Path, config: ValidationConfig) -> tuple[CandidateReport, list[Problem]]:
|
||
text = path.read_text(encoding="utf-8")
|
||
problems: list[Problem] = []
|
||
h2_matches = list(H2_RE.finditer(text))
|
||
headings = tuple(match.group(1).strip() for match in h2_matches)
|
||
|
||
if not h2_matches:
|
||
problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选没有二级标题"))
|
||
if NESTED_HEADING_RE.search(text):
|
||
problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选出现三级或更深标题,违反统一两级结构"))
|
||
if PLACEHOLDER_RE.search(text):
|
||
problems.append(_problem("CANDIDATE_PLACEHOLDER", path, "候选仍含待填占位文字"))
|
||
|
||
sections: list[SectionReport] = []
|
||
all_ids: list[int] = []
|
||
previous_range_end = 0
|
||
|
||
for index, heading_match in enumerate(h2_matches):
|
||
start = heading_match.end()
|
||
end = h2_matches[index + 1].start() if index + 1 < len(h2_matches) else len(text)
|
||
body = text[start:end]
|
||
ranges = RANGE_RE.findall(body)
|
||
setting_matches = list(SETTING_RE.finditer(body))
|
||
|
||
if not ranges and not setting_matches and heading_match.group(1).strip() == "目录":
|
||
continue
|
||
if len(ranges) != 1:
|
||
problems.append(
|
||
_problem("SECTION_RANGE", path, f"章节“{heading_match.group(1).strip()}”必须且只能有一个号段标记")
|
||
)
|
||
continue
|
||
|
||
range_start, range_end = (int(value) for value in ranges[0])
|
||
if range_start > range_end:
|
||
problems.append(_problem("SECTION_RANGE", path, f"章节号段倒置:S{range_start:03d}-S{range_end:03d}"))
|
||
if range_start <= previous_range_end:
|
||
problems.append(_problem("SECTION_RANGE", path, "章节号段未按顺序递增或发生重叠"))
|
||
previous_range_end = max(previous_range_end, range_end)
|
||
|
||
ids = [int(match.group(1)[1:]) for match in setting_matches]
|
||
all_ids.extend(ids)
|
||
if ids != sorted(ids):
|
||
problems.append(_problem("SETTING_ORDER", path, f"章节“{heading_match.group(1).strip()}”的编号未递增"))
|
||
for setting_id in ids:
|
||
if not range_start <= setting_id <= range_end:
|
||
problems.append(
|
||
_problem(
|
||
"SETTING_ID_OUT_OF_RANGE",
|
||
path,
|
||
f"S{setting_id:03d} 不在章节号段 S{range_start:03d}-S{range_end:03d} 内",
|
||
)
|
||
)
|
||
|
||
count = len(ids)
|
||
if not config.min_section_settings <= count <= config.max_section_settings:
|
||
problems.append(
|
||
_problem(
|
||
"SECTION_SETTING_COUNT",
|
||
path,
|
||
f"章节“{heading_match.group(1).strip()}”有 {count} 项,要求 {config.min_section_settings}—{config.max_section_settings}",
|
||
)
|
||
)
|
||
|
||
for item_index, setting_match in enumerate(setting_matches):
|
||
item_end = setting_matches[item_index + 1].start() if item_index + 1 < len(setting_matches) else len(body)
|
||
item_text = body[setting_match.start():item_end]
|
||
if visible_chars(item_text) < config.min_item_chars:
|
||
problems.append(
|
||
_problem("SETTING_ITEM_TOO_SHORT", path, f"{setting_match.group(1)} 内容过短,疑似只有标题或占位句")
|
||
)
|
||
|
||
sections.append(
|
||
SectionReport(
|
||
heading=heading_match.group(1).strip(),
|
||
range_start=range_start,
|
||
range_end=range_end,
|
||
setting_count=count,
|
||
)
|
||
)
|
||
|
||
duplicates = sorted({setting_id for setting_id in all_ids if all_ids.count(setting_id) > 1})
|
||
if duplicates:
|
||
display = ", ".join(f"S{setting_id:03d}" for setting_id in duplicates[:10])
|
||
problems.append(_problem("SETTING_ID_DUPLICATE", path, f"全文编号重复:{display}"))
|
||
|
||
char_count = effective_chars(text)
|
||
if char_count < config.min_chars:
|
||
problems.append(
|
||
_problem("CANDIDATE_LENGTH", path, f"有效字符 {char_count},低于门槛 {config.min_chars}")
|
||
)
|
||
if len(all_ids) < config.min_settings:
|
||
problems.append(
|
||
_problem("CANDIDATE_SETTING_COUNT", path, f"全文共 {len(all_ids)} 项设定,低于门槛 {config.min_settings}")
|
||
)
|
||
|
||
report = CandidateReport(
|
||
file=str(path),
|
||
effective_chars=char_count,
|
||
setting_count=len(all_ids),
|
||
headings=headings,
|
||
sections=tuple(sections),
|
||
)
|
||
return report, problems
|
||
|
||
|
||
def validate_candidates(
|
||
paths: Iterable[Path],
|
||
config: ValidationConfig,
|
||
root_doc: Path | None = None,
|
||
) -> tuple[list[CandidateReport], list[Problem]]:
|
||
candidate_paths = sorted((Path(path) for path in paths), key=lambda item: str(item))
|
||
problems: list[Problem] = []
|
||
reports: list[CandidateReport] = []
|
||
|
||
if len(candidate_paths) < config.min_candidates:
|
||
problems.append(
|
||
Problem(
|
||
code="CANDIDATE_COUNT",
|
||
file="<group>",
|
||
message=f"只有 {len(candidate_paths)} 份候选,要求至少 {config.min_candidates} 份",
|
||
)
|
||
)
|
||
if root_doc is not None:
|
||
if not root_doc.is_file():
|
||
problems.append(_problem("ROOT_FILE_MISSING", root_doc, "根设定文档不存在"))
|
||
else:
|
||
problems.extend(validate_root(root_doc, config))
|
||
|
||
for path in candidate_paths:
|
||
if not path.is_file():
|
||
problems.append(_problem("CANDIDATE_FILE_MISSING", path, "候选文件不存在"))
|
||
continue
|
||
report, file_problems = _parse_candidate(path, config)
|
||
reports.append(report)
|
||
problems.extend(file_problems)
|
||
|
||
if reports:
|
||
baseline = reports[0]
|
||
baseline_shape = tuple(
|
||
(section.heading, section.range_start, section.range_end) for section in baseline.sections
|
||
)
|
||
for report in reports[1:]:
|
||
shape = tuple((section.heading, section.range_start, section.range_end) for section in report.sections)
|
||
if report.headings != baseline.headings or shape != baseline_shape:
|
||
problems.append(
|
||
Problem(
|
||
code="CANDIDATE_STRUCTURE_MISMATCH",
|
||
file=report.file,
|
||
message=f"目录或号段与基准候选 {baseline.file} 不一致",
|
||
)
|
||
)
|
||
return reports, problems
|
||
|
||
|
||
def _build_parser() -> argparse.ArgumentParser:
|
||
parser = argparse.ArgumentParser(description="校验作品设定初始化候选")
|
||
parser.add_argument("files", nargs="+", type=Path, help="候选 Markdown 文件")
|
||
parser.add_argument("--root-doc", type=Path, help="含根设定的唯一前期设计文档")
|
||
parser.add_argument("--min-candidates", type=int, default=1)
|
||
parser.add_argument("--min-settings", type=int, default=100)
|
||
parser.add_argument("--min-section-settings", type=int, default=10)
|
||
parser.add_argument("--max-section-settings", type=int, default=50)
|
||
parser.add_argument("--min-chars", type=int, default=50000)
|
||
parser.add_argument("--min-item-chars", type=int, default=20)
|
||
parser.add_argument("--root-min-chars", type=int, default=30)
|
||
parser.add_argument("--root-max-chars", type=int, default=60)
|
||
parser.add_argument("--json", action="store_true", help="输出稳定 JSON")
|
||
return parser
|
||
|
||
|
||
def main(argv: list[str] | None = None) -> int:
|
||
parser = _build_parser()
|
||
args = parser.parse_args(argv)
|
||
if args.min_section_settings > args.max_section_settings:
|
||
parser.error("--min-section-settings 不能大于 --max-section-settings")
|
||
if args.root_min_chars > args.root_max_chars:
|
||
parser.error("--root-min-chars 不能大于 --root-max-chars")
|
||
|
||
config = ValidationConfig(
|
||
min_candidates=args.min_candidates,
|
||
min_settings=args.min_settings,
|
||
min_section_settings=args.min_section_settings,
|
||
max_section_settings=args.max_section_settings,
|
||
min_chars=args.min_chars,
|
||
min_item_chars=args.min_item_chars,
|
||
root_min_chars=args.root_min_chars,
|
||
root_max_chars=args.root_max_chars,
|
||
)
|
||
reports, problems = validate_candidates(args.files, config, args.root_doc)
|
||
status = "ok" if not problems else "invalid"
|
||
code = "SETTING_INIT_OK" if not problems else "SETTING_INIT_VALIDATION_FAILED"
|
||
payload = {
|
||
"status": status,
|
||
"code": code,
|
||
"reports": [asdict(report) for report in reports],
|
||
"problems": [asdict(problem) for problem in problems],
|
||
}
|
||
|
||
if args.json:
|
||
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
||
else:
|
||
print(f"{code}: {len(reports)} 份候选,{len(problems)} 个问题")
|
||
for report in reports:
|
||
print(f"- {report.file}: {report.setting_count} 项,{report.effective_chars} 有效字符")
|
||
for problem in problems:
|
||
print(f"[{problem.code}] {problem.file}: {problem.message}")
|
||
return 0 if not problems else 1
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|