zizi b0bc7a8745 框架: 技能按动作-对象重组 + 先审后入创作闭环
一、技能重组(动作-对象命名)
- 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为
  clean-book-text/decide-candidate/write-next-chapter/access-database/
  check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持)
- agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名

二、先审后入创作闭环(本次核心)
正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交",
DB 级兜底,编排层跳步即被硬拒。
- candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链
- fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量,
  模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本
- projection_registry.py + example_projection_run(107):投影登记与恢复
- acceptance_state.py:接受前置实时状态重读
- lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格
- DDL 105:example_candidate 增 semantic_status/semantic_report_sha256
- write_canonical.accept:语义兜底+同事务合并增量+登记投影;
  run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链
- claude_runtime:兼容新 CLI modelUsage 信息字段

三、审查修复(独立子代理四维审查后)
- 事实增量 propose→approve 翻态正道,不撞唯一键
- 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记
- 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理

测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。
创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
2026-08-14 10:24:08 +08:00

339 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""校验作品设定初始化的根设定与同构候选,不调用模型、不访问数据库。"""
from __future__ import annotations
import argparse
import json
import re
import sys
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import Iterable
H2_RE = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE)
NESTED_HEADING_RE = re.compile(r"^#{3,6}\s+", re.MULTILINE)
RANGE_RE = re.compile(r"<!--\s*S(\d+)\s*[-—–]\s*S(\d+)\s*-->")
SETTING_RE = re.compile(r"^\s*[-*]\s+(?:\*\*)?(S\d{3,})(?=[^\d])", re.MULTILINE)
ROOT_ITEM_RE = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*(.+)$")
PLACEHOLDER_RE = re.compile(
r"(?:\bTODO\b|\bTBD\b|\[待填\]|待补充|待生成|在此填写|PLACEHOLDER)",
re.IGNORECASE,
)
PROCESS_TERMS = ("Claude", "claude", "子代理", "候选文档", "提示词", "工具调用", "设计过程", "方案来源")
@dataclass(frozen=True)
class ValidationConfig:
min_candidates: int = 1
min_settings: int = 100
min_section_settings: int = 10
max_section_settings: int = 50
min_chars: int = 50000
min_item_chars: int = 20
root_min_chars: int = 30
root_max_chars: int = 60
@dataclass(frozen=True)
class Problem:
code: str
file: str
message: str
@dataclass(frozen=True)
class SectionReport:
heading: str
range_start: int
range_end: int
setting_count: int
@dataclass(frozen=True)
class CandidateReport:
file: str
effective_chars: int
setting_count: int
headings: tuple[str, ...]
sections: tuple[SectionReport, ...]
def effective_chars(text: str) -> int:
"""按非空白字符计数,避免用空行填充篇幅。"""
return len(re.sub(r"\s+", "", text))
def visible_chars(text: str) -> int:
"""去掉常见 Markdown 标记后统计可见字符。"""
cleaned = re.sub(r"[*_`#>\[\]()]", "", text)
return effective_chars(cleaned)
def _problem(code: str, path: Path, message: str) -> Problem:
return Problem(code=code, file=str(path), message=message)
def validate_root(path: Path, config: ValidationConfig) -> list[Problem]:
problems: list[Problem] = []
text = path.read_text(encoding="utf-8")
headings = list(H2_RE.finditer(text))
root_index = next((index for index, match in enumerate(headings) if "根设定" in match.group(1)), None)
if root_index is None:
return [_problem("ROOT_SECTION_MISSING", path, "未找到包含“根设定”的二级标题")]
start = headings[root_index].end()
end = headings[root_index + 1].start() if root_index + 1 < len(headings) else len(text)
body = text[start:end]
if NESTED_HEADING_RE.search(body):
problems.append(_problem("ROOT_STRUCTURE", path, "根设定内部出现三级或更深标题"))
for term in PROCESS_TERMS:
if term in body:
problems.append(_problem("ROOT_PROCESS_LEAK", path, f"根设定包含设计过程词:{term}"))
labels: set[str] = set()
items = 0
for line_number, raw_line in enumerate(body.splitlines(), start=1):
line = raw_line.strip()
if not line or line == "---":
continue
match = ROOT_ITEM_RE.match(line)
if not match:
problems.append(
_problem("ROOT_ITEM_FORMAT", path, f"根设定第 {line_number} 个相对行不是单行加粗标签条目")
)
continue
items += 1
label = match.group(1).strip()
if label in labels:
problems.append(_problem("ROOT_LABEL_DUPLICATE", path, f"根设定标签重复:{label}"))
labels.add(label)
length = visible_chars(line.lstrip("- "))
if not config.root_min_chars <= length <= config.root_max_chars:
problems.append(
_problem(
"ROOT_ITEM_LENGTH",
path,
f"根设定“{label}”为 {length} 个可见字符,要求 {config.root_min_chars}—{config.root_max_chars}",
)
)
if items == 0:
problems.append(_problem("ROOT_EMPTY", path, "根设定没有可校验条目"))
return problems
def _parse_candidate(path: Path, config: ValidationConfig) -> tuple[CandidateReport, list[Problem]]:
text = path.read_text(encoding="utf-8")
problems: list[Problem] = []
h2_matches = list(H2_RE.finditer(text))
headings = tuple(match.group(1).strip() for match in h2_matches)
if not h2_matches:
problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选没有二级标题"))
if NESTED_HEADING_RE.search(text):
problems.append(_problem("CANDIDATE_STRUCTURE", path, "候选出现三级或更深标题,违反统一两级结构"))
if PLACEHOLDER_RE.search(text):
problems.append(_problem("CANDIDATE_PLACEHOLDER", path, "候选仍含待填占位文字"))
sections: list[SectionReport] = []
all_ids: list[int] = []
previous_range_end = 0
for index, heading_match in enumerate(h2_matches):
start = heading_match.end()
end = h2_matches[index + 1].start() if index + 1 < len(h2_matches) else len(text)
body = text[start:end]
ranges = RANGE_RE.findall(body)
setting_matches = list(SETTING_RE.finditer(body))
if not ranges and not setting_matches and heading_match.group(1).strip() == "目录":
continue
if len(ranges) != 1:
problems.append(
_problem("SECTION_RANGE", path, f"章节“{heading_match.group(1).strip()}”必须且只能有一个号段标记")
)
continue
range_start, range_end = (int(value) for value in ranges[0])
if range_start > range_end:
problems.append(_problem("SECTION_RANGE", path, f"章节号段倒置:S{range_start:03d}-S{range_end:03d}"))
if range_start <= previous_range_end:
problems.append(_problem("SECTION_RANGE", path, "章节号段未按顺序递增或发生重叠"))
previous_range_end = max(previous_range_end, range_end)
ids = [int(match.group(1)[1:]) for match in setting_matches]
all_ids.extend(ids)
if ids != sorted(ids):
problems.append(_problem("SETTING_ORDER", path, f"章节“{heading_match.group(1).strip()}”的编号未递增"))
for setting_id in ids:
if not range_start <= setting_id <= range_end:
problems.append(
_problem(
"SETTING_ID_OUT_OF_RANGE",
path,
f"S{setting_id:03d} 不在章节号段 S{range_start:03d}-S{range_end:03d} 内",
)
)
count = len(ids)
if not config.min_section_settings <= count <= config.max_section_settings:
problems.append(
_problem(
"SECTION_SETTING_COUNT",
path,
f"章节“{heading_match.group(1).strip()}”有 {count} 项,要求 {config.min_section_settings}—{config.max_section_settings}",
)
)
for item_index, setting_match in enumerate(setting_matches):
item_end = setting_matches[item_index + 1].start() if item_index + 1 < len(setting_matches) else len(body)
item_text = body[setting_match.start():item_end]
if visible_chars(item_text) < config.min_item_chars:
problems.append(
_problem("SETTING_ITEM_TOO_SHORT", path, f"{setting_match.group(1)} 内容过短,疑似只有标题或占位句")
)
sections.append(
SectionReport(
heading=heading_match.group(1).strip(),
range_start=range_start,
range_end=range_end,
setting_count=count,
)
)
duplicates = sorted({setting_id for setting_id in all_ids if all_ids.count(setting_id) > 1})
if duplicates:
display = ", ".join(f"S{setting_id:03d}" for setting_id in duplicates[:10])
problems.append(_problem("SETTING_ID_DUPLICATE", path, f"全文编号重复:{display}"))
char_count = effective_chars(text)
if char_count < config.min_chars:
problems.append(
_problem("CANDIDATE_LENGTH", path, f"有效字符 {char_count},低于门槛 {config.min_chars}")
)
if len(all_ids) < config.min_settings:
problems.append(
_problem("CANDIDATE_SETTING_COUNT", path, f"全文共 {len(all_ids)} 项设定,低于门槛 {config.min_settings}")
)
report = CandidateReport(
file=str(path),
effective_chars=char_count,
setting_count=len(all_ids),
headings=headings,
sections=tuple(sections),
)
return report, problems
def validate_candidates(
paths: Iterable[Path],
config: ValidationConfig,
root_doc: Path | None = None,
) -> tuple[list[CandidateReport], list[Problem]]:
candidate_paths = sorted((Path(path) for path in paths), key=lambda item: str(item))
problems: list[Problem] = []
reports: list[CandidateReport] = []
if len(candidate_paths) < config.min_candidates:
problems.append(
Problem(
code="CANDIDATE_COUNT",
file="<group>",
message=f"只有 {len(candidate_paths)} 份候选,要求至少 {config.min_candidates} 份",
)
)
if root_doc is not None:
if not root_doc.is_file():
problems.append(_problem("ROOT_FILE_MISSING", root_doc, "根设定文档不存在"))
else:
problems.extend(validate_root(root_doc, config))
for path in candidate_paths:
if not path.is_file():
problems.append(_problem("CANDIDATE_FILE_MISSING", path, "候选文件不存在"))
continue
report, file_problems = _parse_candidate(path, config)
reports.append(report)
problems.extend(file_problems)
if reports:
baseline = reports[0]
baseline_shape = tuple(
(section.heading, section.range_start, section.range_end) for section in baseline.sections
)
for report in reports[1:]:
shape = tuple((section.heading, section.range_start, section.range_end) for section in report.sections)
if report.headings != baseline.headings or shape != baseline_shape:
problems.append(
Problem(
code="CANDIDATE_STRUCTURE_MISMATCH",
file=report.file,
message=f"目录或号段与基准候选 {baseline.file} 不一致",
)
)
return reports, problems
def _build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description="校验作品设定初始化候选")
parser.add_argument("files", nargs="+", type=Path, help="候选 Markdown 文件")
parser.add_argument("--root-doc", type=Path, help="含根设定的唯一前期设计文档")
parser.add_argument("--min-candidates", type=int, default=1)
parser.add_argument("--min-settings", type=int, default=100)
parser.add_argument("--min-section-settings", type=int, default=10)
parser.add_argument("--max-section-settings", type=int, default=50)
parser.add_argument("--min-chars", type=int, default=50000)
parser.add_argument("--min-item-chars", type=int, default=20)
parser.add_argument("--root-min-chars", type=int, default=30)
parser.add_argument("--root-max-chars", type=int, default=60)
parser.add_argument("--json", action="store_true", help="输出稳定 JSON")
return parser
def main(argv: list[str] | None = None) -> int:
parser = _build_parser()
args = parser.parse_args(argv)
if args.min_section_settings > args.max_section_settings:
parser.error("--min-section-settings 不能大于 --max-section-settings")
if args.root_min_chars > args.root_max_chars:
parser.error("--root-min-chars 不能大于 --root-max-chars")
config = ValidationConfig(
min_candidates=args.min_candidates,
min_settings=args.min_settings,
min_section_settings=args.min_section_settings,
max_section_settings=args.max_section_settings,
min_chars=args.min_chars,
min_item_chars=args.min_item_chars,
root_min_chars=args.root_min_chars,
root_max_chars=args.root_max_chars,
)
reports, problems = validate_candidates(args.files, config, args.root_doc)
status = "ok" if not problems else "invalid"
code = "SETTING_INIT_OK" if not problems else "SETTING_INIT_VALIDATION_FAILED"
payload = {
"status": status,
"code": code,
"reports": [asdict(report) for report in reports],
"problems": [asdict(problem) for problem in problems],
}
if args.json:
print(json.dumps(payload, ensure_ascii=False, indent=2))
else:
print(f"{code}: {len(reports)} 份候选,{len(problems)} 个问题")
for report in reports:
print(f"- {report.file}: {report.setting_count} 项,{report.effective_chars} 有效字符")
for problem in problems:
print(f"[{problem.code}] {problem.file}: {problem.message}")
return 0 if not problems else 1
if __name__ == "__main__":
sys.exit(main())