diff --git a/.claude/skills/read-context/SKILL.md b/.claude/skills/read-context/SKILL.md index dba9ddc..d2dcb76 100644 --- a/.claude/skills/read-context/SKILL.md +++ b/.claude/skills/read-context/SKILL.md @@ -11,6 +11,16 @@ SoT 对齐:包结构=专题-03 §4.2 **四层上下文**;字段级 aiContext 裁 ## 步骤 +正文 `continuation` 的 v1 流程在通用四层裁剪之前先执行以下可信读取链,禁止跳步: + +1. **固定检索计划**:从已确认细纲的人物、关系、物品、地点、力量体系和硬事件生成 `RetrievalPlan`;计划身份排除 `runId`,查询范围在写手启动前冻结。 +2. **卡索引**:生产只调用 search skill 的 `work` 授权面,只读 active Canonical entity 和有效 binding;参考作品回放只读预注册 `upgrade_book` 评测卡,固定 `productionRetrievalEligible=false`。 +3. **原文回读**:抽取卡只保留 `sourceVersion/sourceRefs/stateAsOf` 索引。每条历史事实必须顺着块级来源指针,在 `REPEATABLE READ READ ONLY` 快照内读取 `chapter <= asOf` 的原文;缺少指针的卡只记 `unverifiedIndexHint`,不能单独进入事实证据。 +4. **双证据组装**:连续前四章全文先进入 `proseEvidence` 基线,再按卡指针补充历史原文并去重;`factEvidence` 只接收冻结历史原文、正式设定、Canonical 状态和细纲声明的新事实。正式设定、Canonical 状态、细纲新事实必须保留各自不可变权威引用,不强造历史原文。 +5. **冻结与回显**:按 `score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC` 稳定排序,生成 JSON `WriterContext v1` 和 Markdown `RetrievalManifest`。清单与上下文身份排除运行 ID、时间戳和执行节点;预算裁剪必须回显来源与 `token_budget` 原因。 + +对应实现:`scripts/writer_contract.py`、`scripts/retrieve_writer_sources.py`、`scripts/assemble_writer_context.py`。写手只能接收最终冻结的 `WriterContext v1`,不得直接接收卡片全文,也不得自行搜索或回读文件。 + 1. 读 `works/<书>/装配.yaml`:拿槽位绑定与知识绑定。**未绑定的 `knowledge/` 内容一律不读**(=Layer 3 的授权门)。 2. 读 `meta/schemas/` 相关型的 `aiContext`,得出本用途的可见字段集:`true` 可见;`false` 不可见;`[用途…]` 仅列出的用途可见。 3. 按**四层**取数并逐字段裁剪——层是分组骨架,取数规则按**来源**逐条执行: @@ -20,9 +30,9 @@ SoT 对齐:包结构=专题-03 §4.2 **四层上下文**;字段级 aiContext 裁 - 本章细纲:出自大纲当前章条目; - 输出合同:字数区间/frontmatter 要求/新设定申报块等(按 agent 的产出合同); - L0 内部同按稳定度排:细纲在前(同章多轮重生成不变),本回合修改意见最后。 - - **Layer 1 近邻正文**(续写不可省略): - - 上一章末尾 1–2 个场景**原文**(衔接锚,不得摘要化); - - 再前一章只给 frontmatter 摘要(一句话概要+出场+伏笔动作); + - **Layer 1 近邻正文**(正文 continuation v1 不可省略): + - 目标章之前连续四章**全文**作为已验证基线;作品不足四章时从第一章开始,不重复; + - 卡索引命中的来源原文作为补充,按不可变来源引用去重并稳定排序; - `状态.md` **全文**(伏笔台账/时间线/角色最新状态)——"叙事现在时"快照,归近邻层。 - **Layer 2 作品事实**(一致性/规划不可省略;只读已确认): - **设定**:`设定.md` 四节;`generation` 用途裁掉「谜底与真相」「结局方向」「弃案记录」等非 generation 字段; diff --git a/.claude/skills/read-context/scripts/assemble_writer_context.py b/.claude/skills/read-context/scripts/assemble_writer_context.py new file mode 100644 index 0000000..31bb105 --- /dev/null +++ b/.claude/skills/read-context/scripts/assemble_writer_context.py @@ -0,0 +1,459 @@ +#!/usr/bin/env python3 +"""确定性组装 WriterContext v1 与可审阅 RetrievalManifest。 + +组装器不访问数据库、不调用模型。调用方必须先完成检索计划、卡索引和 +sourceRefs 原文回读,再把冻结结果交给本模块。 +""" + +from __future__ import annotations + +import copy +import hashlib +import json +from typing import Any, Mapping, Sequence + +from writer_contract import ( + MANIFEST_VERSION, + canonical_json, + normalize_text, + retrieval_identity, + validate_writer_context, +) + + +class AssemblyError(ValueError): + """上下文不连续、预算不足或证据合同非法时抛出。""" + + +def _hash_text(text: str) -> str: + """对已经归一化的证据文本计算带算法前缀的哈希。""" + + return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _source_key(source_ref: Mapping[str, Any]) -> tuple[str, str, int]: + """来源版本、来源 ID 与偏移共同定义片段身份。""" + + return ( + str(source_ref.get("sourceVersion") or ""), + str(source_ref.get("sourceId") or ""), + int(source_ref.get("startCodePoint") or 0), + ) + + +def _whole_source_key(source_ref: Mapping[str, Any]) -> tuple[str, str]: + """同一块全文已存在时,卡片子区间不再重复注入。""" + + return (str(source_ref.get("sourceVersion") or ""), str(source_ref.get("sourceId") or "")) + + +def _normalize_prose(raw: Mapping[str, Any], *, recent: bool, purpose: str | None = None) -> dict[str, Any]: + """规范化单条原文证据,并机械复核内容哈希和字符区间。""" + + required = {"chapter", "sourceRef", "text"} + if not isinstance(raw, Mapping) or not required.issubset(raw): + raise AssemblyError("原文证据缺少 chapter/sourceRef/text") + chapter = raw["chapter"] + if isinstance(chapter, bool) or not isinstance(chapter, int) or chapter <= 0: + raise AssemblyError("原文证据 chapter 必须是正整数") + source_ref = copy.deepcopy(dict(raw["sourceRef"])) if isinstance(raw["sourceRef"], Mapping) else None + if source_ref is None or not source_ref.get("sourceId") or not source_ref.get("sourceVersion"): + raise AssemblyError("原文证据缺少不可变来源引用") + text = normalize_text(str(raw["text"])) + if not text: + raise AssemblyError("原文证据不能为空") + source_ref["chapter"] = chapter + source_ref.setdefault("startCodePoint", 0) + source_ref.setdefault("endCodePoint", len(text)) + if source_ref["endCodePoint"] > len(text) and source_ref["startCodePoint"] == 0: + raise AssemblyError("原文来源字符区间超过文本长度") + return { + "evidenceId": str(raw.get("evidenceId") or f"prose:{source_ref['sourceId']}:{source_ref['startCodePoint']}"), + "chapter": chapter, + "sourceRef": source_ref, + "contentSha256": _hash_text(text), + "purpose": purpose or str(raw.get("purpose") or "card_source"), + "text": text, + "isRecentBaseline": recent, + } + + +def _normalize_fact(raw: Mapping[str, Any]) -> dict[str, Any]: + """规范化事实证据,并保留来源类型与风险优先级。""" + + required = {"evidenceId", "fact", "sourceType", "sourceRef"} + if not isinstance(raw, Mapping) or not required.issubset(raw): + raise AssemblyError("事实证据缺少 evidenceId/fact/sourceType/sourceRef") + source_type = str(raw["sourceType"]) + if source_type not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}: + raise AssemblyError("事实证据 sourceType 非法") + source_ref = copy.deepcopy(dict(raw["sourceRef"])) if isinstance(raw["sourceRef"], Mapping) else None + if source_ref is None or not source_ref.get("sourceId") or not source_ref.get("sourceVersion"): + raise AssemblyError("事实证据缺少不可变来源引用") + fact = normalize_text(str(raw["fact"])) + if not fact: + raise AssemblyError("事实证据内容不能为空") + risk = str(raw.get("riskLevel") or "medium") + if risk not in {"low", "medium", "high"}: + raise AssemblyError("事实证据 riskLevel 非法") + return { + "evidenceId": str(raw["evidenceId"]), + "fact": fact, + "sourceType": source_type, + "sourceRef": source_ref, + "contentSha256": _hash_text(fact), + "riskLevel": risk, + } + + +def _select_recent_baseline(recent_chapters: Sequence[Mapping[str, Any]], *, as_of: int) -> list[dict[str, Any]]: + """选择冻结章起连续四章全文;作品不足四章时从第一章开始。""" + + by_chapter: dict[int, dict[str, Any]] = {} + for raw in recent_chapters: + normalized = _normalize_prose(raw, recent=True, purpose="recent_full_chapter") + chapter = normalized["chapter"] + if chapter > as_of: + raise AssemblyError("近章原文包含目标章或未来章") + if chapter in by_chapter: + raise AssemblyError(f"第 {chapter} 章完整原文重复") + by_chapter[chapter] = normalized + first = max(1, as_of - 3) + expected = list(range(first, as_of + 1)) + missing = [chapter for chapter in expected if chapter not in by_chapter] + if missing: + raise AssemblyError(f"连续四章基线缺章: {','.join(map(str, missing))}") + return [by_chapter[chapter] for chapter in expected] + + +def _outline_contract(fine_outline: Mapping[str, Any]) -> dict[str, Any]: + """只保留 WriterContext 合同字段,检索用实体列表不会被倾倒给写手。""" + + required = {"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"} + if not isinstance(fine_outline, Mapping) or not required.issubset(fine_outline): + raise AssemblyError("fineOutline 缺少严格合同字段") + return {key: copy.deepcopy(fine_outline[key]) for key in ("sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts")} + + +def _coverage_elements(fine_outline: Mapping[str, Any]) -> list[dict[str, str]]: + """从细纲提取人物、关系、物品、地点和力量体系覆盖目标。""" + + groups = ( + ("entities", None), + ("relations", "character_relation"), + ("items", "item"), + ("locations", "location"), + ("powerSystems", "power_system"), + ) + result: list[dict[str, str]] = [] + seen: set[str] = set() + for field, fallback_type in groups: + values = fine_outline.get(field, []) + if not isinstance(values, list): + raise AssemblyError(f"fineOutline.{field} 必须是数组") + for index, raw in enumerate(values): + if isinstance(raw, str): + item = {"id": f"{field}:{index}", "type": fallback_type or "unknown", "name": raw} + elif isinstance(raw, Mapping): + item = { + "id": str(raw.get("id") or f"{field}:{index}"), + "type": str(raw.get("type") or fallback_type or "unknown"), + "name": str(raw.get("name") or ""), + } + else: + raise AssemblyError(f"fineOutline.{field}[{index}] 类型非法") + if not item["name"]: + raise AssemblyError(f"fineOutline.{field}[{index}] 缺少名称") + if item["id"] not in seen: + seen.add(item["id"]) + result.append(item) + return result + + +def _build_coverage( + fine_outline: Mapping[str, Any], + facts: Sequence[Mapping[str, Any]], + prose: Sequence[Mapping[str, Any]], + cards: Sequence[Mapping[str, Any]], +) -> list[dict[str, Any]]: + """把五类细纲要素映射到最终保留的事实与原文证据。""" + + prose_by_source = { + str(item["sourceRef"]["sourceId"]): item["evidenceId"] + for item in prose + } + declared = fine_outline.get("declaredNewFacts", []) + result: list[dict[str, Any]] = [] + for element in _coverage_elements(fine_outline): + fact_ids = [item["evidenceId"] for item in facts if element["name"] in item["fact"]] + prose_ids: list[str] = [] + for card in cards: + if str(card.get("name")) != element["name"] and str(card.get("type")) != element["type"]: + continue + for ref in card.get("sourceRefs") or []: + evidence_id = prose_by_source.get(str(ref.get("sourceId"))) + if evidence_id and evidence_id not in prose_ids: + prose_ids.append(evidence_id) + is_declared = any( + isinstance(item, Mapping) + and (str(item.get("factId")) == element["id"] or element["name"] in str(item.get("text") or "")) + for item in declared + ) + if is_declared: + status, reason = "declared_new", "" + elif fact_ids and prose_ids: + status, reason = "supported", "" + elif fact_ids: + status, reason = "style_gap", "缺少历史表现原文" + elif prose_ids: + status, reason = "card_gap", "原文可证但缺少冻结事实索引" + else: + status, reason = "unsupported", "没有可信事实证据" + result.append( + { + "elementId": element["id"], + "elementType": element["type"], + "name": element["name"], + "status": status, + "factEvidenceIds": sorted(fact_ids), + "proseEvidenceIds": sorted(prose_ids), + "gapReason": reason, + } + ) + return result + + +def _manifest( + *, + plan_id: str, + facts: Sequence[Mapping[str, Any]], + prose: Sequence[Mapping[str, Any]], + pattern_references: Sequence[Mapping[str, Any]], + omitted: Sequence[Mapping[str, str]], +) -> dict[str, Any]: + """由最终入包来源集合计算稳定 manifest,不含 runId 或时间戳。""" + + unique: dict[tuple[str, str, int], dict[str, Any]] = {} + for item in [*facts, *prose]: + ref = copy.deepcopy(dict(item["sourceRef"])) + unique[_source_key(ref)] = ref + for raw_ref in pattern_references: + ref = copy.deepcopy(dict(raw_ref)) + unique[_source_key(ref)] = ref + sources = [unique[key] for key in sorted(unique)] + omitted_rows = sorted( + [copy.deepcopy(dict(item)) for item in omitted], + key=lambda item: (str(item.get("reason")), str(item.get("sourceId"))), + ) + payload = { + "manifestVersion": MANIFEST_VERSION, + "planId": plan_id, + "sources": sources, + "omittedSources": omitted_rows, + } + return {**payload, "manifestId": retrieval_identity(payload)} + + +def _markdown_manifest(manifest: Mapping[str, Any]) -> str: + """渲染不含运行时元数据的人类审阅清单,保证字节稳定。""" + + lines = [ + "# 正文检索清单", + "", + f"- 计划:`{manifest['planId']}`", + f"- 清单:`{manifest['manifestId']}`", + f"- 纳入来源:{len(manifest['sources'])}", + f"- 排除来源:{len(manifest['omittedSources'])}", + "", + "## 纳入来源", + "", + ] + if manifest["sources"]: + for source in manifest["sources"]: + location = f"第{source['chapter']}章" if "chapter" in source else "权威版本" + lines.append(f"- `{source['sourceId']}` @ `{source['sourceVersion']}`,{location}") + else: + lines.append("- 无") + lines.extend(["", "## 排除来源", ""]) + if manifest["omittedSources"]: + for source in manifest["omittedSources"]: + lines.append(f"- `{source['sourceId']}`:{source['reason']}") + else: + lines.append("- 无") + return "\n".join(lines) + "\n" + + +def _context_size(context: Mapping[str, Any]) -> int: + """用最终规范 JSON 的 Unicode code point 数作为确定性预算单位。""" + + return len(canonical_json(context)) + + +def assemble_context( + *, + run_id: str, + attempt: int, + mode: str, + purpose: str, + quality_policy_version: str, + work_id: int, + target_chapter: int, + as_of: int, + source_version: str, + authorization_snapshot: Mapping[str, Any], + source_status: str, + retrieval_plan: Mapping[str, Any], + retrieval_result: Mapping[str, Any], + fine_outline: Mapping[str, Any], + narrative_state: Mapping[str, Any], + recent_chapters: Sequence[Mapping[str, Any]], + output_contract: Mapping[str, Any], + token_budget: Mapping[str, int], + pattern_references: Sequence[Mapping[str, Any]] = (), + generated_at: str, +) -> dict[str, Any]: + """组装稳定 WriterContext,并返回规范 JSON 与 Markdown manifest。""" + + if retrieval_plan.get("runId") != run_id or retrieval_plan.get("asOf") != as_of: + raise AssemblyError("retrievalPlan 与当前 runId/asOf 不一致") + if retrieval_plan.get("filters", {}).get("workId") != work_id: + raise AssemblyError("retrievalPlan 与当前 workId 不一致") + max_chars = token_budget.get("maxContextChars") + if isinstance(max_chars, bool) or not isinstance(max_chars, int) or max_chars <= 0: + raise AssemblyError("tokenBudget.maxContextChars 必须是正整数") + + baseline = _select_recent_baseline(recent_chapters, as_of=as_of) + baseline_blocks = {_whole_source_key(item["sourceRef"]) for item in baseline} + supplemental: list[dict[str, Any]] = [] + seen_fragments = {_source_key(item["sourceRef"]) for item in baseline} + for raw in retrieval_result.get("proseEvidence", []): + item = _normalize_prose(raw, recent=False) + if item["chapter"] > as_of: + raise AssemblyError("卡来源原文包含目标章或未来章") + if _whole_source_key(item["sourceRef"]) in baseline_blocks: + continue + key = _source_key(item["sourceRef"]) + if key not in seen_fragments: + seen_fragments.add(key) + supplemental.append(item) + supplemental.sort(key=lambda item: _source_key(item["sourceRef"])) + facts = sorted( + [_normalize_fact(item) for item in retrieval_result.get("factEvidence", [])], + key=lambda item: ({"high": 0, "medium": 1, "low": 2}[item["riskLevel"]], item["evidenceId"]), + ) + cards = [copy.deepcopy(dict(item)) for item in retrieval_result.get("cards", [])] + inherited_omitted = [ + {"sourceId": str(item.get("sourceId") or "unknown"), "reason": str(item.get("reason") or "not_relevant")} + for item in retrieval_result.get("manifest", {}).get("omittedSources", []) + if isinstance(item, Mapping) + ] + + selected_facts: list[dict[str, Any]] = [] + # 连续前四章全文是正文实验 v1 的不可裁剪基线。预算容不下时必须失败关闭, + # 不能静默退化成只保留最近一两章。 + selected_prose: list[dict[str, Any]] = list(baseline) + omitted = list(inherited_omitted) + candidates: list[tuple[str, dict[str, Any]]] = [] + candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "high") + candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "medium") + candidates.extend(("fact", item) for item in facts if item["riskLevel"] == "low") + candidates.extend(("prose", item) for item in supplemental) + + def make_context() -> dict[str, Any]: + """用当前选择集生成完整上下文,供预算试算与最终冻结。""" + + ordered_prose = sorted( + selected_prose, + key=lambda item: ( + 0 if item["isRecentBaseline"] else 1, + item["chapter"] if item["isRecentBaseline"] else 0, + _source_key(item["sourceRef"]), + ), + ) + ordered_facts = sorted(selected_facts, key=lambda item: item["evidenceId"]) + manifest = _manifest( + plan_id=str(retrieval_plan["planId"]), + facts=ordered_facts, + prose=ordered_prose, + pattern_references=pattern_references, + omitted=omitted, + ) + context = { + "schemaVersion": "writer-context-v1", + "runId": run_id, + "attempt": attempt, + "mode": mode, + "purpose": purpose, + "qualityPolicyVersion": quality_policy_version, + "workId": work_id, + "targetChapter": target_chapter, + "asOf": as_of, + "contextSnapshot": { + "manifestId": manifest["manifestId"], + "contextSha256": "sha256:" + "0" * 64, + "generatedAt": normalize_text(generated_at), + }, + "sourceVersion": source_version, + "authorizationSnapshot": copy.deepcopy(dict(authorization_snapshot)), + "sourceStatus": source_status, + "retrievalPlan": copy.deepcopy(dict(retrieval_plan)), + "retrievalManifest": manifest, + "fineOutline": _outline_contract(fine_outline), + "narrativeState": copy.deepcopy(dict(narrative_state)), + "factEvidence": ordered_facts, + "proseEvidence": ordered_prose, + "patternReferences": [copy.deepcopy(dict(item)) for item in pattern_references], + "evidenceCoverage": _build_coverage(fine_outline, ordered_facts, ordered_prose, cards), + "outputContract": copy.deepcopy(dict(output_contract)), + "tokenBudget": {"maxContextChars": max_chars, "usedContextChars": 0}, + "omittedSources": sorted(copy.deepcopy(omitted), key=lambda item: (item["reason"], item["sourceId"])), + "acceptanceEligible": mode == "production" and purpose == "production", + } + # usedContextChars 自身位数会影响 JSON 长度,迭代到数值稳定。 + for _ in range(8): + used = _context_size(context) + if context["tokenBudget"]["usedContextChars"] == used: + break + context["tokenBudget"]["usedContextChars"] = used + context["contextSnapshot"]["contextSha256"] = retrieval_identity(context) + return context + + baseline_context = make_context() + if _context_size(baseline_context) > max_chars: + raise AssemblyError("上下文预算不足以容纳细纲硬约束与连续前四章全文基线") + for kind, item in candidates: + target = selected_facts if kind == "fact" else selected_prose + target.append(item) + trial = make_context() + if _context_size(trial) > max_chars: + target.pop() + omitted.append({"sourceId": str(item["sourceRef"]["sourceId"]), "reason": "token_budget"}) + + context = make_context() + # 添加裁剪回显可能占用少量预算;若越界,按最低优先级继续移除。 + while _context_size(context) > max_chars and (selected_prose or selected_facts): + removable_prose = [item for item in selected_prose if not item["isRecentBaseline"]] + if removable_prose: + removed = removable_prose[-1] + selected_prose.remove(removed) + elif selected_facts and selected_facts[-1]["riskLevel"] != "high": + removed = selected_facts.pop() + elif selected_facts: + removed = selected_facts.pop() + else: + raise AssemblyError("上下文预算不足以保留连续前四章全文基线") + omitted.append({"sourceId": str(removed["sourceRef"]["sourceId"]), "reason": "token_budget"}) + context = make_context() + if _context_size(context) > max_chars: + raise AssemblyError("上下文预算不足以保留可用证据") + context["tokenBudget"]["usedContextChars"] = _context_size(context) + context["contextSnapshot"]["contextSha256"] = retrieval_identity(context) + normalized_context = validate_writer_context(context) + return { + "context": normalized_context, + "contextJson": canonical_json(normalized_context), + "manifestMarkdown": _markdown_manifest(normalized_context["retrievalManifest"]), + } + + +__all__ = ["AssemblyError", "assemble_context"] diff --git a/.claude/skills/read-context/scripts/retrieve_writer_sources.py b/.claude/skills/read-context/scripts/retrieve_writer_sources.py new file mode 100644 index 0000000..ba87bdb --- /dev/null +++ b/.claude/skills/read-context/scripts/retrieve_writer_sources.py @@ -0,0 +1,518 @@ +#!/usr/bin/env python3 +"""根据固定卡索引计划回读冻结历史原文。 + +抽取卡只提供定位与历史状态索引。任何由卡承载的历史事实都必须展开 +sourceRefs 并成功读取 asOf 以内的原文;正式设定、Canonical 状态和细纲 +新事实通过独立不可变引用进入事实证据。 +""" + +from __future__ import annotations + +import copy +import hashlib +import pathlib +import sys +from typing import Any, Callable, Mapping, Protocol, Sequence + +from writer_contract import PLAN_VERSION, TIE_BREAK, canonical_json, normalize_text, retrieval_identity + + +class RetrievalError(ValueError): + """检索计划、授权、冻结或来源引用不能证明安全时抛出。""" + + +class CardIndexRepository(Protocol): + """卡索引仓储只接收固定计划,不允许写手自行追加查询。""" + + def search(self, plan: Mapping[str, Any]) -> list[dict[str, Any]]: + """返回带稳定排序字段、历史里程碑和来源指针的卡索引。""" + + +class ProseRepository(Protocol): + """原文仓储只能读取冻结线内的不可变来源引用。""" + + def read_source_refs( + self, + *, + work_id: int, + as_of: int, + source_refs: Sequence[Mapping[str, Any]], + ) -> list[dict[str, Any]]: + """展开已校验的来源指针。""" + + +def _load_search_cards() -> Callable[..., list[dict[str, Any]]]: + """延迟导入 search skill,保持纯数据测试不触发嵌入依赖。""" + + search_scripts = pathlib.Path(__file__).resolve().parents[2] / "search" / "scripts" + sys.path.insert(0, str(search_scripts)) + from search import search_cards + + return search_cards + + +def _chapter(value: Any, path: str) -> int: + """严格解析正整数章号,拒绝 bool 和猜测性字符串。""" + + if isinstance(value, bool): + raise RetrievalError(f"{path} 必须是正整数章号") + if isinstance(value, str) and value.isdigit(): + value = int(value) + if not isinstance(value, int) or value <= 0: + raise RetrievalError(f"{path} 必须是正整数章号") + return value + + +def _milestone_chapter(value: Mapping[str, Any]) -> int: + """兼容实验库中英文里程碑章号字段,但不解析模糊文本。""" + + for key in ("chapter", "章", "chapterNo", "order_no"): + if key in value: + return _chapter(value[key], f"milestone.{key}") + raise RetrievalError("卡里程碑缺少可证明的绝对章号") + + +def stable_sort_cards(cards: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]: + """按合同固定卡片顺序,同分召回不会因数据库执行计划漂移。""" + + copied = [copy.deepcopy(dict(card)) for card in cards] + try: + return sorted( + copied, + key=lambda item: ( + -float(item["score"]), + str(item["sourceVersion"]), + str(item["sourceId"]), + int(item.get("sourceOffset") or 0), + ), + ) + except (KeyError, TypeError, ValueError) as error: + raise RetrievalError("卡缺少稳定排序字段") from error + + +def _outline_elements(fine_outline: Mapping[str, Any]) -> list[dict[str, str]]: + """按合同顺序提取需要检索的人物、关系、物品、地点和力量体系。""" + + groups = ( + ("entities", None), + ("relations", "character_relation"), + ("items", "item"), + ("locations", "location"), + ("powerSystems", "power_system"), + ("stateNeeds", "unknown"), + ) + result: list[dict[str, str]] = [] + seen: set[tuple[str, str]] = set() + for field, fallback_type in groups: + values = fine_outline.get(field, []) + if not isinstance(values, list): + raise RetrievalError(f"fineOutline.{field} 必须是数组") + for index, raw in enumerate(values): + if isinstance(raw, str): + item = {"id": f"{field}:{index}", "type": fallback_type or "unknown", "name": raw} + elif isinstance(raw, Mapping): + item = { + "id": str(raw.get("id") or f"{field}:{index}"), + "type": str(raw.get("type") or fallback_type or "unknown"), + "name": str(raw.get("name") or ""), + } + else: + raise RetrievalError(f"fineOutline.{field}[{index}] 类型非法") + if not item["name"]: + raise RetrievalError(f"fineOutline.{field}[{index}] 缺少 name") + identity = (item["type"], item["name"]) + if identity not in seen: + seen.add(identity) + result.append(item) + return result + + +def build_retrieval_plan( + *, + run_id: str, + work_id: int, + target_chapter: int, + as_of: int, + fine_outline: Mapping[str, Any], + card_index_version: str, + prose_index_version: str, + token_budget: Mapping[str, int], + top_k: int = 5, +) -> dict[str, Any]: + """从已确认细纲确定查询集合,并在执行前冻结计划身份。""" + + target = _chapter(target_chapter, "targetChapter") + freeze = _chapter(as_of, "asOf") + if freeze >= target: + raise RetrievalError("asOf 必须早于 targetChapter") + if not isinstance(work_id, int) or isinstance(work_id, bool) or work_id <= 0: + raise RetrievalError("workId 必须是正整数") + if not isinstance(top_k, int) or isinstance(top_k, bool) or top_k <= 0: + raise RetrievalError("topK 必须是正整数") + hard_constraints = fine_outline.get("hardConstraints", []) + if not isinstance(hard_constraints, list) or any(not isinstance(item, str) for item in hard_constraints): + raise RetrievalError("fineOutline.hardConstraints 必须是字符串数组") + suffix = " ".join(hard_constraints) + queries = [ + { + "queryId": f"query-{index:03d}-{item['id']}", + "text": normalize_text(f"{item['name']} {suffix}".strip()), + "entityTypes": [item["type"]], + "purpose": "fact_and_prose_evidence", + "topK": top_k, + } + for index, item in enumerate(_outline_elements(fine_outline), 1) + ] + plan = { + "planVersion": PLAN_VERSION, + "runId": run_id, + "asOf": freeze, + "queries": queries, + "cardIndexVersion": str(card_index_version), + "proseIndexVersion": str(prose_index_version), + "filters": { + "workId": work_id, + "asOfChapter": freeze, + "sourceStatus": "active", + "authorizationRequired": True, + }, + "tieBreak": TIE_BREAK, + "tokenBudget": copy.deepcopy(dict(token_budget)), + } + plan["planId"] = retrieval_identity(plan) + return plan + + +def _validate_source_ref(ref: Mapping[str, Any], *, as_of: int, path: str) -> dict[str, Any]: + """校验块级原文引用完整且不越过冻结线。""" + + required = {"sourceId", "sourceVersion", "chapter", "blockId", "startCodePoint", "endCodePoint"} + if not isinstance(ref, Mapping) or not required.issubset(ref): + raise RetrievalError(f"{path} 缺少块级来源定位") + chapter = _chapter(ref["chapter"], f"{path}.chapter") + if chapter > as_of: + raise RetrievalError(f"{path} 包含目标章或未来章") + start = ref["startCodePoint"] + end = ref["endCodePoint"] + if any(isinstance(value, bool) or not isinstance(value, int) for value in (ref["blockId"], start, end)): + raise RetrievalError(f"{path} 的块或字符区间非法") + if ref["blockId"] <= 0 or start < 0 or end <= start: + raise RetrievalError(f"{path} 的块或字符区间非法") + if not str(ref["sourceId"]).strip() or not str(ref["sourceVersion"]).strip(): + raise RetrievalError(f"{path} 缺少来源 ID 或版本") + return copy.deepcopy(dict(ref)) + + +def freeze_card(card: Mapping[str, Any], *, as_of: int) -> dict[str, Any]: + """仅用冻结线内里程碑重建卡状态,不透传终态摘要和未来字段。""" + + freeze = _chapter(as_of, "asOf") + milestones = card.get("milestones") + if not isinstance(milestones, list): + raise RetrievalError(f"卡 {card.get('cardId')} 缺少历史里程碑") + kept: list[dict[str, Any]] = [] + for index, milestone in enumerate(milestones): + if not isinstance(milestone, Mapping): + raise RetrievalError(f"卡里程碑[{index}] 必须是对象") + chapter = _milestone_chapter(milestone) + if chapter <= freeze: + normalized = copy.deepcopy(dict(milestone)) + normalized["chapter"] = chapter + for alias in ("章", "chapterNo", "order_no"): + normalized.pop(alias, None) + kept.append(normalized) + kept.sort(key=lambda item: (item["chapter"], str(item.get("id") or ""))) + if not kept: + raise RetrievalError(f"卡 {card.get('cardId')} 在冻结线内没有可证明状态") + refs = [ + _validate_source_ref(ref, as_of=freeze, path=f"card.sourceRefs[{index}]") + for index, ref in enumerate(card.get("sourceRefs") or []) + ] + return { + "cardId": str(card.get("cardId") or ""), + "type": str(card.get("type") or ""), + "name": str(card.get("name") or ""), + "score": float(card.get("score") or 0), + "sourceId": str(card.get("sourceId") or ""), + "sourceVersion": str(card.get("sourceVersion") or ""), + "sourceOffset": int(card.get("sourceOffset") or 0), + "sourceRefs": refs, + "stateAsOf": kept, + "sourceKind": str(card.get("sourceKind") or ""), + "productionRetrievalEligible": bool(card.get("productionRetrievalEligible")), + } + + +class ProductionCardIndexRepository: + """生产卡仓储固定调用 search.py 的 work 授权面。""" + + def __init__(self, *, search_function: Callable[..., list[dict[str, Any]]] | None = None): + self._search = search_function or _load_search_cards() + + def search(self, plan: Mapping[str, Any]) -> list[dict[str, Any]]: + """按计划逐查询召回,并再次失败关闭验证生产资格。""" + + cards: list[dict[str, Any]] = [] + for query in plan.get("queries", []): + entity_types = query.get("entityTypes") or [] + card_type = entity_types[0] if len(entity_types) == 1 and entity_types[0] != "unknown" else None + cards.extend( + self._search( + intent=query["text"], + scope="work", + work_id=plan["filters"]["workId"], + ttype=card_type, + purpose="generation", + top=query["topK"], + ) + ) + unique: dict[str, dict[str, Any]] = {} + for card in stable_sort_cards(cards): + if ( + card.get("sourceKind") != "canonical_entity" + or card.get("sourceStatus") not in {"active", "authorized"} + or card.get("bindingStatus") != "active" + or card.get("productionRetrievalEligible") is not True + ): + raise RetrievalError("生产检索命中非 active Canonical entity 或无效 binding") + unique.setdefault(str(card.get("cardId")), copy.deepcopy(card)) + return stable_sort_cards(list(unique.values())) + + +class ReplayCardIndexRepository: + """回放卡仓储只接受预注册 upgrade_book 评测投影。""" + + def __init__( + self, + cards: Sequence[Mapping[str, Any]], + *, + preregistered_card_ids: Sequence[str], + _validated_replay: bool = False, + ): + if not _validated_replay: + raise RetrievalError("回放仓储必须通过 from_replay_config 完成授权与泄露门禁") + expected = {str(item) for item in preregistered_card_ids} + actual = {str(item.get("cardId")) for item in cards} + if not expected or expected != actual: + raise RetrievalError("回放卡与预注册 ID 不一致") + self._cards = [] + for card in cards: + if ( + card.get("sourceType") != "upgrade_book" + or card.get("evaluationStatus") != "eval_draft" + or card.get("productionRetrievalEligible") is not False + ): + raise RetrievalError("回放只允许不可生产检索的 upgrade_book 评测卡") + self._cards.append(copy.deepcopy(dict(card))) + + @classmethod + def from_replay_config( + cls, + config: Mapping[str, Any], + *, + cards: Sequence[Mapping[str, Any]], + preregistered_card_ids: Sequence[str], + ) -> "ReplayCardIndexRepository": + """复用 replay-eval 的授权、冻结来源和内容泄露审计后构造仓储。""" + + scripts = pathlib.Path(__file__).resolve().parents[2] / "replay-eval" / "scripts" + sys.path.insert(0, str(scripts)) + from audit_leakage import audit_snapshot + from check_snapshot import check_authorization, check_target_sources + + if not isinstance(config, Mapping): + raise RetrievalError("回放配置必须是对象") + target = _chapter(config.get("targetChapter"), "config.targetChapter") + snapshot = config.get("snapshot") + if not isinstance(snapshot, Mapping): + raise RetrievalError("回放配置缺少冻结快照") + as_of = _chapter(snapshot.get("asOfChapter"), "config.snapshot.asOfChapter") + if target != as_of + 1: + raise RetrievalError("回放 targetChapter 必须等于 asOfChapter+1") + authorization_result = check_authorization(config.get("authorization")) + if not authorization_result.get("ok"): + raise RetrievalError("回放授权快照未通过既有门禁") + source_result = check_target_sources(target, config.get("sources", [])) + if not source_result.get("ok"): + raise RetrievalError("回放来源包含目标章、未来章或不可证明区间") + leakage = config.get("leakageAudit") + target_facts = leakage.get("targetFacts") if isinstance(leakage, Mapping) else None + audit_result = audit_snapshot(snapshot, target_facts, as_of=as_of, target=target) + if not audit_result.get("ok"): + raise RetrievalError("回放冻结快照未通过既有内容泄露审计") + return cls( + cards, + preregistered_card_ids=preregistered_card_ids, + _validated_replay=True, + ) + + def search(self, plan: Mapping[str, Any]) -> list[dict[str, Any]]: + """返回预注册集合;计划不能把评测卡转换为生产卡。""" + + del plan + return stable_sort_cards(self._cards) + + +class FrozenProseRepository: + """复用 load_reference_work 的唯一冻结原文 SQL 入口。""" + + def __init__(self, *, dsn: str, tenant_id: int): + self.dsn = dsn + self.tenant_id = tenant_id + + def read_source_refs(self, *, work_id: int, as_of: int, source_refs: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]: + """延迟导入 replay-eval,避免复制 SQL 或建立第二套权限语义。""" + + scripts = pathlib.Path(__file__).resolve().parents[2] / "replay-eval" / "scripts" + sys.path.insert(0, str(scripts)) + from load_reference_work import load_frozen_prose_rows + + return load_frozen_prose_rows( + dsn=self.dsn, + tenant_id=self.tenant_id, + work_id=work_id, + as_of=as_of, + source_refs=source_refs, + ) + + +def _content_hash(text: str) -> str: + """对规范化文本计算带算法前缀的内容哈希。""" + + normalized = normalize_text(text) + return "sha256:" + hashlib.sha256(normalized.encode("utf-8")).hexdigest() + + +def _authoritative_evidence(facts: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]: + """把正式设定、Canonical 状态和细纲新事实转换为独立事实证据。""" + + allowed = {"formal_setting", "canonical_state", "fine_outline_declared_new"} + result: list[dict[str, Any]] = [] + for index, raw in enumerate(facts): + source_type = raw.get("sourceType") + source_ref = raw.get("sourceRef") + fact = normalize_text(str(raw.get("fact") or "")) + if source_type not in allowed or not fact or not isinstance(source_ref, Mapping): + raise RetrievalError(f"authoritativeFacts[{index}] 合同非法") + if not source_ref.get("sourceId") or not source_ref.get("sourceVersion"): + raise RetrievalError(f"authoritativeFacts[{index}] 缺少不可变来源引用") + result.append( + { + "evidenceId": str(raw.get("factId") or f"authoritative:{index}"), + "fact": fact, + "sourceType": source_type, + "sourceRef": copy.deepcopy(dict(source_ref)), + "contentSha256": _content_hash(fact), + "riskLevel": str(raw.get("riskLevel") or "medium"), + } + ) + return result + + +def retrieve_writer_sources( + *, + plan: Mapping[str, Any], + card_repository: CardIndexRepository, + prose_repository: ProseRepository, + authoritative_facts: Sequence[Mapping[str, Any]] = (), +) -> dict[str, Any]: + """执行固定计划,展开卡来源并形成事实/原文两条证据线。""" + + as_of = _chapter(plan.get("asOf"), "plan.asOf") + work_id = plan.get("filters", {}).get("workId") + if not isinstance(work_id, int) or isinstance(work_id, bool) or work_id <= 0: + raise RetrievalError("plan.filters.workId 非法") + cards = [freeze_card(card, as_of=as_of) for card in stable_sort_cards(card_repository.search(plan))] + source_refs: list[dict[str, Any]] = [] + unverified: list[dict[str, Any]] = [] + for card_item in cards: + if card_item["sourceRefs"]: + source_refs.extend(card_item["sourceRefs"]) + else: + unverified.append( + { + "cardId": card_item["cardId"], + "reason": "missing_source_refs", + "classification": "unverifiedIndexHint", + } + ) + deduplicated = { + (str(ref["sourceVersion"]), str(ref["sourceId"]), int(ref["startCodePoint"])): ref + for ref in source_refs + } + ordered_refs = [deduplicated[key] for key in sorted(deduplicated)] + prose_rows = prose_repository.read_source_refs(work_id=work_id, as_of=as_of, source_refs=ordered_refs) if ordered_refs else [] + def ref_key(ref: Mapping[str, Any]) -> tuple[str, str, int]: + """同一块可有多个字符区间,映射键必须包含偏移。""" + + return (str(ref.get("sourceVersion")), str(ref.get("sourceId")), int(ref.get("startCodePoint") or 0)) + + rows_by_source = {ref_key(row.get("sourceRef", {})): row for row in prose_rows} + missing = [ref["sourceId"] for ref in ordered_refs if ref_key(ref) not in rows_by_source] + if missing: + raise RetrievalError(f"卡来源未能回读原文: {','.join(map(str, missing))}") + + prose_evidence: list[dict[str, Any]] = [] + for index, ref in enumerate(ordered_refs): + row = rows_by_source[ref_key(ref)] + text = normalize_text(str(row.get("text") or "")) + if not text: + raise RetrievalError(f"来源 {ref['sourceId']} 原文为空") + prose_evidence.append( + { + "evidenceId": f"prose:card:{index:04d}", + "chapter": _chapter(row.get("chapter"), "prose.chapter"), + "sourceRef": copy.deepcopy(ref), + "contentSha256": _content_hash(text), + "purpose": str(row.get("purpose") or "card_source"), + "text": text, + "isRecentBaseline": False, + } + ) + + fact_evidence = _authoritative_evidence(authoritative_facts) + for card_item in cards: + if not card_item["sourceRefs"]: + continue + latest = card_item["stateAsOf"][-1] + fact_text = normalize_text(str(latest.get("fact") or latest.get("台阶") or canonical_json(latest))) + fact_evidence.append( + { + "evidenceId": f"fact:card:{card_item['cardId']}", + "fact": fact_text, + "sourceType": "historical_prose", + "sourceRef": copy.deepcopy(card_item["sourceRefs"][0]), + "contentSha256": _content_hash(fact_text), + "riskLevel": "medium", + } + ) + fact_evidence.sort(key=lambda item: item["evidenceId"]) + sources = sorted( + [copy.deepcopy(ref) for ref in ordered_refs] + + [copy.deepcopy(item["sourceRef"]) for item in fact_evidence if item["sourceType"] != "historical_prose"], + key=lambda item: (str(item["sourceVersion"]), str(item["sourceId"]), int(item.get("startCodePoint") or 0)), + ) + manifest_payload = { + "manifestVersion": "writer-retrieval-manifest-v1", + "planId": plan["planId"], + "sources": sources, + "omittedSources": [ + {"sourceId": f"card:{item['cardId']}", "reason": item["reason"]} + for item in unverified + ], + } + manifest = {**manifest_payload, "manifestId": retrieval_identity(manifest_payload)} + return { + "cards": cards, + "factEvidence": fact_evidence, + "proseEvidence": prose_evidence, + "unverifiedIndexHints": unverified, + "manifest": manifest, + } + + +__all__ = [ + "RetrievalError", "CardIndexRepository", "ProseRepository", "ProductionCardIndexRepository", + "ReplayCardIndexRepository", "FrozenProseRepository", "stable_sort_cards", "freeze_card", + "build_retrieval_plan", "retrieve_writer_sources", +] diff --git a/.claude/skills/read-context/scripts/test_assemble_writer_context.py b/.claude/skills/read-context/scripts/test_assemble_writer_context.py new file mode 100644 index 0000000..d65db93 --- /dev/null +++ b/.claude/skills/read-context/scripts/test_assemble_writer_context.py @@ -0,0 +1,264 @@ +#!/usr/bin/env python3 +"""事实证据与原文证据双线组装测试。""" + +from __future__ import annotations + +import copy +import hashlib +import pathlib +import sys +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) + +from assemble_writer_context import AssemblyError, assemble_context # noqa: E402 +from writer_contract import retrieval_identity, validate_writer_context # noqa: E402 + + +def digest(text: str) -> str: + """生成测试证据的规范 SHA-256。""" + + return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def prose(chapter: int, block_id: int, text: str, *, recent: bool = False, purpose: str = "card_source") -> dict: + """构造块级可引用原文。""" + + return { + "evidenceId": f"prose:{chapter}:{block_id}", + "chapter": chapter, + "sourceRef": { + "sourceId": f"chapter:{chapter}:block:{block_id}", + "sourceVersion": f"chapter-{chapter}-block-{block_id}-v1", + "chapter": chapter, + "blockId": block_id, + "startCodePoint": 0, + "endCodePoint": len(text), + }, + "contentSha256": digest(text), + "purpose": purpose, + "text": text, + "isRecentBaseline": recent, + } + + +def fact(fact_id: str, text: str, source_type: str = "formal_setting", risk: str = "medium") -> dict: + """构造独立权威事实证据。""" + + return { + "evidenceId": fact_id, + "fact": text, + "sourceType": source_type, + "sourceRef": { + "sourceId": f"{source_type}:{fact_id}", + "sourceVersion": f"{source_type}-v1", + }, + "contentSha256": digest(text), + "riskLevel": risk, + } + + +def plan(run_id: str = "run-a") -> dict: + """构造合法固定检索计划。""" + + value = { + "planVersion": "writer-retrieval-plan-v1", + "runId": run_id, + "asOf": 6, + "queries": [], + "cardIndexVersion": "cards-v1", + "proseIndexVersion": "prose-v1", + "filters": { + "workId": 8, + "asOfChapter": 6, + "sourceStatus": "active", + "authorizationRequired": True, + }, + "tieBreak": "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC", + "tokenBudget": {"maxContextChars": 50000}, + } + value["planId"] = retrieval_identity(value) + return value + + +def fine_outline() -> dict: + """构造含五类覆盖目标的细纲输入。""" + + return { + "sourceRef": {"sourceId": "fine-outline:7", "sourceVersion": "outline-v3", "chapter": 7}, + "hardConstraints": ["甲携剑抵达城门"], + "adjustableBeats": ["可调整过渡方式"], + "declaredNewFacts": [], + "entities": [{"id": "character:甲", "type": "character", "name": "甲"}], + "relations": [{"id": "relation:甲乙", "type": "character_relation", "name": "甲乙关系"}], + "items": [{"id": "item:剑", "type": "item", "name": "剑"}], + "locations": [{"id": "location:城门", "type": "location", "name": "城门"}], + "powerSystems": [{"id": "power:灵力", "type": "power_system", "name": "灵力"}], + } + + +def recent_chapters(last: int = 6) -> list[dict]: + """构造第一章到冻结章的完整章节原文。""" + + return [prose(chapter, 100 + chapter, f"第{chapter}章完整原文。") for chapter in range(1, last + 1)] + + +def base_kwargs() -> dict: + """构造 assemble_context 的公共合法输入。""" + + return { + "run_id": "run-a", + "attempt": 1, + "mode": "production", + "purpose": "production", + "quality_policy_version": "writer-production-v1", + "work_id": 8, + "target_chapter": 7, + "as_of": 6, + "source_version": "raw-file-v1:sha256:" + "a" * 64, + "authorization_snapshot": { + "snapshotId": "auth-1", + "allowedPurpose": "generation", + "verifiedAt": "2026-07-20T00:00:00Z", + }, + "source_status": "active", + "retrieval_plan": plan(), + "retrieval_result": { + "cards": [], + "factEvidence": [], + "proseEvidence": [], + "unverifiedIndexHints": [], + "manifest": { + "manifestVersion": "writer-retrieval-manifest-v1", + "manifestId": "sha256:" + "b" * 64, + "planId": plan()["planId"], + "sources": [], + "omittedSources": [], + }, + }, + "fine_outline": fine_outline(), + "narrative_state": { + "time": "当日", + "location": "城门", + "characterPositions": {"甲": "城外"}, + "immediateSituation": "即将交战", + }, + "recent_chapters": recent_chapters(), + "output_contract": { + "targetChars": 4000, + "minChars": 3600, + "maxChars": 4400, + "frontmatterRequired": False, + "newSettingDeclarationRequired": True, + }, + "token_budget": {"maxContextChars": 50000}, + "generated_at": "2026-07-20T00:00:00Z", + } + + +class AssembleWriterContextTest(unittest.TestCase): + def test_previous_four_full_chapters_are_baseline(self): + result = assemble_context(**base_kwargs()) + baseline = [item for item in result["context"]["proseEvidence"] if item["isRecentBaseline"]] + self.assertEqual([item["chapter"] for item in baseline], [3, 4, 5, 6]) + self.assertTrue(all(item["purpose"] == "recent_full_chapter" for item in baseline)) + validate_writer_context(result["context"]) + + def test_when_fewer_than_four_exist_start_at_chapter_one_without_duplicates(self): + kwargs = base_kwargs() + kwargs.update( + { + "target_chapter": 3, + "as_of": 2, + "recent_chapters": recent_chapters(2), + "retrieval_plan": {**plan(), "asOf": 2, "filters": {**plan()["filters"], "asOfChapter": 2}}, + } + ) + kwargs["retrieval_plan"]["planId"] = retrieval_identity({key: value for key, value in kwargs["retrieval_plan"].items() if key != "planId"}) + kwargs["retrieval_result"]["manifest"]["planId"] = kwargs["retrieval_plan"]["planId"] + result = assemble_context(**kwargs) + chapters = [item["chapter"] for item in result["context"]["proseEvidence"]] + self.assertEqual(chapters, [1, 2]) + + def test_missing_chapter_in_required_continuous_baseline_fails_closed(self): + kwargs = base_kwargs() + kwargs["recent_chapters"] = [item for item in recent_chapters() if item["chapter"] != 4] + with self.assertRaises(AssemblyError): + assemble_context(**kwargs) + + def test_card_prose_is_deduplicated_and_stably_sorted_after_baseline(self): + kwargs = base_kwargs() + duplicate = prose(6, 106, "第6章完整原文。") + supplemental_b = prose(2, 202, "补充乙。") + supplemental_a = prose(1, 201, "补充甲。") + kwargs["retrieval_result"]["proseEvidence"] = [supplemental_b, duplicate, supplemental_a] + result = assemble_context(**kwargs) + evidence = result["context"]["proseEvidence"] + self.assertEqual(sum(item["sourceRef"]["sourceId"] == duplicate["sourceRef"]["sourceId"] for item in evidence), 1) + supplemental = [item for item in evidence if not item["isRecentBaseline"]] + self.assertEqual([item["sourceRef"]["sourceId"] for item in supplemental], ["chapter:1:block:201", "chapter:2:block:202"]) + + def test_fact_and_prose_evidence_are_separate_and_five_types_report_coverage(self): + kwargs = base_kwargs() + kwargs["retrieval_result"]["factEvidence"] = [ + fact("fact:character", "甲保持警惕", risk="high"), + fact("fact:relation", "甲乙关系稳定"), + fact("fact:item", "剑仍在甲手中"), + fact("fact:location", "城门已经关闭"), + fact("fact:power", "灵力消耗受限"), + ] + result = assemble_context(**kwargs) + context = result["context"] + self.assertTrue(all("fact" in item and "text" not in item for item in context["factEvidence"])) + self.assertTrue(all("text" in item and "fact" not in item for item in context["proseEvidence"])) + self.assertEqual( + {item["elementType"] for item in context["evidenceCoverage"]}, + {"character", "character_relation", "item", "location", "power_system"}, + ) + self.assertTrue(all(item["status"] in {"supported", "style_gap"} for item in context["evidenceCoverage"])) + + def test_same_snapshot_and_inputs_have_same_manifest_and_context_hash(self): + first = assemble_context(**base_kwargs()) + second_kwargs = base_kwargs() + second_kwargs["run_id"] = "run-b" + second_kwargs["retrieval_plan"]["runId"] = "run-b" + second = assemble_context(**second_kwargs) + self.assertEqual(first["context"]["retrievalManifest"]["manifestId"], second["context"]["retrievalManifest"]["manifestId"]) + self.assertEqual(first["context"]["contextSnapshot"]["contextSha256"], second["context"]["contextSnapshot"]["contextSha256"]) + self.assertEqual(first["manifestMarkdown"], second["manifestMarkdown"]) + + def test_budget_keeps_hard_outline_recent_prose_and_high_risk_fact_with_trace(self): + kwargs = base_kwargs() + kwargs["retrieval_result"]["factEvidence"] = [ + fact("fact:high", "甲的高风险身份事实" + "甲" * 300, risk="high"), + fact("fact:low", "低风险背景" + "乙" * 300, risk="low"), + ] + kwargs["retrieval_result"]["proseEvidence"] = [prose(1, 900, "很早的补充原文" + "旧" * 500)] + kwargs["recent_chapters"] = [ + prose(chapter, 100 + chapter, f"第{chapter}章" + "近" * 350) + for chapter in range(1, 7) + ] + kwargs["token_budget"] = {"maxContextChars": 7500} + result = assemble_context(**kwargs) + context = result["context"] + self.assertIn("甲携剑抵达城门", context["fineOutline"]["hardConstraints"]) + self.assertIn("fact:high", {item["evidenceId"] for item in context["factEvidence"]}) + baseline = [item["chapter"] for item in context["proseEvidence"] if item["isRecentBaseline"]] + self.assertEqual(baseline, [3, 4, 5, 6]) + self.assertLessEqual(context["tokenBudget"]["usedContextChars"], 7500) + self.assertTrue(context["omittedSources"]) + self.assertIn("token_budget", {item["reason"] for item in context["omittedSources"]}) + + def test_budget_that_cannot_hold_four_full_chapters_fails_closed(self): + kwargs = base_kwargs() + kwargs["recent_chapters"] = [ + prose(chapter, 100 + chapter, f"第{chapter}章" + "近" * 500) + for chapter in range(1, 7) + ] + kwargs["token_budget"] = {"maxContextChars": 3000} + with self.assertRaisesRegex(AssemblyError, "连续前四章全文基线"): + assemble_context(**kwargs) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py b/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py new file mode 100644 index 0000000..c553f14 --- /dev/null +++ b/.claude/skills/read-context/scripts/test_retrieve_writer_sources.py @@ -0,0 +1,399 @@ +#!/usr/bin/env python3 +"""卡索引驱动的冻结原文检索测试。""" + +from __future__ import annotations + +import copy +import pathlib +import sys +import unittest + +SCRIPT_DIR = pathlib.Path(__file__).resolve().parent +sys.path.insert(0, str(SCRIPT_DIR)) +sys.path.insert(0, str(SCRIPT_DIR.parents[1] / "replay-eval" / "scripts")) +sys.path.insert(0, str(SCRIPT_DIR.parents[1] / "search" / "scripts")) + +from load_reference_work import begin_read_snapshot # noqa: E402 +from search import search_cards # noqa: E402 +from retrieve_writer_sources import ( # noqa: E402 + ProductionCardIndexRepository, + ReplayCardIndexRepository, + RetrievalError, + build_retrieval_plan, + freeze_card, + retrieve_writer_sources, + stable_sort_cards, +) + + +SOURCE_VERSION = "raw-file-v1:sha256:" + "a" * 64 + + +def source_ref(chapter: int, block_id: int, offset: int = 0) -> dict: + """构造可回读的冻结原文引用。""" + + return { + "sourceId": f"chapter:{chapter}:block:{block_id}", + "sourceVersion": SOURCE_VERSION, + "chapter": chapter, + "blockId": block_id, + "startCodePoint": offset, + "endCodePoint": offset + 4, + } + + +def card(card_id: str, score: float, chapter: int = 3, offset: int = 0) -> dict: + """构造含历史里程碑和原文指针的抽取卡索引。""" + + return { + "cardId": card_id, + "type": "character", + "name": f"角色{card_id}", + "score": score, + "sourceVersion": SOURCE_VERSION, + "sourceId": f"canonical-entity:{card_id}", + "sourceOffset": offset, + "sourceRefs": [source_ref(chapter, int(card_id), offset)], + "milestones": [ + {"chapter": 1, "fact": "初始状态"}, + {"chapter": chapter, "fact": "冻结线内状态"}, + {"chapter": 9, "fact": "未来终态"}, + ], + "sourceKind": "canonical_entity", + "sourceStatus": "active", + "bindingStatus": "active", + "productionRetrievalEligible": True, + } + + +class FakeCardRepository: + """只返回预置卡片并记录固定计划查询。""" + + def __init__(self, cards): + self.cards = cards + self.plans = [] + + def search(self, plan): + self.plans.append(copy.deepcopy(plan)) + return copy.deepcopy(self.cards) + + +class FakeProseRepository: + """按来源引用返回冻结片段,不访问数据库。""" + + def __init__(self): + self.calls = [] + + def read_source_refs(self, *, work_id, as_of, source_refs): + self.calls.append((work_id, as_of, copy.deepcopy(source_refs))) + rows = [] + for ref in source_refs: + rows.append( + { + "sourceRef": copy.deepcopy(ref), + "chapter": ref["chapter"], + "text": "历史原文", + "purpose": "card_source", + } + ) + return rows + + +class FakeConnection: + """记录事务声明,证明读取器先锁定只读可重复读快照。""" + + def __init__(self): + self.statements = [] + + def execute(self, statement, params=None): + self.statements.append((" ".join(statement.split()), params)) + return self + + +class FakeQueryResult: + """模拟 psycopg 查询结果。""" + + def __init__(self, rows): + self.rows = rows + + def fetchall(self): + return self.rows + + +class FakeSearchConnection: + """为 search_cards 提供字段策略与单条 Canonical 卡结果。""" + + def __init__(self): + self.statements = [] + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, traceback): + return False + + def execute(self, statement, params=None): + normalized = " ".join(statement.split()) + self.statements.append((normalized, params)) + if "muse_meta_schema" in normalized: + return FakeQueryResult([("character", {"秘密": False})]) + return FakeQueryResult( + [ + ( + "entity", + 11, + { + "型": "character", + "名称": "甲", + "一句话摘要": "历史角色", + "字段": { + "秘密": "不可见", + "sourceRefs": [source_ref(3, 11)], + "milestones": [{"chapter": 3, "fact": "历史事实"}], + }, + }, + "active", + 0.9, + 2, + {"sourceRefs": [source_ref(3, 11)]}, + "active", + "active", + ) + ] + ) + + +class RetrieveWriterSourcesTest(unittest.TestCase): + def setUp(self): + self.plan = build_retrieval_plan( + run_id="run-a", + work_id=8, + target_chapter=6, + as_of=5, + fine_outline={ + "entities": [ + {"id": "character:甲", "type": "character", "name": "甲"}, + {"id": "item:剑", "type": "item", "name": "剑"}, + ], + "relations": [], + "locations": [], + "powerSystems": [], + "hardConstraints": ["甲使用剑"], + }, + card_index_version="cards-v1", + prose_index_version="prose-v1", + token_budget={"maxContextChars": 20000}, + ) + + def test_cards_use_documented_stable_order_and_ties_repeat(self): + cards = [card("3", 0.8, offset=2), card("2", 0.8, offset=1), card("1", 0.9)] + expected = ["1", "2", "3"] + for _ in range(3): + self.assertEqual([item["cardId"] for item in stable_sort_cards(cards)], expected) + + def test_plan_identity_excludes_run_id(self): + other = copy.deepcopy(self.plan) + other["runId"] = "run-b" + rebuilt = build_retrieval_plan( + run_id="run-b", + work_id=8, + target_chapter=6, + as_of=5, + fine_outline={ + "entities": [ + {"id": "character:甲", "type": "character", "name": "甲"}, + {"id": "item:剑", "type": "item", "name": "剑"}, + ], + "relations": [], + "locations": [], + "powerSystems": [], + "hardConstraints": ["甲使用剑"], + }, + card_index_version="cards-v1", + prose_index_version="prose-v1", + token_budget={"maxContextChars": 20000}, + ) + self.assertEqual(self.plan["planId"], rebuilt["planId"]) + + def test_state_as_of_uses_only_milestones_at_or_before_freeze(self): + frozen = freeze_card(card("1", 1.0), as_of=5) + self.assertEqual([item["chapter"] for item in frozen["stateAsOf"]], [1, 3]) + self.assertNotIn("未来终态", str(frozen)) + + def test_target_future_and_unprovable_sources_fail_closed(self): + for bad_ref in ( + source_ref(6, 1), + source_ref(7, 1), + {"sourceId": "unknown", "sourceVersion": SOURCE_VERSION}, + ): + unsafe = card("1", 1.0) + unsafe["sourceRefs"] = [bad_ref] + with self.subTest(bad_ref=bad_ref), self.assertRaises(RetrievalError): + retrieve_writer_sources( + plan=self.plan, + card_repository=FakeCardRepository([unsafe]), + prose_repository=FakeProseRepository(), + ) + + def test_cards_expand_to_prose_and_authoritative_facts_keep_independent_refs(self): + prose = FakeProseRepository() + no_ref_card = card("2", 0.7) + no_ref_card["sourceRefs"] = [] + result = retrieve_writer_sources( + plan=self.plan, + card_repository=FakeCardRepository([card("1", 0.9), no_ref_card]), + prose_repository=prose, + authoritative_facts=[ + { + "factId": "setting:1", + "fact": "正式设定事实", + "sourceType": "formal_setting", + "sourceRef": { + "sourceId": "setting:8", + "sourceVersion": "setting-v2", + }, + "riskLevel": "high", + }, + { + "factId": "outline:new:1", + "fact": "细纲声明的新事实", + "sourceType": "fine_outline_declared_new", + "sourceRef": { + "sourceId": "fine-outline:6", + "sourceVersion": "outline-v3", + "chapter": 6, + }, + "riskLevel": "medium", + }, + ], + ) + self.assertEqual(len(prose.calls[0][2]), 1) + self.assertEqual(result["proseEvidence"][0]["text"], "历史原文") + self.assertEqual( + {item["sourceType"] for item in result["factEvidence"]}, + {"historical_prose", "formal_setting", "fine_outline_declared_new"}, + ) + self.assertEqual(result["unverifiedIndexHints"][0]["cardId"], "2") + + def test_snapshot_transaction_is_repeatable_read_and_read_only(self): + connection = FakeConnection() + begin_read_snapshot(connection) + self.assertEqual( + connection.statements[0][0], + "SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY", + ) + + def test_production_repository_only_accepts_active_canonical_binding(self): + calls = [] + + def search_function(**kwargs): + calls.append(kwargs) + return [card("1", 0.9)] + + repository = ProductionCardIndexRepository(search_function=search_function) + result = repository.search(self.plan) + self.assertEqual(result[0]["cardId"], "1") + self.assertEqual(calls[0]["scope"], "work") + + for field, value in ( + ("sourceKind", "eval_draft"), + ("sourceStatus", "draft"), + ("bindingStatus", "inactive"), + ("productionRetrievalEligible", False), + ): + unsafe = card("1", 0.9) + unsafe[field] = value + repository = ProductionCardIndexRepository(search_function=lambda **_: [unsafe]) + with self.subTest(field=field), self.assertRaises(RetrievalError): + repository.search(self.plan) + + def test_search_cards_reuses_active_entity_and_binding_sql(self): + connection = FakeSearchConnection() + result = search_cards( + "甲的历史状态", + scope="work", + work_id=8, + ttype="character", + purpose="generation", + top=5, + connection_factory=lambda _: connection, + embedder=lambda _: [0.1, 0.2], + ) + sql = connection.statements[1][0] + self.assertIn("en.status='active'", sql) + self.assertIn("b.binding_status='active'", sql) + self.assertIn("en.source_action_policy='allowed'", sql) + self.assertEqual(result[0]["sourceKind"], "canonical_entity") + self.assertTrue(result[0]["productionRetrievalEligible"]) + self.assertNotIn("秘密", result[0]["visibleFields"]) + + def test_replay_repository_is_preregistered_upgrade_book_and_never_production_eligible(self): + replay_card = card("11", 0.9) + replay_card.update( + { + "sourceKind": "eval_draft", + "evaluationStatus": "eval_draft", + "sourceType": "upgrade_book", + "productionRetrievalEligible": False, + } + ) + replay_config = { + "targetChapter": 6, + "snapshot": {"asOfChapter": 5, "data": {"chapters": [{"chapter": 5, "text": "安全历史"}]}}, + "sources": [ + {"sourceId": "chapter:5", "sourceVersion": "chapter-v1", "chapter": 5} + ], + "authorization": { + "sourceStatus": "active", + "copyrightStatus": "licensed", + "sourceVersion": SOURCE_VERSION, + "allowedPurpose": ["offline_evaluation"], + "authorizationSnapshot": { + "id": "auth-1", + "version": "v1", + "immutable": True, + "sourceVersion": SOURCE_VERSION, + "sourceStatus": "active", + "allowedPurpose": ["offline_evaluation"], + "checkedAt": "2026-07-20T00:00:00Z", + "revalidationAt": "2026-07-21T00:00:00Z", + }, + }, + "leakageAudit": { + "targetFacts": {"targetChapter": 6, "forbiddenFacts": []} + }, + } + repository = ReplayCardIndexRepository.from_replay_config( + replay_config, + cards=[replay_card], + preregistered_card_ids=["11"], + ) + self.assertFalse(repository.search(self.plan)[0]["productionRetrievalEligible"]) + + denied = copy.deepcopy(replay_config) + denied["authorization"] = {} + with self.assertRaises(RetrievalError): + ReplayCardIndexRepository.from_replay_config( + denied, + cards=[replay_card], + preregistered_card_ids=["11"], + ) + + wrong = copy.deepcopy(replay_card) + wrong["sourceType"] = "extract_chapter" + with self.assertRaises(RetrievalError): + ReplayCardIndexRepository.from_replay_config( + replay_config, + cards=[wrong], + preregistered_card_ids=["11"], + ) + with self.assertRaises(RetrievalError): + ReplayCardIndexRepository.from_replay_config( + replay_config, + cards=[replay_card], + preregistered_card_ids=["12"], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/read-context/scripts/test_writer_contract.py b/.claude/skills/read-context/scripts/test_writer_contract.py new file mode 100644 index 0000000..c3b27a0 --- /dev/null +++ b/.claude/skills/read-context/scripts/test_writer_contract.py @@ -0,0 +1,239 @@ +#!/usr/bin/env python3 +"""正文写手输入输出合同的确定性与失败关闭测试。""" + +from __future__ import annotations + +import copy +import hashlib +import pathlib +import sys +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) + +from writer_contract import ( # noqa: E402 + ContractError, + calculate_target_chars, + canonical_json, + han_count, + normalize_text, + retrieval_identity, + validate_writer_context, + validate_writer_output, +) + + +def valid_context() -> dict: + """构造覆盖全部必填字段的最小合法上下文。""" + + context = { + "schemaVersion": "writer-context-v1", + "runId": "writer-run-001", + "attempt": 1, + "mode": "production", + "purpose": "production", + "qualityPolicyVersion": "writer-production-v1", + "workId": 8, + "targetChapter": 489, + "asOf": 488, + "contextSnapshot": { + "manifestId": "sha256:" + "1" * 64, + "contextSha256": "sha256:" + "2" * 64, + "generatedAt": "2026-07-20T00:00:00Z", + }, + "sourceVersion": "raw-file-v1:sha256:" + "3" * 64, + "authorizationSnapshot": { + "snapshotId": "auth-001", + "allowedPurpose": "generation", + "verifiedAt": "2026-07-20T00:00:00Z", + }, + "sourceStatus": "active", + "retrievalPlan": { + "planVersion": "writer-retrieval-plan-v1", + "planId": "sha256:" + "4" * 64, + "runId": "writer-run-001", + "asOf": 488, + "queries": [], + "cardIndexVersion": "card-v1", + "proseIndexVersion": "prose-v1", + "filters": { + "workId": 8, + "asOfChapter": 488, + "sourceStatus": "active", + "authorizationRequired": True, + }, + "tieBreak": "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC", + "tokenBudget": {"maxContextChars": 20000}, + }, + "retrievalManifest": { + "manifestVersion": "writer-retrieval-manifest-v1", + "manifestId": "sha256:" + "1" * 64, + "planId": "sha256:" + "4" * 64, + "sources": [], + "omittedSources": [], + }, + "fineOutline": { + "sourceRef": { + "sourceId": "fine-outline:489", + "sourceVersion": "fine-outline-v3", + "chapter": 489, + }, + "hardConstraints": ["必须完成围攻突围"], + "adjustableBeats": [], + "declaredNewFacts": [], + }, + "narrativeState": { + "time": "围攻当日", + "location": "圣蒂曼", + "characterPositions": {}, + "immediateSituation": "战斗持续", + }, + "factEvidence": [], + "proseEvidence": [], + "patternReferences": [], + "evidenceCoverage": [], + "outputContract": { + "targetChars": 4000, + "minChars": 3600, + "maxChars": 4400, + "frontmatterRequired": False, + "newSettingDeclarationRequired": True, + }, + "tokenBudget": {"maxContextChars": 20000, "usedContextChars": 0}, + "omittedSources": [], + "acceptanceEligible": True, + } + plan_payload = {key: value for key, value in context["retrievalPlan"].items() if key != "planId"} + context["retrievalPlan"]["planId"] = retrieval_identity(plan_payload) + context["retrievalManifest"]["planId"] = context["retrievalPlan"]["planId"] + manifest_payload = {key: value for key, value in context["retrievalManifest"].items() if key != "manifestId"} + context["retrievalManifest"]["manifestId"] = retrieval_identity(manifest_payload) + context["contextSnapshot"]["manifestId"] = context["retrievalManifest"]["manifestId"] + context["contextSnapshot"]["contextSha256"] = retrieval_identity(context) + return context + + +def valid_output() -> dict: + """构造覆盖全部必填字段的最小合法写手输出。""" + + body = normalize_text("第一段正文。") + digest = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest() + return { + "schemaVersion": "writer-output-v1", + "runId": "writer-run-001", + "attempt": 1, + "mode": "production", + "qualityPolicyVersion": "writer-production-v1", + "contextSnapshotId": "sha256:" + "1" * 64, + "contextSnapshotSha256": "sha256:" + "2" * 64, + "candidateVersion": 1, + "candidateSha256": digest, + "acceptanceEligible": True, + "candidateBody": body, + "claimLedger": [], + "evidenceRequests": [], + "newSettingDeclarations": [], + "selfCheck": {"hardConstraintsCovered": True, "notes": []}, + } + + +class WriterContractTest(unittest.TestCase): + def test_run_id_does_not_change_retrieval_identity(self): + first = {"runId": "run-a", "query": "咖啡\u0301", "nested": {"value": 1}} + second = {"runId": "run-b", "query": "咖啡\u0301", "nested": {"value": 1}} + self.assertEqual(retrieval_identity(first), retrieval_identity(second)) + + def test_text_is_nfc_and_lf_before_offsets_and_hash(self): + decomposed = "Cafe\u0301\r\n第二行\r第三行" + normalized = "Caf\u00e9\n第二行\n第三行" + self.assertEqual(normalize_text(decomposed), normalized) + self.assertEqual(canonical_json({"text": decomposed}), canonical_json({"text": normalized})) + + def test_han_count_does_not_count_markdown_or_non_han_text(self): + self.assertEqual(han_count("# **正文** 123 ABC,扩展𠀀"), 5) + + def test_target_chars_use_half_up_and_hard_bounds(self): + self.assertEqual( + calculate_target_chars( + recent_chapter_han_counts=[2501, 2502, 2503, 2504], + hard_event_count=3, + foreshadowing_action_count=0, + required_scene_count=0, + ), + 2100, + ) + self.assertEqual(calculate_target_chars(explicit_target_chars=1500), 2000) + self.assertEqual(calculate_target_chars(explicit_target_chars=11000), 10000) + self.assertEqual( + calculate_target_chars( + recent_chapter_han_counts=[3000, 3000, 3000], + hard_event_count=0, + foreshadowing_action_count=0, + required_scene_count=0, + ), + 2600, + ) + + def test_evaluation_and_diagnostic_contexts_are_never_acceptable(self): + for purpose in ("evaluation", "diagnostic"): + context = valid_context() + context["mode"] = "diagnostic_only" + context["purpose"] = purpose + context["qualityPolicyVersion"] = "writer-eval-v1" + context["acceptanceEligible"] = True + with self.assertRaises(ContractError): + validate_writer_context(context) + + context["acceptanceEligible"] = False + context["contextSnapshot"]["contextSha256"] = retrieval_identity(context) + validate_writer_context(context) + + def test_unknown_missing_and_wrong_version_fail_closed(self): + for mutation in ("unknown", "missing", "version"): + context = copy.deepcopy(valid_context()) + if mutation == "unknown": + context["unexpected"] = True + elif mutation == "missing": + del context["fineOutline"] + else: + context["schemaVersion"] = "writer-context-v2" + with self.subTest(mutation=mutation), self.assertRaises(ContractError): + validate_writer_context(context) + + output = valid_output() + output["selfCheck"]["unknown"] = True + with self.assertRaises(ContractError): + validate_writer_output(output) + + def test_candidate_hash_and_diagnostic_acceptance_are_checked(self): + output = valid_output() + validate_writer_output(output) + + output["candidateBody"] = "被修改的正文" + with self.assertRaises(ContractError): + validate_writer_output(output) + + diagnostic = valid_output() + diagnostic["mode"] = "diagnostic_only" + diagnostic["qualityPolicyVersion"] = "writer-eval-v1" + diagnostic["acceptanceEligible"] = True + with self.assertRaises(ContractError): + validate_writer_output(diagnostic) + + def test_plan_manifest_and_context_identity_tampering_fails_closed(self): + for path in ("plan", "manifest", "context"): + context = valid_context() + if path == "plan": + context["retrievalPlan"]["cardIndexVersion"] = "tampered" + elif path == "manifest": + context["retrievalManifest"]["sources"].append( + {"sourceId": "setting:1", "sourceVersion": "setting-v1"} + ) + else: + context["fineOutline"]["hardConstraints"].append("被篡改的约束") + with self.subTest(path=path), self.assertRaises(ContractError): + validate_writer_context(context) + + +if __name__ == "__main__": + unittest.main() diff --git a/.claude/skills/read-context/scripts/writer_contract.py b/.claude/skills/read-context/scripts/writer_contract.py new file mode 100644 index 0000000..2eb04e6 --- /dev/null +++ b/.claude/skills/read-context/scripts/writer_contract.py @@ -0,0 +1,549 @@ +#!/usr/bin/env python3 +"""正文写手上下文与输出的严格合同。 + +本模块只处理纯数据,不读取文件、数据库或网络。所有进入写手的文本先做 +Unicode NFC 与换行归一化,所有身份哈希都来自同一份规范 JSON。 +""" + +from __future__ import annotations + +import hashlib +import json +import math +import re +import unicodedata +from decimal import Decimal, ROUND_HALF_UP +from typing import Any, Mapping, Sequence + + +CONTEXT_VERSION = "writer-context-v1" +OUTPUT_VERSION = "writer-output-v1" +PLAN_VERSION = "writer-retrieval-plan-v1" +MANIFEST_VERSION = "writer-retrieval-manifest-v1" +TIE_BREAK = "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC" + +_HASH_RE = re.compile(r"^sha256:[0-9a-f]{64}$") +_RUN_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$") +_VOLATILE_IDENTITY_FIELDS = frozenset( + {"runId", "generatedAt", "timestamp", "executionNode", "contextSha256"} +) + +# Unicode Script=Han 覆盖的标准区段。兼容表意文字也按“汉字”计数, +# 但标点、Markdown、拉丁字母和数字不会落入这些区段。 +_HAN_RANGES = ( + (0x3400, 0x4DBF), + (0x4E00, 0x9FFF), + (0xF900, 0xFAFF), + (0x20000, 0x2EBEF), + (0x2F800, 0x2FA1F), + (0x30000, 0x323AF), +) + + +class ContractError(ValueError): + """输入不符合严格合同时抛出,调用方必须失败关闭。""" + + +def normalize_text(value: str) -> str: + """把文本统一为 NFC 与 LF,供哈希和 Unicode 偏移共同使用。""" + + if not isinstance(value, str): + raise ContractError("待归一化文本必须是字符串") + return unicodedata.normalize("NFC", value.replace("\r\n", "\n").replace("\r", "\n")) + + +def _normalize_json(value: Any) -> Any: + """递归归一化 JSON 值,并拒绝 JSON 之外或不可复现的数值。""" + + if isinstance(value, str): + return normalize_text(value) + if value is None or isinstance(value, (bool, int)): + return value + if isinstance(value, float): + if not math.isfinite(value): + raise ContractError("规范 JSON 不允许 NaN 或 Infinity") + return value + if isinstance(value, list): + return [_normalize_json(item) for item in value] + if isinstance(value, tuple): + return [_normalize_json(item) for item in value] + if isinstance(value, Mapping): + result: dict[str, Any] = {} + for raw_key, item in value.items(): + if not isinstance(raw_key, str): + raise ContractError("规范 JSON 的对象键必须是字符串") + key = normalize_text(raw_key) + if key in result: + raise ContractError(f"对象键在 NFC 归一化后冲突: {key}") + result[key] = _normalize_json(item) + return result + raise ContractError(f"值不是 JSON 类型: {type(value).__name__}") + + +def canonical_json(value: Any) -> str: + """输出 UTF-8 语义、键排序、无额外空白的规范 JSON 文本。""" + + return json.dumps( + _normalize_json(value), + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) + + +def _without_volatile_fields(value: Any) -> Any: + """递归移除运行元数据,防止同一检索输入得到不同身份。""" + + if isinstance(value, list): + return [_without_volatile_fields(item) for item in value] + if isinstance(value, Mapping): + return { + key: _without_volatile_fields(item) + for key, item in value.items() + if key not in _VOLATILE_IDENTITY_FIELDS + } + return value + + +def retrieval_identity(value: Any) -> str: + """计算计划、manifest 或上下文的稳定 SHA-256 身份。""" + + encoded = canonical_json(_without_volatile_fields(value)).encode("utf-8") + return "sha256:" + hashlib.sha256(encoded).hexdigest() + + +def han_count(value: str) -> int: + """统计 Unicode Han code point,不把标点或 Markdown 算入正文长度。""" + + text = normalize_text(value) + return sum(any(start <= ord(char) <= end for start, end in _HAN_RANGES) for char in text) + + +def _round_half_up(value: Decimal) -> int: + """以十进制 ROUND_HALF_UP 规则取整,避免 Python 银行家舍入。""" + + return int(value.quantize(Decimal("1"), rounding=ROUND_HALF_UP)) + + +def _round_to_100(value: Decimal) -> int: + """以百字为单位执行十进制半入取整。""" + + return int((value / Decimal(100)).quantize(Decimal("1"), rounding=ROUND_HALF_UP) * 100) + + +def calculate_target_chars( + *, + explicit_target_chars: int | None = None, + recent_chapter_han_counts: Sequence[int] = (), + default_target_chars: int = 4000, + hard_event_count: int = 3, + foreshadowing_action_count: int = 0, + required_scene_count: int = 0, + min_chars: int = 2000, + max_chars: int = 10000, +) -> int: + """按冻结历史中位数与细纲密度计算确定性目标汉字数。""" + + counts = { + "default_target_chars": default_target_chars, + "hard_event_count": hard_event_count, + "foreshadowing_action_count": foreshadowing_action_count, + "required_scene_count": required_scene_count, + "min_chars": min_chars, + "max_chars": max_chars, + } + if any(isinstance(value, bool) or not isinstance(value, int) for value in counts.values()): + raise ContractError("篇幅参数必须是整数") + if default_target_chars <= 0 or min_chars <= 0 or max_chars < min_chars or any(value < 0 for key, value in counts.items() if "count" in key): + raise ContractError("篇幅边界或细纲计数非法") + if explicit_target_chars is not None: + if isinstance(explicit_target_chars, bool) or not isinstance(explicit_target_chars, int): + raise ContractError("显式 targetChars 必须是整数") + return min(max(explicit_target_chars, min_chars), max_chars) + + valid_counts = list(recent_chapter_han_counts)[-20:] + if any(isinstance(value, bool) or not isinstance(value, int) or value < 500 for value in valid_counts): + raise ContractError("历史章汉字数必须是大于等于 500 的整数") + if len(valid_counts) >= 3: + ordered_counts = sorted(valid_counts) + midpoint = len(ordered_counts) // 2 + if len(ordered_counts) % 2: + baseline = Decimal(ordered_counts[midpoint]) + else: + baseline = (Decimal(ordered_counts[midpoint - 1]) + Decimal(ordered_counts[midpoint])) / 2 + else: + baseline = Decimal(default_target_chars) + density = ( + Decimal(hard_event_count) + + Decimal("0.5") * foreshadowing_action_count + + Decimal("0.5") * required_scene_count + ) + factor = min(Decimal("1.15"), max(Decimal("0.85"), Decimal("0.85") + Decimal("0.05") * (density - 3))) + target = _round_to_100(Decimal(_round_half_up(baseline)) * factor) + return min(max(target, min_chars), max_chars) + + +def _object( + value: Any, + path: str, + required: set[str] | frozenset[str], + optional: set[str] | frozenset[str] = frozenset(), +) -> Mapping[str, Any]: + """校验严格对象,任何缺字段或未知字段都立即失败。""" + + if not isinstance(value, Mapping): + raise ContractError(f"{path} 必须是对象") + missing = sorted(required - set(value)) + unknown = sorted(set(value) - required - optional) + if missing: + raise ContractError(f"{path} 缺少字段: {','.join(missing)}") + if unknown: + raise ContractError(f"{path} 包含未知字段: {','.join(unknown)}") + return value + + +def _string(value: Any, path: str, *, nonempty: bool = True) -> str: + """校验字符串,并在需要时拒绝空值。""" + + if not isinstance(value, str) or (nonempty and not value.strip()): + raise ContractError(f"{path} 必须是非空字符串") + if value != normalize_text(value): + raise ContractError(f"{path} 必须预先归一化为 Unicode NFC/LF") + return value + + +def _integer(value: Any, path: str, *, minimum: int = 0) -> int: + """校验整数,显式排除 bool 这一 Python int 子类。""" + + if isinstance(value, bool) or not isinstance(value, int) or value < minimum: + raise ContractError(f"{path} 必须是大于等于 {minimum} 的整数") + return value + + +def _boolean(value: Any, path: str) -> bool: + """校验严格布尔值。""" + + if not isinstance(value, bool): + raise ContractError(f"{path} 必须是布尔值") + return value + + +def _array(value: Any, path: str) -> list[Any]: + """校验数组并返回原值,拒绝元组等隐式转换。""" + + if not isinstance(value, list): + raise ContractError(f"{path} 必须是数组") + return value + + +def _hash(value: Any, path: str) -> str: + """校验带算法前缀的 SHA-256。""" + + text = _string(value, path) + if not _HASH_RE.fullmatch(text): + raise ContractError(f"{path} 必须是 sha256: 加 64 位小写十六进制") + return text + + +def _source_ref(value: Any, path: str) -> None: + """校验不可变来源引用;历史原文可额外携带块和字符区间。""" + + ref = _object( + value, + path, + frozenset({"sourceId", "sourceVersion"}), + frozenset({"chapter", "blockId", "startCodePoint", "endCodePoint", "contentSha256", "sourceType"}), + ) + _string(ref["sourceId"], f"{path}.sourceId") + _string(ref["sourceVersion"], f"{path}.sourceVersion") + for field in ("chapter", "blockId", "startCodePoint", "endCodePoint"): + if field in ref: + _integer(ref[field], f"{path}.{field}", minimum=0 if "CodePoint" in field else 1) + if "startCodePoint" in ref and "endCodePoint" in ref and ref["endCodePoint"] <= ref["startCodePoint"]: + raise ContractError(f"{path} 字符区间必须是非空左闭右开区间") + if "contentSha256" in ref: + _hash(ref["contentSha256"], f"{path}.contentSha256") + if "sourceType" in ref: + _string(ref["sourceType"], f"{path}.sourceType") + + +def _validate_plan(value: Any, path: str) -> None: + """校验固定检索计划,不允许写手临场扩张查询。""" + + plan = _object( + value, + path, + frozenset( + {"planVersion", "planId", "runId", "asOf", "queries", "cardIndexVersion", "proseIndexVersion", "filters", "tieBreak", "tokenBudget"} + ), + ) + if plan["planVersion"] != PLAN_VERSION: + raise ContractError(f"{path}.planVersion 版本不支持") + _hash(plan["planId"], f"{path}.planId") + if not _RUN_ID_RE.fullmatch(_string(plan["runId"], f"{path}.runId")): + raise ContractError(f"{path}.runId 格式非法") + _integer(plan["asOf"], f"{path}.asOf", minimum=1) + for index, query in enumerate(_array(plan["queries"], f"{path}.queries")): + item = _object(query, f"{path}.queries[{index}]", frozenset({"queryId", "text", "entityTypes", "purpose", "topK"})) + _string(item["queryId"], f"{path}.queries[{index}].queryId") + _string(item["text"], f"{path}.queries[{index}].text") + if any(not isinstance(kind, str) or not kind for kind in _array(item["entityTypes"], f"{path}.queries[{index}].entityTypes")): + raise ContractError(f"{path}.queries[{index}].entityTypes 必须是非空字符串数组") + _string(item["purpose"], f"{path}.queries[{index}].purpose") + _integer(item["topK"], f"{path}.queries[{index}].topK", minimum=1) + _string(plan["cardIndexVersion"], f"{path}.cardIndexVersion") + _string(plan["proseIndexVersion"], f"{path}.proseIndexVersion") + filters = _object(plan["filters"], f"{path}.filters", frozenset({"workId", "asOfChapter", "sourceStatus", "authorizationRequired"})) + _integer(filters["workId"], f"{path}.filters.workId", minimum=1) + _integer(filters["asOfChapter"], f"{path}.filters.asOfChapter", minimum=1) + _string(filters["sourceStatus"], f"{path}.filters.sourceStatus") + _boolean(filters["authorizationRequired"], f"{path}.filters.authorizationRequired") + if plan["tieBreak"] != TIE_BREAK: + raise ContractError(f"{path}.tieBreak 不符合稳定排序合同") + budget = _object(plan["tokenBudget"], f"{path}.tokenBudget", frozenset({"maxContextChars"}), frozenset({"cardChars", "recentProseChars", "historicalProseChars", "patternChars"})) + for key, item in budget.items(): + _integer(item, f"{path}.tokenBudget.{key}", minimum=0) + identity_payload = {key: item for key, item in plan.items() if key != "planId"} + if plan["planId"] != retrieval_identity(identity_payload): + raise ContractError(f"{path}.planId 与计划内容不匹配") + + +def _validate_manifest(value: Any, path: str) -> None: + """校验检索清单的来源与裁剪回显。""" + + manifest = _object(value, path, frozenset({"manifestVersion", "manifestId", "planId", "sources", "omittedSources"})) + if manifest["manifestVersion"] != MANIFEST_VERSION: + raise ContractError(f"{path}.manifestVersion 版本不支持") + _hash(manifest["manifestId"], f"{path}.manifestId") + _hash(manifest["planId"], f"{path}.planId") + for index, source in enumerate(_array(manifest["sources"], f"{path}.sources")): + _source_ref(source, f"{path}.sources[{index}]") + for index, omitted in enumerate(_array(manifest["omittedSources"], f"{path}.omittedSources")): + item = _object(omitted, f"{path}.omittedSources[{index}]", frozenset({"sourceId", "reason"})) + _string(item["sourceId"], f"{path}.omittedSources[{index}].sourceId") + _string(item["reason"], f"{path}.omittedSources[{index}].reason") + identity_payload = {key: item for key, item in manifest.items() if key != "manifestId"} + if manifest["manifestId"] != retrieval_identity(identity_payload): + raise ContractError(f"{path}.manifestId 与来源清单不匹配") + + +def validate_writer_context(value: Any) -> dict[str, Any]: + """校验 WriterContext v1;成功时返回可安全复制的规范 JSON 对象。""" + + required = frozenset( + { + "schemaVersion", "runId", "attempt", "mode", "purpose", "qualityPolicyVersion", + "workId", "targetChapter", "asOf", "contextSnapshot", "sourceVersion", + "authorizationSnapshot", "sourceStatus", "retrievalPlan", "retrievalManifest", + "fineOutline", "narrativeState", "factEvidence", "proseEvidence", + "patternReferences", "evidenceCoverage", "outputContract", "tokenBudget", + "omittedSources", "acceptanceEligible", + } + ) + context = _object(value, "$", required) + if context["schemaVersion"] != CONTEXT_VERSION: + raise ContractError("$.schemaVersion 版本不支持") + run_id = _string(context["runId"], "$.runId") + if not _RUN_ID_RE.fullmatch(run_id): + raise ContractError("$.runId 格式非法") + _integer(context["attempt"], "$.attempt", minimum=1) + if context["mode"] not in {"production", "diagnostic_only"}: + raise ContractError("$.mode 枚举非法") + if context["purpose"] not in {"production", "evaluation", "diagnostic"}: + raise ContractError("$.purpose 枚举非法") + _string(context["qualityPolicyVersion"], "$.qualityPolicyVersion") + _integer(context["workId"], "$.workId", minimum=1) + target = _integer(context["targetChapter"], "$.targetChapter", minimum=1) + as_of = _integer(context["asOf"], "$.asOf", minimum=1) + if as_of >= target: + raise ContractError("$.asOf 必须早于 targetChapter") + + snapshot = _object(context["contextSnapshot"], "$.contextSnapshot", frozenset({"manifestId", "contextSha256", "generatedAt"})) + _hash(snapshot["manifestId"], "$.contextSnapshot.manifestId") + _hash(snapshot["contextSha256"], "$.contextSnapshot.contextSha256") + _string(snapshot["generatedAt"], "$.contextSnapshot.generatedAt") + _string(context["sourceVersion"], "$.sourceVersion") + authorization = _object( + context["authorizationSnapshot"], + "$.authorizationSnapshot", + frozenset({"snapshotId", "allowedPurpose", "verifiedAt"}), + frozenset({"expiresAt", "sourceVersion"}), + ) + for key, item in authorization.items(): + _string(item, f"$.authorizationSnapshot.{key}") + if context["sourceStatus"] not in {"active", "authorized", "frozen_authorized"}: + raise ContractError("$.sourceStatus 不允许生成") + + _validate_plan(context["retrievalPlan"], "$.retrievalPlan") + _validate_manifest(context["retrievalManifest"], "$.retrievalManifest") + if context["retrievalPlan"]["runId"] != run_id or context["retrievalPlan"]["asOf"] != as_of: + raise ContractError("检索计划与上下文的 runId/asOf 不一致") + if context["retrievalManifest"]["planId"] != context["retrievalPlan"]["planId"]: + raise ContractError("检索 manifest 未绑定当前计划") + if context["contextSnapshot"]["manifestId"] != context["retrievalManifest"]["manifestId"]: + raise ContractError("上下文快照未绑定当前 manifest") + + outline = _object(context["fineOutline"], "$.fineOutline", frozenset({"sourceRef", "hardConstraints", "adjustableBeats", "declaredNewFacts"})) + _source_ref(outline["sourceRef"], "$.fineOutline.sourceRef") + for field in ("hardConstraints", "adjustableBeats"): + if any(not isinstance(item, str) or not item for item in _array(outline[field], f"$.fineOutline.{field}")): + raise ContractError(f"$.fineOutline.{field} 必须是非空字符串数组") + for index, fact in enumerate(_array(outline["declaredNewFacts"], "$.fineOutline.declaredNewFacts")): + item = _object(fact, f"$.fineOutline.declaredNewFacts[{index}]", frozenset({"factId", "text", "sourceRef"})) + _string(item["factId"], f"$.fineOutline.declaredNewFacts[{index}].factId") + _string(item["text"], f"$.fineOutline.declaredNewFacts[{index}].text") + _source_ref(item["sourceRef"], f"$.fineOutline.declaredNewFacts[{index}].sourceRef") + + state = _object(context["narrativeState"], "$.narrativeState", frozenset({"time", "location", "characterPositions", "immediateSituation"})) + for field in ("time", "location", "immediateSituation"): + _string(state[field], f"$.narrativeState.{field}", nonempty=False) + positions = _object(state["characterPositions"], "$.narrativeState.characterPositions", frozenset(state["characterPositions"].keys()) if isinstance(state["characterPositions"], Mapping) else frozenset()) + for key, item in positions.items(): + _string(key, "$.narrativeState.characterPositions.") + _string(item, f"$.narrativeState.characterPositions.{key}") + + for index, evidence in enumerate(_array(context["factEvidence"], "$.factEvidence")): + item = _object(evidence, f"$.factEvidence[{index}]", frozenset({"evidenceId", "fact", "sourceType", "sourceRef", "contentSha256", "riskLevel"})) + for field in ("evidenceId", "fact", "sourceType"): + _string(item[field], f"$.factEvidence[{index}].{field}") + if item["sourceType"] not in {"historical_prose", "formal_setting", "canonical_state", "fine_outline_declared_new"}: + raise ContractError(f"$.factEvidence[{index}].sourceType 枚举非法") + _source_ref(item["sourceRef"], f"$.factEvidence[{index}].sourceRef") + if item["sourceType"] == "historical_prose": + required_location = {"chapter", "blockId", "startCodePoint", "endCodePoint"} + if not required_location.issubset(item["sourceRef"]): + raise ContractError(f"$.factEvidence[{index}] 历史事实必须回到章、块和字符区间") + _hash(item["contentSha256"], f"$.factEvidence[{index}].contentSha256") + if item["riskLevel"] not in {"low", "medium", "high"}: + raise ContractError(f"$.factEvidence[{index}].riskLevel 枚举非法") + + for index, evidence in enumerate(_array(context["proseEvidence"], "$.proseEvidence")): + item = _object(evidence, f"$.proseEvidence[{index}]", frozenset({"evidenceId", "chapter", "sourceRef", "contentSha256", "purpose", "text", "isRecentBaseline"})) + _string(item["evidenceId"], f"$.proseEvidence[{index}].evidenceId") + chapter = _integer(item["chapter"], f"$.proseEvidence[{index}].chapter", minimum=1) + if chapter > as_of: + raise ContractError(f"$.proseEvidence[{index}] 超出冻结线") + _source_ref(item["sourceRef"], f"$.proseEvidence[{index}].sourceRef") + if not {"chapter", "blockId", "startCodePoint", "endCodePoint"}.issubset(item["sourceRef"]): + raise ContractError(f"$.proseEvidence[{index}] 必须带章、块和字符区间") + _hash(item["contentSha256"], f"$.proseEvidence[{index}].contentSha256") + _string(item["purpose"], f"$.proseEvidence[{index}].purpose") + text = _string(item["text"], f"$.proseEvidence[{index}].text") + _boolean(item["isRecentBaseline"], f"$.proseEvidence[{index}].isRecentBaseline") + expected = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() + if item["contentSha256"] != expected: + raise ContractError(f"$.proseEvidence[{index}] 文本哈希不匹配") + + for index, reference in enumerate(_array(context["patternReferences"], "$.patternReferences")): + _source_ref(reference, f"$.patternReferences[{index}]") + for index, coverage in enumerate(_array(context["evidenceCoverage"], "$.evidenceCoverage")): + item = _object(coverage, f"$.evidenceCoverage[{index}]", frozenset({"elementId", "elementType", "name", "status", "factEvidenceIds", "proseEvidenceIds", "gapReason"})) + for field in ("elementId", "elementType", "name", "gapReason"): + _string(item[field], f"$.evidenceCoverage[{index}].{field}", nonempty=field != "gapReason") + if item["status"] not in {"supported", "declared_new", "card_gap", "style_gap", "unsupported", "conflict"}: + raise ContractError(f"$.evidenceCoverage[{index}].status 枚举非法") + for field in ("factEvidenceIds", "proseEvidenceIds"): + if any(not isinstance(item_id, str) or not item_id for item_id in _array(item[field], f"$.evidenceCoverage[{index}].{field}")): + raise ContractError(f"$.evidenceCoverage[{index}].{field} 必须是字符串数组") + + output = _object(context["outputContract"], "$.outputContract", frozenset({"targetChars", "minChars", "maxChars", "frontmatterRequired", "newSettingDeclarationRequired"})) + for field in ("targetChars", "minChars", "maxChars"): + _integer(output[field], f"$.outputContract.{field}", minimum=1) + if not output["minChars"] <= output["targetChars"] <= output["maxChars"]: + raise ContractError("$.outputContract 篇幅范围不包含目标值") + _boolean(output["frontmatterRequired"], "$.outputContract.frontmatterRequired") + _boolean(output["newSettingDeclarationRequired"], "$.outputContract.newSettingDeclarationRequired") + budget = _object(context["tokenBudget"], "$.tokenBudget", frozenset({"maxContextChars", "usedContextChars"})) + for field in budget: + _integer(budget[field], f"$.tokenBudget.{field}", minimum=0) + if budget["usedContextChars"] > budget["maxContextChars"]: + raise ContractError("$.tokenBudget 已超预算") + for index, omitted in enumerate(_array(context["omittedSources"], "$.omittedSources")): + item = _object(omitted, f"$.omittedSources[{index}]", frozenset({"sourceId", "reason"})) + _string(item["sourceId"], f"$.omittedSources[{index}].sourceId") + _string(item["reason"], f"$.omittedSources[{index}].reason") + + eligible = _boolean(context["acceptanceEligible"], "$.acceptanceEligible") + if context["purpose"] in {"evaluation", "diagnostic"} or context["mode"] == "diagnostic_only": + if eligible: + raise ContractError("评测或诊断上下文必须 acceptanceEligible=false") + elif context["qualityPolicyVersion"] != "writer-production-v1": + raise ContractError("生产上下文必须绑定 writer-production-v1") + if context["contextSnapshot"]["contextSha256"] != retrieval_identity(context): + raise ContractError("$.contextSnapshot.contextSha256 与上下文内容不匹配") + return json.loads(canonical_json(context)) + + +def validate_writer_output(value: Any) -> dict[str, Any]: + """校验 WriterOutput v1、正文哈希和全部来源引用。""" + + output = _object( + value, + "$", + frozenset( + {"schemaVersion", "runId", "attempt", "mode", "qualityPolicyVersion", "contextSnapshotId", "contextSnapshotSha256", "candidateVersion", "candidateSha256", "acceptanceEligible", "candidateBody", "claimLedger", "evidenceRequests", "newSettingDeclarations", "selfCheck"} + ), + ) + if output["schemaVersion"] != OUTPUT_VERSION: + raise ContractError("$.schemaVersion 版本不支持") + if not _RUN_ID_RE.fullmatch(_string(output["runId"], "$.runId")): + raise ContractError("$.runId 格式非法") + _integer(output["attempt"], "$.attempt", minimum=1) + if output["mode"] not in {"production", "diagnostic_only"}: + raise ContractError("$.mode 枚举非法") + _string(output["qualityPolicyVersion"], "$.qualityPolicyVersion") + _hash(output["contextSnapshotId"], "$.contextSnapshotId") + _hash(output["contextSnapshotSha256"], "$.contextSnapshotSha256") + _integer(output["candidateVersion"], "$.candidateVersion", minimum=1) + body = _string(output["candidateBody"], "$.candidateBody") + candidate_hash = _hash(output["candidateSha256"], "$.candidateSha256") + expected_hash = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest() + if candidate_hash != expected_hash: + raise ContractError("$.candidateSha256 与规范正文不匹配") + eligible = _boolean(output["acceptanceEligible"], "$.acceptanceEligible") + if output["mode"] == "diagnostic_only" and eligible: + raise ContractError("诊断输出必须 acceptanceEligible=false") + + for index, claim in enumerate(_array(output["claimLedger"], "$.claimLedger")): + item = _object( + claim, + f"$.claimLedger[{index}]", + frozenset({"claimId", "candidateSha256", "startCodePoint", "endCodePoint", "factType", "factEvidenceId", "coverageState"}), + frozenset({"proseEvidenceId"}), + ) + for field in ("claimId", "factType", "factEvidenceId", "coverageState"): + _string(item[field], f"$.claimLedger[{index}].{field}") + _hash(item["candidateSha256"], f"$.claimLedger[{index}].candidateSha256") + if item["candidateSha256"] != candidate_hash: + raise ContractError(f"$.claimLedger[{index}] 未绑定当前候选") + start = _integer(item["startCodePoint"], f"$.claimLedger[{index}].startCodePoint") + end = _integer(item["endCodePoint"], f"$.claimLedger[{index}].endCodePoint", minimum=1) + if end <= start or end > len(body): + raise ContractError(f"$.claimLedger[{index}] Unicode 偏移越界") + if "proseEvidenceId" in item and item["proseEvidenceId"] is not None: + _string(item["proseEvidenceId"], f"$.claimLedger[{index}].proseEvidenceId") + for index, request in enumerate(_array(output["evidenceRequests"], "$.evidenceRequests")): + item = _object(request, f"$.evidenceRequests[{index}]", frozenset({"requestId", "query", "reason", "priority"})) + for field in ("requestId", "query", "reason", "priority"): + _string(item[field], f"$.evidenceRequests[{index}].{field}") + for index, declaration in enumerate(_array(output["newSettingDeclarations"], "$.newSettingDeclarations")): + item = _object(declaration, f"$.newSettingDeclarations[{index}]", frozenset({"declarationId", "factType", "text", "startCodePoint", "endCodePoint"})) + for field in ("declarationId", "factType", "text"): + _string(item[field], f"$.newSettingDeclarations[{index}].{field}") + start = _integer(item["startCodePoint"], f"$.newSettingDeclarations[{index}].startCodePoint") + end = _integer(item["endCodePoint"], f"$.newSettingDeclarations[{index}].endCodePoint", minimum=1) + if end <= start or end > len(body): + raise ContractError(f"$.newSettingDeclarations[{index}] Unicode 偏移越界") + self_check = _object(output["selfCheck"], "$.selfCheck", frozenset({"hardConstraintsCovered", "notes"})) + _boolean(self_check["hardConstraintsCovered"], "$.selfCheck.hardConstraintsCovered") + if any(not isinstance(note, str) for note in _array(self_check["notes"], "$.selfCheck.notes")): + raise ContractError("$.selfCheck.notes 必须是字符串数组") + return json.loads(canonical_json(output)) + + +__all__ = [ + "CONTEXT_VERSION", "OUTPUT_VERSION", "PLAN_VERSION", "MANIFEST_VERSION", "TIE_BREAK", + "ContractError", "normalize_text", "canonical_json", "retrieval_identity", "han_count", + "calculate_target_chars", "validate_writer_context", "validate_writer_output", +] diff --git a/.claude/skills/replay-eval/scripts/load_reference_work.py b/.claude/skills/replay-eval/scripts/load_reference_work.py index 29527f5..b5bee44 100644 --- a/.claude/skills/replay-eval/scripts/load_reference_work.py +++ b/.claude/skills/replay-eval/scripts/load_reference_work.py @@ -1,14 +1,16 @@ #!/usr/bin/env python3 """从实验库只读组装回放评测配置。 -本适配器只做 SELECT 和临时文件输出,不写数据库、不读取正文 block 的内容。 -参考作品的目标章只进入审计侧 proxy,历史规划上下文和三臂卡注入严格分开。 +本适配器只做 SELECT 和临时文件输出,不写数据库。正文读取仅允许通过 +`load_frozen_prose_rows()` 在只读快照内读取冻结线以前的 Canonical block; +参考作品目标章仍只进入审计侧 proxy,历史上下文和三臂卡注入严格分开。 """ from __future__ import annotations import argparse import copy +import hashlib import json import re from pathlib import Path @@ -98,6 +100,31 @@ def _card_history(payload: Mapping[str, Any]) -> list[Mapping[str, Any]]: raise AdapterError("卡缺少可按绝对章号冻结的历史字段") +def _structured_source_refs(payload: Mapping[str, Any], history: Sequence[Mapping[str, Any]], as_of: int) -> list[dict[str, Any]]: + """提取卡内结构化原文指针,文本“出处”不能替代块级引用。""" + + candidates: list[Any] = [payload.get("sourceRefs")] + fields = payload.get("字段") + if isinstance(fields, Mapping): + candidates.append(fields.get("sourceRefs")) + candidates.extend(item.get("sourceRefs") for item in history if isinstance(item, Mapping)) + refs: list[dict[str, Any]] = [] + for candidate in candidates: + if not isinstance(candidate, list): + continue + for raw_ref in candidate: + if not isinstance(raw_ref, Mapping): + continue + chapter = normalize_chapter(raw_ref.get("chapter") or raw_ref.get("章")) + if chapter is None or chapter > as_of: + continue + normalized = copy.deepcopy(dict(raw_ref)) + normalized["chapter"] = chapter + normalized.pop("章", None) + refs.append(normalized) + return refs + + def project_card(row: Mapping[str, Any], *, as_of: int, source_version: str) -> dict[str, Any]: """将候选卡投影为截至 as_of 的 eval-only 索引视图。""" @@ -129,7 +156,13 @@ def project_card(row: Mapping[str, Any], *, as_of: int, source_version: str) -> "cardId": card_id, "type": card_type, "name": name, + "score": float(row.get("score") or 0), + "sourceId": f"eval-draft:{card_id}", + "sourceVersion": source_version, + "sourceOffset": 0, "milestones": copy.deepcopy(history), + "stateAsOf": copy.deepcopy(history), + "sourceRefs": _structured_source_refs(payload, history, normalized_as_of), "appearanceChapters": appearances, "derivedState": { "asOfChapter": normalized_as_of, @@ -142,12 +175,116 @@ def project_card(row: Mapping[str, Any], *, as_of: int, source_version: str) -> "chapterRange": f"1-{normalized_as_of}", }, "evaluationStatus": "eval_draft", + "sourceType": "upgrade_book", + "sourceKind": "eval_draft", "upstreamStatus": str(row.get("status") or "unknown"), "productionRetrievalEligible": False, "omittedHistoryCount": len(omitted), } +def begin_read_snapshot(conn: Any) -> None: + """在第一条业务查询前固定可重复读、只读事务。""" + + conn.execute("SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY") + + +def load_frozen_prose_rows( + *, + dsn: str, + tenant_id: int, + work_id: int, + as_of: int, + source_refs: Sequence[Mapping[str, Any]] = (), + chapter_numbers: Sequence[int] = (), +) -> list[dict[str, Any]]: + """从同一只读快照读取冻结线内的 Canonical 历史原文。 + + SQL 只存在于 replay-eval 读取适配器;read-context 的作品与回放仓储都 + 调用本入口,避免再造正文读取旁路。 + """ + + normalized_as_of = _required_chapter(as_of, "as_of") + requested_chapters = {_required_chapter(item, "chapter_numbers[]") for item in chapter_numbers} + block_ids: set[int] = set() + refs_by_block: dict[int, list[Mapping[str, Any]]] = {} + for index, ref in enumerate(source_refs): + if not isinstance(ref, Mapping): + raise AdapterError(f"source_refs[{index}] 必须是对象") + chapter = _required_chapter(ref.get("chapter"), f"source_refs[{index}].chapter") + if chapter > normalized_as_of: + raise AdapterError(f"source_refs[{index}] 包含目标章或未来章") + block_id = ref.get("blockId") + if isinstance(block_id, bool) or not isinstance(block_id, int) or block_id <= 0: + raise AdapterError(f"source_refs[{index}].blockId 必须是正整数") + block_ids.add(block_id) + refs_by_block.setdefault(block_id, []).append(ref) + if any(chapter > normalized_as_of for chapter in requested_chapters): + raise AdapterError("chapter_numbers 包含目标章或未来章") + if not requested_chapters and not block_ids: + return [] + + with psycopg.connect(dsn, row_factory=dict_row) as conn: + begin_read_snapshot(conn) + rows = conn.execute( + """ + SELECT ch.order_no AS chapter,ch.id AS chapter_id,ch.status AS chapter_status, + b.id AS block_id,b.order_no AS block_order,b.revision,b.content_text + FROM muse_content_chapter ch + JOIN muse_content_block b ON b.chapter_id=ch.id + WHERE ch.tenant_id=%s AND ch.work_id=%s AND ch.deleted=FALSE + AND b.tenant_id=%s AND b.work_id=%s AND b.deleted=FALSE + AND ch.order_no<=%s AND ch.status IN ('published','confirmed','canonical') + AND (ch.order_no=ANY(%s) OR b.id=ANY(%s)) + ORDER BY ch.order_no,b.order_no,b.id + """, + (tenant_id, work_id, tenant_id, work_id, normalized_as_of, sorted(requested_chapters), sorted(block_ids)), + ).fetchall() + + result: list[dict[str, Any]] = [] + for row in rows: + text = str(row.get("content_text") or "") + block_id = int(row["block_id"]) + matching_refs = refs_by_block.get(block_id) + if matching_refs: + for ref in matching_refs: + start = int(ref.get("startCodePoint") or 0) + end = int(ref.get("endCodePoint") or len(text)) + if start < 0 or end <= start or end > len(text): + raise AdapterError(f"block {block_id} 的字符区间越界") + fragment = text[start:end] + result.append( + { + "chapter": int(row["chapter"]), + "blockId": block_id, + "blockOrder": int(row["block_order"]), + "sourceRef": copy.deepcopy(dict(ref)), + "text": fragment, + "contentSha256": "sha256:" + hashlib.sha256(fragment.encode("utf-8")).hexdigest(), + } + ) + elif int(row["chapter"]) in requested_chapters: + source_version = f"chapter:{row['chapter']}:block:{block_id}:revision:{row.get('revision') or 0}" + result.append( + { + "chapter": int(row["chapter"]), + "blockId": block_id, + "blockOrder": int(row["block_order"]), + "sourceRef": { + "sourceId": f"chapter:{row['chapter']}:block:{block_id}", + "sourceVersion": source_version, + "chapter": int(row["chapter"]), + "blockId": block_id, + "startCodePoint": 0, + "endCodePoint": len(text), + }, + "text": text, + "contentSha256": "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest(), + } + ) + return result + + def _card_source_version(row: Mapping[str, Any], base_version: str) -> str: """把卡行 revision 纳入来源版本,防止卡内容变更复用旧版本。""" @@ -435,7 +572,7 @@ def load_reference_rows( correct_ids, placebo_ids = _validate_selection(card_selection) selected_ids = [int(item) if str(item).isdigit() else item for item in correct_ids + placebo_ids] with psycopg.connect(dsn, row_factory=dict_row) as conn: - conn.execute("SET TRANSACTION READ ONLY") + begin_read_snapshot(conn) work = conn.execute( """ SELECT id,title,revision,chapter_count,parse_status,import_status diff --git a/.claude/skills/search/scripts/search.py b/.claude/skills/search/scripts/search.py index 442017b..852337e 100644 --- a/.claude/skills/search/scripts/search.py +++ b/.claude/skills/search/scripts/search.py @@ -6,6 +6,7 @@ import json import pathlib import sys +from typing import Any, Callable import click import psycopg @@ -38,6 +39,140 @@ def visible(ai_rule, purpose): return purpose in ai_rule +def _default_embedder(intent: str) -> list[float]: + """复用 embed skill 生成单条查询向量,并统一失败语义。""" + + vectors, bad = embed_texts(_session(), [intent]) + if bad or not vectors: + raise ValueError("查询嵌入失败") + return vectors[0] + + +def _structured_source_refs(payload: dict[str, Any], lineage: Any) -> list[dict[str, Any]]: + """只接收结构化来源指针;人类可读的“出处”文本不能冒充可回读引用。""" + + candidates = [] + if isinstance(lineage, dict): + candidates.append(lineage.get("sourceRefs")) + fields = payload.get("字段") + if isinstance(fields, dict): + candidates.append(fields.get("sourceRefs")) + candidates.append(payload.get("sourceRefs")) + for value in candidates: + if isinstance(value, list) and all(isinstance(item, dict) for item in value): + return value + return [] + + +def _milestones(payload: dict[str, Any]) -> list[dict[str, Any]]: + """从卡字段中提取历史里程碑,终态摘要不会进入冻结投影。""" + + fields = payload.get("字段") + if not isinstance(fields, dict): + return [] + for key in ("演变历程", "演变轨迹", "milestones"): + value = fields.get(key) + if isinstance(value, list) and all(isinstance(item, dict) for item in value): + return value + return [] + + +def search_cards( + intent: str, + *, + scope: str = "admin", + work_id: int | None = None, + ttype: str | None = None, + purpose: str = "generation", + top: int = 5, + dsn: str = DSN, + tenant_id: int = TENANT, + connection_factory: Callable[..., Any] = psycopg.connect, + embedder: Callable[[str], list[float]] = _default_embedder, +) -> list[dict[str, Any]]: + """执行唯一的卡检索语义,CLI 与正文读取器共同调用本函数。 + + 生产作品面只读 active Canonical entity、允许状态和有效绑定;治理面 + 保留原有 draft 能力,但正文生产适配器不会调用治理面。 + """ + + if scope not in {"admin", "work"}: + raise ValueError("scope 只能是 admin 或 work") + if scope == "work" and not work_id: + raise ValueError("scope=work 必须提供 work_id") + if purpose not in {"generation", "planning", "detection", "extraction"}: + raise ValueError("purpose 非法") + if isinstance(top, bool) or not isinstance(top, int) or top <= 0: + raise ValueError("top 必须是正整数") + + qvec = json.dumps(embedder(intent)) + with connection_factory(dsn) as conn: + ai_rules = load_ai_context(conn) + if scope == "admin": + sql = """SELECT 'draft' AS src, d.id, d.draft_payload AS payload, d.status, + 1 - (e.embedding <=> %s::vector) AS score, + d.revision, d.current_canonical_snapshot AS lineage, + NULL::varchar AS binding_status, d.source_status + FROM example_knowledge_embedding e + JOIN muse_knowledge_draft d ON d.id = e.draft_id + WHERE e.tenant_id=%s AND e.deleted=FALSE AND d.deleted=FALSE""" + args = [qvec, tenant_id] + else: + sql = """SELECT 'entity' AS src, en.id, + jsonb_build_object('型', en.entity_type, '名称', en.normalized_name, + '一句话摘要', en.description, '字段', en.attributes) AS payload, + en.status, 1 - (e.embedding <=> %s::vector) AS score, + en.revision, en.lineage_payload AS lineage, + b.binding_status, en.source_status + FROM example_knowledge_embedding e + JOIN muse_knowledge_entity en ON en.id = e.entity_id + JOIN muse_knowledge_binding b ON b.kb_id = en.kb_id AND b.work_id = %s + AND b.binding_status='active' AND b.deleted=FALSE AND b.tenant_id=%s + WHERE e.tenant_id=%s AND e.deleted=FALSE AND en.deleted=FALSE + AND en.status='active' AND en.source_status IN ('active','authorized') + AND en.source_action_policy='allowed'""" + args = [qvec, work_id, tenant_id, tenant_id] + if ttype: + sql += (" AND d.draft_payload->>'型' = %s" if scope == "admin" else " AND en.entity_type = %s") + args.append(ttype) + id_column = "d.id" if scope == "admin" else "en.id" + revision_column = "d.revision" if scope == "admin" else "en.revision" + sql += f" ORDER BY score DESC, {revision_column}::text ASC, {id_column}::text ASC LIMIT %s" + args.append(top) + rows = conn.execute(sql, args).fetchall() + + results = [] + for src, row_id, raw_payload, status, score, revision, lineage, binding_status, source_status in rows: + payload = raw_payload or {} + card_type = payload.get("型") or payload.get("type") or "?" + rules = ai_rules.get(card_type, {}) + fields = payload.get("字段") or {} + visible_fields = {key: item for key, item in fields.items() if visible(rules.get(key), purpose)} + source_version = f"{src}-revision:{revision or 0}" + source_id = f"canonical-entity:{row_id}" if src == "entity" else f"draft:{row_id}" + results.append( + { + "cardId": str(row_id), + "type": card_type, + "name": payload.get("名称"), + "score": float(score), + "summary": payload.get("一句话摘要"), + "visibleFields": visible_fields, + "omittedFields": sorted(set(fields) - set(visible_fields)), + "sourceId": source_id, + "sourceVersion": source_version, + "sourceOffset": 0, + "sourceRefs": _structured_source_refs(payload, lineage), + "milestones": _milestones(payload), + "sourceKind": "canonical_entity" if src == "entity" else "draft", + "sourceStatus": source_status or status, + "bindingStatus": binding_status, + "productionRetrievalEligible": src == "entity" and status == "active" and binding_status == "active", + } + ) + return results + + @click.command() @click.argument("intent") @click.option("--scope", type=click.Choice(["admin", "work"]), default="admin", show_default=True, @@ -49,55 +184,32 @@ def visible(ai_rule, purpose): @click.option("--top", default=5, show_default=True) @click.option("--json", "as_json", is_flag=True) def main(intent, scope, work_id, ttype, purpose, top, as_json): - if scope == "work" and not work_id: - raise click.ClickException("--scope work 必须带 --work-id(授权过滤依赖绑定关系)") + try: + cards = search_cards( + intent, + scope=scope, + work_id=work_id, + ttype=ttype, + purpose=purpose, + top=top, + ) + except ValueError as error: + raise click.ClickException(str(error)) from error - vecs, bad = embed_texts(_session(), [intent]) - if bad: - raise click.ClickException("查询嵌入失败") - qvec = json.dumps(vecs[0]) - - with psycopg.connect(DSN) as conn: - ai_rules = load_ai_context(conn) - if scope == "admin": - # 治理面:draft(pending/confirmed)+entity 全量 - sql = """SELECT 'draft' AS src, d.id, d.draft_payload AS payload, d.status, - 1 - (e.embedding <=> %s::vector) AS score - FROM example_knowledge_embedding e - JOIN muse_knowledge_draft d ON d.id = e.draft_id - WHERE e.tenant_id=%s AND e.deleted=FALSE AND d.deleted=FALSE""" - args = [qvec, TENANT] - else: - # 作品面:仅 active entity 且其 kb 已绑定到该作品(授权在查询层强制) - sql = """SELECT 'entity' AS src, en.id, - jsonb_build_object('型', en.entity_type, '名称', en.normalized_name, - '一句话摘要', en.description, '字段', en.attributes) AS payload, - en.status, 1 - (e.embedding <=> %s::vector) AS score - FROM example_knowledge_embedding e - JOIN muse_knowledge_entity en ON en.id = e.entity_id - JOIN muse_knowledge_binding b ON b.kb_id = en.kb_id AND b.work_id = %s - AND b.binding_status='active' AND b.deleted=FALSE AND b.tenant_id=%s - WHERE e.tenant_id=%s AND e.deleted=FALSE AND en.deleted=FALSE AND en.status='active'""" - args = [qvec, work_id, TENANT, TENANT] - if ttype: - sql += (" AND d.draft_payload->>'型' = %s" if scope == "admin" - else " AND en.entity_type = %s") - args.append(ttype) - sql += " ORDER BY score DESC LIMIT %s" - args.append(top) - rows = conn.execute(sql, args).fetchall() - - results = [] - for src, rid, payload, status, score in rows: - p = payload or {} - t = p.get("型", "?") - rules = ai_rules.get(t, {}) - fields = p.get("字段") or {} - vis = {k: v for k, v in fields.items() if visible(rules.get(k), purpose)} - cut = sorted(set(fields) - set(vis)) - results.append({"来源": f"{src}#{rid}", "型": t, "名称": p.get("名称"), "状态": status, - "相似度": round(float(score), 4), "一句话摘要": p.get("一句话摘要"), - "可见字段": vis, "出处": p.get("出处"), "裁剪回显": cut}) + results = [ + { + "来源": item["sourceId"], + "型": item["type"], + "名称": item["name"], + "状态": item["sourceStatus"], + "相似度": round(item["score"], 4), + "一句话摘要": item["summary"], + "可见字段": item["visibleFields"], + "出处": item["sourceRefs"], + "裁剪回显": item["omittedFields"], + } + for item in cards + ] if as_json: click.echo(json.dumps(results, ensure_ascii=False, indent=1)) diff --git a/docs/superpowers/plans/2026-07-20-writer-agent-v1.md b/docs/superpowers/plans/2026-07-20-writer-agent-v1.md index 3ed26c3..78a3416 100644 --- a/docs/superpowers/plans/2026-07-20-writer-agent-v1.md +++ b/docs/superpowers/plans/2026-07-20-writer-agent-v1.md @@ -1,6 +1,6 @@ # 正文智能体实验台 v1 Implementation Plan -> **For agentic workers:** REQUIRED SUB-SKILL: Use `superpowers:subagent-driven-development` (recommended) or `superpowers:executing-plans` to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. +> **执行约束(创始人 2026-07-20 确认):** 不使用 worktree,不使用 superpower。执行类任务由子代理直接在当前 `main` 工作树实现,主代理负责文件边界、代码审查与机械验证;不得暂存或覆盖用户已有改动。 **Goal:** 在参考作品回放环境中实现一套可机械验证的正文智能体:知识卡只承担索引职责,智能体必须顺着卡片证据回读冻结线内的历史原文,再依据大纲、细纲和事实证据生成正文,并经过审查、盲评及 Gate A/B 验收。 @@ -19,32 +19,30 @@ - `抽取卡 -> 原文` 是强制链路:抽取卡若没有可追踪的历史原文来源,不得单独作为正文硬事实。作者已确认的正式设定、Canonical 状态与细纲声明的新事实可以直接成为事实证据,并标为 `declared_new` 或相应来源类型。 - “完成”至少包含:测试通过、dry-run 通过、真实回放所需配置齐全;没有真实模型回放结果时只能称“实现完成”,不能称 Gate A/B 通过。 -## 环境准备:固定 worktree 本地依赖 +## 环境准备:当前主工作树本地依赖 -- [ ] 在内层 worktree 创建独立虚拟环境,不复用用户主工作树的 `.venv`: +- [ ] 在 `agent-example` 当前 `main` 工作树复用已安装依赖的 `.venv`,先确认解释器版本与依赖可用: ```bash -cd /private/tmp/agent-example-writer-v1 -uv venv --python 3.12 .venv -uv pip install --python .venv/bin/python -r requirements.txt +cd /Users/qingse/Sync/local-git/oh-my-muse/agent-example .venv/bin/python --version ``` -- [ ] 后续所有 Python 命令都在 `/private/tmp/agent-example-writer-v1` 执行并使用 `.venv/bin/python`;若 `requirements.txt` 变化,先重装依赖再验证。 +- [ ] 后续所有 Python 命令都在当前 `agent-example` 主工作树执行并使用 `.venv/bin/python`;若 `requirements.txt` 变化,先重装依赖再验证。 ## 任务 0:回写稳定设计 SoT,标明实验边界 -> 本任务在已建立的外层独立 worktree `/private/tmp/oh-my-muse-writer-sot`(分支 `feature/writer-agent-sot`)执行并独立提交;后续任务均在当前 `agent-example` worktree `/private/tmp/agent-example-writer-v1` 执行。禁止在用户有未提交改动的外层主工作树直接编辑或提交。 +> 本任务直接在外层仓库 `/Users/qingse/Sync/local-git/oh-my-muse` 的当前 `main` 工作树执行。只提交下列精确路径,保留外层和内层工作树中的用户已有改动。 **Files:** -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/专题-01-正文建议接受(Accept Suggestion)实现规范.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/专题-03-AI编排上下文与质量评测实现规范.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/专题-04-生成质量门控与创作健康度设计方案.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/专题-05-AI统一交互协议与外部AgentAdapter设计.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/专题-06-元数据驱动的智能体架构.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/专题-07-知识消费契约与质量闭环.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/design-docs/架构-04-状态机与约束清单.md` -- Modify: `/private/tmp/oh-my-muse-writer-sot/.agents/workflows/ai-development-protocol.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/专题-01-正文建议接受(Accept Suggestion)实现规范.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/专题-03-AI编排上下文与质量评测实现规范.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/专题-04-生成质量门控与创作健康度设计方案.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/专题-05-AI统一交互协议与外部AgentAdapter设计.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/专题-06-元数据驱动的智能体架构.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/专题-07-知识消费契约与质量闭环.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/design-docs/架构-04-状态机与约束清单.md` +- Modify: `/Users/qingse/Sync/local-git/oh-my-muse/.agents/workflows/ai-development-protocol.md` - [ ] 在专题-07 中把“卡是索引,根据卡回读原文”设为正文消费契约,明确卡不能替代原文。 - [ ] 在专题-03 中补齐 `WriterContext v1`、`WriterOutput v1`、`RetrievalManifest` 和冻结语义。 @@ -59,7 +57,7 @@ uv pip install --python .venv/bin/python -r requirements.txt - [ ] 运行文档一致性检查: ```bash -cd /private/tmp/oh-my-muse-writer-sot +cd /Users/qingse/Sync/local-git/oh-my-muse rg -n "卡是索引|WriterContext v1|acceptanceEligible|Gate A|Gate B|Canonical" \ 'design-docs/专题-01-正文建议接受(Accept Suggestion)实现规范.md' \ design-docs/专题-03-AI编排上下文与质量评测实现规范.md \ @@ -74,7 +72,7 @@ rg -n "卡是索引|WriterContext v1|acceptanceEligible|Gate A|Gate B|Canonical" - [ ] Commit: ```bash -git -C /private/tmp/oh-my-muse-writer-sot add \ +git -C /Users/qingse/Sync/local-git/oh-my-muse add \ 'design-docs/专题-01-正文建议接受(Accept Suggestion)实现规范.md' \ design-docs/专题-03-AI编排上下文与质量评测实现规范.md \ design-docs/专题-04-生成质量门控与创作健康度设计方案.md \ @@ -83,7 +81,7 @@ git -C /private/tmp/oh-my-muse-writer-sot add \ design-docs/专题-07-知识消费契约与质量闭环.md \ design-docs/架构-04-状态机与约束清单.md \ .agents/workflows/ai-development-protocol.md -git -C /private/tmp/oh-my-muse-writer-sot commit \ +git -C /Users/qingse/Sync/local-git/oh-my-muse commit \ -m "设计: 固化正文智能体卡索引原文回读契约" ``` diff --git a/meta/schemas/README.md b/meta/schemas/README.md index 2506196..b25c6b6 100644 --- a/meta/schemas/README.md +++ b/meta/schemas/README.md @@ -7,7 +7,7 @@ - **两轴**:`domain` ∈ content / world / narrative / knowledge / ai_context;`scope` ∈ work / chapter / block / entity / relation / event / agent。 - **aiContext 控制项**(阶段一只用这一个):`true` 任何用途都可入 AI 上下文;`false` 一律不入;`[用途…]` 仅列出的用途可入。用途取值:`planning` / `generation` / `detection` / `extraction`。其余控制项(uiVisible/userEditable 等)阶段二随真后端启用。 - **基础字段**(所有型共有,各 schema 不再重复):`名称`、`别名`、`一句话摘要`、`标签`、`来源`(手工 / 抽取@第N章 / 拆书@书名)、`状态`(草稿 / 已确认)。 -- **状态**:`启用`(21 型)/ `待启用`(2 型:pacing、generation_context,首个用到的场景来临时再启用)。范式五型与参考书档案已于拆书场景(A8)启用并补全字段合同。 +- **状态**:23 型均已启用。`generation_context` 已于正文实验台启用;范式五型与参考书档案已于拆书场景(A8)启用并补全字段合同。 - **演进**:增删型或字段先过专题-06 §4.4 的四判据与降级规则;变更靠 git 追溯。 ## 实例落点表(哪个型的实例长在哪) @@ -31,8 +31,8 @@ | event | `知识/事件/*.md` | | reference_work | `knowledge/参考书/*/档案.md`(原文 txt 同目录) | | craft / combat / emotion / scene_pattern / trope | 公共面 `knowledge/范式/{技法,打斗,情感,通用桥段,套路}/`;作品面 `知识/` 对应子目录 | -| generation_context(待启用) | 阶段一以上下文回显形式落 `works/*/评审/`,不建实例文件 | -| pacing(待启用) | 卷中期节奏审计时启用 | +| generation_context | 阶段一以严格 JSON 上下文与 Markdown manifest 回显落 `works/*/评审/`,不建实例文件 | +| pacing | 卷中期节奏审计时使用 | 知识卡「值得立卡」的门槛:有跨章戏份或跨章履约;一次性龙套与单场景道具不立卡,写在章内即可。 diff --git a/meta/schemas/generation_context.yaml b/meta/schemas/generation_context.yaml index 107056e..ab7fd16 100644 --- a/meta/schemas/generation_context.yaml +++ b/meta/schemas/generation_context.yaml @@ -3,7 +3,63 @@ target_type: generation_context domain: ai_context scope: agent 本体分组: 配套 -状态: 待启用 +状态: 启用 判据: AI 上下文组装与输出合同的结构模具,运行时对象、用户不可见 -阶段一落法: 以 read-context 的上下文回显(works/*/评审/上下文-*.md)代替实例,不建卡 -启用条件: 阶段二对齐统一读取器(专题-06 §7)的输出合同时正式启用 +实例落点: 以 read-context 的 JSON 上下文与 Markdown manifest 回显落 works/*/评审/,不建知识卡 +严格合同: + 版本: + WriterContext: writer-context-v1 + WriterOutput: writer-output-v1 + RetrievalPlan: writer-retrieval-plan-v1 + RetrievalManifest: writer-retrieval-manifest-v1 + 规范化: UTF-8、Unicode NFC、LF 换行、对象键排序、无额外空白 + 身份规则: runId、generatedAt、timestamp、executionNode 不参与检索与上下文身份 + 未知字段: 拒绝 + WriterContext必填字段: + - schemaVersion + - runId + - attempt + - mode + - purpose + - qualityPolicyVersion + - workId + - targetChapter + - asOf + - contextSnapshot + - sourceVersion + - authorizationSnapshot + - sourceStatus + - retrievalPlan + - retrievalManifest + - fineOutline + - narrativeState + - factEvidence + - proseEvidence + - patternReferences + - evidenceCoverage + - outputContract + - tokenBudget + - omittedSources + - acceptanceEligible + WriterOutput必填字段: + - schemaVersion + - runId + - attempt + - mode + - qualityPolicyVersion + - contextSnapshotId + - contextSnapshotSha256 + - candidateVersion + - candidateSha256 + - acceptanceEligible + - candidateBody + - claimLedger + - evidenceRequests + - newSettingDeclarations + - selfCheck + 双证据: + factEvidence来源: [historical_prose, formal_setting, canonical_state, fine_outline_declared_new] + proseEvidence来源: 历史 Canonical 原文,必须带章号、块、字符区间与内容哈希 + 接受边界: evaluation、diagnostic、diagnostic_only 一律 acceptanceEligible=false + 稳定排序: score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC +实现校验器: .claude/skills/read-context/scripts/writer_contract.py diff --git a/meta/schemas/outline.yaml b/meta/schemas/outline.yaml index 7204925..1627ddb 100644 --- a/meta/schemas/outline.yaml +++ b/meta/schemas/outline.yaml @@ -13,8 +13,10 @@ scope: work - key: 近三章细纲 说明: 每章:章目标/关键事件/出场角色/伏笔动作(埋·推·收)/章末钩子——续写的直接依据;章细纲字数≈章正文 3–5%(3000 字章→100–150 字);细纲是结构骨架不是缩写,超比例=退回重做 aiContext: true + - { key: targetChars, 说明: 可选的本章目标汉字数;仍受 2000–10000 硬边界约束, aiContext: [generation] } - { key: 未来卷粗纲, 说明: 防续写提前收线,仅规划可见, aiContext: [planning] } - { key: 弃案记录, 说明: 改掉的旧方向,防被 AI 复活, aiContext: false } 设计发现: - 2026-07-09 全书解析设计——存量作品需**规划逆向**(全文→章细纲→卷粗纲→主线,自底向上,与创作期自顶向下互为镜像);SoT 产品-03 §3.7 全书解析产出未含大纲/细纲,「细纲」粒度层级亦为本仓先行,均待回填 - 同日——「细纲/正文字数比例」约束防解析退化成压缩复述,**创始人已拍板@2026-07-09**(章细纲 3–5%、卷粗纲 0.3–0.5%、主线 ≤50 字),已写入上方字段说明;D1 回填 design-docs 时随字段合同一并带走 + - 2026-07-20 正文实验台——新增可选 targetChars;未提供时只依据目标章之前的有效 Canonical 章长中位数与细纲密度确定性计算