zizi b0bc7a8745 框架: 技能按动作-对象重组 + 先审后入创作闭环
一、技能重组(动作-对象命名)
- 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为
  clean-book-text/decide-candidate/write-next-chapter/access-database/
  check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持)
- agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名

二、先审后入创作闭环(本次核心)
正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交",
DB 级兜底,编排层跳步即被硬拒。
- candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链
- fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量,
  模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本
- projection_registry.py + example_projection_run(107):投影登记与恢复
- acceptance_state.py:接受前置实时状态重读
- lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格
- DDL 105:example_candidate 增 semantic_status/semantic_report_sha256
- write_canonical.accept:语义兜底+同事务合并增量+登记投影;
  run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链
- claude_runtime:兼容新 CLI modelUsage 信息字段

三、审查修复(独立子代理四维审查后)
- 事实增量 propose→approve 翻态正道,不撞唯一键
- 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记
- 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理

测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。
创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
2026-08-14 10:24:08 +08:00

582 lines
24 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""正文写手输入输出合同的确定性与失败关闭测试。"""
from __future__ import annotations
import copy
import hashlib
import pathlib
import sys
import unittest
sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent))
from writer_contract import ( # noqa: E402
ContractError,
build_candidate_envelope,
build_writer_creative_input,
calculate_target_chars,
canonical_json,
han_count,
normalize_text,
retrieval_identity,
validate_writer_context,
validate_writer_draft,
validate_writer_output,
)
def valid_context() -> dict:
"""构造覆盖全部必填字段的最小合法上下文。"""
prose_evidence = []
for chapter in range(485, 489):
text = f"第{chapter}章冻结原文。"
prose_evidence.append(
{
"evidenceId": f"prose:{chapter}",
"chapter": chapter,
"sourceRef": {
"sourceId": f"chapter:{chapter}",
"sourceVersion": f"chapter-{chapter}-v1",
"chapter": chapter,
"blockId": 1,
"startCodePoint": 0,
"endCodePoint": len(text),
},
"contentSha256": "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest(),
"purpose": "recent_full_chapter",
"text": text,
"isRecentBaseline": True,
}
)
context = {
"schemaVersion": "writer-context-v1",
"runId": "writer-run-001",
"attempt": 1,
"mode": "production",
"purpose": "production",
"qualityPolicyVersion": "writer-production-v1",
"workId": 8,
"targetChapter": 489,
"asOf": 488,
"contextSnapshot": {
"manifestId": "sha256:" + "1" * 64,
"contextSha256": "sha256:" + "2" * 64,
"generatedAt": "2026-07-20T00:00:00Z",
},
"sourceVersion": "raw-file-v1:sha256:" + "3" * 64,
"authorizationSnapshot": {
"snapshotId": "auth-001",
"allowedPurpose": "generation",
"verifiedAt": "2026-07-20T00:00:00Z",
},
"sourceStatus": "active",
"retrievalPlan": {
"planVersion": "writer-retrieval-plan-v1",
"planId": "sha256:" + "4" * 64,
"runId": "writer-run-001",
"asOf": 488,
"queries": [],
"cardIndexVersion": "card-v1",
"proseIndexVersion": "prose-v1",
"filters": {
"workId": 8,
"asOfChapter": 488,
"sourceStatus": "active",
"authorizationRequired": True,
},
"tieBreak": "score DESC, sourceVersion ASC, sourceId ASC, sourceOffset ASC",
"tokenBudget": {"maxContextChars": 20000},
},
"retrievalManifest": {
"manifestVersion": "writer-retrieval-manifest-v1",
"manifestId": "sha256:" + "1" * 64,
"planId": "sha256:" + "4" * 64,
"sources": [],
"omittedSources": [],
},
"fineOutline": {
"sourceRef": {
"sourceId": "fine-outline:489",
"sourceVersion": "fine-outline-v3",
"chapter": 489,
},
"hardConstraints": ["必须完成围攻突围"],
"adjustableBeats": [],
"declaredNewFacts": [],
},
"narrativeState": {
"time": "围攻当日",
"location": "圣蒂曼",
"characterPositions": {},
"immediateSituation": "战斗持续",
},
"factEvidence": [],
"proseEvidence": prose_evidence,
"patternReferences": [],
"evidenceCoverage": [],
"outputContract": {
"targetChars": 4000,
"minChars": 3600,
"maxChars": 4400,
"frontmatterRequired": False,
},
"tokenBudget": {"maxContextChars": 20000, "usedContextChars": 0},
"omittedSources": [],
"acceptanceEligible": True,
}
plan_payload = {key: value for key, value in context["retrievalPlan"].items() if key != "planId"}
context["retrievalPlan"]["planId"] = retrieval_identity(plan_payload)
context["retrievalManifest"]["planId"] = context["retrievalPlan"]["planId"]
manifest_payload = {key: value for key, value in context["retrievalManifest"].items() if key != "manifestId"}
context["retrievalManifest"]["manifestId"] = retrieval_identity(manifest_payload)
context["contextSnapshot"]["manifestId"] = context["retrievalManifest"]["manifestId"]
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
return context
def valid_draft(*, body: str = "第一段正文。") -> dict:
"""构造 writer 唯一允许的创作输出。"""
return {"candidateBody": body}
def valid_output(context: dict | None = None, *, body: str = "第一段正文。") -> dict:
"""构造由可信 adapter 绑定的候选 envelope。"""
return build_candidate_envelope(context or valid_context(), valid_draft(body=body))
def _resign(context: dict) -> None:
"""改动上下文字段后重算计划/清单/上下文三级身份哈希。"""
plan_payload = {key: value for key, value in context["retrievalPlan"].items() if key != "planId"}
context["retrievalPlan"]["planId"] = retrieval_identity(plan_payload)
context["retrievalManifest"]["planId"] = context["retrievalPlan"]["planId"]
manifest_payload = {key: value for key, value in context["retrievalManifest"].items() if key != "manifestId"}
context["retrievalManifest"]["manifestId"] = retrieval_identity(manifest_payload)
context["contextSnapshot"]["manifestId"] = context["retrievalManifest"]["manifestId"]
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
def first_chapter_context() -> dict:
"""开篇基线章上下文:asOf=0(开篇前冻结线),无历史正文,证据只有设定/大纲/细纲。"""
context = valid_context()
context["targetChapter"] = 1
context["asOf"] = 0
context["retrievalPlan"]["asOf"] = 0
context["retrievalPlan"]["filters"]["asOfChapter"] = 0
context["proseEvidence"] = []
_resign(context)
return context
class WriterContractTest(unittest.TestCase):
def test_first_chapter_context_allows_asof_zero(self):
"""开篇基线章:asOf=0 是合法冻结线,正文基线为空。"""
normalized = validate_writer_context(first_chapter_context())
self.assertEqual(normalized["asOf"], 0)
self.assertEqual(normalized["proseEvidence"], [])
def test_asof_zero_rejects_any_historical_prose(self):
"""asOf=0 冻结线下,连第 1 章正文都属于未来正文,必须失败关闭。"""
context = first_chapter_context()
text = "第一章正文。"
context["proseEvidence"] = [{
"evidenceId": "prose:1", "chapter": 1,
"sourceRef": {"sourceId": "chapter:1", "sourceVersion": "chapter-1-v1",
"chapter": 1, "blockId": 1, "startCodePoint": 0,
"endCodePoint": len(text)},
"contentSha256": "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest(),
"purpose": "recent_full_chapter", "text": text, "isRecentBaseline": True,
}]
_resign(context)
with self.assertRaises(ContractError) as raised:
validate_writer_context(context)
self.assertIn("超出冻结线", str(raised.exception))
def test_index_hints_are_strict_diagnostic_only_and_frozen(self):
"""诊断卡索引提示只能进入不可接受的诊断上下文。"""
hint = {
"cardId": "card-1",
"name": "甲",
"type": "character",
"content": "甲曾在城门出现",
"sourceId": "upgrade-book:card-1",
"sourceVersion": "upgrade-book-v1",
"asOf": 488,
}
context = valid_context()
context.update(
{
"mode": "diagnostic_only",
"purpose": "evaluation",
"qualityPolicyVersion": "writer-eval-v1",
"acceptanceEligible": False,
"evidenceStrategy": "card_index_only",
"indexHints": [hint],
}
)
context["proseEvidence"] = []
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
validate_writer_context(context)
for mutation in ("production", "unknown_field", "future"):
invalid = copy.deepcopy(context)
if mutation == "production":
invalid.update({"mode": "production", "purpose": "production", "acceptanceEligible": True})
elif mutation == "unknown_field":
invalid["indexHints"][0]["reason"] = "不得透传"
else:
invalid["indexHints"][0]["asOf"] = 489
invalid["contextSnapshot"]["contextSha256"] = retrieval_identity(invalid)
with self.subTest(mutation=mutation), self.assertRaises(ContractError):
validate_writer_context(invalid)
def test_card_index_only_requires_explicit_strategy_and_no_fact_or_prose(self):
"""B 臂跳过四章基线必须留在合同字段中,不能隐式旁路。"""
context = valid_context()
context.update(
{
"mode": "diagnostic_only",
"purpose": "diagnostic",
"qualityPolicyVersion": "writer-eval-v1",
"acceptanceEligible": False,
"evidenceStrategy": "card_index_only",
"indexHints": [{
"cardId": "card-1", "name": "甲", "type": "character",
"content": "甲的冻结状态", "sourceId": "card-source:1",
"sourceVersion": "cards-v1", "asOf": 488,
}],
}
)
context["proseEvidence"] = []
context["factEvidence"] = [{
"evidenceId": "fact-1", "fact": "甲在场", "sourceType": "formal_setting",
"sourceRef": {"sourceId": "setting:1", "sourceVersion": "setting-v1"},
"contentSha256": "sha256:" + hashlib.sha256("甲在场".encode()).hexdigest(),
"riskLevel": "high",
}]
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
with self.assertRaises(ContractError):
validate_writer_context(context)
def test_diagnostic_prose_strategies_are_explicit_and_keep_baseline(self):
"""A/C 臂保留连续历史原文,C 臂只额外加入诊断卡索引提示。"""
hint = {
"cardId": "card-1", "name": "甲", "type": "character",
"content": "甲的冻结状态", "sourceId": "card-source:1",
"sourceVersion": "cards-v1", "asOf": 488,
}
for strategy, hints in (
("historical_prose_only", []),
("card_index_plus_prose", [hint]),
):
context = valid_context()
context.update(
{
"mode": "diagnostic_only",
"purpose": "evaluation",
"qualityPolicyVersion": "writer-eval-v1",
"acceptanceEligible": False,
"evidenceStrategy": strategy,
"indexHints": hints,
}
)
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
with self.subTest(strategy=strategy):
self.assertEqual(
[item["chapter"] for item in validate_writer_context(context)["proseEvidence"]],
[485, 486, 487, 488],
)
def test_production_cannot_skip_or_break_recent_baseline(self):
"""生产上下文不能借诊断策略或不连续近章绕过四章基线。"""
for mutation in ("diagnostic_strategy", "missing_baseline"):
context = valid_context()
if mutation == "diagnostic_strategy":
context["evidenceStrategy"] = "historical_prose_only"
else:
context["proseEvidence"].pop(1)
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
with self.subTest(mutation=mutation), self.assertRaises(ContractError):
validate_writer_context(context)
def test_run_id_does_not_change_retrieval_identity(self):
first = {"runId": "run-a", "query": "咖啡\u0301", "nested": {"value": 1}}
second = {"runId": "run-b", "query": "咖啡\u0301", "nested": {"value": 1}}
self.assertEqual(retrieval_identity(first), retrieval_identity(second))
def test_text_is_nfc_and_lf_before_offsets_and_hash(self):
decomposed = "Cafe\u0301\r\n第二行\r第三行"
normalized = "Caf\u00e9\n第二行\n第三行"
self.assertEqual(normalize_text(decomposed), normalized)
self.assertEqual(canonical_json({"text": decomposed}), canonical_json({"text": normalized}))
def test_han_count_does_not_count_markdown_or_non_han_text(self):
self.assertEqual(han_count("# **正文** 123 ABC,扩展𠀀"), 5)
def test_target_chars_use_half_up_and_hard_bounds(self):
self.assertEqual(
calculate_target_chars(
recent_chapter_han_counts=[2501, 2502, 2503, 2504],
hard_event_count=3,
foreshadowing_action_count=0,
required_scene_count=0,
),
2100,
)
self.assertEqual(calculate_target_chars(explicit_target_chars=1500), 2000)
self.assertEqual(calculate_target_chars(explicit_target_chars=11000), 10000)
self.assertEqual(
calculate_target_chars(
recent_chapter_han_counts=[3000, 3000, 3000],
hard_event_count=0,
foreshadowing_action_count=0,
required_scene_count=0,
),
2600,
)
def test_evaluation_and_diagnostic_contexts_are_never_acceptable(self):
for purpose in ("evaluation", "diagnostic"):
context = valid_context()
context["mode"] = "diagnostic_only"
context["purpose"] = purpose
context["qualityPolicyVersion"] = "writer-eval-v1"
context["acceptanceEligible"] = True
with self.assertRaises(ContractError):
validate_writer_context(context)
context["acceptanceEligible"] = False
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
validate_writer_context(context)
def test_unknown_missing_and_wrong_version_fail_closed(self):
for mutation in ("unknown", "missing", "version"):
context = copy.deepcopy(valid_context())
if mutation == "unknown":
context["unexpected"] = True
elif mutation == "missing":
del context["fineOutline"]
else:
context["schemaVersion"] = "writer-context-v2"
with self.subTest(mutation=mutation), self.assertRaises(ContractError):
validate_writer_context(context)
output = valid_output()
output["unexpected"] = True
with self.assertRaises(ContractError):
validate_writer_output(output)
def test_writer_creative_input_excludes_orchestration_fields(self):
context = valid_context()
context["factEvidence"] = [{
"evidenceId": "fact-1",
"fact": "林澈仍在圣蒂曼城内",
"sourceType": "canonical_state",
"sourceRef": {"sourceId": "state:8", "sourceVersion": "state-v1"},
"contentSha256": "sha256:" + "a" * 64,
"riskLevel": "high",
}]
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
creative_input = build_writer_creative_input(context)
self.assertEqual(
set(creative_input),
{
"fineOutline",
"narrativeState",
"factConstraints",
"proseExcerpts",
"patternReferences",
"lengthContract",
"styleConstraints",
},
)
serialized = canonical_json(creative_input)
for forbidden in (
"runId", "contextSnapshot", "authorizationSnapshot", "retrievalPlan",
"retrievalManifest", "candidateVersion", "acceptanceEligible",
"sourceId", "sourceVersion", "contentSha256", "indexHints",
):
self.assertNotIn(forbidden, serialized)
self.assertEqual(creative_input["factConstraints"][0]["text"], "林澈仍在圣蒂曼城内")
self.assertEqual([item["chapter"] for item in creative_input["proseExcerpts"]], [485, 486, 487, 488])
def test_writer_draft_only_allows_nonempty_candidate_body(self):
self.assertEqual(validate_writer_draft(valid_draft()), valid_draft())
for invalid in (
{"candidateBody": ""},
{"candidateBody": "正文", "candidateSha256": "sha256:" + "a" * 64},
{"candidateBody": "正文", "claimLedger": []},
):
with self.subTest(invalid=invalid), self.assertRaises(ContractError):
validate_writer_draft(invalid)
def test_candidate_adapter_normalizes_body_and_binds_hash(self):
context = valid_context()
output = build_candidate_envelope(
context,
valid_draft(body="Cafe\u0301\r\n正文"),
candidate_version=3,
)
self.assertEqual(output["schemaVersion"], "candidate-envelope-v2")
self.assertEqual(output["candidateBody"], "Café\n正文")
self.assertEqual(output["candidateVersion"], 3)
self.assertEqual(
output["candidateSha256"],
"sha256:" + hashlib.sha256("Café\n正文".encode("utf-8")).hexdigest(),
)
validate_writer_output(output)
output["candidateBody"] = "被修改的正文"
with self.assertRaises(ContractError):
validate_writer_output(output)
def test_candidate_adapter_forces_diagnostic_acceptance_false(self):
context = valid_context()
context.update({
"mode": "diagnostic_only",
"purpose": "evaluation",
"qualityPolicyVersion": "writer-eval-v1",
"acceptanceEligible": False,
"evidenceStrategy": "historical_prose_only",
})
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
output = build_candidate_envelope(context, valid_draft())
self.assertFalse(output["acceptanceEligible"])
forged = copy.deepcopy(output)
forged["acceptanceEligible"] = True
with self.assertRaises(ContractError):
validate_writer_output(forged)
def test_plan_manifest_and_context_identity_tampering_fails_closed(self):
for path in ("plan", "manifest", "context"):
context = valid_context()
if path == "plan":
context["retrievalPlan"]["cardIndexVersion"] = "tampered"
elif path == "manifest":
context["retrievalManifest"]["sources"].append(
{"sourceId": "setting:1", "sourceVersion": "setting-v1"}
)
else:
context["fineOutline"]["hardConstraints"].append("被篡改的约束")
with self.subTest(path=path), self.assertRaises(ContractError):
validate_writer_context(context)
class PatternReferenceContractTest(unittest.TestCase):
"""SoT 变更:patternReferences 携带范式卡内容(名字/摘要/写法要点)并投影给写手。"""
def _context_with_pattern(self, references: list) -> dict:
context = valid_context()
context["patternReferences"] = references
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
return context
def test_content_fields_pass_and_project_to_writer(self):
"""内容字段过合同,且写手创作输入真正读到名字/摘要/写法要点。"""
context = self._context_with_pattern(
[
{
"sourceId": "draft:combat-1",
"sourceVersion": "draft-revision:2",
"sourceType": "combat",
"name": "三段式逆转",
"summary": "先压后扬再反转。",
"writingPoints": {"节拍": "压制—喘息—反杀", "钩子": "身份揭破"},
}
]
)
validate_writer_context(context)
creative = build_writer_creative_input(context)
self.assertEqual(
creative["patternReferences"],
[
{
"referenceId": "pattern-1",
"kind": "combat",
"name": "三段式逆转",
"summary": "先压后扬再反转。",
"writingPoints": {"节拍": "压制—喘息—反杀", "钩子": "身份揭破"},
}
],
)
# 来源指针只供审计回读,绝不允许泄进写手创作输入。
serialized = canonical_json(creative)
self.assertNotIn("sourceId", serialized)
self.assertNotIn("sourceVersion", serialized)
def test_content_fields_optional(self):
"""内容字段可选:只带来源指针(历史形状)仍合法,投影只给标签。"""
context = self._context_with_pattern(
[{"sourceId": "draft:trope-1", "sourceVersion": "draft-revision:1", "sourceType": "trope"}]
)
validate_writer_context(context)
creative = build_writer_creative_input(context)
self.assertEqual(
creative["patternReferences"],
[{"referenceId": "pattern-1", "kind": "trope"}],
)
def test_content_oversize_or_unknown_field_fails_closed(self):
"""超量内容、超字段数、未知字段或非对象写法要点,合同一律失败关闭。"""
base = {
"sourceId": "draft:craft-1",
"sourceVersion": "draft-revision:1",
"sourceType": "craft",
}
mutations = {
"name_overlong": {**base, "name": "超长名字" * 20},
"summary_overlong": {**base, "summary": "超长摘要" * 60},
"point_value_overlong": {**base, "writingPoints": {"节拍": "超长写法要点" * 100}},
"too_many_points": {**base, "writingPoints": {f"字段{i}": "要点" for i in range(7)}},
"unknown_field": {**base, "score": 0.9},
"writing_points_not_object": {**base, "writingPoints": ["不是对象"]},
}
for label, ref in mutations.items():
with self.subTest(mutation=label), self.assertRaises(ContractError):
validate_writer_context(self._context_with_pattern([ref]))
def test_strict_source_pointer_unaffected_by_relaxation(self):
"""放宽只针对 patternReferences:事实证据的来源指针夹带 name 仍被拒收。"""
context = valid_context()
context["factEvidence"] = [
{
"evidenceId": "fact-1",
"fact": "林澈仍在圣蒂曼城内",
"sourceType": "canonical_state",
"sourceRef": {
"sourceId": "state:8",
"sourceVersion": "state-v1",
# 内容字段只允许出现在 patternReferences;夹带到其它来源指针必须被拒。
"name": "夹带的名字",
},
"contentSha256": "sha256:" + "a" * 64,
"riskLevel": "high",
}
]
context["contextSnapshot"]["contextSha256"] = retrieval_identity(context)
with self.assertRaises(ContractError):
validate_writer_context(context)
if __name__ == "__main__":
unittest.main()