将角色与 Skill 从 .claude 迁入 .agent,移除 Claude CLI 运行时并接入固定 Opus 角色 profile、完整 schema、预算 deadline、raw 与回执证据链。 同步拆分 Skill 职责、复利 lesson、Gate 回放、Dashboard 人审入口、数据库登记和机械门禁;候选设计正文不包含在本提交中。
295 lines
11 KiB
Python
295 lines
11 KiB
Python
#!/usr/bin/env python3
|
||
"""生产链补证重组装:语义 needs_evidence 时检索可证材料并冻结新 WriterContext。
|
||
|
||
合同(run_writer_pipeline):
|
||
evidence_provider(context, gaps, next_attempt) → 同 runId、新 attempt、
|
||
新 contextSha256、且 WriterCreativeInput 必须变化。
|
||
|
||
策略(最小可接线):
|
||
1. 按缺口 query 在本作品 as_of 内正文 / 知识实体描述做关键词检索;
|
||
2. 命中则追加 factEvidence(sourceType=canonical_state 或 historical_prose);
|
||
3. 命中时追加「补证轮」styleConstraints,约束下一稿用已给摘录支撑相关断言;
|
||
4. 零命中不伪造设定、不禁止写手发明新设定(那是提案,交人闸);只记补证轮元数据;
|
||
5. 重算 contextSha256(retrieval_identity)。
|
||
|
||
编排层看到零新增 factEvidence 时把缺口当新设定提案,当前候选进人闸,不因零命中重写。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import hashlib
|
||
import re
|
||
from typing import Any, Mapping, Sequence
|
||
|
||
from muse_db import connect
|
||
from writer_contract import ( # noqa: E402
|
||
ContractError,
|
||
retrieval_identity,
|
||
validate_writer_context,
|
||
)
|
||
|
||
|
||
class EvidenceReassembleError(Exception):
|
||
"""补证失败关闭:不得静默返回旧上下文。"""
|
||
|
||
|
||
_SPLIT = re.compile(r"[,。;、:!?\s/()()「」《》\[\]【】]+")
|
||
|
||
|
||
def _sha_text(text: str) -> str:
|
||
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||
|
||
|
||
def _query_terms(query: str, *, max_terms: int = 6) -> list[str]:
|
||
"""从缺口 query 抽出可检索片段(长度≥2)。"""
|
||
|
||
parts = [p.strip() for p in _SPLIT.split(query or "") if p and p.strip()]
|
||
terms: list[str] = []
|
||
for part in parts:
|
||
if len(part) < 2:
|
||
continue
|
||
if part not in terms:
|
||
terms.append(part)
|
||
if len(terms) >= max_terms:
|
||
break
|
||
if not terms and query and len(query.strip()) >= 2:
|
||
terms.append(query.strip()[:24])
|
||
return terms
|
||
|
||
|
||
def _search_canonical_snippets(
|
||
*,
|
||
work_id: int,
|
||
as_of: int,
|
||
terms: Sequence[str],
|
||
limit: int = 3,
|
||
) -> list[dict[str, Any]]:
|
||
"""在已确认正文与知识实体描述中找含关键词的短摘录。"""
|
||
|
||
if not terms or work_id < 1 or as_of < 1:
|
||
return []
|
||
hits: list[dict[str, Any]] = []
|
||
with connect(readonly=True) as conn:
|
||
for term in terms:
|
||
if len(hits) >= limit:
|
||
break
|
||
rows = conn.execute(
|
||
"""SELECT c.order_no, b.id, b.revision, b.content_text
|
||
FROM muse_content_chapter c
|
||
JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=false
|
||
WHERE c.work_id=%s AND c.deleted=false AND c.order_no<=%s
|
||
AND b.content_text ILIKE %s
|
||
ORDER BY c.order_no DESC
|
||
LIMIT 2""",
|
||
(work_id, as_of, f"%{term}%"),
|
||
).fetchall()
|
||
for order_no, block_id, revision, body in rows:
|
||
if not isinstance(body, str) or not body:
|
||
continue
|
||
idx = body.find(term)
|
||
if idx < 0:
|
||
# ILIKE 可能大小写不同;退回全文头尾
|
||
snippet = body[:160]
|
||
start = 0
|
||
else:
|
||
start = max(0, idx - 40)
|
||
end = min(len(body), idx + len(term) + 80)
|
||
snippet = body[start:end]
|
||
hits.append(
|
||
{
|
||
"kind": "prose",
|
||
"term": term,
|
||
"chapter": int(order_no),
|
||
"blockId": int(block_id),
|
||
"revision": revision,
|
||
"start": start,
|
||
"end": start + len(snippet),
|
||
"text": snippet,
|
||
}
|
||
)
|
||
if len(hits) >= limit:
|
||
break
|
||
if len(hits) < limit:
|
||
for term in terms:
|
||
if len(hits) >= limit:
|
||
break
|
||
rows = conn.execute(
|
||
"""SELECT e.id, e.normalized_name, e.description, e.revision
|
||
FROM muse_knowledge_entity e
|
||
WHERE e.work_id=%s AND e.deleted=false
|
||
AND (e.normalized_name ILIKE %s
|
||
OR coalesce(e.description,'') ILIKE %s)
|
||
ORDER BY e.id
|
||
LIMIT 2""",
|
||
(work_id, f"%{term}%", f"%{term}%"),
|
||
).fetchall()
|
||
for entity_id, name, description, revision in rows:
|
||
fact = (
|
||
f"实体「{name}」:"
|
||
f"{(description or '').strip()[:200] or '(无描述)'}"
|
||
)
|
||
hits.append(
|
||
{
|
||
"kind": "entity",
|
||
"term": term,
|
||
"entityId": int(entity_id),
|
||
"revision": int(revision or 1),
|
||
"fact": fact,
|
||
}
|
||
)
|
||
if len(hits) >= limit:
|
||
break
|
||
return hits
|
||
|
||
|
||
def _existing_fact_ids(context: Mapping[str, Any]) -> set[str]:
|
||
return {
|
||
str(item.get("evidenceId"))
|
||
for item in (context.get("factEvidence") or [])
|
||
if isinstance(item, Mapping)
|
||
}
|
||
|
||
|
||
def _append_facts_from_hits(
|
||
context: dict[str, Any],
|
||
hits: Sequence[Mapping[str, Any]],
|
||
*,
|
||
gap_id: str,
|
||
attempt: int,
|
||
) -> int:
|
||
"""把检索命中写成 factEvidence;返回新增条数。"""
|
||
|
||
added = 0
|
||
facts = list(context.get("factEvidence") or [])
|
||
known = _existing_fact_ids(context)
|
||
for index, hit in enumerate(hits):
|
||
if hit.get("kind") == "prose":
|
||
text = str(hit["text"])
|
||
evidence_id = f"gap-{gap_id}-prose-{attempt}-{index}"
|
||
if evidence_id in known:
|
||
continue
|
||
facts.append(
|
||
{
|
||
"evidenceId": evidence_id,
|
||
"fact": f"补证摘录(第{hit['chapter']}章,关键词「{hit['term']}」):{text}",
|
||
"sourceType": "historical_prose",
|
||
"sourceRef": {
|
||
"sourceId": f"content-block:{hit['blockId']}",
|
||
"sourceVersion": f"rev{hit['revision']}",
|
||
"chapter": int(hit["chapter"]),
|
||
"blockId": int(hit["blockId"]),
|
||
"startCodePoint": int(hit["start"]),
|
||
"endCodePoint": int(hit["end"]),
|
||
},
|
||
"contentSha256": _sha_text(text),
|
||
"riskLevel": "low",
|
||
}
|
||
)
|
||
known.add(evidence_id)
|
||
added += 1
|
||
elif hit.get("kind") == "entity":
|
||
fact = str(hit["fact"])
|
||
evidence_id = f"gap-{gap_id}-entity-{attempt}-{index}"
|
||
if evidence_id in known:
|
||
continue
|
||
facts.append(
|
||
{
|
||
"evidenceId": evidence_id,
|
||
"fact": fact,
|
||
"sourceType": "canonical_state",
|
||
"sourceRef": {
|
||
"sourceId": f"knowledge-entity:{hit['entityId']}",
|
||
"sourceVersion": f"rev{hit.get('revision', 1)}",
|
||
},
|
||
"contentSha256": _sha_text(fact),
|
||
"riskLevel": "low",
|
||
}
|
||
)
|
||
known.add(evidence_id)
|
||
added += 1
|
||
context["factEvidence"] = facts
|
||
return added
|
||
|
||
|
||
def reassemble_writer_context_for_gaps(
|
||
current_context: Mapping[str, Any],
|
||
gaps: Sequence[Mapping[str, Any]],
|
||
next_attempt: int,
|
||
*,
|
||
work_id: int | None = None,
|
||
) -> dict[str, Any]:
|
||
"""由 production evidence_provider 调用:返回可校验的新 WriterContext。"""
|
||
|
||
if isinstance(next_attempt, bool) or not isinstance(next_attempt, int) or next_attempt < 2:
|
||
raise EvidenceReassembleError("next_attempt 必须是 >=2 的整数")
|
||
if not gaps:
|
||
raise EvidenceReassembleError("补证调用必须携带非空 evidenceGaps")
|
||
try:
|
||
base = validate_writer_context(current_context)
|
||
except ContractError as exc:
|
||
raise EvidenceReassembleError(f"当前上下文非法,无法补证: {exc}") from exc
|
||
if next_attempt <= int(base["attempt"]):
|
||
raise EvidenceReassembleError("next_attempt 必须严格大于当前 attempt")
|
||
|
||
advanced = copy.deepcopy(dict(base))
|
||
advanced["attempt"] = next_attempt
|
||
wid = int(work_id if work_id is not None else advanced.get("workId") or 0)
|
||
as_of = int(advanced.get("asOf") or 0)
|
||
if wid < 1 or as_of < 1:
|
||
raise EvidenceReassembleError("补证需要有效 workId 与 asOf")
|
||
|
||
resolved: list[str] = []
|
||
unresolved: list[str] = []
|
||
total_added = 0
|
||
for gap in gaps:
|
||
if not isinstance(gap, Mapping):
|
||
continue
|
||
gap_id = str(gap.get("gapId") or f"gap-{len(resolved)+len(unresolved)+1}")
|
||
query = str(gap.get("query") or gap.get("reason") or "")
|
||
terms = _query_terms(query)
|
||
hits = _search_canonical_snippets(work_id=wid, as_of=as_of, terms=terms)
|
||
added = _append_facts_from_hits(advanced, hits, gap_id=gap_id, attempt=next_attempt)
|
||
total_added += added
|
||
label = f"{gap_id}:{query[:60]}"
|
||
if added:
|
||
resolved.append(label)
|
||
else:
|
||
unresolved.append(label)
|
||
|
||
# 补证轮只约束「已命中的正典摘录怎么用」。零命中的缺口是新设定提案,
|
||
# 不得写成「下一稿禁止写入」——写手有权设计新设定,是否入库由人决定。
|
||
style = [str(s) for s in (advanced.get("styleConstraints") or [])]
|
||
style = [s for s in style if not s.startswith("补证轮指令:")]
|
||
if resolved:
|
||
style.append(
|
||
"补证轮指令:已追加与缺口相关的前文章节/实体摘录到事实约束;"
|
||
"下一稿只能用已给摘录支撑相关断言,不得外推未出现的量化体系或权限制度。"
|
||
)
|
||
style.append(
|
||
f"补证轮元数据:attempt={next_attempt};检索命中事实 {total_added} 条;"
|
||
f"已解决缺口 {len(resolved)};未解决 {len(unresolved)}。"
|
||
)
|
||
advanced["styleConstraints"] = style
|
||
|
||
# 叙事即时状态也钉一下,避免 creative input 与 style 之外无差分时漏网
|
||
narrative = dict(advanced.get("narrativeState") or {})
|
||
narrative["immediateSituation"] = (
|
||
f"{narrative.get('immediateSituation') or ''} "
|
||
f"[补证轮 attempt={next_attempt}:未解决缺口 {len(unresolved)} 个,"
|
||
f"已注入事实 {total_added} 条]"
|
||
).strip()
|
||
advanced["narrativeState"] = narrative
|
||
|
||
advanced["contextSnapshot"]["contextSha256"] = retrieval_identity(advanced)
|
||
try:
|
||
return validate_writer_context(advanced)
|
||
except ContractError as exc:
|
||
raise EvidenceReassembleError(f"补证后上下文校验失败: {exc}") from exc
|
||
|
||
|
||
__all__ = [
|
||
"EvidenceReassembleError",
|
||
"reassemble_writer_context_for_gaps",
|
||
]
|