363 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""两阶段写手(阶段 F 第一部分):探索与生成分离。
证据背景:烟测实证单阶段派发下多回合探索循环与长篇一次性产出合同冲突——
正文碎片散落在中间回合,最终消息只剩残稿(909 字候选被机械门正确拦截)。
两阶段形态:
1. 探索阶段:派发写作智能体(只读工具),产出探索清单(小结构化 JSON,
不占用长篇输出空间);探索运行独立记账(事件、raw、依赖清单)。
2. 生成阶段:按清单确定性回放资料(只读工具重放,无模型参与),把写手
实际依赖的上下文整理成冻结生成输入,再派发一次(无工具)单次成稿。
职责分界(边界合同):两个派发运行都记在各自的派发运行账下;生产编排
只绑定候选与质量证据。生成阶段沿用 dispatch_writer_bridge 的信封绑定与
回执适配,写作证据引用仍指向生成运行。
"""
from __future__ import annotations
import json
import sys
from collections import Counter
from pathlib import Path
from typing import Any, Callable, Mapping
SCRIPT_DIR = Path(__file__).resolve().parent
DISPATCH_SCRIPTS = SCRIPT_DIR.parents[1] / "dispatch-agent-task" / "scripts"
ASSEMBLE_SCRIPTS = SCRIPT_DIR.parents[1] / "assemble-context" / "scripts"
for _path in (DISPATCH_SCRIPTS, ASSEMBLE_SCRIPTS):
if str(_path) not in sys.path:
sys.path.insert(0, str(_path))
from dispatch_agent_task import run_dispatch # noqa: E402
from pi_runner import ExecutionPolicy # noqa: E402
from read_tools import TOOL_REGISTRY, execute_tool # noqa: E402
from writer_contract import build_writer_creative_input # noqa: E402
from dispatch_writer_bridge import ( # noqa: E402
READ_TOOL_ALLOWLIST,
run_writer_via_dispatch,
writer_session_paths,
)
EXPLORATION_SCHEMA_VERSION = "writer-exploration-manifest-v1"
EXPLORATION_MAX_MATERIALS = 20
EXPLORATION_MAX_DURATION_SECONDS = 1200
EXPLORATION_SESSION_LABEL = "writer-explore"
GENERATION_SESSION_LABEL = "writer-gen"
EXPLORATION_OUTPUT_SCHEMA: dict[str, Any] = {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"additionalProperties": False,
"required": ["schemaVersion", "requiredMaterials"],
"properties": {
"schemaVersion": {"type": "string", "const": EXPLORATION_SCHEMA_VERSION},
"requiredMaterials": {
"type": "array",
"minItems": 1,
"maxItems": EXPLORATION_MAX_MATERIALS,
"items": {
"type": "object",
"additionalProperties": False,
"required": ["tool", "args", "reason"],
"properties": {
"tool": {"type": "string", "enum": list(READ_TOOL_ALLOWLIST)},
"args": {"type": "object"},
"reason": {"type": "string", "minLength": 1},
},
},
},
"styleNotes": {"type": "array", "items": {"type": "string"}},
"continuityNotes": {"type": "string"},
},
}
class ExplorationError(RuntimeError):
"""两阶段写手失败:携带稳定错误码,编排方按失败关闭处理。"""
def __init__(self, code: str, message: str, *, details: Mapping[str, Any] | None = None):
super().__init__(message)
self.code = code
self.details = dict(details or {})
def build_exploration_spec(
context: Mapping[str, Any],
*,
target_chapter: int,
human_instruction: str,
candidate_version: int,
) -> dict[str, Any]:
"""装配探索阶段任务包:只探索不写作,产出探索清单。"""
work_id = context.get("workId")
return {
"specVersion": "agent-task-v1",
"role": "writer",
"taskPrompt": (
f"你是写作智能体的探索阶段。任务:用授权只读工具探索为作品 {work_id} "
f"第{target_chapter}章写正文所需的资料。不要写正文。"
"探索完成后只输出一个 manifest JSON(schema "
f"{EXPLORATION_SCHEMA_VERSION}):requiredMaterials 逐项列出"
"生成阶段需要带入的每份资料(用哪个工具、什么参数、为什么需要);"
"styleNotes 与 continuityNotes 可记录你在探索中观察到的文风与衔接要点。"
"生成阶段正文中的事实只能来自你本次探索实际读到的资料。"
),
"input": {
"workId": work_id,
"targetChapter": target_chapter,
"candidateVersion": candidate_version,
"humanInstruction": (human_instruction or "").strip(),
},
"outputSchema": EXPLORATION_OUTPUT_SCHEMA,
"outputSchemaId": "writer-exploration-manifest-v1",
"toolAllowlist": list(READ_TOOL_ALLOWLIST),
"maxDurationSeconds": EXPLORATION_MAX_DURATION_SECONDS,
}
def parse_exploration_manifest(raw_text: str) -> dict[str, Any]:
"""严格校验探索清单;任何结构偏差失败关闭。"""
try:
data = json.loads(raw_text)
except ValueError as exc:
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID", f"探索清单不是合法 JSON:{exc}"
) from exc
if not isinstance(data, Mapping):
raise ExplorationError("EXPLORATION_MANIFEST_INVALID", "探索清单必须是 JSON 对象")
if data.get("schemaVersion") != EXPLORATION_SCHEMA_VERSION:
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID",
f"探索清单 schemaVersion 必须为 {EXPLORATION_SCHEMA_VERSION}",
)
materials = data.get("requiredMaterials")
if not isinstance(materials, list) or not materials:
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID",
"requiredMaterials 必须是非空列表:生成阶段的事实只能来自探索资料",
)
if len(materials) > EXPLORATION_MAX_MATERIALS:
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID",
f"requiredMaterials 超过上限 {EXPLORATION_MAX_MATERIALS} 条",
)
for index, item in enumerate(materials):
if not isinstance(item, Mapping):
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}] 必须是对象"
)
tool = item.get("tool")
if not isinstance(tool, str) or tool not in TOOL_REGISTRY:
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID",
f"requiredMaterials[{index}].tool 不在只读工具登记表:{tool!r}",
)
if not isinstance(item.get("args"), Mapping):
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}].args 必须是对象"
)
reason = item.get("reason")
if not isinstance(reason, str) or not reason.strip():
raise ExplorationError(
"EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}].reason 必须非空"
)
style_notes = data.get("styleNotes")
if style_notes is not None and not (
isinstance(style_notes, list) and all(isinstance(x, str) for x in style_notes)
):
raise ExplorationError("EXPLORATION_MANIFEST_INVALID", "styleNotes 必须是字符串列表")
continuity = data.get("continuityNotes")
if continuity is not None and not isinstance(continuity, str):
raise ExplorationError("EXPLORATION_MANIFEST_INVALID", "continuityNotes 必须是字符串")
return dict(data)
def replay_manifest_materials(
manifest: Mapping[str, Any],
*,
connect_factory: Callable[..., Any] | None = None,
) -> list[dict[str, Any]]:
"""确定性回放探索清单:无模型参与,逐条重放只读工具取回资料。"""
materials: list[dict[str, Any]] = []
for item in manifest["requiredMaterials"]:
try:
result = execute_tool(item["tool"], item["args"], connect=connect_factory)
except Exception as exc: # 工具层异常一律失败关闭,不带残缺资料进生成
raise ExplorationError(
"EXPLORATION_REPLAY_FAILED",
f"资料回放失败:{item['tool']} 参数 {json.dumps(item['args'], ensure_ascii=False)}:{exc}",
details={"tool": item["tool"], "args": dict(item["args"])},
) from exc
materials.append(
{"tool": item["tool"], "args": dict(item["args"]), "reason": item["reason"], "result": result}
)
return materials
def build_generation_creative_input(
context: Mapping[str, Any],
manifest: Mapping[str, Any],
materials: list[dict[str, Any]],
*,
human_instruction: str = "",
) -> dict[str, Any]:
"""整理生成输入:资料来自探索回放,合同部分来自冻结上下文投影。"""
projected = build_writer_creative_input(context)
creative: dict[str, Any] = {
"inputMode": "two-phase-generation-v1",
"humanInstruction": (human_instruction or "").strip() or "基于探索资料续写本章完整正文。",
"explorationMaterials": materials,
"lengthContract": projected["lengthContract"],
"styleConstraints": projected["styleConstraints"],
}
notes: dict[str, Any] = {}
style_notes = manifest.get("styleNotes")
if style_notes:
notes["styleNotes"] = [str(x) for x in style_notes]
continuity = manifest.get("continuityNotes")
if isinstance(continuity, str) and continuity.strip():
notes["continuityNotes"] = continuity.strip()
if notes:
notes["note"] = "探索笔记是模型观察,仅供参考;正文事实必须以 explorationMaterials 为准。"
creative["explorationNotes"] = notes
return creative
def run_two_phase_writer(
context: Mapping[str, Any],
*,
candidate_version: int,
repo_root: str | Path,
provider: str,
model: str,
thinking: str | None = None,
human_instruction: str = "",
spec_dir: str | Path | None = None,
launcher: Callable[..., Any] | None = None,
connect_factory: Callable[..., Any] | None = None,
) -> tuple[dict[str, Any], Any, tuple[Any, Any], dict[str, Any]]:
"""两阶段派发写作:探索(有工具)→ 回放整理 → 生成(无工具单次成稿)。
返回(候选信封、生成回执适配、证据引用、探索摘要)。
任何阶段失败抛 ExplorationError / DispatchWriterError(失败关闭)。
"""
run_id = str(context.get("runId") or "")
work_id = context.get("workId")
target_chapter = context.get("targetChapter")
if not run_id or not isinstance(work_id, int) or not isinstance(target_chapter, int):
raise ExplorationError(
"EXPLORATION_CONTEXT_INVALID", "WriterContext 缺 runId/workId/targetChapter"
)
spec_root = Path(spec_dir) if spec_dir is not None else SCRIPT_DIR
spec_root.mkdir(parents=True, exist_ok=True)
# 阶段一:探索派发(独立派发运行,独立会话)
explore_spec = build_exploration_spec(
context,
target_chapter=target_chapter,
human_instruction=human_instruction,
candidate_version=candidate_version,
)
explore_spec_file = spec_root / f"{run_id}-exploration-task-v{candidate_version}.json"
explore_spec_file.write_text(
json.dumps(explore_spec, ensure_ascii=False, indent=1), encoding="utf-8"
)
explore_session_id, explore_session_dir = writer_session_paths(
work_id, target_chapter, label=EXPLORATION_SESSION_LABEL
)
explore_session_dir.mkdir(parents=True, mode=0o700, exist_ok=True)
explore_run_id = f"{run_id}-explore-v{candidate_version}"
policy = ExecutionPolicy(provider=provider, model=model, thinking=thinking)
receipt, code = run_dispatch(
explore_spec_file,
repo_root=repo_root,
policy=policy,
run_id=explore_run_id,
trigger_source="user",
trigger_detail={"stage": "writer-exploration", "productionRunId": run_id},
session_id=explore_session_id,
session_dir=explore_session_dir,
enable_read_tools=True,
launcher=launcher,
connect_factory=connect_factory,
)
if code != 0 or receipt.get("status") != "completed":
raise ExplorationError(
str(receipt.get("errorCode") or "EXPLORATION_DISPATCH_FAILED"),
f"写作智能体探索派发未成功:{receipt.get('error') or receipt.get('errorCode')}",
details={"explorationRunId": explore_run_id, "exitCode": code},
)
# 探索产出:结构化输出回读派发运行目录,不另造权威
output_file = Path(str(receipt.get("runDir") or "")) / "output.json"
try:
raw_output = output_file.read_text(encoding="utf-8")
except OSError as exc:
raise ExplorationError(
"EXPLORATION_OUTPUT_MISSING",
f"探索运行未落结构化输出:{output_file}",
details={"explorationRunId": explore_run_id},
) from exc
manifest = parse_exploration_manifest(raw_output)
materials = replay_manifest_materials(manifest, connect_factory=connect_factory)
creative_input = build_generation_creative_input(
context, manifest, materials, human_instruction=human_instruction
)
# 阶段二:生成派发(无工具,单次成稿;信封绑定复用派发桥)
generation_task_prompt = (
f"为第{target_chapter}章写正文候选(候选版本 {candidate_version})。"
f"人的创作指令:{(human_instruction or '').strip() or '基于探索资料续写本章完整正文。'}"
"所需资料已由你的探索阶段整理在冻结创作输入中,不要再调用工具;"
"在单条回复里一次写完本章全部正文并输出完整 JSON。"
"正文中的事实必须来自探索资料。"
)
envelope, writer_receipt, raw_ref = run_writer_via_dispatch(
context,
candidate_version=candidate_version,
repo_root=repo_root,
provider=provider,
model=model,
thinking=thinking,
human_instruction=human_instruction,
task_prompt=generation_task_prompt,
creative_input=creative_input,
enable_read_tools=False,
session_label=GENERATION_SESSION_LABEL,
spec_path=spec_root / f"{run_id}-writer-task-v{candidate_version}-gen.json",
launcher=launcher,
connect_factory=connect_factory,
)
exploration_summary = {
"explorationRunId": explore_run_id,
"explorationSessionId": explore_session_id,
"manifestSchemaVersion": EXPLORATION_SCHEMA_VERSION,
"materialCount": len(materials),
"tools": dict(Counter(item["tool"] for item in materials)),
"styleNotes": len(manifest.get("styleNotes") or []),
"generationRunId": writer_receipt.dispatch_run_id,
}
return envelope, writer_receipt, raw_ref, exploration_summary
__all__ = [
"EXPLORATION_MAX_MATERIALS",
"EXPLORATION_OUTPUT_SCHEMA",
"EXPLORATION_SCHEMA_VERSION",
"EXPLORATION_SESSION_LABEL",
"GENERATION_SESSION_LABEL",
"ExplorationError",
"build_exploration_spec",
"build_generation_creative_input",
"parse_exploration_manifest",
"replay_manifest_materials",
"run_two_phase_writer",
]