#!/usr/bin/env python3 """两阶段写手(阶段 F 第一部分):探索与生成分离。 证据背景:烟测实证单阶段派发下多回合探索循环与长篇一次性产出合同冲突—— 正文碎片散落在中间回合,最终消息只剩残稿(909 字候选被机械门正确拦截)。 两阶段形态: 1. 探索阶段:派发写作智能体(只读工具),产出探索清单(小结构化 JSON, 不占用长篇输出空间);探索运行独立记账(事件、raw、依赖清单)。 2. 生成阶段:按清单确定性回放资料(只读工具重放,无模型参与),把写手 实际依赖的上下文整理成冻结生成输入,再派发一次(无工具)单次成稿。 职责分界(边界合同):两个派发运行都记在各自的派发运行账下;生产编排 只绑定候选与质量证据。生成阶段沿用 dispatch_writer_bridge 的信封绑定与 回执适配,写作证据引用仍指向生成运行。 """ from __future__ import annotations import json import sys from collections import Counter from pathlib import Path from typing import Any, Callable, Mapping SCRIPT_DIR = Path(__file__).resolve().parent DISPATCH_SCRIPTS = SCRIPT_DIR.parents[1] / "dispatch-agent-task" / "scripts" ASSEMBLE_SCRIPTS = SCRIPT_DIR.parents[1] / "assemble-context" / "scripts" for _path in (DISPATCH_SCRIPTS, ASSEMBLE_SCRIPTS): if str(_path) not in sys.path: sys.path.insert(0, str(_path)) from dispatch_agent_task import run_dispatch # noqa: E402 from pi_runner import ExecutionPolicy # noqa: E402 from read_tools import TOOL_REGISTRY, execute_tool # noqa: E402 from writer_contract import build_writer_creative_input # noqa: E402 from dispatch_writer_bridge import ( # noqa: E402 READ_TOOL_ALLOWLIST, run_writer_via_dispatch, writer_session_paths, ) EXPLORATION_SCHEMA_VERSION = "writer-exploration-manifest-v1" EXPLORATION_MAX_MATERIALS = 20 EXPLORATION_MAX_DURATION_SECONDS = 1200 EXPLORATION_SESSION_LABEL = "writer-explore" GENERATION_SESSION_LABEL = "writer-gen" EXPLORATION_OUTPUT_SCHEMA: dict[str, Any] = { "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": False, "required": ["schemaVersion", "requiredMaterials"], "properties": { "schemaVersion": {"type": "string", "const": EXPLORATION_SCHEMA_VERSION}, "requiredMaterials": { "type": "array", "minItems": 1, "maxItems": EXPLORATION_MAX_MATERIALS, "items": { "type": "object", "additionalProperties": False, "required": ["tool", "args", "reason"], "properties": { "tool": {"type": "string", "enum": list(READ_TOOL_ALLOWLIST)}, "args": {"type": "object"}, "reason": {"type": "string", "minLength": 1}, }, }, }, "styleNotes": {"type": "array", "items": {"type": "string"}}, "continuityNotes": {"type": "string"}, }, } class ExplorationError(RuntimeError): """两阶段写手失败:携带稳定错误码,编排方按失败关闭处理。""" def __init__(self, code: str, message: str, *, details: Mapping[str, Any] | None = None): super().__init__(message) self.code = code self.details = dict(details or {}) def build_exploration_spec( context: Mapping[str, Any], *, target_chapter: int, human_instruction: str, candidate_version: int, ) -> dict[str, Any]: """装配探索阶段任务包:只探索不写作,产出探索清单。""" work_id = context.get("workId") return { "specVersion": "agent-task-v1", "role": "writer", "taskPrompt": ( f"你是写作智能体的探索阶段。任务:用授权只读工具探索为作品 {work_id} " f"第{target_chapter}章写正文所需的资料。不要写正文。" "探索完成后只输出一个 manifest JSON(schema " f"{EXPLORATION_SCHEMA_VERSION}):requiredMaterials 逐项列出" "生成阶段需要带入的每份资料(用哪个工具、什么参数、为什么需要);" "styleNotes 与 continuityNotes 可记录你在探索中观察到的文风与衔接要点。" "生成阶段正文中的事实只能来自你本次探索实际读到的资料。" ), "input": { "workId": work_id, "targetChapter": target_chapter, "candidateVersion": candidate_version, "humanInstruction": (human_instruction or "").strip(), }, "outputSchema": EXPLORATION_OUTPUT_SCHEMA, "outputSchemaId": "writer-exploration-manifest-v1", "toolAllowlist": list(READ_TOOL_ALLOWLIST), "maxDurationSeconds": EXPLORATION_MAX_DURATION_SECONDS, } def parse_exploration_manifest(raw_text: str) -> dict[str, Any]: """严格校验探索清单;任何结构偏差失败关闭。""" try: data = json.loads(raw_text) except ValueError as exc: raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"探索清单不是合法 JSON:{exc}" ) from exc if not isinstance(data, Mapping): raise ExplorationError("EXPLORATION_MANIFEST_INVALID", "探索清单必须是 JSON 对象") if data.get("schemaVersion") != EXPLORATION_SCHEMA_VERSION: raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"探索清单 schemaVersion 必须为 {EXPLORATION_SCHEMA_VERSION}", ) materials = data.get("requiredMaterials") if not isinstance(materials, list) or not materials: raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", "requiredMaterials 必须是非空列表:生成阶段的事实只能来自探索资料", ) if len(materials) > EXPLORATION_MAX_MATERIALS: raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"requiredMaterials 超过上限 {EXPLORATION_MAX_MATERIALS} 条", ) for index, item in enumerate(materials): if not isinstance(item, Mapping): raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}] 必须是对象" ) tool = item.get("tool") if not isinstance(tool, str) or tool not in TOOL_REGISTRY: raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}].tool 不在只读工具登记表:{tool!r}", ) if not isinstance(item.get("args"), Mapping): raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}].args 必须是对象" ) reason = item.get("reason") if not isinstance(reason, str) or not reason.strip(): raise ExplorationError( "EXPLORATION_MANIFEST_INVALID", f"requiredMaterials[{index}].reason 必须非空" ) style_notes = data.get("styleNotes") if style_notes is not None and not ( isinstance(style_notes, list) and all(isinstance(x, str) for x in style_notes) ): raise ExplorationError("EXPLORATION_MANIFEST_INVALID", "styleNotes 必须是字符串列表") continuity = data.get("continuityNotes") if continuity is not None and not isinstance(continuity, str): raise ExplorationError("EXPLORATION_MANIFEST_INVALID", "continuityNotes 必须是字符串") return dict(data) def replay_manifest_materials( manifest: Mapping[str, Any], *, connect_factory: Callable[..., Any] | None = None, ) -> list[dict[str, Any]]: """确定性回放探索清单:无模型参与,逐条重放只读工具取回资料。""" materials: list[dict[str, Any]] = [] for item in manifest["requiredMaterials"]: try: result = execute_tool(item["tool"], item["args"], connect=connect_factory) except Exception as exc: # 工具层异常一律失败关闭,不带残缺资料进生成 raise ExplorationError( "EXPLORATION_REPLAY_FAILED", f"资料回放失败:{item['tool']} 参数 {json.dumps(item['args'], ensure_ascii=False)}:{exc}", details={"tool": item["tool"], "args": dict(item["args"])}, ) from exc materials.append( {"tool": item["tool"], "args": dict(item["args"]), "reason": item["reason"], "result": result} ) return materials def build_generation_creative_input( context: Mapping[str, Any], manifest: Mapping[str, Any], materials: list[dict[str, Any]], *, human_instruction: str = "", ) -> dict[str, Any]: """整理生成输入:资料来自探索回放,合同部分来自冻结上下文投影。""" projected = build_writer_creative_input(context) creative: dict[str, Any] = { "inputMode": "two-phase-generation-v1", "humanInstruction": (human_instruction or "").strip() or "基于探索资料续写本章完整正文。", "explorationMaterials": materials, "lengthContract": projected["lengthContract"], "styleConstraints": projected["styleConstraints"], } notes: dict[str, Any] = {} style_notes = manifest.get("styleNotes") if style_notes: notes["styleNotes"] = [str(x) for x in style_notes] continuity = manifest.get("continuityNotes") if isinstance(continuity, str) and continuity.strip(): notes["continuityNotes"] = continuity.strip() if notes: notes["note"] = "探索笔记是模型观察,仅供参考;正文事实必须以 explorationMaterials 为准。" creative["explorationNotes"] = notes return creative def run_two_phase_writer( context: Mapping[str, Any], *, candidate_version: int, repo_root: str | Path, provider: str, model: str, thinking: str | None = None, human_instruction: str = "", spec_dir: str | Path | None = None, launcher: Callable[..., Any] | None = None, connect_factory: Callable[..., Any] | None = None, ) -> tuple[dict[str, Any], Any, tuple[Any, Any], dict[str, Any]]: """两阶段派发写作:探索(有工具)→ 回放整理 → 生成(无工具单次成稿)。 返回(候选信封、生成回执适配、证据引用、探索摘要)。 任何阶段失败抛 ExplorationError / DispatchWriterError(失败关闭)。 """ run_id = str(context.get("runId") or "") work_id = context.get("workId") target_chapter = context.get("targetChapter") if not run_id or not isinstance(work_id, int) or not isinstance(target_chapter, int): raise ExplorationError( "EXPLORATION_CONTEXT_INVALID", "WriterContext 缺 runId/workId/targetChapter" ) spec_root = Path(spec_dir) if spec_dir is not None else SCRIPT_DIR spec_root.mkdir(parents=True, exist_ok=True) # 阶段一:探索派发(独立派发运行,独立会话) explore_spec = build_exploration_spec( context, target_chapter=target_chapter, human_instruction=human_instruction, candidate_version=candidate_version, ) explore_spec_file = spec_root / f"{run_id}-exploration-task-v{candidate_version}.json" explore_spec_file.write_text( json.dumps(explore_spec, ensure_ascii=False, indent=1), encoding="utf-8" ) explore_session_id, explore_session_dir = writer_session_paths( work_id, target_chapter, label=EXPLORATION_SESSION_LABEL ) explore_session_dir.mkdir(parents=True, mode=0o700, exist_ok=True) explore_run_id = f"{run_id}-explore-v{candidate_version}" policy = ExecutionPolicy(provider=provider, model=model, thinking=thinking) receipt, code = run_dispatch( explore_spec_file, repo_root=repo_root, policy=policy, run_id=explore_run_id, trigger_source="user", trigger_detail={"stage": "writer-exploration", "productionRunId": run_id}, session_id=explore_session_id, session_dir=explore_session_dir, enable_read_tools=True, launcher=launcher, connect_factory=connect_factory, ) if code != 0 or receipt.get("status") != "completed": raise ExplorationError( str(receipt.get("errorCode") or "EXPLORATION_DISPATCH_FAILED"), f"写作智能体探索派发未成功:{receipt.get('error') or receipt.get('errorCode')}", details={"explorationRunId": explore_run_id, "exitCode": code}, ) # 探索产出:结构化输出回读派发运行目录,不另造权威 output_file = Path(str(receipt.get("runDir") or "")) / "output.json" try: raw_output = output_file.read_text(encoding="utf-8") except OSError as exc: raise ExplorationError( "EXPLORATION_OUTPUT_MISSING", f"探索运行未落结构化输出:{output_file}", details={"explorationRunId": explore_run_id}, ) from exc manifest = parse_exploration_manifest(raw_output) materials = replay_manifest_materials(manifest, connect_factory=connect_factory) creative_input = build_generation_creative_input( context, manifest, materials, human_instruction=human_instruction ) # 阶段二:生成派发(无工具,单次成稿;信封绑定复用派发桥) generation_task_prompt = ( f"为第{target_chapter}章写正文候选(候选版本 {candidate_version})。" f"人的创作指令:{(human_instruction or '').strip() or '基于探索资料续写本章完整正文。'}" "所需资料已由你的探索阶段整理在冻结创作输入中,不要再调用工具;" "在单条回复里一次写完本章全部正文并输出完整 JSON。" "正文中的事实必须来自探索资料。" ) envelope, writer_receipt, raw_ref = run_writer_via_dispatch( context, candidate_version=candidate_version, repo_root=repo_root, provider=provider, model=model, thinking=thinking, human_instruction=human_instruction, task_prompt=generation_task_prompt, creative_input=creative_input, enable_read_tools=False, session_label=GENERATION_SESSION_LABEL, spec_path=spec_root / f"{run_id}-writer-task-v{candidate_version}-gen.json", launcher=launcher, connect_factory=connect_factory, ) exploration_summary = { "explorationRunId": explore_run_id, "explorationSessionId": explore_session_id, "manifestSchemaVersion": EXPLORATION_SCHEMA_VERSION, "materialCount": len(materials), "tools": dict(Counter(item["tool"] for item in materials)), "styleNotes": len(manifest.get("styleNotes") or []), "generationRunId": writer_receipt.dispatch_run_id, } return envelope, writer_receipt, raw_ref, exploration_summary __all__ = [ "EXPLORATION_MAX_MATERIALS", "EXPLORATION_OUTPUT_SCHEMA", "EXPLORATION_SCHEMA_VERSION", "EXPLORATION_SESSION_LABEL", "GENERATION_SESSION_LABEL", "ExplorationError", "build_exploration_spec", "build_generation_creative_input", "parse_exploration_manifest", "replay_manifest_materials", "run_two_phase_writer", ]