278 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""抽取智能体的框架派发桥(阶段 F 第二部分)。
职责分界(边界合同):智能体框架负责模型事件、raw、逐回合调用,记在派发
运行下;生产编排负责知识草稿落库与业务回执,记在生产抽取运行下。两者以
trigger_detail.productionRunId 关联。
与写手桥的两点不同:
1. 全量正典正文注入任务输入——证据必须是正文逐字片段,而只读工具读正文
会截断,注入是唯一可靠路径;工具白名单留给查重与回读核验。
2. 单次派发(探索与产出同循环)——抽取产出是结构化 JSON,无长篇碎片化
风险,不需要两阶段。
机械校验、修复重派(一轮)、保守收口全部复用既有可信适配层,不新造合同。
"""
from __future__ import annotations
import json
import sys
from pathlib import Path
from typing import Any, Callable, Mapping
SCRIPT_DIR = Path(__file__).resolve().parent
DISPATCH_SCRIPTS = SCRIPT_DIR.parents[1] / "dispatch-agent-task" / "scripts"
if str(DISPATCH_SCRIPTS) not in sys.path:
sys.path.insert(0, str(DISPATCH_SCRIPTS))
from extract_knowledge import ( # noqa: E402
ACTOR,
ExtractionContractError,
_load_chapter,
_propose_chapter_extract_lesson,
normalize_extraction,
persist_extraction,
salvage_extraction,
)
from muse_llm import extract_json # noqa: E402
from record_failed_run import record_failure # noqa: E402
from run_registry import finish_run, new_run_id, start_run # noqa: E402
from dispatch_agent_task import run_dispatch # noqa: E402
from pi_runner import ExecutionPolicy # noqa: E402
from read_tools import TOOL_REGISTRY # noqa: E402
# 抽取探索白名单:查重与回读核验;正文本体注入任务输入,不依赖工具读取。
EXTRACTION_TOOL_ALLOWLIST = ("read_chapter_text", "search_entities")
# 抽取产出是结构化 JSON;严格语义由 normalize_extraction 机械校验,Schema 只做形状兜底。
EXTRACTION_OUTPUT_SCHEMA: dict[str, Any] = {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"required": ["entities", "relations", "state"],
"properties": {
"entities": {"type": "array"},
"relations": {"type": "array"},
"state": {"type": "object"},
},
}
EXTRACTION_MAX_DURATION_SECONDS = 1800
SESSION_ROOT = Path("/tmp/muse-agent-runs/extractor-sessions")
class DispatchExtractionError(RuntimeError):
"""抽取派发失败:携带稳定错误码,编排方按失败关闭处理。"""
def __init__(self, code: str, message: str, *, details: Mapping[str, Any] | None = None):
super().__init__(message)
self.code = code
self.details = dict(details or {})
def extractor_session_paths(work_id: int, chapter_order: int) -> tuple[str, Path]:
"""一章一个抽取智能体会话:修复重派在同一会话内接续。"""
session_id = f"extractor-work{work_id}-ch{chapter_order}"
return session_id, SESSION_ROOT / session_id
def build_extraction_task_spec(
*,
work_id: int,
chapter_order: int,
title: str,
chapter_title: str,
existing_names: list[str],
body: str,
repair_reason: str | None = None,
) -> dict[str, Any]:
"""装配抽取任务包:全量正文进输入,合同条款进任务提示词。"""
task_prompt = (
f"你是抽取智能体。任务:从作品《{title}》第{chapter_order}章"
f"《{chapter_title or ''}》的已接受正文抽取作品私有知识草稿。"
"正文全文在冻结输入的 body 字段;证据必须是正文中的逐字连续片段,不能改写。"
"实体类型只能使用:character、location、faction、power_system、item、event;"
"立卡门槛:具名且有跨章复用或后续履约潜力,一次性龙套与一次性道具不列实体。"
"不要确认知识,不要补写正文没有的事实;低置信内容保留但在 brief/fields 中标注“?”。"
"可用只读工具核对既有实体(查重)与回读正文,但正文以输入 body 为准。"
"只输出一个 JSON 对象:entities(每项 type/name/brief/fields/evidence)、"
"relations(每项 source/target/type/description/evidence)、"
"state(currentSituation/characterStates/foreshadowing/handoff)。"
)
if repair_reason:
task_prompt += (
"\n【机械校验失败,允许一次修复】失败原因:"
+ repair_reason
+ "。只修正证据字段,使每条 evidence 都是正文中的逐字连续片段;"
"删除无法找到逐字证据的条目,不得新增条目、事实、关系或状态。仍只输出同一 JSON 对象。"
)
return {
"specVersion": "agent-task-v1",
"role": "extractor",
"taskPrompt": task_prompt,
"input": {
"workId": work_id,
"chapterOrder": chapter_order,
"title": title,
"chapterTitle": chapter_title or "",
"existingEntityNames": existing_names,
"body": body,
},
"outputSchema": EXTRACTION_OUTPUT_SCHEMA,
"outputSchemaId": "chapter-extraction-v1",
"toolAllowlist": list(EXTRACTION_TOOL_ALLOWLIST),
"maxDurationSeconds": EXTRACTION_MAX_DURATION_SECONDS,
}
def _read_dispatch_output(receipt: Mapping[str, Any], dispatch_run_id: str) -> Any:
"""结构化输出回读派发运行目录的 output.json,不另造权威。"""
output_file = Path(str(receipt.get("runDir") or "")) / "output.json"
try:
raw_text = output_file.read_text(encoding="utf-8")
except OSError as exc:
raise DispatchExtractionError(
"EXTRACTION_OUTPUT_MISSING",
f"抽取派发运行未落结构化输出:{output_file}",
details={"dispatchRunId": dispatch_run_id},
) from exc
try:
return json.loads(raw_text)
except ValueError:
# 模型偶尔包 markdown/解释;用既有提取器兜底,仍失败则失败关闭。
try:
return extract_json(raw_text)
except Exception as exc:
raise DispatchExtractionError(
"EXTRACTION_OUTPUT_INVALID",
f"抽取派发输出不是合法 JSON:{exc}",
details={"dispatchRunId": dispatch_run_id},
) from exc
def run_extraction_via_dispatch(
work_id: int,
chapter_order: int,
*,
repo_root: str | Path,
provider: str,
model: str,
thinking: str | None = None,
run_id: str | None = None,
spec_dir: str | Path | None = None,
launcher: Callable[..., Any] | None = None,
connect_factory: Callable[..., Any] | None = None,
) -> dict[str, Any]:
"""派发抽取智能体并落知识草稿;返回落库摘要。任何失败失败关闭。"""
for name in EXTRACTION_TOOL_ALLOWLIST:
if name not in TOOL_REGISTRY:
raise DispatchExtractionError(
"EXTRACTION_TOOL_UNREGISTERED", f"抽取白名单工具未登记:{name}"
)
active_run = run_id or new_run_id("extract-knowledge", work_id=work_id, target_chapter=chapter_order)
start_run(
run_id=active_run,
work_id=work_id,
target_chapter=chapter_order,
trigger_detail={"stage": "chapter-after-extraction", "mode": "dispatch"},
creator=ACTOR,
)
spec_root = Path(spec_dir) if spec_dir is not None else SCRIPT_DIR
spec_root.mkdir(parents=True, exist_ok=True)
session_id, session_dir = extractor_session_paths(work_id, chapter_order)
session_dir.mkdir(parents=True, mode=0o700, exist_ok=True)
policy = ExecutionPolicy(provider=provider, model=model, thinking=thinking)
try:
(title, chapter_id, chapter_title, _block_id, body), existing = _load_chapter(
work_id, chapter_order
)
existing_names = list(existing[:200])
def _dispatch_once(attempt: int, repair_reason: str | None) -> tuple[Any, Mapping[str, Any]]:
spec = build_extraction_task_spec(
work_id=work_id, chapter_order=chapter_order, title=title,
chapter_title=chapter_title or "", existing_names=existing_names,
body=body, repair_reason=repair_reason,
)
spec_file = spec_root / f"{active_run}-extractor-task-v{attempt}.json"
spec_file.write_text(json.dumps(spec, ensure_ascii=False, indent=1), encoding="utf-8")
dispatch_run_id = f"{active_run}-extractor-v{attempt}"
receipt, code = run_dispatch(
spec_file,
repo_root=repo_root,
policy=policy,
run_id=dispatch_run_id,
trigger_source="user",
trigger_detail={"stage": "extractor-dispatch", "productionRunId": active_run},
session_id=session_id,
session_dir=session_dir,
enable_read_tools=True,
launcher=launcher,
connect_factory=connect_factory,
)
if code != 0 or receipt.get("status") != "completed":
raise DispatchExtractionError(
str(receipt.get("errorCode") or "EXTRACTION_DISPATCH_FAILED"),
f"抽取智能体派发未成功:{receipt.get('error') or receipt.get('errorCode')}",
details={"dispatchRunId": dispatch_run_id, "exitCode": code},
)
return _read_dispatch_output(receipt, dispatch_run_id), receipt
raw_output, receipt = _dispatch_once(1, None)
try:
payload = normalize_extraction(raw_output, body)
except ExtractionContractError as first_error:
# 只允许一轮机械修复重派(证据绑定),对齐直调链语义。
raw_output, receipt = _dispatch_once(2, str(first_error))
try:
payload = normalize_extraction(raw_output, body)
except ExtractionContractError:
payload = salvage_extraction(raw_output, body)
model_ids = receipt.get("actualModelIds") or []
result = persist_extraction(
work_id, chapter_id, chapter_order, active_run, payload,
requested_model=f"{provider}/{model}",
actual_model=str(model_ids[-1]) if model_ids else "",
usage=dict(receipt.get("usage") or {}),
)
finish_run(active_run, "completed", creator=ACTOR,
trigger_detail={"stage": "chapter-after-extraction", "mode": "dispatch",
"drafts": len(result["draft_ids"])})
lesson = _propose_chapter_extract_lesson(
run_id=active_run, work_id=work_id, chapter_order=chapter_order,
draft_count=len(result["draft_ids"]), state_draft_id=result.get("state_draft_id"),
)
return {"run_id": active_run, **result, "lesson": lesson}
except BaseException as exc:
finish_run(active_run, "failed", creator=ACTOR,
trigger_detail={"stage": "chapter-after-extraction", "mode": "dispatch",
"error_type": type(exc).__name__})
try:
record_failure(
active_run,
sample_id=f"extract-ch{chapter_order}",
adapter_role="extractor",
caller="extract-knowledge-dispatch",
failure_type=type(exc).__name__,
)
except Exception:
pass
raise
__all__ = [
"EXTRACTION_MAX_DURATION_SECONDS",
"EXTRACTION_OUTPUT_SCHEMA",
"EXTRACTION_TOOL_ALLOWLIST",
"DispatchExtractionError",
"build_extraction_task_spec",
"extractor_session_paths",
"run_extraction_via_dispatch",
]