278 lines
12 KiB
Python
278 lines
12 KiB
Python
#!/usr/bin/env python3
|
||
"""抽取智能体的框架派发桥(阶段 F 第二部分)。
|
||
|
||
职责分界(边界合同):智能体框架负责模型事件、raw、逐回合调用,记在派发
|
||
运行下;生产编排负责知识草稿落库与业务回执,记在生产抽取运行下。两者以
|
||
trigger_detail.productionRunId 关联。
|
||
|
||
与写手桥的两点不同:
|
||
1. 全量正典正文注入任务输入——证据必须是正文逐字片段,而只读工具读正文
|
||
会截断,注入是唯一可靠路径;工具白名单留给查重与回读核验。
|
||
2. 单次派发(探索与产出同循环)——抽取产出是结构化 JSON,无长篇碎片化
|
||
风险,不需要两阶段。
|
||
|
||
机械校验、修复重派(一轮)、保守收口全部复用既有可信适配层,不新造合同。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import sys
|
||
from pathlib import Path
|
||
from typing import Any, Callable, Mapping
|
||
|
||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||
DISPATCH_SCRIPTS = SCRIPT_DIR.parents[1] / "dispatch-agent-task" / "scripts"
|
||
if str(DISPATCH_SCRIPTS) not in sys.path:
|
||
sys.path.insert(0, str(DISPATCH_SCRIPTS))
|
||
|
||
from extract_knowledge import ( # noqa: E402
|
||
ACTOR,
|
||
ExtractionContractError,
|
||
_load_chapter,
|
||
_propose_chapter_extract_lesson,
|
||
normalize_extraction,
|
||
persist_extraction,
|
||
salvage_extraction,
|
||
)
|
||
from muse_llm import extract_json # noqa: E402
|
||
from record_failed_run import record_failure # noqa: E402
|
||
from run_registry import finish_run, new_run_id, start_run # noqa: E402
|
||
from dispatch_agent_task import run_dispatch # noqa: E402
|
||
from pi_runner import ExecutionPolicy # noqa: E402
|
||
from read_tools import TOOL_REGISTRY # noqa: E402
|
||
|
||
# 抽取探索白名单:查重与回读核验;正文本体注入任务输入,不依赖工具读取。
|
||
EXTRACTION_TOOL_ALLOWLIST = ("read_chapter_text", "search_entities")
|
||
|
||
# 抽取产出是结构化 JSON;严格语义由 normalize_extraction 机械校验,Schema 只做形状兜底。
|
||
EXTRACTION_OUTPUT_SCHEMA: dict[str, Any] = {
|
||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||
"type": "object",
|
||
"required": ["entities", "relations", "state"],
|
||
"properties": {
|
||
"entities": {"type": "array"},
|
||
"relations": {"type": "array"},
|
||
"state": {"type": "object"},
|
||
},
|
||
}
|
||
|
||
EXTRACTION_MAX_DURATION_SECONDS = 1800
|
||
SESSION_ROOT = Path("/tmp/muse-agent-runs/extractor-sessions")
|
||
|
||
|
||
class DispatchExtractionError(RuntimeError):
|
||
"""抽取派发失败:携带稳定错误码,编排方按失败关闭处理。"""
|
||
|
||
def __init__(self, code: str, message: str, *, details: Mapping[str, Any] | None = None):
|
||
super().__init__(message)
|
||
self.code = code
|
||
self.details = dict(details or {})
|
||
|
||
|
||
def extractor_session_paths(work_id: int, chapter_order: int) -> tuple[str, Path]:
|
||
"""一章一个抽取智能体会话:修复重派在同一会话内接续。"""
|
||
|
||
session_id = f"extractor-work{work_id}-ch{chapter_order}"
|
||
return session_id, SESSION_ROOT / session_id
|
||
|
||
|
||
def build_extraction_task_spec(
|
||
*,
|
||
work_id: int,
|
||
chapter_order: int,
|
||
title: str,
|
||
chapter_title: str,
|
||
existing_names: list[str],
|
||
body: str,
|
||
repair_reason: str | None = None,
|
||
) -> dict[str, Any]:
|
||
"""装配抽取任务包:全量正文进输入,合同条款进任务提示词。"""
|
||
|
||
task_prompt = (
|
||
f"你是抽取智能体。任务:从作品《{title}》第{chapter_order}章"
|
||
f"《{chapter_title or ''}》的已接受正文抽取作品私有知识草稿。"
|
||
"正文全文在冻结输入的 body 字段;证据必须是正文中的逐字连续片段,不能改写。"
|
||
"实体类型只能使用:character、location、faction、power_system、item、event;"
|
||
"立卡门槛:具名且有跨章复用或后续履约潜力,一次性龙套与一次性道具不列实体。"
|
||
"不要确认知识,不要补写正文没有的事实;低置信内容保留但在 brief/fields 中标注“?”。"
|
||
"可用只读工具核对既有实体(查重)与回读正文,但正文以输入 body 为准。"
|
||
"只输出一个 JSON 对象:entities(每项 type/name/brief/fields/evidence)、"
|
||
"relations(每项 source/target/type/description/evidence)、"
|
||
"state(currentSituation/characterStates/foreshadowing/handoff)。"
|
||
)
|
||
if repair_reason:
|
||
task_prompt += (
|
||
"\n【机械校验失败,允许一次修复】失败原因:"
|
||
+ repair_reason
|
||
+ "。只修正证据字段,使每条 evidence 都是正文中的逐字连续片段;"
|
||
"删除无法找到逐字证据的条目,不得新增条目、事实、关系或状态。仍只输出同一 JSON 对象。"
|
||
)
|
||
return {
|
||
"specVersion": "agent-task-v1",
|
||
"role": "extractor",
|
||
"taskPrompt": task_prompt,
|
||
"input": {
|
||
"workId": work_id,
|
||
"chapterOrder": chapter_order,
|
||
"title": title,
|
||
"chapterTitle": chapter_title or "",
|
||
"existingEntityNames": existing_names,
|
||
"body": body,
|
||
},
|
||
"outputSchema": EXTRACTION_OUTPUT_SCHEMA,
|
||
"outputSchemaId": "chapter-extraction-v1",
|
||
"toolAllowlist": list(EXTRACTION_TOOL_ALLOWLIST),
|
||
"maxDurationSeconds": EXTRACTION_MAX_DURATION_SECONDS,
|
||
}
|
||
|
||
|
||
def _read_dispatch_output(receipt: Mapping[str, Any], dispatch_run_id: str) -> Any:
|
||
"""结构化输出回读派发运行目录的 output.json,不另造权威。"""
|
||
|
||
output_file = Path(str(receipt.get("runDir") or "")) / "output.json"
|
||
try:
|
||
raw_text = output_file.read_text(encoding="utf-8")
|
||
except OSError as exc:
|
||
raise DispatchExtractionError(
|
||
"EXTRACTION_OUTPUT_MISSING",
|
||
f"抽取派发运行未落结构化输出:{output_file}",
|
||
details={"dispatchRunId": dispatch_run_id},
|
||
) from exc
|
||
try:
|
||
return json.loads(raw_text)
|
||
except ValueError:
|
||
# 模型偶尔包 markdown/解释;用既有提取器兜底,仍失败则失败关闭。
|
||
try:
|
||
return extract_json(raw_text)
|
||
except Exception as exc:
|
||
raise DispatchExtractionError(
|
||
"EXTRACTION_OUTPUT_INVALID",
|
||
f"抽取派发输出不是合法 JSON:{exc}",
|
||
details={"dispatchRunId": dispatch_run_id},
|
||
) from exc
|
||
|
||
|
||
def run_extraction_via_dispatch(
|
||
work_id: int,
|
||
chapter_order: int,
|
||
*,
|
||
repo_root: str | Path,
|
||
provider: str,
|
||
model: str,
|
||
thinking: str | None = None,
|
||
run_id: str | None = None,
|
||
spec_dir: str | Path | None = None,
|
||
launcher: Callable[..., Any] | None = None,
|
||
connect_factory: Callable[..., Any] | None = None,
|
||
) -> dict[str, Any]:
|
||
"""派发抽取智能体并落知识草稿;返回落库摘要。任何失败失败关闭。"""
|
||
|
||
for name in EXTRACTION_TOOL_ALLOWLIST:
|
||
if name not in TOOL_REGISTRY:
|
||
raise DispatchExtractionError(
|
||
"EXTRACTION_TOOL_UNREGISTERED", f"抽取白名单工具未登记:{name}"
|
||
)
|
||
|
||
active_run = run_id or new_run_id("extract-knowledge", work_id=work_id, target_chapter=chapter_order)
|
||
start_run(
|
||
run_id=active_run,
|
||
work_id=work_id,
|
||
target_chapter=chapter_order,
|
||
trigger_detail={"stage": "chapter-after-extraction", "mode": "dispatch"},
|
||
creator=ACTOR,
|
||
)
|
||
spec_root = Path(spec_dir) if spec_dir is not None else SCRIPT_DIR
|
||
spec_root.mkdir(parents=True, exist_ok=True)
|
||
session_id, session_dir = extractor_session_paths(work_id, chapter_order)
|
||
session_dir.mkdir(parents=True, mode=0o700, exist_ok=True)
|
||
policy = ExecutionPolicy(provider=provider, model=model, thinking=thinking)
|
||
|
||
try:
|
||
(title, chapter_id, chapter_title, _block_id, body), existing = _load_chapter(
|
||
work_id, chapter_order
|
||
)
|
||
existing_names = list(existing[:200])
|
||
|
||
def _dispatch_once(attempt: int, repair_reason: str | None) -> tuple[Any, Mapping[str, Any]]:
|
||
spec = build_extraction_task_spec(
|
||
work_id=work_id, chapter_order=chapter_order, title=title,
|
||
chapter_title=chapter_title or "", existing_names=existing_names,
|
||
body=body, repair_reason=repair_reason,
|
||
)
|
||
spec_file = spec_root / f"{active_run}-extractor-task-v{attempt}.json"
|
||
spec_file.write_text(json.dumps(spec, ensure_ascii=False, indent=1), encoding="utf-8")
|
||
dispatch_run_id = f"{active_run}-extractor-v{attempt}"
|
||
receipt, code = run_dispatch(
|
||
spec_file,
|
||
repo_root=repo_root,
|
||
policy=policy,
|
||
run_id=dispatch_run_id,
|
||
trigger_source="user",
|
||
trigger_detail={"stage": "extractor-dispatch", "productionRunId": active_run},
|
||
session_id=session_id,
|
||
session_dir=session_dir,
|
||
enable_read_tools=True,
|
||
launcher=launcher,
|
||
connect_factory=connect_factory,
|
||
)
|
||
if code != 0 or receipt.get("status") != "completed":
|
||
raise DispatchExtractionError(
|
||
str(receipt.get("errorCode") or "EXTRACTION_DISPATCH_FAILED"),
|
||
f"抽取智能体派发未成功:{receipt.get('error') or receipt.get('errorCode')}",
|
||
details={"dispatchRunId": dispatch_run_id, "exitCode": code},
|
||
)
|
||
return _read_dispatch_output(receipt, dispatch_run_id), receipt
|
||
|
||
raw_output, receipt = _dispatch_once(1, None)
|
||
try:
|
||
payload = normalize_extraction(raw_output, body)
|
||
except ExtractionContractError as first_error:
|
||
# 只允许一轮机械修复重派(证据绑定),对齐直调链语义。
|
||
raw_output, receipt = _dispatch_once(2, str(first_error))
|
||
try:
|
||
payload = normalize_extraction(raw_output, body)
|
||
except ExtractionContractError:
|
||
payload = salvage_extraction(raw_output, body)
|
||
|
||
model_ids = receipt.get("actualModelIds") or []
|
||
result = persist_extraction(
|
||
work_id, chapter_id, chapter_order, active_run, payload,
|
||
requested_model=f"{provider}/{model}",
|
||
actual_model=str(model_ids[-1]) if model_ids else "",
|
||
usage=dict(receipt.get("usage") or {}),
|
||
)
|
||
finish_run(active_run, "completed", creator=ACTOR,
|
||
trigger_detail={"stage": "chapter-after-extraction", "mode": "dispatch",
|
||
"drafts": len(result["draft_ids"])})
|
||
lesson = _propose_chapter_extract_lesson(
|
||
run_id=active_run, work_id=work_id, chapter_order=chapter_order,
|
||
draft_count=len(result["draft_ids"]), state_draft_id=result.get("state_draft_id"),
|
||
)
|
||
return {"run_id": active_run, **result, "lesson": lesson}
|
||
except BaseException as exc:
|
||
finish_run(active_run, "failed", creator=ACTOR,
|
||
trigger_detail={"stage": "chapter-after-extraction", "mode": "dispatch",
|
||
"error_type": type(exc).__name__})
|
||
try:
|
||
record_failure(
|
||
active_run,
|
||
sample_id=f"extract-ch{chapter_order}",
|
||
adapter_role="extractor",
|
||
caller="extract-knowledge-dispatch",
|
||
failure_type=type(exc).__name__,
|
||
)
|
||
except Exception:
|
||
pass
|
||
raise
|
||
|
||
|
||
__all__ = [
|
||
"EXTRACTION_MAX_DURATION_SECONDS",
|
||
"EXTRACTION_OUTPUT_SCHEMA",
|
||
"EXTRACTION_TOOL_ALLOWLIST",
|
||
"DispatchExtractionError",
|
||
"build_extraction_task_spec",
|
||
"extractor_session_paths",
|
||
"run_extraction_via_dispatch",
|
||
]
|