muse-agent-example/数据库/旧库迁移/知识记录分流.py
zizi cf0f4fb985 W20–W24:保存知识方法、迁移框架、审校修订与案例行为的集成成果
接续 88dd570,保存 W20–W24 已实现的共享接口、业务入口、迁移、工作台、测试与文档。
W20/W22/W23 保持 in_progress,W21/W24 保持 verified;此提交不宣称方法或规则正式启用、多轮返修、真实角色评测完成。

W25 新增实验、标定、逐调用交付与角色执行及其迁移/测试/索引留在实施工作树,原有私人和旧实现保留项不纳入。

验证:离线 571、前端 33 通过;PG 469 项通过、2 项浏览器未启用,2 项误带入的 W25 用例已移出本提交;最终任务与交付边界 37 项通过。make 检查、最终类型、84 项资源及 diff 检查通过。未重跑浏览器或 Pi 宿主,不以合成调用认证外部模型效果。
独立整体审查四维通过;证据保存在 R2-20260909/提交W20-W24。
2026-09-13 16:24:42 +08:00

509 lines
23 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""K01—K08 纯分流;只给转换建议,确认与当前可用性不由本函数授权。"""
from __future__ import annotations
from dataclasses import asdict, dataclass, field, replace
from typing import Any
from 关联已确认 import 生成关联建议, 请求历史关联
from 建立映射 import 读取JSON, 旧快照, 映射配置, 迁移错误
from 转换结构 import 取一致值, 转换动态字段
方法五型 = {"craft", "combat", "emotion", "scene_pattern", "trope"}
实体六型 = {"character", "location", "item", "faction", "power_system", "event"}
已知类型 = 方法五型 | 实体六型 | {"character_relation", "style", "pacing"}
抽取来源 = {"chapter_extract", "parse_book", "upgrade_book"}
历史状态 = {"rejected", "discarded", "superseded", "deleted"}
载荷键 = {
"type",
"型",
"name",
"名称",
"brief",
"一句话摘要",
"description",
"描述",
"evidence",
"证据",
"attributes",
"fields",
"字段",
"aliases",
"别名",
"source",
"target",
"来源",
"出处",
"实例",
"状态",
"目标库",
"可见范围",
"_work_id",
"章序",
"甲方",
"乙方",
"甲方draft",
"乙方draft",
"甲方名称",
"乙方名称",
"关系类型",
}
@dataclass(frozen=True)
class 分流结果:
source_key: str
source_hash: str
disposition: str
rule_id: str
reason_code: str
original: dict
owner: str = ""
type_id: str = ""
scope: str = ""
target_ref: str = ""
state: str = ""
content: dict = field(default_factory=dict)
references: tuple[dict, ...] = ()
conflicts: tuple[dict, ...] = ()
detail: str = ""
def 导出(self) -> dict:
return asdict(self)
def _载荷(行: dict, 表: str) -> dict:
原值 = 取一致值(行, ("draft_payload", "payload_json"), 默认=None)
if 原值 is None:
描述 = 行.get("description")
if 表 == "muse_knowledge_entity":
return {
"type": 行.get("entity_type"),
"name": 行.get("normalized_name"),
"brief": "" if 描述 is None else 描述,
"fields": 行.get("attributes", {}),
}
if 表 == "muse_knowledge_relation":
return {
"source": 行.get("source_entity_id"),
"target": 行.get("target_entity_id"),
"关系类型": 行.get("relation_type"),
"brief": "" if 描述 is None else 描述,
"fields": 行.get("attributes", {}),
}
return {}
try:
值 = 读取JSON(原值) if isinstance(原值, str) else 原值
except (ValueError, RecursionError) as 错:
raise 迁移错误("payload_invalid", "旧载荷不是有效JSON") from 错
if not isinstance(值, dict):
raise 迁移错误("payload_invalid", "旧载荷必须是对象")
return 值
def _标识字符串(值: Any) -> str:
if type(值) not in (str, int) or not str(值).strip():
raise 迁移错误("field_type_mismatch", "标识必须为非空字符串或整数,不能是bool")
return str(值)
def _名称引用(值: Any) -> str:
if not isinstance(值, str) or not 值.strip():
raise 迁移错误("field_type_mismatch", "端点名称必须是非空字符串")
return "name:" + 值.strip()
def _端点引用(值: dict, 字段: dict, 边: str, *, 行: dict) -> list[str]:
前缀 = "source" if 边 == "甲方" else "target"
引用 = []
for 容器, 键, 类别 in (
(行, 前缀 + "_entity_id", "entity"),
(值, 前缀, "typed"),
(值, 边 + "draft", "draft"),
(字段, 边 + "draft", "draft"),
(值, 边, "name"),
(字段, 边, "name"),
(值, 边 + "名称", "name"),
(字段, 边 + "名称", "name"),
):
if 键 not in 容器 or (容器 is 行 and 容器[键] is None):
continue
项 = 容器[键]
if 类别 == "typed":
if type(项) is int:
引用.append("entity:" + str(项))
continue
if isinstance(项, dict):
if set(项) - {"name", "名称"}:
raise 迁移错误("unknown_fields", "端点名称对象含未知字段")
if not 项:
raise 迁移错误("missing_endpoint", "端点名称对象为空")
项 = 取一致值(项, ("name", "名称"))
引用.append(_名称引用(项))
elif 类别 == "name":
引用.append(_名称引用(项))
else:
引用.append(类别 + ":" + _标识字符串(项))
return list(dict.fromkeys(引用))
def _解析端点(映射, 源, work_id, 载荷, 字段, 边, 行, 作品) -> str:
try:
return 映射.端点(源, work_id, _端点引用(载荷, 字段, 边, 行=行), 作品=作品)
except 迁移错误 as 错:
# 只暴露错误侧,不把旧字段值、原文或引用串拼进错误详情。
raise 迁移错误(错.code, "relation.source" if 边 == "甲方" else "relation.target") from None
def _来源提示(载荷, 行, 作品, 来源, *, 关系: bool) -> tuple[dict, Any]:
提示: dict = {}
章 = 载荷.get("章序")
if "章序" in 载荷:
_标识字符串(章)
if not 关系 and any(k in 载荷 for k in ("source", "来源", "出处")):
值 = 取一致值(载荷, ("source", "来源", "出处"))
if isinstance(值, dict):
if set(值) - {"workId", "chapter", "chapterId"}:
raise 迁移错误("unknown_fields", "旧来源对象含未知字段")
for v in 值.values():
_标识字符串(v)
if 作品["use"] != "author" or 行.get("source_type") != "chapter_extract":
raise 迁移错误("unknown_source", "该用途与来源不能解释章级来源声明")
if "workId" in 值 and (
str(值["workId"]) != str(行.get("work_id"))
or str(值["workId"]) != str(来源.get("work_id"))
):
raise 迁移错误("source_scope_conflict", "source.workId")
if "chapterId" in 值 and str(值["chapterId"]) != str(行.get("source_id")):
raise 迁移错误("source_scope_conflict", "source.chapterId")
if "chapter" in 值:
if 章 is not None and str(章) != str(值["chapter"]):
raise 迁移错误("scope_conflict", "source.chapter")
章 = 值["chapter"]
提示.update(值)
elif not isinstance(值, str):
raise 迁移错误("field_type_mismatch", "来源只接受对象或原字符串提示")
if any(k in 载荷 for k in ("evidence", "证据")):
证据 = 取一致值(载荷, ("evidence", "证据"))
if 证据 is not None and not isinstance(证据, str):
raise 迁移错误("field_type_mismatch", "证据必须是字符串或空值")
提示["evidence"] = 证据
return 提示, 章
def _来源引用(来源: dict, 章: Any) -> dict:
引用 = {"target_ref": 来源["target_ref"], "revision": 来源["revision"]}
if 章 is not None:
章节 = 来源.get("chapters", {})
定位 = 章节.get(str(章)) if isinstance(章节, dict) else None
if not isinstance(定位, str) or not 定位:
raise 迁移错误("missing_location", "旧章号缺少明确来源定位")
引用["locator"] = 定位
return 引用
def 分流知识记录(
快照: 旧快照, 映射: 映射配置, *, 已知源: dict[str, 旧快照] | None = None
) -> 分流结果:
行, 源 = 快照.原行, 快照.source
表 = 源.table.rsplit(".", 1)[-1]
规则 = ""
def 结果(归宿, 原因, **参数):
return 分流结果(源.记录键, 快照.源哈希, 归宿, 规则, 原因, 快照.导出(), **参数)
try:
if 源.system == "file":
from 读取旧文件 import 解码文件
from 转换正文 import 转换文件正文
规则 = "F01"
解码文件(快照)
路由 = 映射.文件路由(快照)
if 路由["route"] == "archive":
return 结果("历史保留", "file_archive", state="historical")
规则 = "F02"
作品 = 映射.作品(源, 路由["legacy_work_id"], "author")
return 结果(
"待转换",
"body_candidate",
owner="B05",
type_id="document",
scope=作品["target_ref"],
target_ref=路由["target"]["chapter_id"],
state="pending",
content=转换文件正文(快照, 路由),
)
状态 = 行.get("status", "pending")
if not isinstance(状态, str):
raise 迁移错误("state_conflict", "旧状态类型无法解释")
if "deleted" in 行 and (
not isinstance(行["deleted"], (str, int, bool))
or 行["deleted"] not in (True, False, 0, 1, "true", "false", "0", "1")
):
raise 迁移错误("state_conflict", "删除标记无法解释,不能当作未删除")
if 行.get("deleted") in (True, 1, "true", "1") or 状态 in 历史状态:
return 结果("历史保留", "inactive_history", state=状态)
来源状态, 来源策略 = 行.get("source_status"), 行.get("source_action_policy")
if 来源状态 is not None and not isinstance(来源状态, str):
raise 迁移错误("permission_conflict", "来源状态类型无法解释")
if 来源策略 is not None and not isinstance(来源策略, str):
raise 迁移错误("permission_conflict", "来源授权类型无法解释")
if 来源状态 in {"withdrawn", "revoked", "deleted"} or 来源策略 in {"denied", "blocked"}:
return 结果("历史保留", "source_revoked", state=状态)
if 请求历史关联(快照, 映射):
规则 = "K06"
return 结果(
"待转换", "linked_existing_confirmation", **生成关联建议(快照, 映射, 已知源 or {})
)
if 表 in {"example_run", "example_run_receipt", "example_llm_call", "runs", "events"}:
return 结果("历史保留", "legacy_runtime", owner="S02", state=状态)
if 表 in {"reviews", "revisions", "example_user_decision", "example_candidate_cas"}:
return 结果("历史保留", "legacy_decision", owner="S01", state=状态)
延后 = {
"example_quality_result": "deferred_W22_23",
"example_ai_flavor_case": "deferred_W24",
"example_candidate": "deferred_W25",
"example_lesson": "deferred_W26_27",
"lessons": "deferred_W26_27",
}
if 表 in 延后:
return 结果("隔离", 延后[表], state=状态)
if 表 not in {
"muse_knowledge_draft",
"muse_knowledge_entity",
"muse_knowledge_relation",
"cards",
"snapshots",
"example_knowledge_embedding",
}:
return 结果("隔离", "unknown_table")
载荷 = _载荷(行, 表)
work_id = 行.get("work_id")
if work_id is None:
raise 迁移错误("unknown_use", "旧作品身份缺失,不能猜测作用域")
作品 = 映射.作品(源, work_id, 映射.选择用途(快照))
owner = {"author": "B02", "reference": "B03", "method": "B04"}[作品["use"]]
if 表 == "example_knowledge_embedding" or (表 == "cards" and 行.get("kind") == "embedding"):
规则 = "K08"
指针 = [("draft:" + str(行["draft_id"]))] if 行.get("draft_id") is not None else []
if 行.get("entity_id") is not None:
指针.append("entity:" + str(行["entity_id"]))
指向 = 映射.端点(源, work_id, 指针, 作品=作品)
return 结果(
"待转换",
"index_identity_only",
owner="B09",
type_id="embedding",
scope=作品["target_ref"],
target_ref=快照.建议身份(
"B09", "embedding", scope=作品["target_ref"], author_id=作品["author_id"]
),
state="historical",
references=({"target_ref": 指向},),
)
if (
行.get("source_status") not in {"active", "authorized"}
or 行.get("source_action_policy") != "allowed"
):
raise 迁移错误("permission_conflict", "来源当前资格不明确或不允许")
if set(载荷) - 载荷键:
raise 迁移错误(
"unknown_fields", "未知载荷字段:" + "/".join(sorted(set(载荷) - 载荷键))
)
if "_work_id" in 载荷 and str(载荷["_work_id"]) != str(work_id):
raise 迁移错误("scope_conflict", "载荷与表级作品身份不同")
类型 = 取一致值(载荷, ("type", "型"), 原因="type_conflict")
if 类型 is not None and not isinstance(类型, str):
raise 迁移错误("type_conflict", "类型必须是明确字符串")
草稿型 = 行.get("draft_type", "relation" if 表 == "muse_knowledge_relation" else "entity")
if 草稿型 not in ("entity", "relation"):
raise 迁移错误("unknown_draft_type", "未登记的草稿生命周期")
关系 = 草稿型 == "relation" or 类型 == "character_relation"
关系类型 = None
if 关系:
关系字段 = 取一致值(载荷, ("fields", "字段", "attributes"), 默认={})
if not isinstance(关系字段, dict):
raise 迁移错误("field_type_mismatch", "关系字段不是对象")
候选关系 = {}
if "关系类型" in 载荷:
候选关系["顶层"] = 载荷["关系类型"]
if "关系类型" in 关系字段:
候选关系["字段"] = 关系字段["关系类型"]
if 草稿型 == "relation" and 类型 not in {None, "character_relation"}:
候选关系["旧type"] = 类型
关系类型 = 取一致值(候选关系, tuple(候选关系))
类型 = "character_relation"
else:
别名 = 映射.值.get("type_aliases", {})
类型 = 别名.get(类型, 类型) if isinstance(类型, str) else 类型
if not isinstance(类型, str) or 类型 not in 已知类型:
raise 迁移错误("unknown_type", "没有可解释的类型与生命周期")
目标库 = 载荷.get("目标库")
if 目标库 == "公共范式库" and 作品["use"] != "method":
raise 迁移错误("scope_conflict", "公共描述与明确作品用途冲突")
if 目标库 == "本书作品库" and 作品["use"] == "method":
raise 迁移错误("scope_conflict", "本书描述不能替代方法库作用域")
来源类型 = 行.get("source_type", "")
if not isinstance(来源类型, str) or not 来源类型:
raise 迁移错误("unknown_source", "旧来源类型缺失或类型不符")
来源 = 映射.来源(源, 来源类型, 行.get("source_id"))
if 作品["use"] != "method" and str(来源.get("work_id")) != str(work_id):
raise 迁移错误("source_scope_conflict", "来源与作品没有明确一致的归属")
if str(work_id) == "0":
if not (
草稿型 == "entity"
and 来源类型 == "parse_book"
and 类型 in 方法五型
and 作品["use"] == "method"
and 目标库 in (None, "公共范式库")
):
raise 迁移错误("zero_work_not_pattern", "0只承接明确的公共五型范式")
规则 = "K01"
elif 关系:
if 作品["use"] == "method":
raise 迁移错误("unknown_use", "关系卡不能凭类型自动进入方法库")
规则 = "K05" if 草稿型 == "relation" else "K04"
elif 类型 in 实体六型 and 作品["use"] in {"author", "reference"}:
if 来源类型 not in 抽取来源:
raise 迁移错误("unknown_source", "未登记的抽取来源")
规则 = "K02" if 作品["use"] == "author" else "K03"
elif 类型 in {"style", "pacing"} or 作品["use"] == "method":
规则 = "K07"
if 作品["use"] == "author":
owner = "B06" if 类型 == "style" else "B01"
else:
raise 迁移错误("unknown_use", "该类型与用途没有确定分流规则")
字段 = 取一致值(载荷, ("fields", "字段", "attributes"), 默认={})
if not isinstance(字段, dict):
raise 迁移错误("field_type_mismatch", "字段不是对象")
关系内容 = {}
if 关系:
if not isinstance(关系类型, str) or not 关系类型:
raise 迁移错误("relation_type_missing", "关系类型缺失")
# 字段内端点也先经过同一解析器,不能被通用结构错误遮掉错误侧。
关系内容 = {
"source": _解析端点(映射, 源, work_id, 载荷, 字段, "甲方", 行, 作品),
"target": _解析端点(映射, 源, work_id, 载荷, 字段, "乙方", 行, 作品),
"relation_type": 关系类型,
}
内容: dict[str, Any] = {
"name": 取一致值(载荷, ("name", "名称"), 默认=""),
"brief": 取一致值(载荷, ("brief", "一句话摘要", "description", "描述"), 默认=""),
"fields": 转换动态字段(字段, 类型, 映射.值.get("schemas", {})),
}
if not isinstance(内容["name"], str) or (not 关系 and not 内容["name"].strip()):
raise 迁移错误("missing_required", "对象名称缺失或类型不符")
if not isinstance(内容["brief"], str):
raise 迁移错误("field_type_mismatch", "描述类型不符")
别名 = 取一致值(载荷, ("aliases", "别名"), 默认=[])
if not isinstance(别名, list) or not all(isinstance(v, str) and v.strip() for v in 别名):
raise 迁移错误("field_type_mismatch", "别名必须是明确字符串列表")
内容["aliases"] = list(dict.fromkeys(别名))
内容.update(关系内容)
提示, 章 = _来源提示(载荷, 行, 作品, 来源, 关系=关系)
if 提示:
内容["source_hint"] = 提示
引用 = [_来源引用(来源, 章)]
实例 = 载荷.get("实例", [])
if not isinstance(实例, list):
raise 迁移错误("field_type_mismatch", "实例必须是列表")
for 项 in 实例:
if not isinstance(项, dict) or not isinstance(项.get("书"), str):
raise 迁移错误("missing_source", "实例需要明确书源")
实例源 = 映射.实例来源(源, 项["书"])
章 = 取一致值(项, ("章", "章序", "章节"))
引用.append({**_来源引用(实例源, 章), "instance": 项})
target_ref = 快照.建议身份(
owner, 类型, scope=作品["target_ref"], author_id=作品["author_id"]
)
新状态 = "pending"
镜像 = 表 in {"cards", "snapshots"}
if (
状态 == "confirmed"
or 镜像
or 表 in {"muse_knowledge_entity", "muse_knowledge_relation"}
):
规则 = "K08" if 镜像 else "K06"
前缀 = "relation:" if 关系 else "entity:"
目标引用 = [
前缀 + str(行[k])
for k in ("entity_id", "target_object_id")
if 行.get(k) is not None
]
if 表 in {"muse_knowledge_entity", "muse_knowledge_relation"}:
目标引用.append(前缀 + 源.id)
if 镜像:
目标引用 = [("card:" if 表 == "cards" else "snapshot:") + 源.id]
确认 = 映射.确认(快照, 目标引用, 作品=作品)
if 确认 is None:
return 结果(
"历史保留",
"missing_confirmation",
owner=owner,
type_id=类型,
scope=作品["target_ref"],
state="migration_review_required",
content=内容,
)
target_ref, 新状态 = (
确认["target_ref"],
"mirror_history" if 镜像 else "confirmed_history",
)
引用.append({"decision_ref": 确认["decision_ref"]})
elif 状态 != "pending" or 载荷.get("状态", "草稿") not in ("草稿", "pending"):
raise 迁移错误("state_conflict", "表级状态与载荷状态不一致")
return 结果(
"待转换",
"classified",
owner=owner,
type_id=类型,
scope=作品["target_ref"],
target_ref=target_ref,
state=新状态,
content=内容,
references=tuple(引用),
)
except 迁移错误 as 错:
return 结果("隔离", 错.code, detail=错.detail)
def 分流批次(
快照: list[旧快照], 映射: 映射配置, *, 已存正式源: dict[str, 旧快照] | None = None
) -> list[分流结果]:
哈希: dict[str, set[str]] = {}
for 项 in 快照:
哈希.setdefault(项.source.记录键, set()).add(项.源哈希)
全源 = {**(已存正式源 or {}), **{项.source.记录键: 项 for 项 in 快照}}
结果 = [分流知识记录(项, 映射, 已知源=全源) for 项 in 快照]
对象目标: dict[str, set[tuple[str, str, str]]] = {}
for 源, 项 in zip(快照, 结果, strict=True):
if 项.disposition == "待转换" and 项.target_ref:
对象目标.setdefault(源.source.对象键, set()).add((项.owner, 项.scope, 项.target_ref))
已核对 = []
# 整批先判断记录漂移和对象归属,不能按输入顺序抢占或分裂同一旧对象。
for 源, 项 in zip(快照, 结果, strict=True):
目标 = sorted(对象目标.get(源.source.对象键, set()))
if len(哈希[项.source_key]) > 1:
代码, 说明 = "drift", "同批相同源身份存在不同内容"
elif len(目标) > 1:
代码, 说明 = "identity_conflict", "同一旧对象的不同版本指向多个目标或作用域"
else:
已核对.append(项)
continue
已核对.append(
replace(
项,
disposition="隔离",
reason_code=代码,
owner="",
target_ref="",
state="",
content={},
references=(),
detail=说明,
conflicts=tuple({"owner": o, "scope": s, "target_ref": t} for o, s, t in 目标),
)
)
return 已核对