muse-agent-example/tests/迁移/test_混合知识分流.py
zizi cf0f4fb985 W20–W24:保存知识方法、迁移框架、审校修订与案例行为的集成成果
接续 88dd570,保存 W20–W24 已实现的共享接口、业务入口、迁移、工作台、测试与文档。
W20/W22/W23 保持 in_progress,W21/W24 保持 verified;此提交不宣称方法或规则正式启用、多轮返修、真实角色评测完成。

W25 新增实验、标定、逐调用交付与角色执行及其迁移/测试/索引留在实施工作树,原有私人和旧实现保留项不纳入。

验证:离线 571、前端 33 通过;PG 469 项通过、2 项浏览器未启用,2 项误带入的 W25 用例已移出本提交;最终任务与交付边界 37 项通过。make 检查、最终类型、84 项资源及 diff 检查通过。未重跑浏览器或 Pi 宿主,不以合成调用认证外部模型效果。
独立整体审查四维通过;证据保存在 R2-20260909/提交W20-W24。
2026-09-13 16:24:42 +08:00

550 lines
22 KiB
Python

"""合成快照的纯分流证据;不代替清单中 TC-knowledge-* 的隔离 PG 导入验收。"""
from __future__ import annotations
import copy
import json
import subprocess
import sys
from pathlib import Path
import pytest
from 建立映射 import 稳定JSON, 旧快照, 映射配置, 身份映射, 迁移错误
from 知识记录分流 import 分流批次, 分流知识记录
@pytest.fixture
def 映射原值():
作品 = [
{"legacy_id": 0, "use": "method", "target_ref": "author-library:a"},
{"legacy_id": 7, "use": "author", "target_ref": "author-work:7"},
{"legacy_id": 8, "use": "reference", "target_ref": "reference:8"},
{"legacy_id": 9, "use": "method", "target_ref": "author-library:a"},
]
来源 = [
{
"source_type": "parse_book",
"legacy_id": 8,
"work_id": 8,
"target_ref": "source:8",
"revision": "source-v1",
"aliases": ["甲书"],
"chapters": {"12": "source:8:chapter12"},
},
{
"source_type": "parse_book",
"legacy_id": 10,
"work_id": 10,
"target_ref": "source:10",
"revision": "source-v2",
"aliases": ["乙书"],
"chapters": {"12": "source:10:chapter12"},
},
{
"source_type": "chapter_extract",
"legacy_id": 70,
"work_id": 7,
"target_ref": "chapter:70",
"revision": "body-v1",
},
]
对象 = [
{"work_id": 7, "refs": ["entity:11", "draft:101", "name:甲"], "target_ref": "world:甲"},
{"work_id": 7, "refs": ["entity:12", "draft:102", "name:乙"], "target_ref": "world:乙"},
]
通用 = {"system": "pg", "database": "synthetic-legacy"}
def 模式(属性):
return {"type": "object", "properties": 属性, "additionalProperties": False}
return {
"works": [{**通用, "author_id": "author:a", **r} for r in 作品],
"sources": [{**通用, **r} for r in 来源],
"objects": [{**通用, "use": "author", "scope": "author-work:7", **r} for r in 对象],
"schemas": {
"character": 模式({"性格": {"type": "string"}}),
"location": 模式({}),
"craft": 模式({"原理": {"type": "string"}}),
"character_relation": 模式(
{k: {"type": "string"} for k in ("甲方", "乙方", "关系类型")}
),
"style": 模式({}),
"pacing": 模式({}),
},
}
def _快照(载荷=None, *, 源=None, **覆盖):
原行 = {
"work_id": 7,
"status": "pending",
"draft_type": "entity",
"source_type": "chapter_extract",
"source_id": 70,
"source_status": "active",
"source_action_policy": "allowed",
"draft_payload": 载荷
if 载荷 is not None
else {"type": "character", "name": "甲", "fields": {"性格": "果断"}},
**覆盖,
}
return 旧快照.从载荷(
{
"source": {
"system": "pg",
"database": "synthetic-legacy",
"table": "muse_knowledge_draft",
"id": "1",
"revision": "1",
**(源 or {}),
},
"record": 原行,
}
)
def _判定(原值, 快照):
结果 = 分流知识记录(快照, 映射配置(原值))
assert 结果.original == 快照.导出()
assert 结果.source_hash == 快照.源哈希
return 结果
def _公共(**覆盖):
return _快照(
{"型": "craft", "名称": "合成铺垫法", "字段": {"原理": "先建立可核对的条件"}, **覆盖},
work_id=0,
source_type="parse_book",
source_id=8,
)
def test_公共五型保持候选与明确作者库__21a001(映射原值):
结果 = _判定(映射原值, _公共())
assert (结果.rule_id, 结果.owner, 结果.state) == ("K01", "B04", "pending")
assert 结果.scope == "author-library:a" and 结果.type_id == "craft"
assert 结果.references == ({"target_ref": "source:8", "revision": "source-v1"},)
def test_作者实体字段与来源不丢失__21a002(映射原值):
结果 = _判定(映射原值, _快照())
assert (结果.rule_id, 结果.owner, 结果.state) == ("K02", "B02", "pending")
assert 结果.content["fields"] == {"性格": "果断"}
assert 结果.references[0]["target_ref"] == "chapter:70"
重复用途 = {**映射原值["works"][1], "use": "reference", "target_ref": "reference:7"}
映射原值["works"].append(重复用途)
assert _判定(映射原值, _快照()).reason_code == "unknown_use"
快照 = _快照()
映射原值["routes"] = [
{
"system": "pg",
"database": "synthetic-legacy",
"source_key": 快照.source.记录键,
"source_hash": 快照.源哈希,
"use": "author",
}
]
assert _判定(映射原值, 快照).owner == "B02"
def test_同模具参考实体不进入作者事实__21a003(映射原值):
结果 = _判定(映射原值, _快照(work_id=8, source_type="parse_book", source_id=8))
assert (结果.rule_id, 结果.owner, 结果.scope) == ("K03", "B03", "reference:8")
assert _判定(映射原值, _快照(work_id=8)).reason_code == "source_scope_conflict"
def test_实体形关系和显式关系共享身份与端点__21a004(映射原值):
a = _快照(
{
"type": "character_relation",
"name": "甲乙师徒",
"fields": {"甲方": "甲", "乙方": "乙", "关系类型": "师徒"},
}
)
b = _快照(
{"type": "师徒", "source": 11, "target": 12}, draft_type="relation", 源={"revision": "2"}
)
x, y = _判定(映射原值, a), _判定(映射原值, b)
assert (x.rule_id, y.rule_id) == ("K04", "K05")
assert x.target_ref == y.target_ref
for r in (x, y):
assert r.owner == "B02" and r.type_id == "character_relation"
assert (r.content["source"], r.content["target"], r.content["relation_type"]) == (
"world:甲",
"world:乙",
"师徒",
)
@pytest.mark.parametrize(
"问题,原因",
[
("缺失", "missing_endpoint"),
("同名", "ambiguous_endpoint"),
("编号矛盾", "endpoint_conflict"),
],
)
def test_关系端点缺失或多义不得猜测__21a005(映射原值, 问题, 原因):
载荷 = {"type": "师徒", "甲方": "甲", "乙方": "乙"}
if 问题 == "缺失":
映射原值["objects"].pop()
elif 问题 == "同名":
映射原值["objects"].append({**映射原值["objects"][0], "target_ref": "world:重名甲"})
else:
载荷["source"] = 12
结果 = _判定(映射原值, _快照(载荷, draft_type="relation"))
assert 结果.disposition == "隔离" and 结果.reason_code == 原因
assert 结果.target_ref == ""
def test_确认历史核对目标并按源身份幂等__21a006(映射原值):
快照 = _快照(status="confirmed", entity_id=11)
确认 = {
"system": "pg",
"database": "synthetic-legacy",
"source_key": 快照.source.记录键,
"source_hash": 快照.源哈希,
"target_ref": "world:甲",
"decision_ref": "legacy-decision:1",
}
映射原值["confirmations"] = [确认]
结果 = _判定(映射原值, 快照)
assert 结果.state == "confirmed_history" and 结果.target_ref == "world:甲"
台账 = 身份映射()
assert 台账.登记(快照, 结果.target_ref, 结果.rule_id, scope=结果.scope) is True
assert 台账.登记(快照, 结果.target_ref, 结果.rule_id, scope=结果.scope) is False
assert len(台账.记录) == 1
坏 = _快照(status="confirmed", entity_id=99)
确认.update(source_hash=坏.源哈希)
assert _判定(映射原值, 坏).state == "migration_review_required"
assert _判定(映射原值, 坏).target_ref == ""
def test_双键冲突不按优先级吞值__21a007(映射原值):
快照 = _公共(type="character")
结果 = _判定(映射原值, 快照)
assert 结果.disposition == "隔离" and 结果.reason_code == "type_conflict"
assert 结果.original["record"]["draft_payload"]["型"] == "craft"
assert 结果.original["record"]["draft_payload"]["type"] == "character"
def test_零哨兵不能成为作品或普通公共实体__21a008(映射原值):
结果 = _判定(
映射原值,
_快照(
{"type": "location", "name": "旧城", "fields": {}},
work_id=0,
source_type="parse_book",
source_id=8,
),
)
assert 结果.reason_code == "zero_work_not_pattern" and not 结果.target_ref
def test_跨书实例分别绑定来源和章定位__21a009(映射原值):
快照 = _公共(
实例=[
{"书": "甲书", "章": 12, "引文": "合成甲"},
{"书": "乙书", "章": 12, "引文": "合成乙"},
]
)
结果 = _判定(映射原值, 快照)
assert [r["locator"] for r in 结果.references[1:]] == [
"source:8:chapter12",
"source:10:chapter12",
]
del 映射原值["sources"][1]["chapters"]["12"]
assert _判定(映射原值, 快照).reason_code == "missing_location"
@pytest.mark.parametrize(
"覆盖",
[
{"deleted": True},
{"status": "rejected"},
{"status": "discarded"},
{"status": "superseded"},
{"source_status": "withdrawn"},
{"source_action_policy": "denied"},
],
)
def test_删除拒绝和撤权不复活__21a00a(映射原值, 覆盖):
结果 = _判定(映射原值, _快照(**覆盖))
assert 结果.disposition == "历史保留" and not 结果.target_ref and not 结果.content
@pytest.mark.parametrize(
"字段,原因",
[
({"陌生字段": {"原值": [1, 2]}}, "unknown_fields"),
({"性格": 3}, "field_type_mismatch"),
({}, "missing_required"),
],
)
def test_未知或不合法动态字段原样隔离__21a00b(映射原值, 字段, 原因):
映射原值["schemas"]["character"]["required"] = ["性格"]
快照 = _快照({"type": "character", "name": "甲", "fields": 字段})
结果 = _判定(映射原值, 快照)
assert 结果.reason_code == 原因 and 结果.disposition == "隔离"
assert 结果.original["record"]["draft_payload"]["fields"] == 字段
def test_来源命名空间区分同号对象__21a00c(映射原值):
for 类别 in ("works", "sources", "objects"):
映射原值[类别] += [{**r, "system": "sqlite"} for r in 映射原值[类别]]
x = _判定(映射原值, _快照())
y = _判定(映射原值, _快照(源={"system": "sqlite"}))
assert x.source_key != y.source_key and x.target_ref != y.target_ref
assert x.source_hash == y.source_hash
未知 = _判定(映射原值, _快照(源={"database": "another-database"}))
assert 未知.reason_code == "unknown_use"
@pytest.mark.parametrize(
"类型,作品号,owner",
[
("style", 7, "B06"),
("pacing", 7, "B01"),
("style", 8, "B03"),
("pacing", 8, "B03"),
("character", 9, "B04"),
],
)
def test_通用写法和本书风格按用途分流__21a00d(映射原值, 类型, 作品号, owner):
来源号, 来源型 = (70, "chapter_extract") if 作品号 == 7 else (8, "parse_book")
结果 = _判定(
映射原值,
_快照(
{"type": 类型, "name": "合成写法", "fields": {}},
work_id=作品号,
source_id=来源号,
source_type=来源型,
),
)
assert 结果.rule_id == "K07" and 结果.owner == owner and 结果.state == "pending"
def test_向量只承接指针且内容镜像不复制候选__21a00e(映射原值):
向量 = _快照(
源={"table": "example_knowledge_embedding"},
entity_id=11,
embedding=[0.2, 0.3],
model="synthetic-v1",
dimensions=2,
)
结果 = _判定(映射原值, 向量)
assert (结果.rule_id, 结果.owner, 结果.state) == ("K08", "B09", "historical")
assert 结果.content == {} and 结果.references == ({"target_ref": "world:甲"},)
卡 = _快照(源={"table": "cards"})
assert _判定(映射原值, 卡).state == "migration_review_required"
映射原值["objects"][0]["refs"].append("card:1")
映射原值["confirmations"] = [
{
"system": "pg",
"database": "synthetic-legacy",
"source_key": 卡.source.记录键,
"source_hash": 卡.源哈希,
"target_ref": "world:甲",
"decision_ref": "legacy-decision:1",
}
]
镜像 = _判定(映射原值, 卡)
assert (
镜像.rule_id == "K08" and 镜像.state == "mirror_history" and 镜像.target_ref == "world:甲"
)
def test_不可变快照和相同身份漂移__21a00f(映射原值):
原 = _快照()
值 = 原.原行
值["draft_payload"]["name"] = "不可污染冻结原文"
assert 原.原行["draft_payload"]["name"] == "甲"
坏 = _快照({"type": "character", "name": "乙", "fields": {"性格": "果断"}})
台账 = 身份映射()
台账.登记(原, "world:甲", "K02", scope="author-work:7")
with pytest.raises(迁移错误, match="drift"):
台账.登记(坏, "world:甲", "K02", scope="author-work:7")
assert all(
r.disposition == "隔离" and r.reason_code == "drift"
for r in 分流批次([原, 坏], 映射配置(映射原值))
)
with pytest.raises(迁移错误, match="snapshot_hash_mismatch"):
旧快照.从载荷({**原.导出(), "source_hash": "0" * 64})
新版本 = _快照(源={"revision": "2"})
assert 原.建议身份(
"B02", "character", scope="author-work:7", author_id="author:a"
) == 新版本.建议身份("B02", "character", scope="author-work:7", author_id="author:a")
with pytest.raises(迁移错误, match="identity_conflict"):
台账.登记(新版本, "world:另一个", "K02", scope="author-work:7")
def test_坏载荷与不可解释结构不崩溃不联网__21a010(映射原值):
for 载荷 in ('{"type":"character","type":"craft"}', "null", '{"type": NaN}'):
assert _判定(映射原值, _快照(载荷)).disposition == "隔离"
for 引用 in ("https://example.invalid/schema", "#/$defs/missing"):
映射原值["schemas"]["character"]["properties"]["性格"] = {"$ref": 引用}
assert _判定(映射原值, _快照()).reason_code == "schema_invalid"
assert _判定(映射原值, _快照(source_status=[])).reason_code == "permission_conflict"
for 坏标记 in (None, [], 2, "未知"):
assert _判定(映射原值, _快照(deleted=坏标记)).reason_code == "state_conflict"
assert _判定(映射原值, _快照(draft_type="unknown")).reason_code == "unknown_draft_type"
def test_命令只生成计划并保护输入和已有输出__21a011(映射原值, tmp_path):
输入, 映射, 输出 = [tmp_path / n for n in ("快照.json", "映射.json", "计划.json")]
输入.write_text(稳定JSON([_公共().导出()]), encoding="utf-8")
映射.write_text(稳定JSON(映射原值), encoding="utf-8")
原文 = 输入.read_bytes()
脚本 = Path(__file__).resolve().parents[2] / "数据库/旧库迁移/入口.py"
命令 = [
sys.executable,
str(脚本),
"分流",
"--快照",
str(输入),
"--映射",
str(映射),
"--输出",
str(输出),
]
for _ in range(2):
进程 = subprocess.run(命令, capture_output=True, text=True, timeout=15)
assert 进程.returncode == 0, 进程.stderr
assert "合成铺垫法" not in 进程.stdout
计划 = json.loads(输出.read_text())
assert 计划["stage"] == "classification_only" and 计划["dispositions"] == {"待转换": 1}
assert 计划["results"][0]["original"] == _公共().导出()
assert (输出.stat().st_mode & 0o777) == 0o600
assert subprocess.run([*命令[:-1], str(输入)], capture_output=True).returncode == 1
输出.write_text("已有产物", encoding="utf-8")
assert subprocess.run(命令, capture_output=True).returncode == 1
assert 输出.read_text() == "已有产物" and 输入.read_bytes() == 原文
assert not list(tmp_path.glob(".迁移分流-*"))
def test_运行历史不成为新任务且延后语义隔离__21a012(映射原值):
运行 = _判定(映射原值, _快照(源={"table": "example_run"}, status="completed"))
assert 运行.disposition == "历史保留" and 运行.owner == "S02" and not 运行.target_ref
经验 = _判定(映射原值, _快照(源={"table": "example_lesson"}))
assert 经验.disposition == "隔离" and 经验.reason_code == "deferred_W26_27"
无映射 = copy.deepcopy(映射原值)
无映射["works"] = []
assert _判定(无映射, _快照()).reason_code == "unknown_use"
def test_双用途端点和确认不能串入作者作用域__21a013(映射原值):
快照 = _快照({"type": "师徒", "source": 11, "target": 12}, draft_type="relation")
映射原值["works"].append(
{**映射原值["works"][1], "use": "reference", "target_ref": "reference:7"}
)
映射原值["routes"] = [
{
"system": "pg",
"database": "synthetic-legacy",
"source_key": 快照.source.记录键,
"source_hash": 快照.源哈希,
"use": "reference",
}
]
assert _判定(映射原值, 快照).reason_code == "missing_endpoint"
映射原值["objects"] += [
{
**r,
"use": "reference",
"scope": "reference:7",
"target_ref": "reference:" + r["target_ref"],
}
for r in 映射原值["objects"]
]
关系 = _判定(映射原值, 快照)
assert 关系.owner == "B03" and 关系.content["source"] == "reference:world:甲"
作者映射 = copy.deepcopy(映射原值)
作者映射["routes"][0]["use"] = "author"
assert _判定(作者映射, 快照).target_ref != 关系.target_ref
映射原值["routes"][0]["source_hash"] = "0" * 64
assert _判定(映射原值, 快照).reason_code == "unknown_use"
已确认 = _快照(status="confirmed", entity_id=11)
映射原值["routes"][0]["source_hash"] = 已确认.源哈希
映射原值["confirmations"] = [
{
"system": "pg",
"database": "synthetic-legacy",
"source_key": 已确认.source.记录键,
"source_hash": 已确认.源哈希,
"target_ref": "world:甲",
"decision_ref": "legacy-decision:1",
}
]
assert _判定(映射原值, 已确认).state == "migration_review_required"
映射原值["confirmations"][0]["target_ref"] = "reference:world:甲"
assert _判定(映射原值, 已确认).state == "confirmed_history"
def test_缺来源编号和错误映射不能自洽补造__21a014(映射原值):
del 映射原值["sources"][2]["legacy_id"]
assert _判定(映射原值, _快照(source_id=None)).reason_code == "missing_source"
assert _判定(映射原值, _快照(source_type=[])).reason_code == "unknown_source"
with pytest.raises(迁移错误, match="source_invalid"):
_快照(id=2)
台账 = 身份映射()
台账.登记(_快照(), "world:甲", "K02", scope="author-work:7")
with pytest.raises(迁移错误, match="drift"):
台账.登记(_快照(), "world:甲", "K02", scope="author-work:另一个")
@pytest.mark.parametrize("表", ["muse_knowledge_entity", "muse_knowledge_relation"])
@pytest.mark.parametrize("描述", [" 保留原描述\n第二行。", None, 0])
def test_正式表描述进入归一结果且保全其他原列__21b001(映射原值, 表, 描述):
数据 = _快照(源={"table": 表}).导出()
数据.pop("source_hash")
行 = 数据["record"]
行.pop("draft_payload")
行.pop("draft_type")
行.update(
description=描述, confidence=0.91, scope="旧作用域", lineage_payload={"来源轨迹": [8, 10]}
)
if 表 == "muse_knowledge_entity":
行.update(entity_type="character", normalized_name="甲", attributes={"性格": "果断"})
else:
行.update(source_entity_id=11, target_entity_id=12, relation_type="师徒", attributes={})
结果 = _判定(映射原值, 旧快照.从载荷(数据))
if 描述 == 0:
assert 结果.disposition == "隔离" and 结果.reason_code == "field_type_mismatch"
else:
assert 结果.content["brief"] == ("" if 描述 is None else 描述)
assert 结果.state == "migration_review_required"
assert 结果.original["record"]["lineage_payload"] == {"来源轨迹": [8, 10]}
assert 结果.original["record"]["confidence"] == 0.91
def test_同旧对象跨版本异目标在分流阶段全部隔离__21b002(映射原值):
一 = _快照(status="confirmed", entity_id=11)
二 = _快照(源={"revision": "2"}, status="confirmed", entity_id=12)
映射原值["confirmations"] = [
{
"system": "pg",
"database": "synthetic-legacy",
"source_key": x.source.记录键,
"source_hash": x.源哈希,
"target_ref": t,
"decision_ref": "legacy-decision:" + x.source.revision,
}
for x, t in ((一, "world:甲"), (二, "world:乙"))
]
配置 = 映射配置(映射原值)
# 两条单独输入均能关联声明;合批必须先隔离,不能等导入顺序抢占。
assert all(分流知识记录(x, 配置).state == "confirmed_history" for x in (一, 二))
结果 = 分流批次([一, 二], 配置)
assert len(结果) == 2
for r in 结果:
assert r.disposition == "隔离" and r.reason_code == "identity_conflict"
assert r.target_ref == "" and r.content == {} and r.references == ()
assert {x["target_ref"] for x in r.conflicts} == {"world:甲", "world:乙"}
assert {r.source_key for r in 结果} == {一.source.记录键, 二.source.记录键}
映射原值["confirmations"][1]["target_ref"] = "world:甲"
二同 = _快照(源={"revision": "2"}, status="confirmed", entity_id=11)
映射原值["confirmations"][1]["source_hash"] = 二同.源哈希
重复 = 分流批次([一, 二同], 映射配置(映射原值))
assert all(r.disposition == "待转换" and r.target_ref == "world:甲" for r in 重复)