"""合成快照的纯分流证据;不代替清单中 TC-knowledge-* 的隔离 PG 导入验收。""" from __future__ import annotations import copy import json import subprocess import sys from pathlib import Path import pytest from 建立映射 import 稳定JSON, 旧快照, 映射配置, 身份映射, 迁移错误 from 知识记录分流 import 分流批次, 分流知识记录 @pytest.fixture def 映射原值(): 作品 = [ {"legacy_id": 0, "use": "method", "target_ref": "author-library:a"}, {"legacy_id": 7, "use": "author", "target_ref": "author-work:7"}, {"legacy_id": 8, "use": "reference", "target_ref": "reference:8"}, {"legacy_id": 9, "use": "method", "target_ref": "author-library:a"}, ] 来源 = [ { "source_type": "parse_book", "legacy_id": 8, "work_id": 8, "target_ref": "source:8", "revision": "source-v1", "aliases": ["甲书"], "chapters": {"12": "source:8:chapter12"}, }, { "source_type": "parse_book", "legacy_id": 10, "work_id": 10, "target_ref": "source:10", "revision": "source-v2", "aliases": ["乙书"], "chapters": {"12": "source:10:chapter12"}, }, { "source_type": "chapter_extract", "legacy_id": 70, "work_id": 7, "target_ref": "chapter:70", "revision": "body-v1", }, ] 对象 = [ {"work_id": 7, "refs": ["entity:11", "draft:101", "name:甲"], "target_ref": "world:甲"}, {"work_id": 7, "refs": ["entity:12", "draft:102", "name:乙"], "target_ref": "world:乙"}, ] 通用 = {"system": "pg", "database": "synthetic-legacy"} def 模式(属性): return {"type": "object", "properties": 属性, "additionalProperties": False} return { "works": [{**通用, "author_id": "author:a", **r} for r in 作品], "sources": [{**通用, **r} for r in 来源], "objects": [{**通用, "use": "author", "scope": "author-work:7", **r} for r in 对象], "schemas": { "character": 模式({"性格": {"type": "string"}}), "location": 模式({}), "craft": 模式({"原理": {"type": "string"}}), "character_relation": 模式( {k: {"type": "string"} for k in ("甲方", "乙方", "关系类型")} ), "style": 模式({}), "pacing": 模式({}), }, } def _快照(载荷=None, *, 源=None, **覆盖): 原行 = { "work_id": 7, "status": "pending", "draft_type": "entity", "source_type": "chapter_extract", "source_id": 70, "source_status": "active", "source_action_policy": "allowed", "draft_payload": 载荷 if 载荷 is not None else {"type": "character", "name": "甲", "fields": {"性格": "果断"}}, **覆盖, } return 旧快照.从载荷( { "source": { "system": "pg", "database": "synthetic-legacy", "table": "muse_knowledge_draft", "id": "1", "revision": "1", **(源 or {}), }, "record": 原行, } ) def _判定(原值, 快照): 结果 = 分流知识记录(快照, 映射配置(原值)) assert 结果.original == 快照.导出() assert 结果.source_hash == 快照.源哈希 return 结果 def _公共(**覆盖): return _快照( {"型": "craft", "名称": "合成铺垫法", "字段": {"原理": "先建立可核对的条件"}, **覆盖}, work_id=0, source_type="parse_book", source_id=8, ) def test_公共五型保持候选与明确作者库__21a001(映射原值): 结果 = _判定(映射原值, _公共()) assert (结果.rule_id, 结果.owner, 结果.state) == ("K01", "B04", "pending") assert 结果.scope == "author-library:a" and 结果.type_id == "craft" assert 结果.references == ({"target_ref": "source:8", "revision": "source-v1"},) def test_作者实体字段与来源不丢失__21a002(映射原值): 结果 = _判定(映射原值, _快照()) assert (结果.rule_id, 结果.owner, 结果.state) == ("K02", "B02", "pending") assert 结果.content["fields"] == {"性格": "果断"} assert 结果.references[0]["target_ref"] == "chapter:70" 重复用途 = {**映射原值["works"][1], "use": "reference", "target_ref": "reference:7"} 映射原值["works"].append(重复用途) assert _判定(映射原值, _快照()).reason_code == "unknown_use" 快照 = _快照() 映射原值["routes"] = [ { "system": "pg", "database": "synthetic-legacy", "source_key": 快照.source.记录键, "source_hash": 快照.源哈希, "use": "author", } ] assert _判定(映射原值, 快照).owner == "B02" def test_同模具参考实体不进入作者事实__21a003(映射原值): 结果 = _判定(映射原值, _快照(work_id=8, source_type="parse_book", source_id=8)) assert (结果.rule_id, 结果.owner, 结果.scope) == ("K03", "B03", "reference:8") assert _判定(映射原值, _快照(work_id=8)).reason_code == "source_scope_conflict" def test_实体形关系和显式关系共享身份与端点__21a004(映射原值): a = _快照( { "type": "character_relation", "name": "甲乙师徒", "fields": {"甲方": "甲", "乙方": "乙", "关系类型": "师徒"}, } ) b = _快照( {"type": "师徒", "source": 11, "target": 12}, draft_type="relation", 源={"revision": "2"} ) x, y = _判定(映射原值, a), _判定(映射原值, b) assert (x.rule_id, y.rule_id) == ("K04", "K05") assert x.target_ref == y.target_ref for r in (x, y): assert r.owner == "B02" and r.type_id == "character_relation" assert (r.content["source"], r.content["target"], r.content["relation_type"]) == ( "world:甲", "world:乙", "师徒", ) @pytest.mark.parametrize( "问题,原因", [ ("缺失", "missing_endpoint"), ("同名", "ambiguous_endpoint"), ("编号矛盾", "endpoint_conflict"), ], ) def test_关系端点缺失或多义不得猜测__21a005(映射原值, 问题, 原因): 载荷 = {"type": "师徒", "甲方": "甲", "乙方": "乙"} if 问题 == "缺失": 映射原值["objects"].pop() elif 问题 == "同名": 映射原值["objects"].append({**映射原值["objects"][0], "target_ref": "world:重名甲"}) else: 载荷["source"] = 12 结果 = _判定(映射原值, _快照(载荷, draft_type="relation")) assert 结果.disposition == "隔离" and 结果.reason_code == 原因 assert 结果.target_ref == "" def test_确认历史核对目标并按源身份幂等__21a006(映射原值): 快照 = _快照(status="confirmed", entity_id=11) 确认 = { "system": "pg", "database": "synthetic-legacy", "source_key": 快照.source.记录键, "source_hash": 快照.源哈希, "target_ref": "world:甲", "decision_ref": "legacy-decision:1", } 映射原值["confirmations"] = [确认] 结果 = _判定(映射原值, 快照) assert 结果.state == "confirmed_history" and 结果.target_ref == "world:甲" 台账 = 身份映射() assert 台账.登记(快照, 结果.target_ref, 结果.rule_id, scope=结果.scope) is True assert 台账.登记(快照, 结果.target_ref, 结果.rule_id, scope=结果.scope) is False assert len(台账.记录) == 1 坏 = _快照(status="confirmed", entity_id=99) 确认.update(source_hash=坏.源哈希) assert _判定(映射原值, 坏).state == "migration_review_required" assert _判定(映射原值, 坏).target_ref == "" def test_双键冲突不按优先级吞值__21a007(映射原值): 快照 = _公共(type="character") 结果 = _判定(映射原值, 快照) assert 结果.disposition == "隔离" and 结果.reason_code == "type_conflict" assert 结果.original["record"]["draft_payload"]["型"] == "craft" assert 结果.original["record"]["draft_payload"]["type"] == "character" def test_零哨兵不能成为作品或普通公共实体__21a008(映射原值): 结果 = _判定( 映射原值, _快照( {"type": "location", "name": "旧城", "fields": {}}, work_id=0, source_type="parse_book", source_id=8, ), ) assert 结果.reason_code == "zero_work_not_pattern" and not 结果.target_ref def test_跨书实例分别绑定来源和章定位__21a009(映射原值): 快照 = _公共( 实例=[ {"书": "甲书", "章": 12, "引文": "合成甲"}, {"书": "乙书", "章": 12, "引文": "合成乙"}, ] ) 结果 = _判定(映射原值, 快照) assert [r["locator"] for r in 结果.references[1:]] == [ "source:8:chapter12", "source:10:chapter12", ] del 映射原值["sources"][1]["chapters"]["12"] assert _判定(映射原值, 快照).reason_code == "missing_location" @pytest.mark.parametrize( "覆盖", [ {"deleted": True}, {"status": "rejected"}, {"status": "discarded"}, {"status": "superseded"}, {"source_status": "withdrawn"}, {"source_action_policy": "denied"}, ], ) def test_删除拒绝和撤权不复活__21a00a(映射原值, 覆盖): 结果 = _判定(映射原值, _快照(**覆盖)) assert 结果.disposition == "历史保留" and not 结果.target_ref and not 结果.content @pytest.mark.parametrize( "字段,原因", [ ({"陌生字段": {"原值": [1, 2]}}, "unknown_fields"), ({"性格": 3}, "field_type_mismatch"), ({}, "missing_required"), ], ) def test_未知或不合法动态字段原样隔离__21a00b(映射原值, 字段, 原因): 映射原值["schemas"]["character"]["required"] = ["性格"] 快照 = _快照({"type": "character", "name": "甲", "fields": 字段}) 结果 = _判定(映射原值, 快照) assert 结果.reason_code == 原因 and 结果.disposition == "隔离" assert 结果.original["record"]["draft_payload"]["fields"] == 字段 def test_来源命名空间区分同号对象__21a00c(映射原值): for 类别 in ("works", "sources", "objects"): 映射原值[类别] += [{**r, "system": "sqlite"} for r in 映射原值[类别]] x = _判定(映射原值, _快照()) y = _判定(映射原值, _快照(源={"system": "sqlite"})) assert x.source_key != y.source_key and x.target_ref != y.target_ref assert x.source_hash == y.source_hash 未知 = _判定(映射原值, _快照(源={"database": "another-database"})) assert 未知.reason_code == "unknown_use" @pytest.mark.parametrize( "类型,作品号,owner", [ ("style", 7, "B06"), ("pacing", 7, "B01"), ("style", 8, "B03"), ("pacing", 8, "B03"), ("character", 9, "B04"), ], ) def test_通用写法和本书风格按用途分流__21a00d(映射原值, 类型, 作品号, owner): 来源号, 来源型 = (70, "chapter_extract") if 作品号 == 7 else (8, "parse_book") 结果 = _判定( 映射原值, _快照( {"type": 类型, "name": "合成写法", "fields": {}}, work_id=作品号, source_id=来源号, source_type=来源型, ), ) assert 结果.rule_id == "K07" and 结果.owner == owner and 结果.state == "pending" def test_向量只承接指针且内容镜像不复制候选__21a00e(映射原值): 向量 = _快照( 源={"table": "example_knowledge_embedding"}, entity_id=11, embedding=[0.2, 0.3], model="synthetic-v1", dimensions=2, ) 结果 = _判定(映射原值, 向量) assert (结果.rule_id, 结果.owner, 结果.state) == ("K08", "B09", "historical") assert 结果.content == {} and 结果.references == ({"target_ref": "world:甲"},) 卡 = _快照(源={"table": "cards"}) assert _判定(映射原值, 卡).state == "migration_review_required" 映射原值["objects"][0]["refs"].append("card:1") 映射原值["confirmations"] = [ { "system": "pg", "database": "synthetic-legacy", "source_key": 卡.source.记录键, "source_hash": 卡.源哈希, "target_ref": "world:甲", "decision_ref": "legacy-decision:1", } ] 镜像 = _判定(映射原值, 卡) assert ( 镜像.rule_id == "K08" and 镜像.state == "mirror_history" and 镜像.target_ref == "world:甲" ) def test_不可变快照和相同身份漂移__21a00f(映射原值): 原 = _快照() 值 = 原.原行 值["draft_payload"]["name"] = "不可污染冻结原文" assert 原.原行["draft_payload"]["name"] == "甲" 坏 = _快照({"type": "character", "name": "乙", "fields": {"性格": "果断"}}) 台账 = 身份映射() 台账.登记(原, "world:甲", "K02", scope="author-work:7") with pytest.raises(迁移错误, match="drift"): 台账.登记(坏, "world:甲", "K02", scope="author-work:7") assert all( r.disposition == "隔离" and r.reason_code == "drift" for r in 分流批次([原, 坏], 映射配置(映射原值)) ) with pytest.raises(迁移错误, match="snapshot_hash_mismatch"): 旧快照.从载荷({**原.导出(), "source_hash": "0" * 64}) 新版本 = _快照(源={"revision": "2"}) assert 原.建议身份( "B02", "character", scope="author-work:7", author_id="author:a" ) == 新版本.建议身份("B02", "character", scope="author-work:7", author_id="author:a") with pytest.raises(迁移错误, match="identity_conflict"): 台账.登记(新版本, "world:另一个", "K02", scope="author-work:7") def test_坏载荷与不可解释结构不崩溃不联网__21a010(映射原值): for 载荷 in ('{"type":"character","type":"craft"}', "null", '{"type": NaN}'): assert _判定(映射原值, _快照(载荷)).disposition == "隔离" for 引用 in ("https://example.invalid/schema", "#/$defs/missing"): 映射原值["schemas"]["character"]["properties"]["性格"] = {"$ref": 引用} assert _判定(映射原值, _快照()).reason_code == "schema_invalid" assert _判定(映射原值, _快照(source_status=[])).reason_code == "permission_conflict" for 坏标记 in (None, [], 2, "未知"): assert _判定(映射原值, _快照(deleted=坏标记)).reason_code == "state_conflict" assert _判定(映射原值, _快照(draft_type="unknown")).reason_code == "unknown_draft_type" def test_命令只生成计划并保护输入和已有输出__21a011(映射原值, tmp_path): 输入, 映射, 输出 = [tmp_path / n for n in ("快照.json", "映射.json", "计划.json")] 输入.write_text(稳定JSON([_公共().导出()]), encoding="utf-8") 映射.write_text(稳定JSON(映射原值), encoding="utf-8") 原文 = 输入.read_bytes() 脚本 = Path(__file__).resolve().parents[2] / "数据库/旧库迁移/入口.py" 命令 = [ sys.executable, str(脚本), "分流", "--快照", str(输入), "--映射", str(映射), "--输出", str(输出), ] for _ in range(2): 进程 = subprocess.run(命令, capture_output=True, text=True, timeout=15) assert 进程.returncode == 0, 进程.stderr assert "合成铺垫法" not in 进程.stdout 计划 = json.loads(输出.read_text()) assert 计划["stage"] == "classification_only" and 计划["dispositions"] == {"待转换": 1} assert 计划["results"][0]["original"] == _公共().导出() assert (输出.stat().st_mode & 0o777) == 0o600 assert subprocess.run([*命令[:-1], str(输入)], capture_output=True).returncode == 1 输出.write_text("已有产物", encoding="utf-8") assert subprocess.run(命令, capture_output=True).returncode == 1 assert 输出.read_text() == "已有产物" and 输入.read_bytes() == 原文 assert not list(tmp_path.glob(".迁移分流-*")) def test_运行历史不成为新任务且延后语义隔离__21a012(映射原值): 运行 = _判定(映射原值, _快照(源={"table": "example_run"}, status="completed")) assert 运行.disposition == "历史保留" and 运行.owner == "S02" and not 运行.target_ref 经验 = _判定(映射原值, _快照(源={"table": "example_lesson"})) assert 经验.disposition == "隔离" and 经验.reason_code == "deferred_W26_27" 无映射 = copy.deepcopy(映射原值) 无映射["works"] = [] assert _判定(无映射, _快照()).reason_code == "unknown_use" def test_双用途端点和确认不能串入作者作用域__21a013(映射原值): 快照 = _快照({"type": "师徒", "source": 11, "target": 12}, draft_type="relation") 映射原值["works"].append( {**映射原值["works"][1], "use": "reference", "target_ref": "reference:7"} ) 映射原值["routes"] = [ { "system": "pg", "database": "synthetic-legacy", "source_key": 快照.source.记录键, "source_hash": 快照.源哈希, "use": "reference", } ] assert _判定(映射原值, 快照).reason_code == "missing_endpoint" 映射原值["objects"] += [ { **r, "use": "reference", "scope": "reference:7", "target_ref": "reference:" + r["target_ref"], } for r in 映射原值["objects"] ] 关系 = _判定(映射原值, 快照) assert 关系.owner == "B03" and 关系.content["source"] == "reference:world:甲" 作者映射 = copy.deepcopy(映射原值) 作者映射["routes"][0]["use"] = "author" assert _判定(作者映射, 快照).target_ref != 关系.target_ref 映射原值["routes"][0]["source_hash"] = "0" * 64 assert _判定(映射原值, 快照).reason_code == "unknown_use" 已确认 = _快照(status="confirmed", entity_id=11) 映射原值["routes"][0]["source_hash"] = 已确认.源哈希 映射原值["confirmations"] = [ { "system": "pg", "database": "synthetic-legacy", "source_key": 已确认.source.记录键, "source_hash": 已确认.源哈希, "target_ref": "world:甲", "decision_ref": "legacy-decision:1", } ] assert _判定(映射原值, 已确认).state == "migration_review_required" 映射原值["confirmations"][0]["target_ref"] = "reference:world:甲" assert _判定(映射原值, 已确认).state == "confirmed_history" def test_缺来源编号和错误映射不能自洽补造__21a014(映射原值): del 映射原值["sources"][2]["legacy_id"] assert _判定(映射原值, _快照(source_id=None)).reason_code == "missing_source" assert _判定(映射原值, _快照(source_type=[])).reason_code == "unknown_source" with pytest.raises(迁移错误, match="source_invalid"): _快照(id=2) 台账 = 身份映射() 台账.登记(_快照(), "world:甲", "K02", scope="author-work:7") with pytest.raises(迁移错误, match="drift"): 台账.登记(_快照(), "world:甲", "K02", scope="author-work:另一个") @pytest.mark.parametrize("表", ["muse_knowledge_entity", "muse_knowledge_relation"]) @pytest.mark.parametrize("描述", [" 保留原描述\n第二行。", None, 0]) def test_正式表描述进入归一结果且保全其他原列__21b001(映射原值, 表, 描述): 数据 = _快照(源={"table": 表}).导出() 数据.pop("source_hash") 行 = 数据["record"] 行.pop("draft_payload") 行.pop("draft_type") 行.update( description=描述, confidence=0.91, scope="旧作用域", lineage_payload={"来源轨迹": [8, 10]} ) if 表 == "muse_knowledge_entity": 行.update(entity_type="character", normalized_name="甲", attributes={"性格": "果断"}) else: 行.update(source_entity_id=11, target_entity_id=12, relation_type="师徒", attributes={}) 结果 = _判定(映射原值, 旧快照.从载荷(数据)) if 描述 == 0: assert 结果.disposition == "隔离" and 结果.reason_code == "field_type_mismatch" else: assert 结果.content["brief"] == ("" if 描述 is None else 描述) assert 结果.state == "migration_review_required" assert 结果.original["record"]["lineage_payload"] == {"来源轨迹": [8, 10]} assert 结果.original["record"]["confidence"] == 0.91 def test_同旧对象跨版本异目标在分流阶段全部隔离__21b002(映射原值): 一 = _快照(status="confirmed", entity_id=11) 二 = _快照(源={"revision": "2"}, status="confirmed", entity_id=12) 映射原值["confirmations"] = [ { "system": "pg", "database": "synthetic-legacy", "source_key": x.source.记录键, "source_hash": x.源哈希, "target_ref": t, "decision_ref": "legacy-decision:" + x.source.revision, } for x, t in ((一, "world:甲"), (二, "world:乙")) ] 配置 = 映射配置(映射原值) # 两条单独输入均能关联声明;合批必须先隔离,不能等导入顺序抢占。 assert all(分流知识记录(x, 配置).state == "confirmed_history" for x in (一, 二)) 结果 = 分流批次([一, 二], 配置) assert len(结果) == 2 for r in 结果: assert r.disposition == "隔离" and r.reason_code == "identity_conflict" assert r.target_ref == "" and r.content == {} and r.references == () assert {x["target_ref"] for x in r.conflicts} == {"world:甲", "world:乙"} assert {r.source_key for r in 结果} == {一.source.记录键, 二.source.记录键} 映射原值["confirmations"][1]["target_ref"] = "world:甲" 二同 = _快照(源={"revision": "2"}, status="confirmed", entity_id=11) 映射原值["confirmations"][1]["source_hash"] = 二同.源哈希 重复 = 分流批次([一, 二同], 映射配置(映射原值)) assert all(r.disposition == "待转换" and r.target_ref == "world:甲" for r in 重复)