98 lines
3.7 KiB
Python
98 lines
3.7 KiB
Python
"""分章切窗与实体归并的确定性边界:覆盖、漂移、误并防护与样本观察陈述。"""
|
|
|
|
import pytest
|
|
|
|
from muse.资料研究.分章切窗 import 分章, 切窗, 漂移检查, 窗计划哈希, 覆盖核对
|
|
from muse.资料研究.实体归并 import 归并, 收集实体
|
|
from muse.资料研究.样本选题 import 样本比较, 选书交接单, 选题依据哈希
|
|
from muse.资料研究.模型 import 资料错误
|
|
|
|
|
|
def _文本(章数: int = 5) -> str:
|
|
return "".join(f"第{章}回 风起\n内容{'甲' * 20}\n" for 章 in range(1, 章数 + 1))
|
|
|
|
|
|
def test_切窗覆盖与无重叠__18a01a():
|
|
文本 = _文本()
|
|
窗口列表, 哈希 = 切窗(文本, 窗口章数=2)
|
|
assert [窗.窗ID for 窗 in 窗口列表] == ["w001", "w002", "w003"]
|
|
统计 = 覆盖核对(文本, 窗口列表)
|
|
assert 统计["重叠码点"] == 0
|
|
assert 统计["覆盖码点"] == len(文本)
|
|
assert 哈希 == 窗计划哈希(窗口列表)
|
|
|
|
|
|
def test_切窗漂移拒绝写入__18a02b():
|
|
文本 = _文本(3)
|
|
窗口列表, _ = 切窗(文本, 窗口章数=2)
|
|
篡改 = 文本.replace("甲" * 20, "乙" * 20)
|
|
with pytest.raises(资料错误, match="漂移"):
|
|
漂移检查(篡改, 窗口列表[0])
|
|
漂移检查(文本, 窗口列表[0]) # 原文不变时放行
|
|
|
|
|
|
def test_无章界整篇单窗__18a03c():
|
|
文本 = "没有章界的一段长文。" * 10
|
|
章 = 分章(文本)
|
|
assert len(章) == 1
|
|
窗口列表, _ = 切窗(文本)
|
|
assert len(窗口列表) == 1
|
|
|
|
|
|
def test_同名异型歧义不误并__18a04d():
|
|
窗口结果 = [
|
|
{
|
|
"window_id": "w001",
|
|
"entities": [
|
|
{"name": "林深", "type": "人物", "evidence": ["林深把伞收在了门边"]},
|
|
{"name": "青云镇", "type": "地点", "evidence": ["青云镇的雨"]},
|
|
],
|
|
},
|
|
{
|
|
"window_id": "w002",
|
|
"entities": [
|
|
{"name": "林深", "type": "地名", "aliases": ["林深镇"], "evidence": ["林深界碑"]},
|
|
],
|
|
},
|
|
]
|
|
结果 = 归并(收集实体(窗口结果))
|
|
名组 = [g for g in 结果["合并组"] if "林深" in g["aliases"] or g["canonical"] == "林深"]
|
|
歧义名 = [g["名"] for g in 结果["歧义"]]
|
|
assert "林深" in 歧义名, f"同名异型应保留歧义:{结果}"
|
|
assert not any(g["canonical"] == "林深" for g in 名组), "同名异型不得直接合并"
|
|
|
|
|
|
def test_同型别名带证据合并__18a05e():
|
|
窗口结果 = [
|
|
{
|
|
"window_id": "w001",
|
|
"entities": [{"name": "林深", "type": "人物", "evidence": ["林深把伞收在了门边"]}],
|
|
},
|
|
{
|
|
"window_id": "w002",
|
|
"entities": [
|
|
{"name": "林深", "type": "人物", "aliases": ["老林"], "evidence": ["老林摇头"]}
|
|
],
|
|
},
|
|
]
|
|
结果 = 归并(收集实体(窗口结果))
|
|
assert len(结果["合并组"]) == 1
|
|
组 = 结果["合并组"][0]
|
|
assert 组["canonical"] == "林深" and "老林" in 组["aliases"]
|
|
assert 组["windows"] == ["w001", "w002"]
|
|
assert len(组["evidence"]) == 2
|
|
assert 结果["歧义"] == []
|
|
|
|
|
|
def test_样本比较是观察陈述__18a06f():
|
|
A = {"source_id": "s-a", "revision": 1, "content_hash": "h-a", "字数": 100, "章数": 3}
|
|
B = {"source_id": "s-b", "revision": 1, "content_hash": "h-b", "字数": 250, "章数": 7}
|
|
差异 = 样本比较(A, B)
|
|
assert 差异["差异"]["字数"] == {"A": 100, "B": 250}
|
|
assert "因果" in 差异["说明"]
|
|
交接 = 选书交接单(A, "题材接近,节奏可借鉴")
|
|
assert 交接["source_id"] == "s-a" and 交接["content_hash"] == "h-a"
|
|
assert 选题依据哈希(交接)
|
|
with pytest.raises(资料错误):
|
|
选书交接单(A, " ")
|