"""分章切窗与实体归并的确定性边界:覆盖、漂移、误并防护与样本观察陈述。""" import pytest from muse.资料研究.分章切窗 import 分章, 切窗, 漂移检查, 窗计划哈希, 覆盖核对 from muse.资料研究.实体归并 import 归并, 收集实体 from muse.资料研究.样本选题 import 样本比较, 选书交接单, 选题依据哈希 from muse.资料研究.模型 import 资料错误 def _文本(章数: int = 5) -> str: return "".join(f"第{章}回 风起\n内容{'甲' * 20}\n" for 章 in range(1, 章数 + 1)) @pytest.mark.case_id( "NC-w18-18a01a", environment="离线单元", given="确定性输入或隔离数据库", when="切窗覆盖与无重叠", then=["行为可复验", "失败与歧义如实呈现"], contract="docs/系统架构/新版设计/模块设计/B03-资料研究.md", ) def test_切窗覆盖与无重叠__18a01a(): 文本 = _文本() 窗口列表, 哈希 = 切窗(文本, 窗口章数=2) assert [窗.窗ID for 窗 in 窗口列表] == ["w001", "w002", "w003"] 统计 = 覆盖核对(文本, 窗口列表) assert 统计["重叠码点"] == 0 assert 统计["覆盖码点"] == len(文本) assert 哈希 == 窗计划哈希(窗口列表) @pytest.mark.case_id( "NC-w18-18a02b", environment="离线单元", given="确定性输入或隔离数据库", when="切窗漂移拒绝写入", then=["行为可复验", "失败与歧义如实呈现"], contract="docs/系统架构/新版设计/模块设计/B03-资料研究.md", ) def test_切窗漂移拒绝写入__18a02b(): 文本 = _文本(3) 窗口列表, _ = 切窗(文本, 窗口章数=2) 篡改 = 文本.replace("甲" * 20, "乙" * 20) with pytest.raises(资料错误, match="漂移"): 漂移检查(篡改, 窗口列表[0]) 漂移检查(文本, 窗口列表[0]) # 原文不变时放行 @pytest.mark.case_id( "NC-w18-18a03c", environment="离线单元", given="确定性输入或隔离数据库", when="无章界整篇单窗", then=["行为可复验", "失败与歧义如实呈现"], contract="docs/系统架构/新版设计/模块设计/B03-资料研究.md", ) def test_无章界整篇单窗__18a03c(): 文本 = "没有章界的一段长文。" * 10 章 = 分章(文本) assert len(章) == 1 窗口列表, _ = 切窗(文本) assert len(窗口列表) == 1 @pytest.mark.case_id( "NC-w18-18a04d", environment="离线单元", given="确定性输入或隔离数据库", when="同名异型歧义不误并", then=["行为可复验", "失败与歧义如实呈现"], contract="docs/系统架构/新版设计/模块设计/B03-资料研究.md", ) def test_同名异型歧义不误并__18a04d(): 窗口结果 = [ { "window_id": "w001", "entities": [ {"name": "林深", "type": "人物", "evidence": ["林深把伞收在了门边"]}, {"name": "青云镇", "type": "地点", "evidence": ["青云镇的雨"]}, ], }, { "window_id": "w002", "entities": [ {"name": "林深", "type": "地名", "aliases": ["林深镇"], "evidence": ["林深界碑"]}, ], }, ] 结果 = 归并(收集实体(窗口结果)) 名组 = [g for g in 结果["合并组"] if "林深" in g["aliases"] or g["canonical"] == "林深"] 歧义名 = [g["名"] for g in 结果["歧义"]] assert all(项["处理"] == "歧义只登记,由作者人工判定" for 项 in 结果["歧义"]) assert "林深" in 歧义名, f"同名异型应保留歧义:{结果}" assert not any(g["canonical"] == "林深" for g in 名组), "同名异型不得直接合并" @pytest.mark.case_id( "NC-w18-18a05e", environment="离线单元", given="确定性输入或隔离数据库", when="同型别名带证据合并", then=["行为可复验", "失败与歧义如实呈现"], contract="docs/系统架构/新版设计/模块设计/B03-资料研究.md", ) def test_同型别名带证据合并__18a05e(): 窗口结果 = [ { "window_id": "w001", "entities": [{"name": "林深", "type": "人物", "evidence": ["林深把伞收在了门边"]}], }, { "window_id": "w002", "entities": [ {"name": "林深", "type": "人物", "aliases": ["老林"], "evidence": ["老林摇头"]} ], }, ] 结果 = 归并(收集实体(窗口结果)) assert len(结果["合并组"]) == 1 组 = 结果["合并组"][0] assert 组["canonical"] == "林深" and "老林" in 组["aliases"] assert 组["windows"] == ["w001", "w002"] assert len(组["evidence"]) == 2 assert 结果["歧义"] == [] @pytest.mark.case_id( "NC-w18-18a06f", environment="离线单元", given="确定性输入或隔离数据库", when="样本比较是观察陈述", then=["行为可复验", "失败与歧义如实呈现"], contract="docs/系统架构/新版设计/模块设计/B03-资料研究.md", ) def test_样本比较是观察陈述__18a06f(): A = {"source_id": "s-a", "revision": 1, "content_hash": "h-a", "字数": 100, "章数": 3} B = {"source_id": "s-b", "revision": 1, "content_hash": "h-b", "字数": 250, "章数": 7} 差异 = 样本比较(A, B) assert 差异["差异"]["字数"] == {"A": 100, "B": 250} assert "因果" in 差异["说明"] 交接 = 选书交接单(A, "题材接近,节奏可借鉴") assert 交接["source_id"] == "s-a" and 交接["content_hash"] == "h-a" assert 选题依据哈希(交接) with pytest.raises(资料错误): 选书交接单(A, " ")