From 00aa5e323878c505112c7b9f450f3cd702456754 Mon Sep 17 00:00:00 2001 From: zizi Date: Mon, 21 Sep 2026 21:09:28 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BC=98=E5=8C=96(=E4=BA=A4=E4=BB=98=E8=BF=9E?= =?UTF-8?q?=E8=BD=BD):=20=E5=85=A8=E4=B9=A6=E6=95=B4=E7=90=86=E9=87=87?= =?UTF-8?q?=E7=94=A8=E6=89=B9=E9=87=8F=E5=BD=92=E5=B1=9E=E4=B8=8E=E6=89=B9?= =?UTF-8?q?=E9=87=8F=E6=AD=A3=E6=96=87=E8=AF=BB=E5=8F=96=E5=B9=B6=E7=A7=BB?= =?UTF-8?q?=E5=87=BA=E7=BA=AF=E8=AE=A1=E7=AE=97?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- src/muse/交付连载/全书整理.py | 19 ++- src/muse/交付连载/接口.py | 5 +- src/muse/作品规划/存储.py | 13 ++ src/muse/作品规划/接口.py | 23 ++++ src/muse/正文写作/存储.py | 16 +++ src/muse/正文写作/接口.py | 33 +++++- tests/单元/test_全书工作面批量读取.py | 163 ++++++++++++++++++++++++++ tests/用例清单.json | 64 ++++++++++ 8 files changed, 330 insertions(+), 6 deletions(-) create mode 100644 tests/单元/test_全书工作面批量读取.py diff --git a/src/muse/交付连载/全书整理.py b/src/muse/交付连载/全书整理.py index 0788c5b..527e66c 100644 --- a/src/muse/交付连载/全书整理.py +++ b/src/muse/交付连载/全书整理.py @@ -10,18 +10,26 @@ from muse.正文写作.接口 import ( 准备派生候选, 可见文本, 应用选段, + 批量读取当前正文依据, 段落哈希, 段落文本, - 读取当前正文依据, 读取正文依据, ) -def 全书工作面(连, author, work_id, *, names=None, exceptions=()): +def 读取全书数据(连, author, work_id): + """同事务批量获取目录与全书当前正文依据。""" directory = 读取作品范围(连, author, work_id) + chapter_ids = [c["chapter_id"] for c in directory["chapters"]] + body_map = 批量读取当前正文依据(连, author, work_id, chapter_ids) + return directory, body_map + + +def 整理全书工作面(work_id, directory, body_map, *, names=None, exceptions=()): + """纯内存整理工作面结构、段落偏移与文字疑点,不占用数据库连接。""" chapters, problems = [], [] for chapter in directory["chapters"]: - body = 读取当前正文依据(连, author, chapter["chapter_id"]) + body = body_map.get(chapter["chapter_id"]) if body is None: chapters.append({**chapter, "state": "missing", "revision": None}) problems.append({"chapter_id": chapter["chapter_id"], "kind": "missing_body"}) @@ -67,6 +75,11 @@ def 全书工作面(连, author, work_id, *, names=None, exceptions=()): } +def 全书工作面(连, author, work_id, *, names=None, exceptions=()): + directory, body_map = 读取全书数据(连, author, work_id) + return 整理全书工作面(work_id, directory, body_map, names=names, exceptions=exceptions) + + def 跨章候选身份(author, req, chapter): return str( uuid5( diff --git a/src/muse/交付连载/接口.py b/src/muse/交付连载/接口.py index 32a26eb..6856dc8 100644 --- a/src/muse/交付连载/接口.py +++ b/src/muse/交付连载/接口.py @@ -49,11 +49,12 @@ class 交付服务: raise 交付错误("SCOPE_DENIED", "交付需要生产用途的本作者身份") def 全书回看(self, 身份, work_id, *, names=None, exceptions=()): - from muse.交付连载.全书整理 import 全书工作面 + from muse.交付连载.全书整理 import 整理全书工作面, 读取全书数据 self.核对身份(身份) with self.数据库.连接(只读=True) as 连: - return 全书工作面(连, 身份.作者, work_id, names=names, exceptions=exceptions) + directory, body_map = 读取全书数据(连, 身份.作者, work_id) + return 整理全书工作面(work_id, directory, body_map, names=names, exceptions=exceptions) def 提出跨章修订(self, 身份, 命令ID, 请求: 全书修订): self.核对身份(身份) diff --git a/src/muse/作品规划/存储.py b/src/muse/作品规划/存储.py index bcf4295..a883b2a 100644 --- a/src/muse/作品规划/存储.py +++ b/src/muse/作品规划/存储.py @@ -242,6 +242,19 @@ class 作品存储: .fetchone() ) + def 批量读取章节归属(self, chapter_ids: list[str]) -> list[dict]: + if not chapter_ids: + return [] + return ( + self.连.cursor(row_factory=dict_row) + .execute( + "SELECT c.chapter_id,c.work_id,c.title,c.position,w.author_id,w.directory_revision " + "FROM muse_chapter c JOIN muse_work w USING(work_id) WHERE c.chapter_id=ANY(%s)", + (chapter_ids,), + ) + .fetchall() + ) + class 规划存储: def __init__(self, 连): diff --git a/src/muse/作品规划/接口.py b/src/muse/作品规划/接口.py index 6e9cef5..e97bb20 100644 --- a/src/muse/作品规划/接口.py +++ b/src/muse/作品规划/接口.py @@ -282,6 +282,28 @@ def 读取章节归属(连, 作者: str, chapter_id: str) -> dict: return {k: v for k, v in 行.items() if k != "author_id"} +def 批量核对章节归属( + 连, 作者: str, chapter_ids: list[str], *, work_id: str | None = None +) -> dict[str, dict]: + """同事务批量核对章节归属;任一不存在或越界整批拒绝,返回按 chapter_id 索引的字典。""" + if not chapter_ids: + return {} + 唯一ID = list(dict.fromkeys(chapter_ids)) + rows = 作品存储(连).批量读取章节归属(唯一ID) + if len(rows) != len(唯一ID): + 查到ID = {r["chapter_id"] for r in rows} + 缺失 = [cid for cid in 唯一ID if cid not in 查到ID] + raise 作品错误("CHAPTER_NOT_FOUND", f"章节不存在: {sorted(缺失)}") + 结果 = {} + for 行 in rows: + if 行["author_id"] != 作者: + raise 作品错误("SCOPE_DENIED", "章节不属于当前作者") + if work_id is not None and 行["work_id"] != work_id: + raise 作品错误("SCOPE_DENIED", "章节不属于指定作品") + 结果[行["chapter_id"]] = {k: v for k, v in 行.items() if k != "author_id"} + return 结果 + + def 读取有界作品范围(连, 作者: str, work_id: str, *, 数量上限: int, 字节上限: int) -> dict: """在同一语句限制目录传输;超限只交回数量/大小,调用方决定拒绝口径。""" from muse.作品规划.存储 import 作品存储 @@ -312,6 +334,7 @@ __all__ = [ "作品错误", "登记作品参与者", "读取章节归属", + "批量核对章节归属", "创作进度服务", "读取作品范围", "探索服务", diff --git a/src/muse/正文写作/存储.py b/src/muse/正文写作/存储.py index 83d306f..d9601d9 100644 --- a/src/muse/正文写作/存储.py +++ b/src/muse/正文写作/存储.py @@ -93,6 +93,22 @@ class 正文存储: .fetchone() ) + def 批量读取当前(self, chapter_ids: list[str], 分支: str = "main") -> list[dict]: + if not chapter_ids: + return [] + return ( + self.连.cursor(row_factory=dict_row) + .execute( + "SELECT d.document_id,d.chapter_id,d.branch_id,v.revision,v.document," + "v.document_hash,v.visible_text_hash,v.restored_from " + "FROM muse_document d JOIN muse_document_version v ON v.document_id=d.document_id " + "AND v.revision=d.current_revision " + "WHERE d.chapter_id=ANY(%s) AND d.branch_id=%s", + (chapter_ids, 分支), + ) + .fetchall() + ) + def 读取元数据(self, document_id: str, 版本: int) -> dict | None: """复检只返回版本和大小;数据库侧长度计算不承诺物理 I/O 上界。""" return ( diff --git a/src/muse/正文写作/接口.py b/src/muse/正文写作/接口.py index 477faea..4139bd9 100644 --- a/src/muse/正文写作/接口.py +++ b/src/muse/正文写作/接口.py @@ -3,7 +3,7 @@ from dataclasses import asdict from uuid import NAMESPACE_URL, uuid5 -from muse.作品规划.接口 import 读取章节归属 +from muse.作品规划.接口 import 批量核对章节归属, 读取章节归属 from muse.基础设施.数据库.连接 import 数据库工厂 from muse.正式变更.接口 import 作者动作, 参与者目录, 变更命令, 变更错误, 正式变更服务 from muse.正文写作.人工保存 import 文稿依赖, 正文保存入口, 正文参与者 @@ -66,6 +66,36 @@ def 读取当前正文依据(连, 作者: str, chapter_id: str, *, 分支: str = } +def 批量读取当前正文依据( + 连, + 作者: str, + work_id: str, + chapter_ids: list[str], + *, + 分支: str = "main", +) -> dict[str, dict]: + """批量核对章位归属后读取当前正式正文;未保存不返回条目,由调用方作为缺文处理。""" + if not chapter_ids: + return {} + 归属 = 批量核对章节归属(连, 作者, chapter_ids, work_id=work_id) + 存储 = 正文存储(连) + 行列表 = 存储.批量读取当前(list(归属.keys()), 分支=分支) + 结果 = {} + for 行 in 行列表: + cid = 行["chapter_id"] + id_ = 文稿身份(cid, 分支) + 结果[cid] = { + "work_id": work_id, + "document_id": id_, + "chapter_id": cid, + "revision": 行["revision"], + "current_revision": 行["revision"], + "document_hash": 行["document_hash"], + "document": 存储.恢复草稿(行), + } + return 结果 + + def 读取正文依据(连, 作者: str, chapter_id: str, revision: int, *, 分支: str = "main") -> dict: """事实及其他同事务消费者读取确切正文;同时返回当前指针供来源复检。""" 章节 = 读取章节归属(连, 作者, chapter_id) @@ -440,6 +470,7 @@ __all__ = [ "读取正文依据", "读取候选依据", "读取当前正文依据", + "批量读取当前正文依据", "文稿身份", "创建写作任务", "写入模型候选", diff --git a/tests/单元/test_全书工作面批量读取.py b/tests/单元/test_全书工作面批量读取.py new file mode 100644 index 0000000..5fef308 --- /dev/null +++ b/tests/单元/test_全书工作面批量读取.py @@ -0,0 +1,163 @@ +"""全书工作面批量读取与归属校验的独立机制验证。""" + +from dataclasses import asdict + +import pytest + +from muse.交付连载.全书整理 import 整理全书工作面 +from muse.作品规划.模型 import 作品错误 +from muse.正文写作.模型 import 正文草稿, 正文错误, 段落 +from muse.正文写作.正文格式 import 可见文本哈希, 正文结构哈希 + + +class 模拟连: + def __init__(self, chapters, *, author="author-1", work_id="work-1", missing=(), corrupt=()): + self.chapters = chapters + self.author = author + self.work_id = work_id + self.missing = set(missing) + self.corrupt = set(corrupt) + self.queries = [] + + def cursor(self, **_): + return self + + def execute(self, sql, params=()): + self.queries.append((sql, params)) + if "FROM muse_work w" in sql or "c.chapter_id=ANY(%s)" in sql: + # 批量读取章节归属 + ids = params[0] + rows = [] + for cid in ids: + if cid.startswith("nonexistent"): + continue + a = "wrong-author" if cid.startswith("wrong-author") else self.author + w = "wrong-work" if cid.startswith("wrong-work") else self.work_id + rows.append( + { + "chapter_id": cid, + "work_id": w, + "title": f"Title {cid}", + "position": 1, + "author_id": a, + "directory_revision": 1, + } + ) + self._rows = rows + return self + + if "FROM muse_document d JOIN muse_document_version v" in sql: + # 批量读取当前正文 + ids = params[0] + branch = params[1] + rows = [] + for cid in ids: + if cid in self.missing: + continue + draft = 正文草稿((段落(f"{cid}:p0", ()),)) + doc_hash = "bad_hash" if cid in self.corrupt else 正文结构哈希(draft) + vis_hash = 可见文本哈希(draft) + rows.append( + { + "document_id": f"doc-{cid}", + "chapter_id": cid, + "branch_id": branch, + "revision": 1, + "document": asdict(draft), + "document_hash": doc_hash, + "visible_text_hash": vis_hash, + "restored_from": None, + } + ) + self._rows = rows + return self + + self._rows = [] + return self + + def fetchall(self): + return getattr(self, "_rows", []) + + def fetchone(self): + rows = getattr(self, "_rows", []) + return rows[0] if rows else None + + +@pytest.mark.case_id("TC-DELIVERY-BATCH-AUTH-CHECK") +def test_批量核对章节归属异常分支(): + from muse.作品规划.接口 import 批量核对章节归属 + + # 1. 空列表直接返回 + 连 = 模拟连([]) + assert 批量核对章节归属(连, "author-1", []) == {} + + # 2. 章节不存在整批拒绝 + 连 = 模拟连([]) + with pytest.raises(作品错误) as exc: + 批量核对章节归属(连, "author-1", ["c1", "nonexistent-1"]) + assert exc.value.错误码 == "CHAPTER_NOT_FOUND" + + # 3. 章节不属于当前作者整批拒绝 + with pytest.raises(作品错误) as exc: + 批量核对章节归属(连, "author-1", ["c1", "wrong-author-c2"]) + assert exc.value.错误码 == "SCOPE_DENIED" + assert "章节不属于当前作者" in exc.value.说明 + + # 4. 章节不属于指定作品整批拒绝 + with pytest.raises(作品错误) as exc: + 批量核对章节归属(连, "author-1", ["c1", "wrong-work-c2"], work_id="work-1") + assert exc.value.错误码 == "SCOPE_DENIED" + assert "章节不属于指定作品" in exc.value.说明 + + +@pytest.mark.case_id("TC-DELIVERY-BATCH-CORRUPT-DETECT") +def test_批量读取正文损坏则报错(): + from muse.正文写作.接口 import 批量读取当前正文依据 + + 连 = 模拟连(["c1"], corrupt=["c1"]) + with pytest.raises(正文错误, match="已保存正文与其结构或文本哈希不一致"): + 批量读取当前正文依据(连, "author-1", "work-1", ["c1"]) + + +@pytest.mark.case_id("TC-DELIVERY-BATCH-READ-WORKBENCH") +def test_全书批量读取查询次数为常数上界(): + from muse.正文写作.接口 import 批量读取当前正文依据 + + # 100 章批量获取,SQL 查询次数恒定为 1 次批量归属 + 1 次批量正文 + c_ids = [f"ch-{i}" for i in range(100)] + 连 = 模拟连(c_ids) + 结果 = 批量读取当前正文依据(连, "author-1", "work-1", c_ids) + assert len(结果) == 100 + assert len(连.queries) == 2 # 1次批量核对归属 + 1次批量正文读取,而非 200 次! + + +@pytest.mark.case_id("TC-DELIVERY-WORKBENCH-MEM-SEPARATED") +def test_全书工作面组装正确区分状态(): + # 模拟包含正常章、缺正文章、空正文章 + directory = { + "revision": 5, + "chapters": [ + {"chapter_id": "c1", "title": "第一章", "position": 1}, + {"chapter_id": "c2", "title": "第二章", "position": 2}, + ], + } + # c1 正常,c2 缺失 + draft = 正文草稿((段落("c1:p0", ()),)) + body_map = { + "c1": { + "work_id": "work-1", + "document_id": "doc-c1", + "chapter_id": "c1", + "revision": 1, + "current_revision": 1, + "document_hash": 正文结构哈希(draft), + "document": draft, + } + } + res = 整理全书工作面("work-1", directory, body_map) + assert res["work_id"] == "work-1" + assert res["directory_revision"] == 5 + assert len(res["chapters"]) == 2 + assert res["chapters"][0]["state"] == "empty" # 空内容 + assert res["chapters"][1]["state"] == "missing" # 缺正文 + assert not res["can_freeze"] # 有 problem 不能冻结 diff --git a/tests/用例清单.json b/tests/用例清单.json index 57ed403..c9e5732 100644 --- a/tests/用例清单.json +++ b/tests/用例清单.json @@ -35317,6 +35317,70 @@ "数据库" ] }, + { + "case_id": "TC-DELIVERY-BATCH-AUTH-CHECK", + "file": "tests/单元/test_全书工作面批量读取.py", + "symbol": "test_批量核对章节归属异常分支", + "parameter_ids": [], + "node_ids": [ + "tests/单元/test_全书工作面批量读取.py::test_批量核对章节归属异常分支" + ], + "fixtures": [ + "request", + "测试资源接缝", + "源码资源", + "离线防护" + ], + "markers": [] + }, + { + "case_id": "TC-DELIVERY-BATCH-CORRUPT-DETECT", + "file": "tests/单元/test_全书工作面批量读取.py", + "symbol": "test_批量读取正文损坏则报错", + "parameter_ids": [], + "node_ids": [ + "tests/单元/test_全书工作面批量读取.py::test_批量读取正文损坏则报错" + ], + "fixtures": [ + "request", + "测试资源接缝", + "源码资源", + "离线防护" + ], + "markers": [] + }, + { + "case_id": "TC-DELIVERY-BATCH-READ-WORKBENCH", + "file": "tests/单元/test_全书工作面批量读取.py", + "symbol": "test_全书批量读取查询次数为常数上界", + "parameter_ids": [], + "node_ids": [ + "tests/单元/test_全书工作面批量读取.py::test_全书批量读取查询次数为常数上界" + ], + "fixtures": [ + "request", + "测试资源接缝", + "源码资源", + "离线防护" + ], + "markers": [] + }, + { + "case_id": "TC-DELIVERY-WORKBENCH-MEM-SEPARATED", + "file": "tests/单元/test_全书工作面批量读取.py", + "symbol": "test_全书工作面组装正确区分状态", + "parameter_ids": [], + "node_ids": [ + "tests/单元/test_全书工作面批量读取.py::test_全书工作面组装正确区分状态" + ], + "fixtures": [ + "request", + "测试资源接缝", + "源码资源", + "离线防护" + ], + "markers": [] + }, { "case_id": "TC-O08-SSE-001", "file": "tests/单元/test_任务事件收尾.py",