优化(交付连载): 全书整理采用批量归属与批量正文读取并移出纯计算

This commit is contained in:
zizi 2026-09-21 21:09:28 +08:00
parent 41faed9f31
commit 00aa5e3238
8 changed files with 330 additions and 6 deletions

View File

@ -10,18 +10,26 @@ from muse.正文写作.接口 import (
准备派生候选,
可见文本,
应用选段,
批量读取当前正文依据,
段落哈希,
段落文本,
读取当前正文依据,
读取正文依据,
)
def 全书工作面(连, author, work_id, *, names=None, exceptions=()):
def 读取全书数据(连, author, work_id):
"""同事务批量获取目录与全书当前正文依据。"""
directory = 读取作品范围(连, author, work_id)
chapter_ids = [c["chapter_id"] for c in directory["chapters"]]
body_map = 批量读取当前正文依据(连, author, work_id, chapter_ids)
return directory, body_map
def 整理全书工作面(work_id, directory, body_map, *, names=None, exceptions=()):
"""纯内存整理工作面结构、段落偏移与文字疑点,不占用数据库连接。"""
chapters, problems = [], []
for chapter in directory["chapters"]:
body = 读取当前正文依据(连, author, chapter["chapter_id"])
body = body_map.get(chapter["chapter_id"])
if body is None:
chapters.append({**chapter, "state": "missing", "revision": None})
problems.append({"chapter_id": chapter["chapter_id"], "kind": "missing_body"})
@ -67,6 +75,11 @@ def 全书工作面(连, author, work_id, *, names=None, exceptions=()):
}
def 全书工作面(连, author, work_id, *, names=None, exceptions=()):
directory, body_map = 读取全书数据(连, author, work_id)
return 整理全书工作面(work_id, directory, body_map, names=names, exceptions=exceptions)
def 跨章候选身份(author, req, chapter):
return str(
uuid5(

View File

@ -49,11 +49,12 @@ class 交付服务:
raise 交付错误("SCOPE_DENIED", "交付需要生产用途的本作者身份")
def 全书回看(self, 身份, work_id, *, names=None, exceptions=()):
from muse.交付连载.全书整理 import 全书工作面
from muse.交付连载.全书整理 import 整理全书工作面, 读取全书数据
self.核对身份(身份)
with self.数据库.连接(只读=True) as 连:
return 全书工作面(连, 身份.作者, work_id, names=names, exceptions=exceptions)
directory, body_map = 读取全书数据(连, 身份.作者, work_id)
return 整理全书工作面(work_id, directory, body_map, names=names, exceptions=exceptions)
def 提出跨章修订(self, 身份, 命令ID, 请求: 全书修订):
self.核对身份(身份)

View File

@ -242,6 +242,19 @@ class 作品存储:
.fetchone()
)
def 批量读取章节归属(self, chapter_ids: list[str]) -> list[dict]:
if not chapter_ids:
return []
return (
self.连.cursor(row_factory=dict_row)
.execute(
"SELECT c.chapter_id,c.work_id,c.title,c.position,w.author_id,w.directory_revision "
"FROM muse_chapter c JOIN muse_work w USING(work_id) WHERE c.chapter_id=ANY(%s)",
(chapter_ids,),
)
.fetchall()
)
class 规划存储:
def __init__(self, 连):

View File

@ -282,6 +282,28 @@ def 读取章节归属(连, 作者: str, chapter_id: str) -> dict:
return {k: v for k, v in 行.items() if k != "author_id"}
def 批量核对章节归属(
连, 作者: str, chapter_ids: list[str], *, work_id: str | None = None
) -> dict[str, dict]:
"""同事务批量核对章节归属;任一不存在或越界整批拒绝,返回按 chapter_id 索引的字典。"""
if not chapter_ids:
return {}
唯一ID = list(dict.fromkeys(chapter_ids))
rows = 作品存储(连).批量读取章节归属(唯一ID)
if len(rows) != len(唯一ID):
查到ID = {r["chapter_id"] for r in rows}
缺失 = [cid for cid in 唯一ID if cid not in 查到ID]
raise 作品错误("CHAPTER_NOT_FOUND", f"章节不存在: {sorted(缺失)}")
结果 = {}
for 行 in rows:
if 行["author_id"] != 作者:
raise 作品错误("SCOPE_DENIED", "章节不属于当前作者")
if work_id is not None and 行["work_id"] != work_id:
raise 作品错误("SCOPE_DENIED", "章节不属于指定作品")
结果[行["chapter_id"]] = {k: v for k, v in 行.items() if k != "author_id"}
return 结果
def 读取有界作品范围(连, 作者: str, work_id: str, *, 数量上限: int, 字节上限: int) -> dict:
"""在同一语句限制目录传输;超限只交回数量/大小,调用方决定拒绝口径。"""
from muse.作品规划.存储 import 作品存储
@ -312,6 +334,7 @@ __all__ = [
"作品错误",
"登记作品参与者",
"读取章节归属",
"批量核对章节归属",
"创作进度服务",
"读取作品范围",
"探索服务",

View File

@ -93,6 +93,22 @@ class 正文存储:
.fetchone()
)
def 批量读取当前(self, chapter_ids: list[str], 分支: str = "main") -> list[dict]:
if not chapter_ids:
return []
return (
self.连.cursor(row_factory=dict_row)
.execute(
"SELECT d.document_id,d.chapter_id,d.branch_id,v.revision,v.document,"
"v.document_hash,v.visible_text_hash,v.restored_from "
"FROM muse_document d JOIN muse_document_version v ON v.document_id=d.document_id "
"AND v.revision=d.current_revision "
"WHERE d.chapter_id=ANY(%s) AND d.branch_id=%s",
(chapter_ids, 分支),
)
.fetchall()
)
def 读取元数据(self, document_id: str, 版本: int) -> dict | None:
"""复检只返回版本和大小;数据库侧长度计算不承诺物理 I/O 上界。"""
return (

View File

@ -3,7 +3,7 @@
from dataclasses import asdict
from uuid import NAMESPACE_URL, uuid5
from muse.作品规划.接口 import 读取章节归属
from muse.作品规划.接口 import 批量核对章节归属, 读取章节归属
from muse.基础设施.数据库.连接 import 数据库工厂
from muse.正式变更.接口 import 作者动作, 参与者目录, 变更命令, 变更错误, 正式变更服务
from muse.正文写作.人工保存 import 文稿依赖, 正文保存入口, 正文参与者
@ -66,6 +66,36 @@ def 读取当前正文依据(连, 作者: str, chapter_id: str, *, 分支: str =
}
def 批量读取当前正文依据(
连,
作者: str,
work_id: str,
chapter_ids: list[str],
*,
分支: str = "main",
) -> dict[str, dict]:
"""批量核对章位归属后读取当前正式正文;未保存不返回条目,由调用方作为缺文处理。"""
if not chapter_ids:
return {}
归属 = 批量核对章节归属(连, 作者, chapter_ids, work_id=work_id)
存储 = 正文存储(连)
行列表 = 存储.批量读取当前(list(归属.keys()), 分支=分支)
结果 = {}
for 行 in 行列表:
cid = 行["chapter_id"]
id_ = 文稿身份(cid, 分支)
结果[cid] = {
"work_id": work_id,
"document_id": id_,
"chapter_id": cid,
"revision": 行["revision"],
"current_revision": 行["revision"],
"document_hash": 行["document_hash"],
"document": 存储.恢复草稿(行),
}
return 结果
def 读取正文依据(连, 作者: str, chapter_id: str, revision: int, *, 分支: str = "main") -> dict:
"""事实及其他同事务消费者读取确切正文;同时返回当前指针供来源复检。"""
章节 = 读取章节归属(连, 作者, chapter_id)
@ -440,6 +470,7 @@ __all__ = [
"读取正文依据",
"读取候选依据",
"读取当前正文依据",
"批量读取当前正文依据",
"文稿身份",
"创建写作任务",
"写入模型候选",

View File

@ -0,0 +1,163 @@
"""全书工作面批量读取与归属校验的独立机制验证。"""
from dataclasses import asdict
import pytest
from muse.交付连载.全书整理 import 整理全书工作面
from muse.作品规划.模型 import 作品错误
from muse.正文写作.模型 import 正文草稿, 正文错误, 段落
from muse.正文写作.正文格式 import 可见文本哈希, 正文结构哈希
class 模拟连:
def __init__(self, chapters, *, author="author-1", work_id="work-1", missing=(), corrupt=()):
self.chapters = chapters
self.author = author
self.work_id = work_id
self.missing = set(missing)
self.corrupt = set(corrupt)
self.queries = []
def cursor(self, **_):
return self
def execute(self, sql, params=()):
self.queries.append((sql, params))
if "FROM muse_work w" in sql or "c.chapter_id=ANY(%s)" in sql:
# 批量读取章节归属
ids = params[0]
rows = []
for cid in ids:
if cid.startswith("nonexistent"):
continue
a = "wrong-author" if cid.startswith("wrong-author") else self.author
w = "wrong-work" if cid.startswith("wrong-work") else self.work_id
rows.append(
{
"chapter_id": cid,
"work_id": w,
"title": f"Title {cid}",
"position": 1,
"author_id": a,
"directory_revision": 1,
}
)
self._rows = rows
return self
if "FROM muse_document d JOIN muse_document_version v" in sql:
# 批量读取当前正文
ids = params[0]
branch = params[1]
rows = []
for cid in ids:
if cid in self.missing:
continue
draft = 正文草稿((段落(f"{cid}:p0", ()),))
doc_hash = "bad_hash" if cid in self.corrupt else 正文结构哈希(draft)
vis_hash = 可见文本哈希(draft)
rows.append(
{
"document_id": f"doc-{cid}",
"chapter_id": cid,
"branch_id": branch,
"revision": 1,
"document": asdict(draft),
"document_hash": doc_hash,
"visible_text_hash": vis_hash,
"restored_from": None,
}
)
self._rows = rows
return self
self._rows = []
return self
def fetchall(self):
return getattr(self, "_rows", [])
def fetchone(self):
rows = getattr(self, "_rows", [])
return rows[0] if rows else None
@pytest.mark.case_id("TC-DELIVERY-BATCH-AUTH-CHECK")
def test_批量核对章节归属异常分支():
from muse.作品规划.接口 import 批量核对章节归属
# 1. 空列表直接返回
连 = 模拟连([])
assert 批量核对章节归属(连, "author-1", []) == {}
# 2. 章节不存在整批拒绝
连 = 模拟连([])
with pytest.raises(作品错误) as exc:
批量核对章节归属(连, "author-1", ["c1", "nonexistent-1"])
assert exc.value.错误码 == "CHAPTER_NOT_FOUND"
# 3. 章节不属于当前作者整批拒绝
with pytest.raises(作品错误) as exc:
批量核对章节归属(连, "author-1", ["c1", "wrong-author-c2"])
assert exc.value.错误码 == "SCOPE_DENIED"
assert "章节不属于当前作者" in exc.value.说明
# 4. 章节不属于指定作品整批拒绝
with pytest.raises(作品错误) as exc:
批量核对章节归属(连, "author-1", ["c1", "wrong-work-c2"], work_id="work-1")
assert exc.value.错误码 == "SCOPE_DENIED"
assert "章节不属于指定作品" in exc.value.说明
@pytest.mark.case_id("TC-DELIVERY-BATCH-CORRUPT-DETECT")
def test_批量读取正文损坏则报错():
from muse.正文写作.接口 import 批量读取当前正文依据
连 = 模拟连(["c1"], corrupt=["c1"])
with pytest.raises(正文错误, match="已保存正文与其结构或文本哈希不一致"):
批量读取当前正文依据(连, "author-1", "work-1", ["c1"])
@pytest.mark.case_id("TC-DELIVERY-BATCH-READ-WORKBENCH")
def test_全书批量读取查询次数为常数上界():
from muse.正文写作.接口 import 批量读取当前正文依据
# 100 章批量获取,SQL 查询次数恒定为 1 次批量归属 + 1 次批量正文
c_ids = [f"ch-{i}" for i in range(100)]
连 = 模拟连(c_ids)
结果 = 批量读取当前正文依据(连, "author-1", "work-1", c_ids)
assert len(结果) == 100
assert len(连.queries) == 2 # 1次批量核对归属 + 1次批量正文读取,而非 200 次!
@pytest.mark.case_id("TC-DELIVERY-WORKBENCH-MEM-SEPARATED")
def test_全书工作面组装正确区分状态():
# 模拟包含正常章、缺正文章、空正文章
directory = {
"revision": 5,
"chapters": [
{"chapter_id": "c1", "title": "第一章", "position": 1},
{"chapter_id": "c2", "title": "第二章", "position": 2},
],
}
# c1 正常,c2 缺失
draft = 正文草稿((段落("c1:p0", ()),))
body_map = {
"c1": {
"work_id": "work-1",
"document_id": "doc-c1",
"chapter_id": "c1",
"revision": 1,
"current_revision": 1,
"document_hash": 正文结构哈希(draft),
"document": draft,
}
}
res = 整理全书工作面("work-1", directory, body_map)
assert res["work_id"] == "work-1"
assert res["directory_revision"] == 5
assert len(res["chapters"]) == 2
assert res["chapters"][0]["state"] == "empty" # 空内容
assert res["chapters"][1]["state"] == "missing" # 缺正文
assert not res["can_freeze"] # 有 problem 不能冻结

View File

@ -35317,6 +35317,70 @@
"数据库"
]
},
{
"case_id": "TC-DELIVERY-BATCH-AUTH-CHECK",
"file": "tests/单元/test_全书工作面批量读取.py",
"symbol": "test_批量核对章节归属异常分支",
"parameter_ids": [],
"node_ids": [
"tests/单元/test_全书工作面批量读取.py::test_批量核对章节归属异常分支"
],
"fixtures": [
"request",
"测试资源接缝",
"源码资源",
"离线防护"
],
"markers": []
},
{
"case_id": "TC-DELIVERY-BATCH-CORRUPT-DETECT",
"file": "tests/单元/test_全书工作面批量读取.py",
"symbol": "test_批量读取正文损坏则报错",
"parameter_ids": [],
"node_ids": [
"tests/单元/test_全书工作面批量读取.py::test_批量读取正文损坏则报错"
],
"fixtures": [
"request",
"测试资源接缝",
"源码资源",
"离线防护"
],
"markers": []
},
{
"case_id": "TC-DELIVERY-BATCH-READ-WORKBENCH",
"file": "tests/单元/test_全书工作面批量读取.py",
"symbol": "test_全书批量读取查询次数为常数上界",
"parameter_ids": [],
"node_ids": [
"tests/单元/test_全书工作面批量读取.py::test_全书批量读取查询次数为常数上界"
],
"fixtures": [
"request",
"测试资源接缝",
"源码资源",
"离线防护"
],
"markers": []
},
{
"case_id": "TC-DELIVERY-WORKBENCH-MEM-SEPARATED",
"file": "tests/单元/test_全书工作面批量读取.py",
"symbol": "test_全书工作面组装正确区分状态",
"parameter_ids": [],
"node_ids": [
"tests/单元/test_全书工作面批量读取.py::test_全书工作面组装正确区分状态"
],
"fixtures": [
"request",
"测试资源接缝",
"源码资源",
"离线防护"
],
"markers": []
},
{
"case_id": "TC-O08-SSE-001",
"file": "tests/单元/test_任务事件收尾.py",