优化(交付连载): 全书整理采用批量归属与批量正文读取并移出纯计算
This commit is contained in:
parent
41faed9f31
commit
00aa5e3238
@ -10,18 +10,26 @@ from muse.正文写作.接口 import (
|
||||
准备派生候选,
|
||||
可见文本,
|
||||
应用选段,
|
||||
批量读取当前正文依据,
|
||||
段落哈希,
|
||||
段落文本,
|
||||
读取当前正文依据,
|
||||
读取正文依据,
|
||||
)
|
||||
|
||||
|
||||
def 全书工作面(连, author, work_id, *, names=None, exceptions=()):
|
||||
def 读取全书数据(连, author, work_id):
|
||||
"""同事务批量获取目录与全书当前正文依据。"""
|
||||
directory = 读取作品范围(连, author, work_id)
|
||||
chapter_ids = [c["chapter_id"] for c in directory["chapters"]]
|
||||
body_map = 批量读取当前正文依据(连, author, work_id, chapter_ids)
|
||||
return directory, body_map
|
||||
|
||||
|
||||
def 整理全书工作面(work_id, directory, body_map, *, names=None, exceptions=()):
|
||||
"""纯内存整理工作面结构、段落偏移与文字疑点,不占用数据库连接。"""
|
||||
chapters, problems = [], []
|
||||
for chapter in directory["chapters"]:
|
||||
body = 读取当前正文依据(连, author, chapter["chapter_id"])
|
||||
body = body_map.get(chapter["chapter_id"])
|
||||
if body is None:
|
||||
chapters.append({**chapter, "state": "missing", "revision": None})
|
||||
problems.append({"chapter_id": chapter["chapter_id"], "kind": "missing_body"})
|
||||
@ -67,6 +75,11 @@ def 全书工作面(连, author, work_id, *, names=None, exceptions=()):
|
||||
}
|
||||
|
||||
|
||||
def 全书工作面(连, author, work_id, *, names=None, exceptions=()):
|
||||
directory, body_map = 读取全书数据(连, author, work_id)
|
||||
return 整理全书工作面(work_id, directory, body_map, names=names, exceptions=exceptions)
|
||||
|
||||
|
||||
def 跨章候选身份(author, req, chapter):
|
||||
return str(
|
||||
uuid5(
|
||||
|
||||
@ -49,11 +49,12 @@ class 交付服务:
|
||||
raise 交付错误("SCOPE_DENIED", "交付需要生产用途的本作者身份")
|
||||
|
||||
def 全书回看(self, 身份, work_id, *, names=None, exceptions=()):
|
||||
from muse.交付连载.全书整理 import 全书工作面
|
||||
from muse.交付连载.全书整理 import 整理全书工作面, 读取全书数据
|
||||
|
||||
self.核对身份(身份)
|
||||
with self.数据库.连接(只读=True) as 连:
|
||||
return 全书工作面(连, 身份.作者, work_id, names=names, exceptions=exceptions)
|
||||
directory, body_map = 读取全书数据(连, 身份.作者, work_id)
|
||||
return 整理全书工作面(work_id, directory, body_map, names=names, exceptions=exceptions)
|
||||
|
||||
def 提出跨章修订(self, 身份, 命令ID, 请求: 全书修订):
|
||||
self.核对身份(身份)
|
||||
|
||||
@ -242,6 +242,19 @@ class 作品存储:
|
||||
.fetchone()
|
||||
)
|
||||
|
||||
def 批量读取章节归属(self, chapter_ids: list[str]) -> list[dict]:
|
||||
if not chapter_ids:
|
||||
return []
|
||||
return (
|
||||
self.连.cursor(row_factory=dict_row)
|
||||
.execute(
|
||||
"SELECT c.chapter_id,c.work_id,c.title,c.position,w.author_id,w.directory_revision "
|
||||
"FROM muse_chapter c JOIN muse_work w USING(work_id) WHERE c.chapter_id=ANY(%s)",
|
||||
(chapter_ids,),
|
||||
)
|
||||
.fetchall()
|
||||
)
|
||||
|
||||
|
||||
class 规划存储:
|
||||
def __init__(self, 连):
|
||||
|
||||
@ -282,6 +282,28 @@ def 读取章节归属(连, 作者: str, chapter_id: str) -> dict:
|
||||
return {k: v for k, v in 行.items() if k != "author_id"}
|
||||
|
||||
|
||||
def 批量核对章节归属(
|
||||
连, 作者: str, chapter_ids: list[str], *, work_id: str | None = None
|
||||
) -> dict[str, dict]:
|
||||
"""同事务批量核对章节归属;任一不存在或越界整批拒绝,返回按 chapter_id 索引的字典。"""
|
||||
if not chapter_ids:
|
||||
return {}
|
||||
唯一ID = list(dict.fromkeys(chapter_ids))
|
||||
rows = 作品存储(连).批量读取章节归属(唯一ID)
|
||||
if len(rows) != len(唯一ID):
|
||||
查到ID = {r["chapter_id"] for r in rows}
|
||||
缺失 = [cid for cid in 唯一ID if cid not in 查到ID]
|
||||
raise 作品错误("CHAPTER_NOT_FOUND", f"章节不存在: {sorted(缺失)}")
|
||||
结果 = {}
|
||||
for 行 in rows:
|
||||
if 行["author_id"] != 作者:
|
||||
raise 作品错误("SCOPE_DENIED", "章节不属于当前作者")
|
||||
if work_id is not None and 行["work_id"] != work_id:
|
||||
raise 作品错误("SCOPE_DENIED", "章节不属于指定作品")
|
||||
结果[行["chapter_id"]] = {k: v for k, v in 行.items() if k != "author_id"}
|
||||
return 结果
|
||||
|
||||
|
||||
def 读取有界作品范围(连, 作者: str, work_id: str, *, 数量上限: int, 字节上限: int) -> dict:
|
||||
"""在同一语句限制目录传输;超限只交回数量/大小,调用方决定拒绝口径。"""
|
||||
from muse.作品规划.存储 import 作品存储
|
||||
@ -312,6 +334,7 @@ __all__ = [
|
||||
"作品错误",
|
||||
"登记作品参与者",
|
||||
"读取章节归属",
|
||||
"批量核对章节归属",
|
||||
"创作进度服务",
|
||||
"读取作品范围",
|
||||
"探索服务",
|
||||
|
||||
@ -93,6 +93,22 @@ class 正文存储:
|
||||
.fetchone()
|
||||
)
|
||||
|
||||
def 批量读取当前(self, chapter_ids: list[str], 分支: str = "main") -> list[dict]:
|
||||
if not chapter_ids:
|
||||
return []
|
||||
return (
|
||||
self.连.cursor(row_factory=dict_row)
|
||||
.execute(
|
||||
"SELECT d.document_id,d.chapter_id,d.branch_id,v.revision,v.document,"
|
||||
"v.document_hash,v.visible_text_hash,v.restored_from "
|
||||
"FROM muse_document d JOIN muse_document_version v ON v.document_id=d.document_id "
|
||||
"AND v.revision=d.current_revision "
|
||||
"WHERE d.chapter_id=ANY(%s) AND d.branch_id=%s",
|
||||
(chapter_ids, 分支),
|
||||
)
|
||||
.fetchall()
|
||||
)
|
||||
|
||||
def 读取元数据(self, document_id: str, 版本: int) -> dict | None:
|
||||
"""复检只返回版本和大小;数据库侧长度计算不承诺物理 I/O 上界。"""
|
||||
return (
|
||||
|
||||
@ -3,7 +3,7 @@
|
||||
from dataclasses import asdict
|
||||
from uuid import NAMESPACE_URL, uuid5
|
||||
|
||||
from muse.作品规划.接口 import 读取章节归属
|
||||
from muse.作品规划.接口 import 批量核对章节归属, 读取章节归属
|
||||
from muse.基础设施.数据库.连接 import 数据库工厂
|
||||
from muse.正式变更.接口 import 作者动作, 参与者目录, 变更命令, 变更错误, 正式变更服务
|
||||
from muse.正文写作.人工保存 import 文稿依赖, 正文保存入口, 正文参与者
|
||||
@ -66,6 +66,36 @@ def 读取当前正文依据(连, 作者: str, chapter_id: str, *, 分支: str =
|
||||
}
|
||||
|
||||
|
||||
def 批量读取当前正文依据(
|
||||
连,
|
||||
作者: str,
|
||||
work_id: str,
|
||||
chapter_ids: list[str],
|
||||
*,
|
||||
分支: str = "main",
|
||||
) -> dict[str, dict]:
|
||||
"""批量核对章位归属后读取当前正式正文;未保存不返回条目,由调用方作为缺文处理。"""
|
||||
if not chapter_ids:
|
||||
return {}
|
||||
归属 = 批量核对章节归属(连, 作者, chapter_ids, work_id=work_id)
|
||||
存储 = 正文存储(连)
|
||||
行列表 = 存储.批量读取当前(list(归属.keys()), 分支=分支)
|
||||
结果 = {}
|
||||
for 行 in 行列表:
|
||||
cid = 行["chapter_id"]
|
||||
id_ = 文稿身份(cid, 分支)
|
||||
结果[cid] = {
|
||||
"work_id": work_id,
|
||||
"document_id": id_,
|
||||
"chapter_id": cid,
|
||||
"revision": 行["revision"],
|
||||
"current_revision": 行["revision"],
|
||||
"document_hash": 行["document_hash"],
|
||||
"document": 存储.恢复草稿(行),
|
||||
}
|
||||
return 结果
|
||||
|
||||
|
||||
def 读取正文依据(连, 作者: str, chapter_id: str, revision: int, *, 分支: str = "main") -> dict:
|
||||
"""事实及其他同事务消费者读取确切正文;同时返回当前指针供来源复检。"""
|
||||
章节 = 读取章节归属(连, 作者, chapter_id)
|
||||
@ -440,6 +470,7 @@ __all__ = [
|
||||
"读取正文依据",
|
||||
"读取候选依据",
|
||||
"读取当前正文依据",
|
||||
"批量读取当前正文依据",
|
||||
"文稿身份",
|
||||
"创建写作任务",
|
||||
"写入模型候选",
|
||||
|
||||
163
tests/单元/test_全书工作面批量读取.py
Normal file
163
tests/单元/test_全书工作面批量读取.py
Normal file
@ -0,0 +1,163 @@
|
||||
"""全书工作面批量读取与归属校验的独立机制验证。"""
|
||||
|
||||
from dataclasses import asdict
|
||||
|
||||
import pytest
|
||||
|
||||
from muse.交付连载.全书整理 import 整理全书工作面
|
||||
from muse.作品规划.模型 import 作品错误
|
||||
from muse.正文写作.模型 import 正文草稿, 正文错误, 段落
|
||||
from muse.正文写作.正文格式 import 可见文本哈希, 正文结构哈希
|
||||
|
||||
|
||||
class 模拟连:
|
||||
def __init__(self, chapters, *, author="author-1", work_id="work-1", missing=(), corrupt=()):
|
||||
self.chapters = chapters
|
||||
self.author = author
|
||||
self.work_id = work_id
|
||||
self.missing = set(missing)
|
||||
self.corrupt = set(corrupt)
|
||||
self.queries = []
|
||||
|
||||
def cursor(self, **_):
|
||||
return self
|
||||
|
||||
def execute(self, sql, params=()):
|
||||
self.queries.append((sql, params))
|
||||
if "FROM muse_work w" in sql or "c.chapter_id=ANY(%s)" in sql:
|
||||
# 批量读取章节归属
|
||||
ids = params[0]
|
||||
rows = []
|
||||
for cid in ids:
|
||||
if cid.startswith("nonexistent"):
|
||||
continue
|
||||
a = "wrong-author" if cid.startswith("wrong-author") else self.author
|
||||
w = "wrong-work" if cid.startswith("wrong-work") else self.work_id
|
||||
rows.append(
|
||||
{
|
||||
"chapter_id": cid,
|
||||
"work_id": w,
|
||||
"title": f"Title {cid}",
|
||||
"position": 1,
|
||||
"author_id": a,
|
||||
"directory_revision": 1,
|
||||
}
|
||||
)
|
||||
self._rows = rows
|
||||
return self
|
||||
|
||||
if "FROM muse_document d JOIN muse_document_version v" in sql:
|
||||
# 批量读取当前正文
|
||||
ids = params[0]
|
||||
branch = params[1]
|
||||
rows = []
|
||||
for cid in ids:
|
||||
if cid in self.missing:
|
||||
continue
|
||||
draft = 正文草稿((段落(f"{cid}:p0", ()),))
|
||||
doc_hash = "bad_hash" if cid in self.corrupt else 正文结构哈希(draft)
|
||||
vis_hash = 可见文本哈希(draft)
|
||||
rows.append(
|
||||
{
|
||||
"document_id": f"doc-{cid}",
|
||||
"chapter_id": cid,
|
||||
"branch_id": branch,
|
||||
"revision": 1,
|
||||
"document": asdict(draft),
|
||||
"document_hash": doc_hash,
|
||||
"visible_text_hash": vis_hash,
|
||||
"restored_from": None,
|
||||
}
|
||||
)
|
||||
self._rows = rows
|
||||
return self
|
||||
|
||||
self._rows = []
|
||||
return self
|
||||
|
||||
def fetchall(self):
|
||||
return getattr(self, "_rows", [])
|
||||
|
||||
def fetchone(self):
|
||||
rows = getattr(self, "_rows", [])
|
||||
return rows[0] if rows else None
|
||||
|
||||
|
||||
@pytest.mark.case_id("TC-DELIVERY-BATCH-AUTH-CHECK")
|
||||
def test_批量核对章节归属异常分支():
|
||||
from muse.作品规划.接口 import 批量核对章节归属
|
||||
|
||||
# 1. 空列表直接返回
|
||||
连 = 模拟连([])
|
||||
assert 批量核对章节归属(连, "author-1", []) == {}
|
||||
|
||||
# 2. 章节不存在整批拒绝
|
||||
连 = 模拟连([])
|
||||
with pytest.raises(作品错误) as exc:
|
||||
批量核对章节归属(连, "author-1", ["c1", "nonexistent-1"])
|
||||
assert exc.value.错误码 == "CHAPTER_NOT_FOUND"
|
||||
|
||||
# 3. 章节不属于当前作者整批拒绝
|
||||
with pytest.raises(作品错误) as exc:
|
||||
批量核对章节归属(连, "author-1", ["c1", "wrong-author-c2"])
|
||||
assert exc.value.错误码 == "SCOPE_DENIED"
|
||||
assert "章节不属于当前作者" in exc.value.说明
|
||||
|
||||
# 4. 章节不属于指定作品整批拒绝
|
||||
with pytest.raises(作品错误) as exc:
|
||||
批量核对章节归属(连, "author-1", ["c1", "wrong-work-c2"], work_id="work-1")
|
||||
assert exc.value.错误码 == "SCOPE_DENIED"
|
||||
assert "章节不属于指定作品" in exc.value.说明
|
||||
|
||||
|
||||
@pytest.mark.case_id("TC-DELIVERY-BATCH-CORRUPT-DETECT")
|
||||
def test_批量读取正文损坏则报错():
|
||||
from muse.正文写作.接口 import 批量读取当前正文依据
|
||||
|
||||
连 = 模拟连(["c1"], corrupt=["c1"])
|
||||
with pytest.raises(正文错误, match="已保存正文与其结构或文本哈希不一致"):
|
||||
批量读取当前正文依据(连, "author-1", "work-1", ["c1"])
|
||||
|
||||
|
||||
@pytest.mark.case_id("TC-DELIVERY-BATCH-READ-WORKBENCH")
|
||||
def test_全书批量读取查询次数为常数上界():
|
||||
from muse.正文写作.接口 import 批量读取当前正文依据
|
||||
|
||||
# 100 章批量获取,SQL 查询次数恒定为 1 次批量归属 + 1 次批量正文
|
||||
c_ids = [f"ch-{i}" for i in range(100)]
|
||||
连 = 模拟连(c_ids)
|
||||
结果 = 批量读取当前正文依据(连, "author-1", "work-1", c_ids)
|
||||
assert len(结果) == 100
|
||||
assert len(连.queries) == 2 # 1次批量核对归属 + 1次批量正文读取,而非 200 次!
|
||||
|
||||
|
||||
@pytest.mark.case_id("TC-DELIVERY-WORKBENCH-MEM-SEPARATED")
|
||||
def test_全书工作面组装正确区分状态():
|
||||
# 模拟包含正常章、缺正文章、空正文章
|
||||
directory = {
|
||||
"revision": 5,
|
||||
"chapters": [
|
||||
{"chapter_id": "c1", "title": "第一章", "position": 1},
|
||||
{"chapter_id": "c2", "title": "第二章", "position": 2},
|
||||
],
|
||||
}
|
||||
# c1 正常,c2 缺失
|
||||
draft = 正文草稿((段落("c1:p0", ()),))
|
||||
body_map = {
|
||||
"c1": {
|
||||
"work_id": "work-1",
|
||||
"document_id": "doc-c1",
|
||||
"chapter_id": "c1",
|
||||
"revision": 1,
|
||||
"current_revision": 1,
|
||||
"document_hash": 正文结构哈希(draft),
|
||||
"document": draft,
|
||||
}
|
||||
}
|
||||
res = 整理全书工作面("work-1", directory, body_map)
|
||||
assert res["work_id"] == "work-1"
|
||||
assert res["directory_revision"] == 5
|
||||
assert len(res["chapters"]) == 2
|
||||
assert res["chapters"][0]["state"] == "empty" # 空内容
|
||||
assert res["chapters"][1]["state"] == "missing" # 缺正文
|
||||
assert not res["can_freeze"] # 有 problem 不能冻结
|
||||
@ -35317,6 +35317,70 @@
|
||||
"数据库"
|
||||
]
|
||||
},
|
||||
{
|
||||
"case_id": "TC-DELIVERY-BATCH-AUTH-CHECK",
|
||||
"file": "tests/单元/test_全书工作面批量读取.py",
|
||||
"symbol": "test_批量核对章节归属异常分支",
|
||||
"parameter_ids": [],
|
||||
"node_ids": [
|
||||
"tests/单元/test_全书工作面批量读取.py::test_批量核对章节归属异常分支"
|
||||
],
|
||||
"fixtures": [
|
||||
"request",
|
||||
"测试资源接缝",
|
||||
"源码资源",
|
||||
"离线防护"
|
||||
],
|
||||
"markers": []
|
||||
},
|
||||
{
|
||||
"case_id": "TC-DELIVERY-BATCH-CORRUPT-DETECT",
|
||||
"file": "tests/单元/test_全书工作面批量读取.py",
|
||||
"symbol": "test_批量读取正文损坏则报错",
|
||||
"parameter_ids": [],
|
||||
"node_ids": [
|
||||
"tests/单元/test_全书工作面批量读取.py::test_批量读取正文损坏则报错"
|
||||
],
|
||||
"fixtures": [
|
||||
"request",
|
||||
"测试资源接缝",
|
||||
"源码资源",
|
||||
"离线防护"
|
||||
],
|
||||
"markers": []
|
||||
},
|
||||
{
|
||||
"case_id": "TC-DELIVERY-BATCH-READ-WORKBENCH",
|
||||
"file": "tests/单元/test_全书工作面批量读取.py",
|
||||
"symbol": "test_全书批量读取查询次数为常数上界",
|
||||
"parameter_ids": [],
|
||||
"node_ids": [
|
||||
"tests/单元/test_全书工作面批量读取.py::test_全书批量读取查询次数为常数上界"
|
||||
],
|
||||
"fixtures": [
|
||||
"request",
|
||||
"测试资源接缝",
|
||||
"源码资源",
|
||||
"离线防护"
|
||||
],
|
||||
"markers": []
|
||||
},
|
||||
{
|
||||
"case_id": "TC-DELIVERY-WORKBENCH-MEM-SEPARATED",
|
||||
"file": "tests/单元/test_全书工作面批量读取.py",
|
||||
"symbol": "test_全书工作面组装正确区分状态",
|
||||
"parameter_ids": [],
|
||||
"node_ids": [
|
||||
"tests/单元/test_全书工作面批量读取.py::test_全书工作面组装正确区分状态"
|
||||
],
|
||||
"fixtures": [
|
||||
"request",
|
||||
"测试资源接缝",
|
||||
"源码资源",
|
||||
"离线防护"
|
||||
],
|
||||
"markers": []
|
||||
},
|
||||
{
|
||||
"case_id": "TC-O08-SSE-001",
|
||||
"file": "tests/单元/test_任务事件收尾.py",
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user