- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
528 lines
20 KiB
Python
528 lines
20 KiB
Python
"""PG原生作品/章/block合同;合成文本,不读取旧库或私人快照。"""
|
||
|
||
from copy import deepcopy
|
||
|
||
import pytest
|
||
from pydantic import TypeAdapter
|
||
|
||
from muse.正文写作.接口 import 可见文本, 可见文本哈希, 正文结构哈希, 正文草稿
|
||
from PG作品映射 import PG目标ID, 核对PG组, 指针
|
||
from 建立映射 import 数据集命名空间, 旧快照, 映射配置, 迁移错误
|
||
from 知识记录分流 import 分流批次
|
||
from 转换PG正文 import 转换PG正文
|
||
|
||
|
||
def 源(表, id_, *, tenant=0, **值):
|
||
return 旧快照.从载荷(
|
||
{
|
||
"source": {
|
||
"system": "pg",
|
||
"database": 数据集命名空间("pg-content-test", schema="public", tenant_id=tenant),
|
||
"table": "public.muse_content_" + 表,
|
||
"id": str(id_),
|
||
"revision": "1",
|
||
},
|
||
"record": {"id": id_, "revision": 1, "tenant_id": tenant, "deleted": False, **值},
|
||
}
|
||
)
|
||
|
||
|
||
def 合成PG组(*, use="author", 多块=True):
|
||
w = 源(
|
||
"work",
|
||
12,
|
||
owner_user_id=1,
|
||
title=" 合成作品 ",
|
||
status="writing",
|
||
description="原描述",
|
||
chapter_count=3,
|
||
)
|
||
c1 = 源(
|
||
"chapter",
|
||
101,
|
||
tenant=1,
|
||
work_id=12,
|
||
title="首章",
|
||
order_no=4,
|
||
status="confirmed",
|
||
goal_snapshot={"旧键": "原样"},
|
||
)
|
||
c2 = 源("chapter", 102, work_id=12, title="空块章", order_no=8, status="draft")
|
||
c3 = 源("chapter", 103, work_id=12, title="无块章", order_no=9, status="draft")
|
||
b1 = 源(
|
||
"block",
|
||
201,
|
||
tenant=1,
|
||
work_id=12,
|
||
chapter_id=101,
|
||
order_no=3,
|
||
block_type="scene",
|
||
content_doc=None,
|
||
content_text="\ufeff 甲\r\n\n乙 \r",
|
||
)
|
||
b2 = 源(
|
||
"block",
|
||
202,
|
||
tenant=1,
|
||
work_id=12,
|
||
chapter_id=101,
|
||
order_no=7,
|
||
block_type="scene",
|
||
content_doc=None,
|
||
content_text=" 丙\n",
|
||
)
|
||
b3 = 源(
|
||
"block",
|
||
203,
|
||
work_id=12,
|
||
chapter_id=102,
|
||
order_no=1,
|
||
block_type="scene",
|
||
content_doc=None,
|
||
content_text="",
|
||
)
|
||
blocks = [b1, b2] if 多块 else [b1]
|
||
rows = [w, c1, c2, c3, *blocks, b3]
|
||
value = {
|
||
"works": [],
|
||
"pg_author_id": "migration-author",
|
||
"pg_works": [
|
||
{
|
||
"work": 指针(w),
|
||
"use": use,
|
||
"chapters": [
|
||
{
|
||
"chapter": 指针(c1),
|
||
"blocks": [指针(b) for b in blocks],
|
||
"block_separator": "\n",
|
||
},
|
||
{"chapter": 指针(c2), "blocks": [指针(b3)], "block_separator": "\n"},
|
||
{"chapter": 指针(c3), "blocks": [], "block_separator": "\n"},
|
||
],
|
||
}
|
||
],
|
||
}
|
||
return rows, value
|
||
|
||
|
||
def test_PG原身份_显式用途_跨namespace父子边_来源字段保全__19a2b9():
|
||
rows, value = 合成PG组()
|
||
before = [s.导出() for s in rows]
|
||
result = 分流批次(list(reversed(rows)), 映射配置(value))
|
||
assert all(r.disposition == "待转换" for r in result)
|
||
assert {r.type_id for r in result} == {"pg_profile", "pg_directory", "pg_body"}
|
||
assert [s.导出() for s in rows] == before
|
||
assert all(r.original["source"]["system"] == "pg" for r in result)
|
||
assert rows[0].原行["description"] == "原描述"
|
||
value["pg_works"][0]["use"] = "reference"
|
||
assert {r.owner for r in 分流批次(rows, 映射配置(value))} == {"B03"}
|
||
del value["pg_works"][0]["use"]
|
||
with pytest.raises(迁移错误, match="pg_mapping_invalid"):
|
||
映射配置(value)
|
||
|
||
|
||
def test_PG多block拼接_可见文本结构哈希_每块段落映射_空块不等无块__41c291():
|
||
rows, _ = 合成PG组()
|
||
blocks = rows[4:6]
|
||
result = 转换PG正文(blocks)
|
||
doc = TypeAdapter(正文草稿).validate_python(result["document"])
|
||
assert 可见文本(doc) == "\n".join(b.原行["content_text"] for b in blocks)
|
||
assert result["visible_text_hash"] == 可见文本哈希(doc)
|
||
assert result["document_hash"] == 正文结构哈希(doc)
|
||
assert [b["source_key"] for b in result["block_map"]] == [b.source.记录键 for b in blocks]
|
||
assert [b["order_no"] for b in result["block_map"]] == [3, 7]
|
||
for old, mapping, paragraph in zip(blocks, result["block_map"], doc.paragraphs, strict=True):
|
||
assert mapping["source_hash"] == old.源哈希
|
||
assert mapping["paragraph_map"][0]["source_scene_id"] == str(old.原行["id"])
|
||
assert mapping["paragraph_map"][0]["paragraph_id"] == paragraph.paragraph_id
|
||
empty = 转换PG正文([rows[-1]])
|
||
assert 可见文本(TypeAdapter(正文草稿).validate_python(empty["document"])) == ""
|
||
assert len(empty["block_map"]) == 1
|
||
with pytest.raises(迁移错误, match="pg_no_blocks"):
|
||
转换PG正文([])
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"change,code",
|
||
[
|
||
("hash", "pg_source_drift"),
|
||
("order", "pg_parent_conflict"),
|
||
("parent", "pg_parent_conflict"),
|
||
("unknown", "pg_unknown_fields"),
|
||
("owner", "pg_owner_conflict"),
|
||
],
|
||
)
|
||
def test_PG拒绝漂移错序错父未知字段与伪身份__36cc92(change, code):
|
||
rows, value = 合成PG组()
|
||
if change == "hash":
|
||
value["pg_works"][0]["work"]["source_hash"] = "0" * 64
|
||
elif change == "order":
|
||
value["pg_works"][0]["chapters"][0]["blocks"].reverse()
|
||
else:
|
||
i = 0 if change == "owner" else 4
|
||
raw = rows[i].导出()
|
||
raw.pop("source_hash")
|
||
raw["record"].update(
|
||
{
|
||
"parent": {"chapter_id": 102},
|
||
"unknown": {"未知列": 1},
|
||
"owner": {"owner_user_id": 2},
|
||
}[change]
|
||
)
|
||
rows[i] = 旧快照.从载荷(raw)
|
||
if i == 0:
|
||
value["pg_works"][0]["work"] = 指针(rows[i])
|
||
else:
|
||
value["pg_works"][0]["chapters"][0]["blocks"][0] = 指针(rows[i])
|
||
mapping = 映射配置(value)
|
||
with pytest.raises(迁移错误, match=code):
|
||
核对PG组(mapping, rows[4], {s.source.记录键: s for s in rows})
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"doc", [{}, [], {"type": "doc", "content": []}, {"format_version": 9, "paragraphs": []}]
|
||
)
|
||
def test_PG未知文档不能丢结构只拿text__43830b(doc):
|
||
block = 源(
|
||
"block",
|
||
1,
|
||
work_id=1,
|
||
chapter_id=2,
|
||
order_no=1,
|
||
content_doc=doc,
|
||
content_text="不能掩盖未知结构",
|
||
)
|
||
original = block.导出()
|
||
with pytest.raises(迁移错误, match="pg_document_invalid"):
|
||
转换PG正文([block])
|
||
assert block.导出() == original
|
||
|
||
|
||
def test_PG_document_v1保旧段样式_重新定段ID防跨块碰撞__75309d():
|
||
doc = {
|
||
"format_version": 1,
|
||
"paragraphs": [
|
||
{"paragraph_id": "p1", "content": [{"type": "text", "text": "甲", "marks": ["strong"]}]}
|
||
],
|
||
}
|
||
blocks = [
|
||
源(
|
||
"block",
|
||
i,
|
||
work_id=1,
|
||
chapter_id=2,
|
||
order_no=i,
|
||
content_doc=deepcopy(doc),
|
||
content_text="甲",
|
||
)
|
||
for i in (1, 2)
|
||
]
|
||
result = 转换PG正文(blocks)
|
||
parsed = TypeAdapter(正文草稿).validate_python(result["document"])
|
||
assert 可见文本(parsed) == "甲\n甲"
|
||
assert len({p.paragraph_id for p in parsed.paragraphs}) == 2
|
||
assert all(p.content[0].marks == ("strong",) for p in parsed.paragraphs)
|
||
assert all(b["paragraph_map"][0]["source_paragraph_id"] == "p1" for b in result["block_map"])
|
||
|
||
|
||
def test_PG目标身份不是tenant副本或输入作者猜测__d0cbd7():
|
||
rows, _ = 合成PG组()
|
||
w = rows[0]
|
||
assert PG目标ID(w, "a", "work") == PG目标ID(w, "a", "work")
|
||
assert PG目标ID(w, "a", "work") != PG目标ID(w, "b", "work")
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"use,large", [("author", False), ("reference", False), ("reference", True)]
|
||
)
|
||
def test_PG分片导入与共享重放_离线公开口替身__2ae4dd(monkeypatch, tmp_path, use, large):
|
||
"""只验证迁移编排及公开口参数;不冒称真实PG/S01执行证据。"""
|
||
from contextlib import nullcontext
|
||
from dataclasses import asdict
|
||
from types import SimpleNamespace
|
||
|
||
import 导入新库
|
||
from muse.作品规划.接口 import 档案结构
|
||
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
|
||
from muse.正式变更.接口 import 变更错误, 固定哈希
|
||
from muse.正文写作.接口 import 段落
|
||
from muse.资料研究.接口 import 导入结果
|
||
from PG台账接线 import 核对PG台账, 检查PG冻结
|
||
from 映射台账 import 迁移命令ID
|
||
|
||
class 公开口替身:
|
||
def __init__(self):
|
||
self.配置 = SimpleNamespace(数据库=None, 运行用途=用途.生产)
|
||
self.正式 = self
|
||
self.calls, self.receipts, self.profiles, self.directories = [], {}, {}, {}
|
||
self.candidates, self.sources = {}, {}
|
||
|
||
def 要求作品(self):
|
||
return self
|
||
|
||
要求正文 = 要求数据库 = 要求知识方法 = 要求故事世界 = 要求审校 = 要求资料 = 要求作品
|
||
|
||
def 连接(self, **_):
|
||
return nullcontext(self)
|
||
|
||
def transaction(self):
|
||
return nullcontext(self)
|
||
|
||
def 读取回执(self, 身份, cmd):
|
||
if cmd not in self.receipts:
|
||
raise 变更错误("RECEIPT_NOT_FOUND", "离线替身无回执")
|
||
return self.receipts[cmd]
|
||
|
||
def 新档案结构(self, 身份, work, schema, version):
|
||
return 档案结构(schema, version, "fixture-schema-hash", "fixture-projection")
|
||
|
||
def _receipt(self, cmd, id_, row):
|
||
r = {"target_ref": id_, "action": "manual_save", "results": [row]}
|
||
self.receipts[cmd] = r
|
||
return r
|
||
|
||
def 保存档案(self, 身份, cmd, req):
|
||
self.calls.append("profile")
|
||
row = {
|
||
"work_id": req.work_id,
|
||
"revision": 1,
|
||
"schema_binding": asdict(req.schema),
|
||
"content_hash": 固定哈希(req.content),
|
||
}
|
||
self.profiles[req.work_id] = row
|
||
self.directories[req.work_id] = [
|
||
{"work_id": req.work_id, "revision": 0, "chapters": [], "nodes": []}
|
||
]
|
||
return self._receipt(cmd, "work:" + req.work_id, row)
|
||
|
||
def 读取档案版本依据(self, 身份, work, version):
|
||
assert version == 1
|
||
return self.profiles[work]
|
||
|
||
def 读取目录(self, 身份, work):
|
||
return self.directories[work][-1]
|
||
|
||
def 添加章节(self, 身份, cmd, work, chapter, title, *, 预期目录版本):
|
||
self.calls.append("directory")
|
||
old = self.directories[work][-1]
|
||
assert old["revision"] == 预期目录版本
|
||
chapters = [*old["chapters"], {"chapter_id": chapter, "title": title}]
|
||
nodes = [*old["nodes"], {"node_id": chapter, "title": title, "kind": "chapter"}]
|
||
self.directories[work].append(
|
||
{
|
||
"work_id": work,
|
||
"revision": 预期目录版本 + 1,
|
||
"chapters": chapters,
|
||
"nodes": nodes,
|
||
}
|
||
)
|
||
return self._receipt(
|
||
cmd,
|
||
"directory:" + work,
|
||
{
|
||
"work_id": work,
|
||
"revision": 预期目录版本 + 1,
|
||
"chapter_ids": [c["chapter_id"] for c in chapters],
|
||
},
|
||
)
|
||
|
||
def 读取节点目录(self, 身份, work, *, 版本):
|
||
return self.directories[work][版本]
|
||
|
||
def 读取正文(self, 身份, chapter, **_):
|
||
work = next(
|
||
w
|
||
for w, ds in self.directories.items()
|
||
if any(c["chapter_id"] == chapter for c in ds[-1]["chapters"])
|
||
)
|
||
return {
|
||
"work_id": work,
|
||
"revision": 0,
|
||
"document_hash": 正文结构哈希(正文草稿((段落("empty", ()),))),
|
||
}
|
||
|
||
def 创建人工候选(self, 身份, cmd, chapter, version, draft, *, 分支):
|
||
self.calls.append("body")
|
||
assert version == 0 and 分支 == "main"
|
||
id_ = "offline-candidate-" + str(len(self.candidates))
|
||
self.candidates[id_] = {
|
||
"chapter_id": chapter,
|
||
"branch_id": 分支,
|
||
"base_revision": 0,
|
||
"origin": "author",
|
||
"revision": 1,
|
||
"document": asdict(draft),
|
||
"candidate_hash": "fixture-candidate-hash",
|
||
}
|
||
return self._receipt(
|
||
cmd,
|
||
id_,
|
||
{"candidate_id": id_, "revision": 1, "candidate_hash": "fixture-candidate-hash"},
|
||
)
|
||
|
||
def 读取候选(self, 身份, id_, *, 版本):
|
||
assert 版本 == 1
|
||
return self.candidates[id_]
|
||
|
||
def 导入(self, 连, author, req):
|
||
import hashlib
|
||
|
||
self.calls.append("reference")
|
||
assert req.authorized_uses == ()
|
||
id_ = "00000000-0000-0000-0000-000000000001"
|
||
h = hashlib.sha256(req.content.encode()).hexdigest()
|
||
self.sources[id_] = {
|
||
"source_id": id_,
|
||
"revision": 1,
|
||
"content": req.content,
|
||
"content_hash": h,
|
||
"import_result": {"导入者": author},
|
||
"kind": "reference",
|
||
"origin": req.origin,
|
||
"authorized_uses": [],
|
||
}
|
||
return 导入结果(id_, 1, h, False)
|
||
|
||
def 读取版本(self, 连, id_, version):
|
||
assert version == 1
|
||
return self.sources[id_]
|
||
|
||
def 列出来源(self, 连):
|
||
from uuid import UUID
|
||
|
||
return [{**s, "source_id": UUID(s["source_id"])} for s in self.sources.values()]
|
||
|
||
app = 公开口替身()
|
||
identity = 调用身份("migration-author", None, 用途.生产, 内容用途.规划)
|
||
records = {}
|
||
|
||
class 台账替身:
|
||
def __init__(self, *args):
|
||
self.装配, self.身份, self.PG参考根缓存 = app, identity, {}
|
||
|
||
def 读取(self, key):
|
||
return records.get(key)
|
||
|
||
def 占用(self, source, mapping_hash, route):
|
||
key = source.source.记录键
|
||
if key not in records:
|
||
records[key] = {
|
||
"source_key": key,
|
||
"source_hash": source.源哈希,
|
||
"mapping_hash": mapping_hash,
|
||
"command_id": 迁移命令ID(identity.作者, key),
|
||
"original": source.导出(),
|
||
"route": route.导出(),
|
||
"state": "reserved",
|
||
"frozen_request": None,
|
||
"receipt": None,
|
||
"targets": [],
|
||
"reason_code": "",
|
||
}
|
||
else:
|
||
assert records[key]["source_hash"] == source.源哈希
|
||
assert records[key]["mapping_hash"] == mapping_hash
|
||
assert records[key]["route"] == route.导出()
|
||
return records[key]
|
||
|
||
def 冻结请求(self, key, req):
|
||
检查PG冻结(records[key], req)
|
||
records[key]["frozen_request"] = deepcopy(req)
|
||
return req
|
||
|
||
def 保存终态(self, key, state, *, 目标=None, 回执=None, 原因=""):
|
||
if state == "mapped":
|
||
assert 核对PG台账(self, records[key], 回执) == 目标
|
||
records[key].update(state=state, targets=目标 or [], receipt=回执, reason_code=原因)
|
||
return records[key]
|
||
|
||
def 列出(self, keys):
|
||
return [records[k] for k in keys]
|
||
|
||
monkeypatch.setattr(导入新库, "数据库工厂", lambda *args: app)
|
||
monkeypatch.setattr(导入新库, "核对隔离目标", lambda *args, **kw: None)
|
||
monkeypatch.setattr(导入新库, "迁移台账", 台账替身)
|
||
from 入口 import _读文件, 文件上限
|
||
from 建立映射 import 稳定JSON
|
||
|
||
rows, value = 合成PG组(use=use)
|
||
if large:
|
||
blocks = [
|
||
源(
|
||
"block",
|
||
1000 + i,
|
||
tenant=1,
|
||
work_id=12,
|
||
chapter_id=101,
|
||
order_no=i + 1,
|
||
block_type="scene",
|
||
content_doc=None,
|
||
content_text="甲" * 16666,
|
||
)
|
||
for i in range(500)
|
||
]
|
||
rows = [*rows[:4], *blocks, rows[-1]]
|
||
value["pg_works"][0]["chapters"][0]["blocks"] = [指针(b) for b in blocks]
|
||
mapping_path = tmp_path / "mapping.json"
|
||
mapping_path.write_text(稳定JSON(value), encoding="utf-8")
|
||
mapping = 映射配置(_读文件(mapping_path))
|
||
params = dict(允许实例=None, 管理引用=None, 目标回执=None)
|
||
|
||
def import_chunk(chunk, index):
|
||
path = tmp_path / (str(index) + ".json")
|
||
path.write_text(稳定JSON([s.导出() for s in chunk]), encoding="utf-8")
|
||
assert path.stat().st_size <= 文件上限
|
||
snapshots = [旧快照.从载荷(p) for p in _读文件(path)]
|
||
result = 导入新库.导入快照(app, identity, snapshots, mapping, **params)
|
||
assert len(稳定JSON(result).encode()) < 16384
|
||
return result
|
||
|
||
first = import_chunk(rows[:4], "metadata")
|
||
assert first["states"] == ({"mapped": 4} if use == "author" else {"reserved": 4})
|
||
assert first["fragment_wait"]["can_continue"] is (use == "reference")
|
||
partial = import_chunk([rows[4]], "partial")
|
||
assert partial["fragment_wait"]["can_continue"] is True
|
||
assert partial["fragment_wait"]["reason_code"] == "pg_dependency_missing"
|
||
assert records[rows[4].source.记录键]["state"] == "reserved"
|
||
if not large:
|
||
from PG台账接线 import PG分片等待
|
||
|
||
known = {r.source.记录键: r for r in rows[:5]}
|
||
ledger = 台账替身()
|
||
reserved = records[rows[4].source.记录键]
|
||
reserved["frozen_request"] = {"interrupted": True}
|
||
assert not PG分片等待(ledger, mapping, known, rows[:5])["can_continue"]
|
||
reserved["frozen_request"] = None
|
||
work = records[rows[0].source.记录键]
|
||
previous = work["state"]
|
||
work["state"] = "quarantined"
|
||
assert not PG分片等待(ledger, mapping, known, rows[:5])["can_continue"]
|
||
work["state"] = previous
|
||
chunks = [rows[5:205], rows[205:405], rows[405:]] if large else [rows[5:]]
|
||
for i, chunk in enumerate(chunks):
|
||
final = import_chunk(chunk, "body-" + str(i))
|
||
assert final["fragment_wait"]["can_continue"] is False
|
||
if large:
|
||
frozen = records[rows[0].source.记录键]["frozen_request"]
|
||
assert len(frozen["payload"]["request"]["content"].encode()) > 24945859
|
||
assert len(稳定JSON(frozen).encode()) > 文件上限
|
||
assert "甲" not in 稳定JSON(final)
|
||
# 根在台账内部,成员没有复制整书;重放也只读原分片,不产巨型CLI文件。
|
||
assert all(
|
||
set(records[s.source.记录键]["frozen_request"]["payload"]) == {"member_of"}
|
||
for s in rows[1:]
|
||
)
|
||
assert all(r["state"] == "mapped" for r in records.values())
|
||
assert app.calls == (
|
||
["profile", "directory", "directory", "directory", "body", "body"]
|
||
if use == "author"
|
||
else ["reference"]
|
||
)
|
||
original_calls = list(app.calls)
|
||
receipt_snapshot = {k: deepcopy(r["receipt"]) for k, r in records.items()}
|
||
for i, chunk in enumerate([rows[:4], [rows[4]], *chunks]):
|
||
import_chunk(chunk, "replay-" + str(i))
|
||
assert app.calls == original_calls
|
||
assert {k: r["receipt"] for k, r in records.items()} == receipt_snapshot
|
||
assert all(r["original"] == s.导出() for s in rows for r in [records[s.source.记录键]])
|