- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具) - 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺 - 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口 - 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影) - 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威) - R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
287 lines
11 KiB
Python
287 lines
11 KiB
Python
"""PG作品清单与原生父子身份;用途只来自显式映射,不由tenant决定。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
|
||
from 建立映射 import 稳定JSON, 读取JSON, 旧快照, 源身份, 迁移错误
|
||
|
||
PG内容表 = {"muse_content_work", "muse_content_chapter", "muse_content_block"}
|
||
|
||
|
||
def 是PG内容(源):
|
||
return 源.source.system == "pg" and 源.source.table.split(".")[-1] in PG内容表
|
||
|
||
|
||
def 指针(源):
|
||
return {"source_key": 源.source.记录键, "source_hash": 源.源哈希}
|
||
|
||
|
||
def _指针(p):
|
||
if (
|
||
not isinstance(p, dict)
|
||
or set(p) != {"source_key", "source_hash"}
|
||
or any(not isinstance(p[k], str) or not p[k] for k in p)
|
||
or not re.fullmatch(r"[0-9a-f]{64}", p["source_hash"])
|
||
):
|
||
raise 迁移错误("pg_mapping_invalid", "PG来源必须明确原source_key与source_hash")
|
||
|
||
|
||
def 建立PG索引(组表):
|
||
索引, 原生对象 = {}, set()
|
||
for 组 in 组表:
|
||
if set(组) != {"work", "use", "chapters"} or 组["use"] not in {"author", "reference"}:
|
||
raise 迁移错误("pg_mapping_invalid", "PG作品必须显式选择author/reference及完整章节清单")
|
||
_指针(组["work"])
|
||
if PG源身份(组["work"]).table.split(".")[-1] != "muse_content_work":
|
||
raise 迁移错误("pg_mapping_invalid", "work指针必须绑定原作品表")
|
||
if not isinstance(组["chapters"], list):
|
||
raise 迁移错误("pg_mapping_invalid", "PG章节清单必须是数组")
|
||
成员: list[tuple[dict, dict | None]] = [(组["work"], None)]
|
||
for 章 in 组["chapters"]:
|
||
if (
|
||
not isinstance(章, dict)
|
||
or set(章) != {"chapter", "blocks", "block_separator"}
|
||
or 章.get("block_separator") != "\n"
|
||
or not isinstance(章["blocks"], list)
|
||
):
|
||
raise 迁移错误("pg_mapping_invalid", "PG章必须绑定原章与完整block清单")
|
||
_指针(章["chapter"])
|
||
if PG源身份(章["chapter"]).table.split(".")[-1] != "muse_content_chapter":
|
||
raise 迁移错误("pg_mapping_invalid", "chapter指针必须绑定原章表")
|
||
成员.append((章["chapter"], 章))
|
||
for 块 in 章["blocks"]:
|
||
_指针(块)
|
||
if PG源身份(块).table.split(".")[-1] != "muse_content_block":
|
||
raise 迁移错误("pg_mapping_invalid", "blocks指针必须绑定原block表")
|
||
成员.append((块, 章))
|
||
for p, 章 in 成员:
|
||
identity = PG源身份(p)
|
||
try:
|
||
namespace = 读取JSON(identity.database)
|
||
except ValueError:
|
||
namespace = identity.database
|
||
if isinstance(namespace, dict):
|
||
namespace = {k: v for k, v in namespace.items() if k != "tenant_id"}
|
||
native = 稳定JSON([identity.system, namespace, identity.table, identity.id])
|
||
if native in 原生对象:
|
||
raise 迁移错误(
|
||
"pg_mapping_invalid", "原表id全库唯一;不能按namespace复制同一原生对象"
|
||
)
|
||
原生对象.add(native)
|
||
if p["source_key"] in 索引:
|
||
raise 迁移错误("pg_mapping_invalid", "PG源不能重复或同时指定多个作品用途")
|
||
索引[p["source_key"]] = (组, 章, p)
|
||
return 索引
|
||
|
||
|
||
def PG绑定(映射, 源):
|
||
项 = 映射.PG索引.get(源.source.记录键)
|
||
if 项 is None:
|
||
raise 迁移错误("pg_mapping_missing", "PG作品/章/block缺少显式用途与原来源清单")
|
||
if 项[2] != 指针(源):
|
||
raise 迁移错误("pg_source_drift", "PG映射与原记录哈希不同")
|
||
return 项[0], 项[1]
|
||
|
||
|
||
def PG依赖指针(映射, 源):
|
||
组, 章 = PG绑定(映射, 源)
|
||
# 完整章清单固定目录顺序;正文只装载本章,参考书导入装载整书。
|
||
结果 = [组["work"], *(c["chapter"] for c in 组["chapters"])]
|
||
章节 = (
|
||
组["chapters"]
|
||
if 组["use"] == "reference"
|
||
else ([章] if 章 and 源.source.table.split(".")[-1] == "muse_content_block" else [])
|
||
)
|
||
for c in 章节:
|
||
结果.extend(c["blocks"])
|
||
return 结果
|
||
|
||
|
||
def 取PG源(已知源, p, 表):
|
||
源 = 已知源.get(p["source_key"])
|
||
if 源 is None:
|
||
raise 迁移错误("pg_dependency_missing", "PG显式依赖尚未提供或留存")
|
||
if 指针(源) != p or 源.source.system != "pg" or 源.source.table.split(".")[-1] != 表:
|
||
raise 迁移错误("pg_source_drift", "PG依赖的原表、身份或哈希不同")
|
||
return 源
|
||
|
||
|
||
def _原记录(源, 字段):
|
||
行 = 源.原行
|
||
通用 = {
|
||
"id",
|
||
"revision",
|
||
"creator",
|
||
"create_time",
|
||
"updater",
|
||
"update_time",
|
||
"deleted",
|
||
"tenant_id",
|
||
"command_id",
|
||
}
|
||
if set(行) - 通用 - set(字段):
|
||
raise 迁移错误("pg_unknown_fields", "PG原表存在未登记列;保原行而不忽略")
|
||
if (
|
||
type(行.get("id")) is not int
|
||
or type(行.get("tenant_id")) is not int
|
||
or type(行.get("revision")) is not int
|
||
):
|
||
raise 迁移错误("pg_record_invalid", "PG原id、tenant和revision必须是整数")
|
||
if 源.source.database.startswith("{"):
|
||
namespace = 读取JSON(源.source.database)
|
||
if not isinstance(namespace, dict) or namespace.get("tenant_id") != 行["tenant_id"]:
|
||
raise 迁移错误("pg_parent_conflict", "原namespace与原行tenant不同,不能覆盖审计身份")
|
||
if 行.get("deleted") is not False:
|
||
raise 迁移错误("pg_inactive_source", "删除或资格不明的源只保历史,不创建内容")
|
||
return 行
|
||
|
||
|
||
def _同库(a, b):
|
||
# namespace逐字保留;跨namespace只由两端带哈希的显式父子边连接。
|
||
from 建立映射 import 读取JSON
|
||
|
||
if a.source.database == b.source.database:
|
||
return
|
||
try:
|
||
x, y = 读取JSON(a.source.database), 读取JSON(b.source.database)
|
||
if (
|
||
not isinstance(x, dict)
|
||
or not isinstance(y, dict)
|
||
or set(x) != {"dataset", "schema", "tenant_id"}
|
||
or set(y) != set(x)
|
||
):
|
||
raise ValueError
|
||
if any(x[k] != y[k] for k in ("dataset", "schema")):
|
||
raise ValueError
|
||
if x["tenant_id"] != a.原行["tenant_id"] or y["tenant_id"] != b.原行["tenant_id"]:
|
||
raise ValueError
|
||
except (ValueError, KeyError, TypeError):
|
||
raise 迁移错误(
|
||
"pg_parent_conflict", "跨namespace父子边必须属于同一数据集/schema且保原tenant"
|
||
) from None
|
||
|
||
|
||
def 核对PG组(映射, 源, 已知源):
|
||
组, 本章 = PG绑定(映射, 源)
|
||
作品 = 取PG源(已知源, 组["work"], "muse_content_work")
|
||
行 = _原记录(
|
||
作品,
|
||
{
|
||
"owner_user_id",
|
||
"title",
|
||
"description",
|
||
"genre",
|
||
"cover_image_url",
|
||
"summary",
|
||
"status",
|
||
"work_schema_id",
|
||
"import_status",
|
||
"parse_status",
|
||
"word_count",
|
||
"chapter_count",
|
||
},
|
||
)
|
||
if 行.get("owner_user_id") != 1 or type(行.get("owner_user_id")) is not int:
|
||
raise 迁移错误("pg_owner_conflict", "本单作者迁移合同只承接原owner_user_id=1,不推导新作者")
|
||
if not isinstance(行.get("title"), str) or not 行["title"].strip():
|
||
raise 迁移错误("pg_record_invalid", "作品标题缺失")
|
||
章节, 上序 = [], None
|
||
for c in 组["chapters"]:
|
||
章 = 取PG源(已知源, c["chapter"], "muse_content_chapter")
|
||
r = _原记录(
|
||
章,
|
||
{
|
||
"work_id",
|
||
"title",
|
||
"order_no",
|
||
"status",
|
||
"goal_snapshot",
|
||
"outline_snapshot",
|
||
"parse_review_status",
|
||
},
|
||
)
|
||
_同库(章, 作品)
|
||
if (
|
||
r.get("work_id") != 行["id"]
|
||
or type(r.get("order_no")) is not int
|
||
or (上序 is not None and r["order_no"] <= 上序)
|
||
):
|
||
raise 迁移错误("pg_parent_conflict", "章归属或显式顺序与原work_id/order_no不同")
|
||
if not isinstance(r.get("title"), str) or not r["title"].strip():
|
||
raise 迁移错误("pg_record_invalid", "章标题缺失")
|
||
上序 = r["order_no"]
|
||
块组, 上块序 = [], None
|
||
if (组["use"] == "reference" and 本章 is None) or (
|
||
源.source.table.split(".")[-1] == "muse_content_block" and c == 本章
|
||
):
|
||
for p in c["blocks"]:
|
||
块 = 取PG源(已知源, p, "muse_content_block")
|
||
b = _原记录(
|
||
块,
|
||
{
|
||
"work_id",
|
||
"chapter_id",
|
||
"order_no",
|
||
"block_type",
|
||
"title",
|
||
"content_doc",
|
||
"content_text",
|
||
"word_count",
|
||
},
|
||
)
|
||
_同库(块, 章)
|
||
if (
|
||
b.get("work_id") != 行["id"]
|
||
or b.get("chapter_id") != r["id"]
|
||
or type(b.get("order_no")) is not int
|
||
or (上块序 is not None and b["order_no"] <= 上块序)
|
||
):
|
||
raise 迁移错误(
|
||
"pg_parent_conflict", "block归属或顺序与原work/chapter/order不同"
|
||
)
|
||
上块序 = b["order_no"]
|
||
块组.append(块)
|
||
章节.append((章, 块组))
|
||
return 组, 作品, 章节
|
||
|
||
|
||
def PG源身份(p):
|
||
try:
|
||
值 = 读取JSON(p["source_key"])
|
||
if not isinstance(值, list) or len(值) != 5:
|
||
raise ValueError
|
||
源 = 源身份(*值)
|
||
if 源.system != "pg" or 源.table.split(".")[-1] not in PG内容表:
|
||
raise ValueError
|
||
return 源
|
||
except (ValueError, TypeError):
|
||
raise 迁移错误("pg_mapping_invalid", "PG来源指针必须是完整原生内容表记录键") from None
|
||
|
||
|
||
def PG目标ID(源: 旧快照 | 源身份, 作者: str, 种类: str):
|
||
from uuid import NAMESPACE_URL, uuid5
|
||
|
||
identity = 源.source if isinstance(源, 旧快照) else 源
|
||
return str(
|
||
uuid5(
|
||
NAMESPACE_URL,
|
||
"muse:legacy:" + 稳定JSON([identity.对象键, "B01", 种类, "author:" + 作者, 作者]),
|
||
)
|
||
)
|
||
|
||
|
||
def PG排序(映射, 源):
|
||
if not 是PG内容(源):
|
||
return (3, "", 0, 0)
|
||
try:
|
||
组, 章 = PG绑定(映射, 源)
|
||
if 章 is None:
|
||
return (0, 组["work"]["source_key"], 0, 0)
|
||
i = 组["chapters"].index(章)
|
||
if 指针(源) == 章["chapter"]:
|
||
return (1, 组["work"]["source_key"], i, 0)
|
||
return (2, 组["work"]["source_key"], i, 章["blocks"].index(指针(源)))
|
||
except 迁移错误:
|
||
return (3, "", 0, 0)
|