zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

287 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""PG作品清单与原生父子身份;用途只来自显式映射,不由tenant决定。"""
from __future__ import annotations
import re
from 建立映射 import 稳定JSON, 读取JSON, 旧快照, 源身份, 迁移错误
PG内容表 = {"muse_content_work", "muse_content_chapter", "muse_content_block"}
def 是PG内容(源):
return 源.source.system == "pg" and 源.source.table.split(".")[-1] in PG内容表
def 指针(源):
return {"source_key": 源.source.记录键, "source_hash": 源.源哈希}
def _指针(p):
if (
not isinstance(p, dict)
or set(p) != {"source_key", "source_hash"}
or any(not isinstance(p[k], str) or not p[k] for k in p)
or not re.fullmatch(r"[0-9a-f]{64}", p["source_hash"])
):
raise 迁移错误("pg_mapping_invalid", "PG来源必须明确原source_key与source_hash")
def 建立PG索引(组表):
索引, 原生对象 = {}, set()
for 组 in 组表:
if set(组) != {"work", "use", "chapters"} or 组["use"] not in {"author", "reference"}:
raise 迁移错误("pg_mapping_invalid", "PG作品必须显式选择author/reference及完整章节清单")
_指针(组["work"])
if PG源身份(组["work"]).table.split(".")[-1] != "muse_content_work":
raise 迁移错误("pg_mapping_invalid", "work指针必须绑定原作品表")
if not isinstance(组["chapters"], list):
raise 迁移错误("pg_mapping_invalid", "PG章节清单必须是数组")
成员: list[tuple[dict, dict | None]] = [(组["work"], None)]
for 章 in 组["chapters"]:
if (
not isinstance(章, dict)
or set(章) != {"chapter", "blocks", "block_separator"}
or 章.get("block_separator") != "\n"
or not isinstance(章["blocks"], list)
):
raise 迁移错误("pg_mapping_invalid", "PG章必须绑定原章与完整block清单")
_指针(章["chapter"])
if PG源身份(章["chapter"]).table.split(".")[-1] != "muse_content_chapter":
raise 迁移错误("pg_mapping_invalid", "chapter指针必须绑定原章表")
成员.append((章["chapter"], 章))
for 块 in 章["blocks"]:
_指针(块)
if PG源身份(块).table.split(".")[-1] != "muse_content_block":
raise 迁移错误("pg_mapping_invalid", "blocks指针必须绑定原block表")
成员.append((块, 章))
for p, 章 in 成员:
identity = PG源身份(p)
try:
namespace = 读取JSON(identity.database)
except ValueError:
namespace = identity.database
if isinstance(namespace, dict):
namespace = {k: v for k, v in namespace.items() if k != "tenant_id"}
native = 稳定JSON([identity.system, namespace, identity.table, identity.id])
if native in 原生对象:
raise 迁移错误(
"pg_mapping_invalid", "原表id全库唯一;不能按namespace复制同一原生对象"
)
原生对象.add(native)
if p["source_key"] in 索引:
raise 迁移错误("pg_mapping_invalid", "PG源不能重复或同时指定多个作品用途")
索引[p["source_key"]] = (组, 章, p)
return 索引
def PG绑定(映射, 源):
项 = 映射.PG索引.get(源.source.记录键)
if 项 is None:
raise 迁移错误("pg_mapping_missing", "PG作品/章/block缺少显式用途与原来源清单")
if 项[2] != 指针(源):
raise 迁移错误("pg_source_drift", "PG映射与原记录哈希不同")
return 项[0], 项[1]
def PG依赖指针(映射, 源):
组, 章 = PG绑定(映射, 源)
# 完整章清单固定目录顺序;正文只装载本章,参考书导入装载整书。
结果 = [组["work"], *(c["chapter"] for c in 组["chapters"])]
章节 = (
组["chapters"]
if 组["use"] == "reference"
else ([章] if 章 and 源.source.table.split(".")[-1] == "muse_content_block" else [])
)
for c in 章节:
结果.extend(c["blocks"])
return 结果
def 取PG源(已知源, p, 表):
源 = 已知源.get(p["source_key"])
if 源 is None:
raise 迁移错误("pg_dependency_missing", "PG显式依赖尚未提供或留存")
if 指针(源) != p or 源.source.system != "pg" or 源.source.table.split(".")[-1] != 表:
raise 迁移错误("pg_source_drift", "PG依赖的原表、身份或哈希不同")
return 源
def _原记录(源, 字段):
行 = 源.原行
通用 = {
"id",
"revision",
"creator",
"create_time",
"updater",
"update_time",
"deleted",
"tenant_id",
"command_id",
}
if set(行) - 通用 - set(字段):
raise 迁移错误("pg_unknown_fields", "PG原表存在未登记列;保原行而不忽略")
if (
type(行.get("id")) is not int
or type(行.get("tenant_id")) is not int
or type(行.get("revision")) is not int
):
raise 迁移错误("pg_record_invalid", "PG原id、tenant和revision必须是整数")
if 源.source.database.startswith("{"):
namespace = 读取JSON(源.source.database)
if not isinstance(namespace, dict) or namespace.get("tenant_id") != 行["tenant_id"]:
raise 迁移错误("pg_parent_conflict", "原namespace与原行tenant不同,不能覆盖审计身份")
if 行.get("deleted") is not False:
raise 迁移错误("pg_inactive_source", "删除或资格不明的源只保历史,不创建内容")
return 行
def _同库(a, b):
# namespace逐字保留;跨namespace只由两端带哈希的显式父子边连接。
from 建立映射 import 读取JSON
if a.source.database == b.source.database:
return
try:
x, y = 读取JSON(a.source.database), 读取JSON(b.source.database)
if (
not isinstance(x, dict)
or not isinstance(y, dict)
or set(x) != {"dataset", "schema", "tenant_id"}
or set(y) != set(x)
):
raise ValueError
if any(x[k] != y[k] for k in ("dataset", "schema")):
raise ValueError
if x["tenant_id"] != a.原行["tenant_id"] or y["tenant_id"] != b.原行["tenant_id"]:
raise ValueError
except (ValueError, KeyError, TypeError):
raise 迁移错误(
"pg_parent_conflict", "跨namespace父子边必须属于同一数据集/schema且保原tenant"
) from None
def 核对PG组(映射, 源, 已知源):
组, 本章 = PG绑定(映射, 源)
作品 = 取PG源(已知源, 组["work"], "muse_content_work")
行 = _原记录(
作品,
{
"owner_user_id",
"title",
"description",
"genre",
"cover_image_url",
"summary",
"status",
"work_schema_id",
"import_status",
"parse_status",
"word_count",
"chapter_count",
},
)
if 行.get("owner_user_id") != 1 or type(行.get("owner_user_id")) is not int:
raise 迁移错误("pg_owner_conflict", "本单作者迁移合同只承接原owner_user_id=1,不推导新作者")
if not isinstance(行.get("title"), str) or not 行["title"].strip():
raise 迁移错误("pg_record_invalid", "作品标题缺失")
章节, 上序 = [], None
for c in 组["chapters"]:
章 = 取PG源(已知源, c["chapter"], "muse_content_chapter")
r = _原记录(
章,
{
"work_id",
"title",
"order_no",
"status",
"goal_snapshot",
"outline_snapshot",
"parse_review_status",
},
)
_同库(章, 作品)
if (
r.get("work_id") != 行["id"]
or type(r.get("order_no")) is not int
or (上序 is not None and r["order_no"] <= 上序)
):
raise 迁移错误("pg_parent_conflict", "章归属或显式顺序与原work_id/order_no不同")
if not isinstance(r.get("title"), str) or not r["title"].strip():
raise 迁移错误("pg_record_invalid", "章标题缺失")
上序 = r["order_no"]
块组, 上块序 = [], None
if (组["use"] == "reference" and 本章 is None) or (
源.source.table.split(".")[-1] == "muse_content_block" and c == 本章
):
for p in c["blocks"]:
块 = 取PG源(已知源, p, "muse_content_block")
b = _原记录(
块,
{
"work_id",
"chapter_id",
"order_no",
"block_type",
"title",
"content_doc",
"content_text",
"word_count",
},
)
_同库(块, 章)
if (
b.get("work_id") != 行["id"]
or b.get("chapter_id") != r["id"]
or type(b.get("order_no")) is not int
or (上块序 is not None and b["order_no"] <= 上块序)
):
raise 迁移错误(
"pg_parent_conflict", "block归属或顺序与原work/chapter/order不同"
)
上块序 = b["order_no"]
块组.append(块)
章节.append((章, 块组))
return 组, 作品, 章节
def PG源身份(p):
try:
值 = 读取JSON(p["source_key"])
if not isinstance(值, list) or len(值) != 5:
raise ValueError
源 = 源身份(*值)
if 源.system != "pg" or 源.table.split(".")[-1] not in PG内容表:
raise ValueError
return 源
except (ValueError, TypeError):
raise 迁移错误("pg_mapping_invalid", "PG来源指针必须是完整原生内容表记录键") from None
def PG目标ID(源: 旧快照 | 源身份, 作者: str, 种类: str):
from uuid import NAMESPACE_URL, uuid5
identity = 源.source if isinstance(源, 旧快照) else 源
return str(
uuid5(
NAMESPACE_URL,
"muse:legacy:" + 稳定JSON([identity.对象键, "B01", 种类, "author:" + 作者, 作者]),
)
)
def PG排序(映射, 源):
if not 是PG内容(源):
return (3, "", 0, 0)
try:
组, 章 = PG绑定(映射, 源)
if 章 is None:
return (0, 组["work"]["source_key"], 0, 0)
i = 组["chapters"].index(章)
if 指针(源) == 章["chapter"]:
return (1, 组["work"]["source_key"], i, 0)
return (2, 组["work"]["source_key"], i, 章["blocks"].index(指针(源)))
except 迁移错误:
return (3, "", 0, 0)