"""固定定稿格式的文本保真、成品核对及标准包结构。""" import copy import hashlib from io import BytesIO from uuid import UUID from xml.etree import ElementTree as ET from zipfile import ZIP_STORED, ZipFile import pytest from muse.交付连载.导出格式 import EPUB, Markdown, Word, 导出, 校验, 纯文本 from muse.交付连载.模型 import 交付错误 格式 = ("txt", "md", "docx", "epub") 读取器 = {"txt": 纯文本.读取, "md": Markdown.读取, "docx": Word.读取, "epub": EPUB.读取} @pytest.fixture def 固定定稿(): paragraphs = [ " 原文标记:夜色里,青禾握着旧钥匙。 ", "", "", "半角 空格\t制表\r单CR\n单LF\r\n双字符换行。", "e\u0301 ≠ é;家庭👨‍👩‍👧‍👦;😀;汉字𠮷。", "# 标题 *不是强调* `代码` [链接](https://example.invalid) ", '章节 {"paragraph_bytes":[999]}\n', "\t  \u00a0 ", "", ] return { "delivery_id": UUID("00000000-0000-4000-8000-000000000027"), "manifest_hash": "a" * 64, "manifest": { "materials": { "title": '雾港 #1 <>& "题名"', "author_name": "青禾 & 海风", "introduction": "简介首行\r\n第二行 连续空白\t结尾。", "synopsis": "梗概与正文独立。", "characters": "青禾:守门人。\n海风:旅人。", } }, "chapters": [ { "chapter_id": 'c-2<&"', "title": "第一章 # 标题 <>& *正文*", "paragraphs": paragraphs, "text": "\n".join(paragraphs), }, { "chapter_id": "c-1", "title": "第二章", "paragraphs": ["", "最后一盏灯。", ""], "text": "\n最后一盏灯。\n", }, ], } @pytest.mark.case_id( "NC-w27-27d001", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["四格式可见正文重读保留段落与Unicode"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", 格式) def test_四格式可见正文重读保留段落与Unicode__27d001(固定定稿, format): data = 导出(format, 固定定稿) actual = 纯文本.读取(data, 固定定稿) if format == "txt" else 读取器[format](data) expected_chapters = copy.deepcopy(固定定稿["chapters"]) if format == "txt": for chapter in expected_chapters: chapter["paragraphs"] = None chapter.pop("chapter_id") assert actual["chapters"] == expected_chapters assert actual["materials"] == 固定定稿["manifest"]["materials"] proof = 校验(format, data, 固定定稿) assert proof["verified"] is True assert proof["sha256"] == hashlib.sha256(data).hexdigest() assert proof["chapter_ids"] == (None if format == "txt" else ['c-2<&"', "c-1"]) assert proof["chapters"][0]["paragraph_count"] == (None if format == "txt" else 9) assert ( proof["chapters"][0]["text_sha256"] == hashlib.sha256(固定定稿["chapters"][0]["text"].encode("utf-8")).hexdigest() ) @pytest.mark.case_id( "NC-w27-27d002", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["同一定稿重复导出字节相同且不修改输入"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", 格式) def test_同一定稿重复导出字节相同且不修改输入__27d002(固定定稿, format): before = copy.deepcopy(固定定稿) first = 导出(format, 固定定稿) assert first == 导出(format, 固定定稿) assert 固定定稿 == before def _修改条目(data: bytes, name: str, modify) -> bytes: output = BytesIO() with ZipFile(BytesIO(data)) as source, ZipFile(output, "w") as target: for info in source.infolist(): value = source.read(info.filename) target.writestr(info, modify(value) if info.filename == name else value) return output.getvalue() @pytest.mark.case_id( "NC-w27-27d003", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["篡改实际正文而保留身份必须被拒绝"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", 格式) def test_篡改实际正文而保留身份必须被拒绝__27d003(固定定稿, format): data = 导出(format, 固定定稿) def replace(value): assert "原文标记".encode() in value return value.replace("原文标记".encode(), "篡改标记".encode()) if format in ("docx", "epub"): name = "word/document.xml" if format == "docx" else "EPUB/chapter-000001.xhtml" damaged = _修改条目(data, name, replace) else: damaged = replace(data) with pytest.raises(交付错误) as error: 校验(format, damaged, 固定定稿) assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED" @pytest.mark.case_id( "NC-w27-27d004", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["同一清单哈希不能掩盖成品顺序缺章段落或材料变化"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", 格式) @pytest.mark.parametrize("mutation", ("reorder", "missing", "paragraph", "materials")) def test_同一清单哈希不能掩盖成品顺序缺章段落或材料变化__27d004(固定定稿, format, mutation): changed = copy.deepcopy(固定定稿) if mutation == "reorder": changed["chapters"].reverse() elif mutation == "missing": changed["chapters"].pop() elif mutation == "paragraph": # 可见文本相同,段落边界不同;不能靠整章文本哈希替代段落核对。 changed["chapters"][0]["paragraphs"] = [changed["chapters"][0]["text"]] else: changed["manifest"]["materials"]["synopsis"] = "材料被换成另一版本。" data = 导出(format, changed) if format == "txt" and mutation == "paragraph": assert 校验(format, data, 固定定稿)["verified"] return with pytest.raises(交付错误) as error: 校验(format, data, 固定定稿) assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED" @pytest.mark.case_id( "NC-w27-27d005", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["XML格式拒绝无法表示字符而非静默丢弃"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", ("md", "docx", "epub")) @pytest.mark.parametrize("char", ("\x00", "\x0b", "\uffff")) def test_XML格式拒绝无法表示字符而非静默丢弃__27d005(固定定稿, format, char): 固定定稿["chapters"][0]["paragraphs"][0] += char 固定定稿["chapters"][0]["text"] = "\n".join(固定定稿["chapters"][0]["paragraphs"]) with pytest.raises(交付错误) as error: 导出(format, 固定定稿) assert error.value.错误码 == "EXPORT_INVALID_TEXT" @pytest.mark.case_id( "NC-w27-27d006", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["TXT保留XML不支持的合法UTF8控制字符"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) def test_TXT保留XML不支持的合法UTF8控制字符__27d006(固定定稿): 固定定稿["chapters"][0]["paragraphs"][0] += "\x00\x0b\uffff" 固定定稿["chapters"][0]["text"] = "\n".join(固定定稿["chapters"][0]["paragraphs"]) actual = 纯文本.读取(导出("txt", 固定定稿), 固定定稿) assert actual["chapters"][0]["text"] == 固定定稿["chapters"][0]["text"] @pytest.mark.case_id( "NC-w27-27d007", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["孤立代理码和正文段落不一致明确失败"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", 格式) def test_孤立代理码和正文段落不一致明确失败__27d007(固定定稿, format): 固定定稿["chapters"][0]["paragraphs"][0] += "\ud800" with pytest.raises(交付错误) as mismatch: 导出(format, 固定定稿) assert mismatch.value.错误码 == "EXPORT_INVALID_TEXT" 固定定稿["chapters"][0]["text"] = "\n".join(固定定稿["chapters"][0]["paragraphs"]) with pytest.raises(交付错误) as surrogate: 导出(format, 固定定稿) assert surrogate.value.错误码 == "EXPORT_INVALID_TEXT" @pytest.mark.case_id( "NC-w27-27d008", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["截断成品明确失败"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", 格式) def test_截断成品明确失败__27d008(固定定稿, format): data = 导出(format, 固定定稿) with pytest.raises(交付错误) as error: 校验(format, data[: len(data) // 2], 固定定稿) assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED" @pytest.mark.case_id( "NC-w27-27d009", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["DOCX真实页面与文本节点满足阅读结构"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) def test_DOCX真实页面与文本节点满足阅读结构__27d009(固定定稿): data = 导出("docx", 固定定稿) ns = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"} w = "{" + ns["w"] + "}" with ZipFile(BytesIO(data)) as archive: document = ET.fromstring(archive.read("word/document.xml")) styles = ET.fromstring(archive.read("word/styles.xml")) assert document.find("w:body/w:sectPr/w:pgSz", ns).attrib == { w + "w": "11906", w + "h": "16838", } chapter_style = styles.find("w:style[@w:styleId='Heading1']", ns) assert chapter_style.find("w:pPr/w:pageBreakBefore", ns) is not None assert chapter_style.find("w:pPr/w:outlineLvl", ns).get(w + "val") == "0" assert styles.find("w:docDefaults/w:rPrDefault/w:rPr/w:rFonts", ns).get(w + "eastAsia") body = document.findall("w:body/w:sdt/w:sdtContent", ns)[5:] assert [len(b.findall("w:p", ns)) - 1 for b in body] == [9, 3] assert document.findall(".//w:br[@w:type='textWrapping']", ns) assert document.findall(".//w:br", ns) assert not document.findall(".//w:vanish", ns) assert all(i.date_time == (1980, 1, 1, 0, 0, 0) for i in archive.infolist()) @pytest.mark.case_id( "NC-w27-27d010", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["EPUB真实mimetype包关系目录与spine一致"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) def test_EPUB真实mimetype包关系目录与spine一致__27d010(固定定稿): data = 导出("epub", 固定定稿) ns = {"o": "http://www.idpf.org/2007/opf", "x": "http://www.w3.org/1999/xhtml"} with ZipFile(BytesIO(data)) as archive: first = archive.infolist()[0] assert (first.filename, first.compress_type, first.extra) == ("mimetype", ZIP_STORED, b"") assert archive.read("mimetype") == b"application/epub+zip" package = ET.fromstring(archive.read("EPUB/package.opf")) items = {n.get("id"): n.get("href") for n in package.findall("o:manifest/o:item", ns)} order = [n.get("idref") for n in package.findall("o:spine/o:itemref", ns)] nav = ET.fromstring(archive.read("EPUB/nav.xhtml")) links = [n.get("href") for n in nav.findall(".//x:nav/x:ol/x:li/x:a", ns)] assert links == [items[key] for key in order] assert order == ["materials", "chapter-000001", "chapter-000002"] paragraphs = ET.fromstring(archive.read("EPUB/chapter-000001.xhtml")).findall( "x:body/x:section/x:p", ns ) assert [p.text or "" for p in paragraphs] == 固定定稿["chapters"][0]["paragraphs"] @pytest.mark.case_id( "NC-w27-27d011", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["字体隐藏或EPUB关系被破坏不能冒充文本校验通过"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", ("docx", "epub")) def test_字体隐藏或EPUB关系被破坏不能冒充文本校验通过__27d011(固定定稿, format): data = 导出(format, 固定定稿) if format == "docx": data = _修改条目( data, "word/styles.xml", lambda b: b.replace(b"", b"") ) else: data = _修改条目( data, "EPUB/package.opf", lambda b: b.replace(b'properties="nav"', b'properties="cover-image"'), ) with pytest.raises(交付错误) as error: 校验(format, data, 固定定稿) assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED" @pytest.mark.case_id( "NC-w27-27d012", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["未知格式明确拒绝"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) def test_未知格式明确拒绝__27d012(固定定稿): with pytest.raises(交付错误) as error: 导出("pdf", 固定定稿) assert error.value.错误码 == "EXPORT_FORMAT_UNSUPPORTED" @pytest.mark.case_id( "NC-w27-27d013", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["改变渲染而未改变文本节点的篡改也被拒绝"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) @pytest.mark.parametrize("format", ("md", "epub")) def test_改变渲染而未改变文本节点的篡改也被拒绝__27d013(固定定稿, format): data = 导出(format, 固定定稿) if format == "md": data = data.replace(b"*", b"*") else: data = _修改条目( data, "EPUB/chapter-000001.xhtml", lambda b: b.replace(b"

", "

额外正文".encode(), 1), ) with pytest.raises(交付错误) as error: 校验(format, data, 固定定稿) assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED" @pytest.mark.case_id( "NC-w27-27d014", environment="离线", given="本作者完整目录、确切正文版本与独立隔离数据", when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复", then=["TXT是直接交稿文本且不包含工程头"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) def test_TXT是直接交稿文本且不包含工程头__27d014(固定定稿): data = 导出("txt", 固定定稿) assert data.startswith('雾港 #1 <>& "题名"\n\n青禾 & 海风\n\n简介\n'.encode()) assert not data.startswith(b"Muse") assert b'"delivery_id"' not in data and "UTF-8字节数".encode() not in data assert 固定定稿["chapters"][0]["text"].encode() in data assert data.endswith((固定定稿["chapters"][1]["text"] + "\n").encode()) @pytest.mark.case_id( "NC-w27-27d015", environment="离线", given="同一可见TXT及不同外部清单身份", when="核对实际文本而不编码的身份改变", then=["身份标为not_encoded,回执不报告虚构的章节身份"], contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md", ) def test_TXT不把外部清单身份冒充文件内验证结果__27d015(固定定稿): data = 导出("txt", 固定定稿) changed = copy.deepcopy(固定定稿) changed["chapters"][0]["chapter_id"] = "another-chapter" changed["delivery_id"] = UUID("00000000-0000-4000-8000-000000000028") changed["manifest_hash"] = "b" * 64 proof = 校验("txt", data, changed) assert proof["verified"] is True # 只证明相同的可见文稿。 assert proof["identities"] == "not_encoded" and proof["chapter_ids"] is None assert "identities" not in proof["verified_scope"] assert all("chapter_id" not in row for row in proof["chapters"]) actual = 纯文本.读取(data, changed) assert "delivery_id" not in actual and "manifest_hash" not in actual