muse-agent-example/tests/单元/test_交付格式.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

402 lines
18 KiB
Python
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""固定定稿格式的文本保真、成品核对及标准包结构。"""
import copy
import hashlib
from io import BytesIO
from uuid import UUID
from xml.etree import ElementTree as ET
from zipfile import ZIP_STORED, ZipFile
import pytest
from muse.交付连载.导出格式 import EPUB, Markdown, Word, 导出, 校验, 纯文本
from muse.交付连载.模型 import 交付错误
格式 = ("txt", "md", "docx", "epub")
读取器 = {"txt": 纯文本.读取, "md": Markdown.读取, "docx": Word.读取, "epub": EPUB.读取}
@pytest.fixture
def 固定定稿():
paragraphs = [
" 原文标记:夜色里,青禾握着旧钥匙。 ",
"",
"",
"半角 空格\t制表\r单CR\n单LF\r\n双字符换行。",
"e\u0301 ≠ é;家庭👨‍👩‍👧‍👦;😀;汉字𠮷。",
"# 标题 *不是强调* `代码` [链接](https://example.invalid) <script>&#13;</script>",
'章节 {"paragraph_bytes":[999]}\n<!-- muse-chapter-end -->',
"\t  \u00a0 ",
"",
]
return {
"delivery_id": UUID("00000000-0000-4000-8000-000000000027"),
"manifest_hash": "a" * 64,
"manifest": {
"materials": {
"title": '雾港 #1 <>& "题名"',
"author_name": "青禾 & 海风",
"introduction": "简介首行\r\n第二行 连续空白\t结尾。",
"synopsis": "梗概与正文独立。",
"characters": "青禾:守门人。\n海风:旅人。",
}
},
"chapters": [
{
"chapter_id": 'c-2<&"',
"title": "第一章 # 标题 <>& *正文*",
"paragraphs": paragraphs,
"text": "\n".join(paragraphs),
},
{
"chapter_id": "c-1",
"title": "第二章",
"paragraphs": ["", "最后一盏灯。", ""],
"text": "\n最后一盏灯。\n",
},
],
}
@pytest.mark.case_id(
"NC-w27-27d001",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["四格式可见正文重读保留段落与Unicode"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", 格式)
def test_四格式可见正文重读保留段落与Unicode__27d001(固定定稿, format):
data = 导出(format, 固定定稿)
actual = 纯文本.读取(data, 固定定稿) if format == "txt" else 读取器[format](data)
expected_chapters = copy.deepcopy(固定定稿["chapters"])
if format == "txt":
for chapter in expected_chapters:
chapter["paragraphs"] = None
chapter.pop("chapter_id")
assert actual["chapters"] == expected_chapters
assert actual["materials"] == 固定定稿["manifest"]["materials"]
proof = 校验(format, data, 固定定稿)
assert proof["verified"] is True
assert proof["sha256"] == hashlib.sha256(data).hexdigest()
assert proof["chapter_ids"] == (None if format == "txt" else ['c-2<&"', "c-1"])
assert proof["chapters"][0]["paragraph_count"] == (None if format == "txt" else 9)
assert (
proof["chapters"][0]["text_sha256"]
== hashlib.sha256(固定定稿["chapters"][0]["text"].encode("utf-8")).hexdigest()
)
@pytest.mark.case_id(
"NC-w27-27d002",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["同一定稿重复导出字节相同且不修改输入"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", 格式)
def test_同一定稿重复导出字节相同且不修改输入__27d002(固定定稿, format):
before = copy.deepcopy(固定定稿)
first = 导出(format, 固定定稿)
assert first == 导出(format, 固定定稿)
assert 固定定稿 == before
def _修改条目(data: bytes, name: str, modify) -> bytes:
output = BytesIO()
with ZipFile(BytesIO(data)) as source, ZipFile(output, "w") as target:
for info in source.infolist():
value = source.read(info.filename)
target.writestr(info, modify(value) if info.filename == name else value)
return output.getvalue()
@pytest.mark.case_id(
"NC-w27-27d003",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["篡改实际正文而保留身份必须被拒绝"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", 格式)
def test_篡改实际正文而保留身份必须被拒绝__27d003(固定定稿, format):
data = 导出(format, 固定定稿)
def replace(value):
assert "原文标记".encode() in value
return value.replace("原文标记".encode(), "篡改标记".encode())
if format in ("docx", "epub"):
name = "word/document.xml" if format == "docx" else "EPUB/chapter-000001.xhtml"
damaged = _修改条目(data, name, replace)
else:
damaged = replace(data)
with pytest.raises(交付错误) as error:
校验(format, damaged, 固定定稿)
assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED"
@pytest.mark.case_id(
"NC-w27-27d004",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["同一清单哈希不能掩盖成品顺序缺章段落或材料变化"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", 格式)
@pytest.mark.parametrize("mutation", ("reorder", "missing", "paragraph", "materials"))
def test_同一清单哈希不能掩盖成品顺序缺章段落或材料变化__27d004(固定定稿, format, mutation):
changed = copy.deepcopy(固定定稿)
if mutation == "reorder":
changed["chapters"].reverse()
elif mutation == "missing":
changed["chapters"].pop()
elif mutation == "paragraph":
# 可见文本相同,段落边界不同;不能靠整章文本哈希替代段落核对。
changed["chapters"][0]["paragraphs"] = [changed["chapters"][0]["text"]]
else:
changed["manifest"]["materials"]["synopsis"] = "材料被换成另一版本。"
data = 导出(format, changed)
if format == "txt" and mutation == "paragraph":
assert 校验(format, data, 固定定稿)["verified"]
return
with pytest.raises(交付错误) as error:
校验(format, data, 固定定稿)
assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED"
@pytest.mark.case_id(
"NC-w27-27d005",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["XML格式拒绝无法表示字符而非静默丢弃"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", ("md", "docx", "epub"))
@pytest.mark.parametrize("char", ("\x00", "\x0b", "\uffff"))
def test_XML格式拒绝无法表示字符而非静默丢弃__27d005(固定定稿, format, char):
固定定稿["chapters"][0]["paragraphs"][0] += char
固定定稿["chapters"][0]["text"] = "\n".join(固定定稿["chapters"][0]["paragraphs"])
with pytest.raises(交付错误) as error:
导出(format, 固定定稿)
assert error.value.错误码 == "EXPORT_INVALID_TEXT"
@pytest.mark.case_id(
"NC-w27-27d006",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["TXT保留XML不支持的合法UTF8控制字符"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
def test_TXT保留XML不支持的合法UTF8控制字符__27d006(固定定稿):
固定定稿["chapters"][0]["paragraphs"][0] += "\x00\x0b\uffff"
固定定稿["chapters"][0]["text"] = "\n".join(固定定稿["chapters"][0]["paragraphs"])
actual = 纯文本.读取(导出("txt", 固定定稿), 固定定稿)
assert actual["chapters"][0]["text"] == 固定定稿["chapters"][0]["text"]
@pytest.mark.case_id(
"NC-w27-27d007",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["孤立代理码和正文段落不一致明确失败"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", 格式)
def test_孤立代理码和正文段落不一致明确失败__27d007(固定定稿, format):
固定定稿["chapters"][0]["paragraphs"][0] += "\ud800"
with pytest.raises(交付错误) as mismatch:
导出(format, 固定定稿)
assert mismatch.value.错误码 == "EXPORT_INVALID_TEXT"
固定定稿["chapters"][0]["text"] = "\n".join(固定定稿["chapters"][0]["paragraphs"])
with pytest.raises(交付错误) as surrogate:
导出(format, 固定定稿)
assert surrogate.value.错误码 == "EXPORT_INVALID_TEXT"
@pytest.mark.case_id(
"NC-w27-27d008",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["截断成品明确失败"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", 格式)
def test_截断成品明确失败__27d008(固定定稿, format):
data = 导出(format, 固定定稿)
with pytest.raises(交付错误) as error:
校验(format, data[: len(data) // 2], 固定定稿)
assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED"
@pytest.mark.case_id(
"NC-w27-27d009",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["DOCX真实页面与文本节点满足阅读结构"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
def test_DOCX真实页面与文本节点满足阅读结构__27d009(固定定稿):
data = 导出("docx", 固定定稿)
ns = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
w = "{" + ns["w"] + "}"
with ZipFile(BytesIO(data)) as archive:
document = ET.fromstring(archive.read("word/document.xml"))
styles = ET.fromstring(archive.read("word/styles.xml"))
assert document.find("w:body/w:sectPr/w:pgSz", ns).attrib == {
w + "w": "11906",
w + "h": "16838",
}
chapter_style = styles.find("w:style[@w:styleId='Heading1']", ns)
assert chapter_style.find("w:pPr/w:pageBreakBefore", ns) is not None
assert chapter_style.find("w:pPr/w:outlineLvl", ns).get(w + "val") == "0"
assert styles.find("w:docDefaults/w:rPrDefault/w:rPr/w:rFonts", ns).get(w + "eastAsia")
body = document.findall("w:body/w:sdt/w:sdtContent", ns)[5:]
assert [len(b.findall("w:p", ns)) - 1 for b in body] == [9, 3]
assert document.findall(".//w:br[@w:type='textWrapping']", ns)
assert document.findall(".//w:br", ns)
assert not document.findall(".//w:vanish", ns)
assert all(i.date_time == (1980, 1, 1, 0, 0, 0) for i in archive.infolist())
@pytest.mark.case_id(
"NC-w27-27d010",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["EPUB真实mimetype包关系目录与spine一致"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
def test_EPUB真实mimetype包关系目录与spine一致__27d010(固定定稿):
data = 导出("epub", 固定定稿)
ns = {"o": "http://www.idpf.org/2007/opf", "x": "http://www.w3.org/1999/xhtml"}
with ZipFile(BytesIO(data)) as archive:
first = archive.infolist()[0]
assert (first.filename, first.compress_type, first.extra) == ("mimetype", ZIP_STORED, b"")
assert archive.read("mimetype") == b"application/epub+zip"
package = ET.fromstring(archive.read("EPUB/package.opf"))
items = {n.get("id"): n.get("href") for n in package.findall("o:manifest/o:item", ns)}
order = [n.get("idref") for n in package.findall("o:spine/o:itemref", ns)]
nav = ET.fromstring(archive.read("EPUB/nav.xhtml"))
links = [n.get("href") for n in nav.findall(".//x:nav/x:ol/x:li/x:a", ns)]
assert links == [items[key] for key in order]
assert order == ["materials", "chapter-000001", "chapter-000002"]
paragraphs = ET.fromstring(archive.read("EPUB/chapter-000001.xhtml")).findall(
"x:body/x:section/x:p", ns
)
assert [p.text or "" for p in paragraphs] == 固定定稿["chapters"][0]["paragraphs"]
@pytest.mark.case_id(
"NC-w27-27d011",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["字体隐藏或EPUB关系被破坏不能冒充文本校验通过"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", ("docx", "epub"))
def test_字体隐藏或EPUB关系被破坏不能冒充文本校验通过__27d011(固定定稿, format):
data = 导出(format, 固定定稿)
if format == "docx":
data = _修改条目(
data, "word/styles.xml", lambda b: b.replace(b"<w:rPr>", b"<w:rPr><w:vanish/>")
)
else:
data = _修改条目(
data,
"EPUB/package.opf",
lambda b: b.replace(b'properties="nav"', b'properties="cover-image"'),
)
with pytest.raises(交付错误) as error:
校验(format, data, 固定定稿)
assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED"
@pytest.mark.case_id(
"NC-w27-27d012",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["未知格式明确拒绝"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
def test_未知格式明确拒绝__27d012(固定定稿):
with pytest.raises(交付错误) as error:
导出("pdf", 固定定稿)
assert error.value.错误码 == "EXPORT_FORMAT_UNSUPPORTED"
@pytest.mark.case_id(
"NC-w27-27d013",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["改变渲染而未改变文本节点的篡改也被拒绝"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
@pytest.mark.parametrize("format", ("md", "epub"))
def test_改变渲染而未改变文本节点的篡改也被拒绝__27d013(固定定稿, format):
data = 导出(format, 固定定稿)
if format == "md":
data = data.replace(b"&#42;", b"*")
else:
data = _修改条目(
data,
"EPUB/chapter-000001.xhtml",
lambda b: b.replace(b"</p>", "</p>额外正文".encode(), 1),
)
with pytest.raises(交付错误) as error:
校验(format, data, 固定定稿)
assert error.value.错误码 == "EXPORT_VERIFICATION_FAILED"
@pytest.mark.case_id(
"NC-w27-27d014",
environment="离线",
given="本作者完整目录、确切正文版本与独立隔离数据",
when="通过定稿、S02导出、跨章候选或连载反馈实际入口执行并检查失败恢复",
then=["TXT是直接交稿文本且不包含工程头"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
def test_TXT是直接交稿文本且不包含工程头__27d014(固定定稿):
data = 导出("txt", 固定定稿)
assert data.startswith('雾港 #1 <>& "题名"\n\n青禾 & 海风\n\n简介\n'.encode())
assert not data.startswith(b"Muse")
assert b'"delivery_id"' not in data and "UTF-8字节数".encode() not in data
assert 固定定稿["chapters"][0]["text"].encode() in data
assert data.endswith((固定定稿["chapters"][1]["text"] + "\n").encode())
@pytest.mark.case_id(
"NC-w27-27d015",
environment="离线",
given="同一可见TXT及不同外部清单身份",
when="核对实际文本而不编码的身份改变",
then=["身份标为not_encoded,回执不报告虚构的章节身份"],
contract="docs/系统架构/新版设计/模块设计/B08-交付连载.md",
)
def test_TXT不把外部清单身份冒充文件内验证结果__27d015(固定定稿):
data = 导出("txt", 固定定稿)
changed = copy.deepcopy(固定定稿)
changed["chapters"][0]["chapter_id"] = "another-chapter"
changed["delivery_id"] = UUID("00000000-0000-4000-8000-000000000028")
changed["manifest_hash"] = "b" * 64
proof = 校验("txt", data, changed)
assert proof["verified"] is True # 只证明相同的可见文稿。
assert proof["identities"] == "not_encoded" and proof["chapter_ids"] is None
assert "identities" not in proof["verified_scope"]
assert all("chapter_id" not in row for row in proof["chapters"])
actual = 纯文本.读取(data, changed)
assert "delivery_id" not in actual and "manifest_hash" not in actual