180 lines
9.4 KiB
Python
180 lines
9.4 KiB
Python
#!/usr/bin/env python3
|
||
"""clean_detect 失败关闭路径的纯离线回归测试。
|
||
|
||
红线:不连接数据库,不调用网络或真实 LLM。
|
||
chat_governed 与 extract_json 全部 mock/patch,只验证 clean_detect 在三种情形下的行为合同:
|
||
1. 治理链耗尽 → (None,None,None) → 写空占位+skipped,绝不返回假成功
|
||
2. JSON 解析失败 → 失败关闭,该窗不清洗、不拿坏数据当成功
|
||
3. 正常路径 → 正确解析 deletions(逐字原文+理由),不改写正文、不做删除
|
||
"""
|
||
import json
|
||
import pathlib
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
from unittest.mock import patch
|
||
|
||
from click.testing import CliRunner
|
||
|
||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||
SCRIPT_DIR = PROJECT_ROOT / "muse" / "content" / "entity" / "skills" / "ingest" / "clean-book-text" / "scripts"
|
||
sys.path.insert(0, str(SCRIPT_DIR))
|
||
import clean_detect # noqa: E402
|
||
|
||
|
||
class CleanDetectOfflineTest(unittest.TestCase):
|
||
"""覆盖 clean_detect 三条失败关闭/正常路径。"""
|
||
|
||
def setUp(self):
|
||
# 每个测试用独立 tmp 目录,patch OUT 使产物落在 tmp 而非真实 /tmp/muse-clean
|
||
self._tmp = tempfile.TemporaryDirectory()
|
||
self.tmp_path = pathlib.Path(self._tmp.name)
|
||
self.work_id = 99999 # 用不存在的 work_id 避免碰撞真实产物
|
||
self.work_dir = self.tmp_path / str(self.work_id)
|
||
self.work_dir.mkdir(parents=True)
|
||
|
||
# 写 manifest 和窗文本文件(模拟 clean_prep 的产出)
|
||
self.window_file = self.work_dir / "window-001.txt"
|
||
self.window_file.write_text(
|
||
"【第1章 | 开始】\n这是正文。\n一秒记住♂粒÷小÷说→网\n", encoding="utf-8")
|
||
manifest = {
|
||
"title": "测试书",
|
||
"windows": [
|
||
{"win": 1, "from": 1, "to": 1, "file": str(self.window_file)},
|
||
],
|
||
}
|
||
(self.work_dir / "manifest.json").write_text(
|
||
json.dumps(manifest, ensure_ascii=False), encoding="utf-8")
|
||
|
||
self.runner = CliRunner()
|
||
self.patcher = patch.object(clean_detect, "OUT", self.tmp_path)
|
||
self.patcher.start()
|
||
|
||
def tearDown(self):
|
||
self.patcher.stop()
|
||
self._tmp.cleanup()
|
||
|
||
def _invoke(self):
|
||
return self.runner.invoke(clean_detect.main, ["--work-id", str(self.work_id)])
|
||
|
||
# ── 情形1:治理链耗尽 ──────────────────────────────────────────────
|
||
|
||
def test_governance_chain_exhausted_writes_empty_placeholder_and_skips(self):
|
||
"""chat_governed 返回 (None,None,None) 时:写空占位+skipped,
|
||
不产生有效 deletions,绝不返回假成功。"""
|
||
with patch.object(clean_detect, "chat_governed", return_value=(None, None, None)):
|
||
result = self._invoke()
|
||
|
||
self.assertEqual(0, result.exit_code, f"命令应正常退出但输出:{result.output}")
|
||
# 产物文件存在
|
||
out_file = self.work_dir / "deletions-001.json"
|
||
self.assertTrue(out_file.exists(), "应写出空占位产物")
|
||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||
# 核心断言:deletions 为空、有 skipped 标记
|
||
self.assertEqual([], data["deletions"], "治理链耗尽时 deletions 必须为空")
|
||
self.assertIn("skipped", data, "必须记 skipped 标记")
|
||
self.assertIn("治理链", data["skipped"], "skipped 原因应说明治理链耗尽")
|
||
# 不能出现 model 字段(那是成功路径才写的,避免被下游误当成功)
|
||
self.assertNotIn("model", data, "失败路径不应写 model 字段")
|
||
# 输出中应有跳过与不清洗提示
|
||
self.assertIn("跳过", result.output)
|
||
self.assertIn("不清洗", result.output)
|
||
|
||
# ── 情形2:JSON 解析失败 ────────────────────────────────────────────
|
||
|
||
def test_json_parse_failure_fails_closed(self):
|
||
"""chat_governed 成功但 extract_json 抛 ValueError 时:
|
||
失败关闭,不拿坏数据当成功。"""
|
||
fake_content = "这不是JSON,是模型的胡言乱语"
|
||
fake_usage = {"prompt_tokens": 100, "completion_tokens": 50}
|
||
with patch.object(clean_detect, "chat_governed",
|
||
return_value=(fake_content, fake_usage, "MiniMax-M3")), \
|
||
patch.object(clean_detect, "extract_json",
|
||
side_effect=ValueError("无法解析")):
|
||
result = self._invoke()
|
||
|
||
self.assertEqual(0, result.exit_code, f"命令应正常退出但输出:{result.output}")
|
||
out_file = self.work_dir / "deletions-001.json"
|
||
self.assertTrue(out_file.exists(), "应写出失败关闭产物")
|
||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||
self.assertEqual([], data["deletions"], "解析失败时 deletions 必须为空")
|
||
self.assertIn("skipped", data, "必须记 skipped 标记")
|
||
self.assertIn("JSON", data["skipped"], "skipped 原因应说明 JSON 解析失败")
|
||
self.assertNotIn("model", data, "失败路径不应写 model 字段")
|
||
self.assertIn("解析失败", result.output)
|
||
self.assertIn("不清洗", result.output)
|
||
|
||
# ── 情形3:正常路径 ─────────────────────────────────────────────────
|
||
|
||
def test_valid_deletions_parsed_correctly_without_rewrite_or_delete(self):
|
||
"""chat_governed 返回合法 deletions JSON 时:正确解析产出(逐字原文+理由),
|
||
clean_detect 本身不改写正文、不做删除(删除是 clean_apply 的事)。"""
|
||
deletions = [
|
||
{"chapter_order": 1, "exact": "一秒记住♂粒÷小÷说→网", "reason": "书站广告变体"},
|
||
]
|
||
fake_content = json.dumps({"deletions": deletions}, ensure_ascii=False)
|
||
fake_usage = {"prompt_tokens": 5000, "completion_tokens": 200}
|
||
with patch.object(clean_detect, "chat_governed",
|
||
return_value=(fake_content, fake_usage, "MiniMax-M3")):
|
||
result = self._invoke()
|
||
|
||
self.assertEqual(0, result.exit_code, f"命令应正常退出但输出:{result.output}")
|
||
out_file = self.work_dir / "deletions-001.json"
|
||
self.assertTrue(out_file.exists())
|
||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||
# 核心断言:deletions 正确解析,逐字原文与理由原样保留
|
||
self.assertEqual(1, len(data["deletions"]))
|
||
self.assertEqual("一秒记住♂粒÷小÷说→网", data["deletions"][0]["exact"])
|
||
self.assertEqual("书站广告变体", data["deletions"][0]["reason"])
|
||
self.assertEqual(1, data["deletions"][0]["chapter_order"])
|
||
# 成功路径写 model 字段
|
||
self.assertEqual("MiniMax-M3", data["model"])
|
||
# 不应有 skipped 标记
|
||
self.assertNotIn("skipped", data, "成功路径不应有 skipped 标记")
|
||
# clean_detect 不改写正文:源文件内容不变(删除是 clean_apply 的职责)
|
||
original_text = self.window_file.read_text(encoding="utf-8")
|
||
self.assertIn("一秒记住♂粒÷小÷说→网", original_text,
|
||
"源 txt 不应被改动(clean_detect 只探测不删除)")
|
||
self.assertIn("这是正文。", original_text, "正文不应被改动")
|
||
# 输出中应有检出提示
|
||
self.assertIn("检出 1 段", result.output)
|
||
|
||
# ── 情形3补充:空 deletions(该窗无垃圾)是成功路径而非 skipped ──────
|
||
|
||
def test_empty_deletions_is_success_not_skip(self):
|
||
"""模型返回空 deletions 表示该窗干净——是成功路径(写 model),不是 skipped。"""
|
||
fake_content = '{"deletions": []}'
|
||
fake_usage = {"prompt_tokens": 3000, "completion_tokens": 30}
|
||
with patch.object(clean_detect, "chat_governed",
|
||
return_value=(fake_content, fake_usage, "MiniMax-M3")):
|
||
result = self._invoke()
|
||
|
||
self.assertEqual(0, result.exit_code)
|
||
out_file = self.work_dir / "deletions-001.json"
|
||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||
self.assertEqual([], data["deletions"])
|
||
self.assertEqual("MiniMax-M3", data["model"], "空 deletions 仍是成功路径,应写 model")
|
||
self.assertNotIn("skipped", data, "空 deletions 不是失败,不应有 skipped")
|
||
self.assertIn("检出 0 段", result.output)
|
||
|
||
# ── 断点续跑:已有产物时不调用 LLM ──────────────────────────────────
|
||
|
||
def test_existing_output_skipped_without_force(self):
|
||
"""已有产物文件时不调用 chat_governed(断点续跑幂等),--force 才重跑。"""
|
||
out_file = self.work_dir / "deletions-001.json"
|
||
out_file.write_text('{"deletions": [], "model": "old"}', encoding="utf-8")
|
||
|
||
with patch.object(clean_detect, "chat_governed") as mock_gov:
|
||
result = self._invoke()
|
||
mock_gov.assert_not_called()
|
||
|
||
self.assertEqual(0, result.exit_code)
|
||
self.assertIn("已有产物", result.output)
|
||
# 文件内容不变
|
||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||
self.assertEqual("old", data["model"])
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main(verbosity=2)
|