muse-agent-example/tests/skills/deconstruct-book/test_parse_llm_offline.py
zizi c9f69d9d6d 治理: Skill 测试治理第一阶段——harness 控制平面 + 实现测试迁出运行时目录
范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):

1. 新增 harness/ 控制平面
   - skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
   - run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
     空跑与 skip-only 失败关闭、AST 测试形状门
   - manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
   - manifests/test-inventory.json:81 个测试资产登记
   - specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责

2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
   - 71 个测试文件迁移并修复项目根与临时目录运行导入
   - 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
   - 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
   - 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变

3. 运行时文档清理
   - 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
     只保留运行时合同;业务运行合同、额度、授权与离线模式均保留

4. SoT 同步
   - AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
   - 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
   - humanization 覆盖矩阵:活动测试路径同步迁移

验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。

已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
2026-08-19 01:50:20 +08:00

197 lines
9.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""parse_llm 章级细纲修复的纯离线回归测试。
红线:不连接数据库,不调用网络或真实 LLM。测试只验证细纲专用压缩、失败状态流和
incomplete-only 的目标章筛选,避免放量时用真实额度验证本可机械证明的行为。
"""
import pathlib
import sys
import tempfile
import unittest
from unittest.mock import Mock, patch
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "deconstruct-book" / "scripts"
sys.path.insert(0, str(SCRIPT_DIR))
import parse_llm as pll # noqa: E402
class _RowsConnection:
"""只实现目标章查询所需的最小假连接,并记录查询次数。"""
def __init__(self, rows):
self.rows = rows
self.queries = []
def execute(self, sql, args):
self.queries.append((sql, args))
return self
def fetchall(self):
return self.rows
class ParseLlmOfflineTest(unittest.TestCase):
"""覆盖章级比例失败后的最小修复合同。"""
def test_outline_repair_only_replaces_outline_and_uses_short_prompt(self):
"""专用调用只压细纲,首轮实体与线索对象必须原样保留。"""
entities = [{"type": "character", "name": "甲", "brief": "主角"}]
hints = [{"type": "craft", "name": "伏笔", "clue": "埋后收", "evidence": "本章"}]
first = {"outline": "旧细纲" * 80, "entities": entities, "hints": hints}
source = "正文" * 1000
with patch.object(pll, "m3_json", return_value=({"outline": "新纲" * 30}, {"prompt_tokens": 9})) as call:
repaired, usage, error = pll.repair_outline(first, source, "MiniMax-M3")
self.assertIsNone(error)
self.assertEqual("新纲" * 30, repaired["outline"])
self.assertIs(entities, repaired["entities"])
self.assertIs(hints, repaired["hints"])
self.assertEqual(9, usage["prompt_tokens"])
prompt = call.call_args.args[0]
cap = pll.outline_target_cap(source)
self.assertLess(len(prompt), 1200)
self.assertNotIn(source, prompt)
self.assertNotIn("entities", prompt)
self.assertNotIn("hints", prompt)
self.assertIn(f"非空白字符不得超过 {cap}", prompt)
self.assertIn('{"outline": "压缩后的细纲"}', prompt)
def test_outline_repair_retries_finitely_and_never_truncates_or_reingests_over_limit(self):
"""两轮仍超目标时保持首轮失败,不截断、不用超长结果伪造 done。"""
first = {"outline": "首轮" * 100, "entities": [{"name": "甲"}], "hints": [{"name": "线索"}]}
source = "正文" * 1000
over_1 = "第一次仍超" * 30
over_2 = "第二次仍超" * 40
with patch.object(pll, "m3_json", side_effect=[
({"outline": over_1}, {"completion_tokens": 10}),
({"outline": over_2}, {"completion_tokens": 11}),
]) as llm_call, patch.object(pll, "ingest", return_value=(False, "细纲比例 9.0% 超标")) as ingest_call:
ok, output, usage = pll.ingest_scaffold_with_repair(7, 1286, first, source, "MiniMax-M3")
self.assertFalse(ok)
self.assertEqual(2, llm_call.call_count)
self.assertEqual(1, ingest_call.call_count)
self.assertIn(f"压缩输出 {pll.non_whitespace_len(over_2)} 字", output)
self.assertIn("保持 failed", output)
self.assertNotIn(over_2[:pll.outline_target_cap(source)], str(ingest_call.call_args_list))
self.assertEqual(21, usage["completion_tokens"])
def test_compression_retry_breaks_total_cap_into_short_phrase_budget(self):
"""真实 M3 忽略单一总上限后,重试须同时给上一版长度与逐短语预算。"""
first = pll.outline_compression_prompt("甲" * 89, 60, 1)
second = pll.outline_compression_prompt("甲" * 89, 60, 2)
self.assertIn("上一版共 89 个非空白字符", second)
self.assertIn("最多 4 个无标签短语", first)
self.assertIn("最多 2 个无标签短语", second)
self.assertIn("总计仍不得超过 60", second)
def test_final_repair_may_use_eight_percent_hard_cap_without_false_failure(self):
"""最后一轮已满足既有 8% 机械硬门时允许入库,4.5% 只是优选目标。"""
first = {"outline": "首轮" * 100, "entities": [], "hints": []}
source = "正文" * 1000
target = pll.outline_target_cap(source)
hard = pll.outline_hard_cap(source)
self.assertLess(target, hard)
between = "纲" * (target + 10)
with patch.object(pll, "m3_json", side_effect=[
({"outline": "仍超目标" * 30}, {}),
({"outline": between}, {}),
]):
repaired, _usage, error = pll.repair_outline(first, source, "MiniMax-M3")
self.assertIsNone(error)
self.assertEqual(between, repaired["outline"])
def test_incomplete_only_selects_once_while_default_keeps_full_range(self):
"""增量模式一次筛出 pending/failed;默认模式仍按原范围逐章兼容。"""
conn = _RowsConnection([(3,), (8,), (13,)])
self.assertEqual([3, 8, 13], pll.chapter_orders(conn, 7, 1, 20, incomplete_only=True))
self.assertEqual(1, len(conn.queries))
sql, args = conn.queries[0]
self.assertIn("scaffold_status, 'pending') IN ('pending', 'failed')", sql)
self.assertEqual((pll.TENANT, 7, 1, 20), args)
untouched = _RowsConnection([])
self.assertEqual(list(range(1, 21)), pll.chapter_orders(untouched, 7, 1, 20, incomplete_only=False))
self.assertEqual([], untouched.queries)
def test_incomplete_only_skips_full_range_task_initialization(self):
"""任务表已存在的增量恢复不得再对全书逐行执行 init-tasks。"""
with patch.object(pll.subprocess, "run") as run:
pll.initialize_tasks(7, 1, 2250, incomplete_only=True)
run.assert_not_called()
pll.initialize_tasks(7, 1, 2250, incomplete_only=False)
run.assert_called_once()
self.assertIn("init-tasks", run.call_args.args[0])
def test_failed_chapter_reuses_complete_strict_cache_without_full_extraction(self):
"""failed 章命中完整首轮载荷时,直接复用且不得再次执行正文完整抽取。"""
cached = {
"outline": "仍然超长的首轮细纲",
"entities": [{"type": "character", "name": "甲", "brief": "主角"}],
"hints": [{"type": "craft", "name": "伏笔", "clue": "埋后收", "evidence": "本章"}],
}
full_extract = Mock(side_effect=AssertionError("cache hit 不应调用完整 m3_json 抽取"))
with tempfile.TemporaryDirectory() as tmp, patch.object(pll, "TMP", pathlib.Path(tmp)):
cache_path = pathlib.Path(tmp) / "7-1286-scaffold.json"
cache_path.write_text(__import__("json").dumps(cached), encoding="utf-8")
data, usage, trace = pll.resolve_scaffold_payload(7, 1286, "failed", full_extract)
self.assertEqual(cached, data)
self.assertEqual({}, usage)
self.assertIn("cache hit", trace)
self.assertIn(str(cache_path), trace)
full_extract.assert_not_called()
def test_pending_chapter_never_reuses_cache(self):
"""pending 即使存在形状完整的同名文件,也必须重新执行完整抽取。"""
fresh = {"outline": "新细纲", "entities": [], "hints": []}
full_extract = Mock(return_value=(fresh, {"prompt_tokens": 11}))
with tempfile.TemporaryDirectory() as tmp, patch.object(pll, "TMP", pathlib.Path(tmp)):
(pathlib.Path(tmp) / "7-9-scaffold.json").write_text(
'{"outline":"旧细纲","entities":[],"hints":[]}', encoding="utf-8")
data, usage, trace = pll.resolve_scaffold_payload(7, 9, "pending", full_extract)
self.assertEqual(fresh, data)
self.assertEqual({"prompt_tokens": 11}, usage)
self.assertIn("cache miss", trace)
self.assertIn("status=pending", trace)
full_extract.assert_called_once_with()
def test_missing_or_invalid_failed_cache_falls_back_to_full_extraction(self):
"""failed 缓存缺失、非严格对象或字段类型不合理时,都走完整抽取回退。"""
cases = {
"missing": None,
"root-list": '[]',
"bad-outline": '{"outline":[],"entities":[],"hints":[]}',
"bad-entities": '{"outline":"纲","entities":[1],"hints":[]}',
"bad-hints": '{"outline":"纲","entities":[],"hints":"线索"}',
"broken-json": '{"outline":',
}
for name, cache_text in cases.items():
with self.subTest(name=name), tempfile.TemporaryDirectory() as tmp, \
patch.object(pll, "TMP", pathlib.Path(tmp)):
if cache_text is not None:
(pathlib.Path(tmp) / "7-10-scaffold.json").write_text(cache_text, encoding="utf-8")
fresh = {"outline": f"新细纲-{name}", "entities": [], "hints": []}
full_extract = Mock(return_value=(fresh, {"completion_tokens": 7}))
data, usage, trace = pll.resolve_scaffold_payload(7, 10, "failed", full_extract)
self.assertEqual(fresh, data)
self.assertEqual({"completion_tokens": 7}, usage)
self.assertIn("cache miss", trace)
full_extract.assert_called_once_with()
if __name__ == "__main__":
unittest.main(verbosity=2)