范围(不含 design-story-foundation、docs/、humanization/README.md 等进行中改动):
1. 新增 harness/ 控制平面
- skill_harness.py 静态审计:32 个运行时 Skill 的 frontmatter/manifest/文档污染,当前 0 问题
- run_selected.py 选择性执行器:manifest 与磁盘一一对账、依赖阻断、
空跑与 skip-only 失败关闭、AST 测试形状门
- manifests/skills.json:32 个 Skill 的合同责任方与协作领域登记
- manifests/test-inventory.json:81 个测试资产登记
- specs/skill-testing.md 与 README.md:测试分层、证据边界与 harness 职责
2. 实现测试从 .claude/skills/*/scripts/ 迁至 tests/skills/<skill>/
- 71 个测试文件迁移并修复项目根与临时目录运行导入
- 数据库触发器测试宽泛异常收窄为 psycopg.errors.RaiseException
- 抽取离线大测试拆出真实 PG smoke(默认阻断,不计入离线通过)
- 抽取 presence 去重边界拆出独立测试:493 + 78 = 571 项检查不变
3. 运行时文档清理
- 13 个 SKILL.md 移除自测/离线验证段落、测试命令与测试文件事实源表述,
只保留运行时合同;业务运行合同、额度、授权与离线模式均保留
4. SoT 同步
- AGENTS.md:新增 Skill 领域索引(7 个合同责任方分组,覆盖 32 个运行时 Skill)
- 领域 07:测试入口改由 harness/manifests/ 登记,SKILL.md 不承载测试命令
- humanization 覆盖矩阵:活动测试路径同步迁移
验证证据: harness 自测 15 项 + runner 自测 13 项通过;静态审计 32 Skill / 0 问题;
73 个非数据库测试通过;8 个集成条目中 6 个 PostgreSQL 项被依赖门明确阻断;
py_compile 与 git diff --check 通过。未连接 PostgreSQL、网络、真实模型或额度。
已知边界: 真正 skill_behavior_eval 仍为 0,尚未验证任何 Skill 自然语言行为;
evaluate-frozen-replay 的 raw 存储边界冲突留待单独治理。
197 lines
9.4 KiB
Python
197 lines
9.4 KiB
Python
#!/usr/bin/env python3
|
||
"""parse_llm 章级细纲修复的纯离线回归测试。
|
||
|
||
红线:不连接数据库,不调用网络或真实 LLM。测试只验证细纲专用压缩、失败状态流和
|
||
incomplete-only 的目标章筛选,避免放量时用真实额度验证本可机械证明的行为。
|
||
"""
|
||
import pathlib
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
from unittest.mock import Mock, patch
|
||
|
||
|
||
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
||
SCRIPT_DIR = PROJECT_ROOT / ".claude" / "skills" / "deconstruct-book" / "scripts"
|
||
sys.path.insert(0, str(SCRIPT_DIR))
|
||
import parse_llm as pll # noqa: E402
|
||
|
||
|
||
class _RowsConnection:
|
||
"""只实现目标章查询所需的最小假连接,并记录查询次数。"""
|
||
|
||
def __init__(self, rows):
|
||
self.rows = rows
|
||
self.queries = []
|
||
|
||
def execute(self, sql, args):
|
||
self.queries.append((sql, args))
|
||
return self
|
||
|
||
def fetchall(self):
|
||
return self.rows
|
||
|
||
|
||
class ParseLlmOfflineTest(unittest.TestCase):
|
||
"""覆盖章级比例失败后的最小修复合同。"""
|
||
|
||
def test_outline_repair_only_replaces_outline_and_uses_short_prompt(self):
|
||
"""专用调用只压细纲,首轮实体与线索对象必须原样保留。"""
|
||
entities = [{"type": "character", "name": "甲", "brief": "主角"}]
|
||
hints = [{"type": "craft", "name": "伏笔", "clue": "埋后收", "evidence": "本章"}]
|
||
first = {"outline": "旧细纲" * 80, "entities": entities, "hints": hints}
|
||
source = "正文" * 1000
|
||
|
||
with patch.object(pll, "m3_json", return_value=({"outline": "新纲" * 30}, {"prompt_tokens": 9})) as call:
|
||
repaired, usage, error = pll.repair_outline(first, source, "MiniMax-M3")
|
||
|
||
self.assertIsNone(error)
|
||
self.assertEqual("新纲" * 30, repaired["outline"])
|
||
self.assertIs(entities, repaired["entities"])
|
||
self.assertIs(hints, repaired["hints"])
|
||
self.assertEqual(9, usage["prompt_tokens"])
|
||
prompt = call.call_args.args[0]
|
||
cap = pll.outline_target_cap(source)
|
||
self.assertLess(len(prompt), 1200)
|
||
self.assertNotIn(source, prompt)
|
||
self.assertNotIn("entities", prompt)
|
||
self.assertNotIn("hints", prompt)
|
||
self.assertIn(f"非空白字符不得超过 {cap}", prompt)
|
||
self.assertIn('{"outline": "压缩后的细纲"}', prompt)
|
||
|
||
def test_outline_repair_retries_finitely_and_never_truncates_or_reingests_over_limit(self):
|
||
"""两轮仍超目标时保持首轮失败,不截断、不用超长结果伪造 done。"""
|
||
first = {"outline": "首轮" * 100, "entities": [{"name": "甲"}], "hints": [{"name": "线索"}]}
|
||
source = "正文" * 1000
|
||
over_1 = "第一次仍超" * 30
|
||
over_2 = "第二次仍超" * 40
|
||
|
||
with patch.object(pll, "m3_json", side_effect=[
|
||
({"outline": over_1}, {"completion_tokens": 10}),
|
||
({"outline": over_2}, {"completion_tokens": 11}),
|
||
]) as llm_call, patch.object(pll, "ingest", return_value=(False, "细纲比例 9.0% 超标")) as ingest_call:
|
||
ok, output, usage = pll.ingest_scaffold_with_repair(7, 1286, first, source, "MiniMax-M3")
|
||
|
||
self.assertFalse(ok)
|
||
self.assertEqual(2, llm_call.call_count)
|
||
self.assertEqual(1, ingest_call.call_count)
|
||
self.assertIn(f"压缩输出 {pll.non_whitespace_len(over_2)} 字", output)
|
||
self.assertIn("保持 failed", output)
|
||
self.assertNotIn(over_2[:pll.outline_target_cap(source)], str(ingest_call.call_args_list))
|
||
self.assertEqual(21, usage["completion_tokens"])
|
||
|
||
def test_compression_retry_breaks_total_cap_into_short_phrase_budget(self):
|
||
"""真实 M3 忽略单一总上限后,重试须同时给上一版长度与逐短语预算。"""
|
||
first = pll.outline_compression_prompt("甲" * 89, 60, 1)
|
||
second = pll.outline_compression_prompt("甲" * 89, 60, 2)
|
||
self.assertIn("上一版共 89 个非空白字符", second)
|
||
self.assertIn("最多 4 个无标签短语", first)
|
||
self.assertIn("最多 2 个无标签短语", second)
|
||
self.assertIn("总计仍不得超过 60", second)
|
||
|
||
def test_final_repair_may_use_eight_percent_hard_cap_without_false_failure(self):
|
||
"""最后一轮已满足既有 8% 机械硬门时允许入库,4.5% 只是优选目标。"""
|
||
first = {"outline": "首轮" * 100, "entities": [], "hints": []}
|
||
source = "正文" * 1000
|
||
target = pll.outline_target_cap(source)
|
||
hard = pll.outline_hard_cap(source)
|
||
self.assertLess(target, hard)
|
||
between = "纲" * (target + 10)
|
||
|
||
with patch.object(pll, "m3_json", side_effect=[
|
||
({"outline": "仍超目标" * 30}, {}),
|
||
({"outline": between}, {}),
|
||
]):
|
||
repaired, _usage, error = pll.repair_outline(first, source, "MiniMax-M3")
|
||
|
||
self.assertIsNone(error)
|
||
self.assertEqual(between, repaired["outline"])
|
||
|
||
def test_incomplete_only_selects_once_while_default_keeps_full_range(self):
|
||
"""增量模式一次筛出 pending/failed;默认模式仍按原范围逐章兼容。"""
|
||
conn = _RowsConnection([(3,), (8,), (13,)])
|
||
self.assertEqual([3, 8, 13], pll.chapter_orders(conn, 7, 1, 20, incomplete_only=True))
|
||
self.assertEqual(1, len(conn.queries))
|
||
sql, args = conn.queries[0]
|
||
self.assertIn("scaffold_status, 'pending') IN ('pending', 'failed')", sql)
|
||
self.assertEqual((pll.TENANT, 7, 1, 20), args)
|
||
|
||
untouched = _RowsConnection([])
|
||
self.assertEqual(list(range(1, 21)), pll.chapter_orders(untouched, 7, 1, 20, incomplete_only=False))
|
||
self.assertEqual([], untouched.queries)
|
||
|
||
def test_incomplete_only_skips_full_range_task_initialization(self):
|
||
"""任务表已存在的增量恢复不得再对全书逐行执行 init-tasks。"""
|
||
with patch.object(pll.subprocess, "run") as run:
|
||
pll.initialize_tasks(7, 1, 2250, incomplete_only=True)
|
||
run.assert_not_called()
|
||
|
||
pll.initialize_tasks(7, 1, 2250, incomplete_only=False)
|
||
run.assert_called_once()
|
||
self.assertIn("init-tasks", run.call_args.args[0])
|
||
|
||
def test_failed_chapter_reuses_complete_strict_cache_without_full_extraction(self):
|
||
"""failed 章命中完整首轮载荷时,直接复用且不得再次执行正文完整抽取。"""
|
||
cached = {
|
||
"outline": "仍然超长的首轮细纲",
|
||
"entities": [{"type": "character", "name": "甲", "brief": "主角"}],
|
||
"hints": [{"type": "craft", "name": "伏笔", "clue": "埋后收", "evidence": "本章"}],
|
||
}
|
||
full_extract = Mock(side_effect=AssertionError("cache hit 不应调用完整 m3_json 抽取"))
|
||
|
||
with tempfile.TemporaryDirectory() as tmp, patch.object(pll, "TMP", pathlib.Path(tmp)):
|
||
cache_path = pathlib.Path(tmp) / "7-1286-scaffold.json"
|
||
cache_path.write_text(__import__("json").dumps(cached), encoding="utf-8")
|
||
data, usage, trace = pll.resolve_scaffold_payload(7, 1286, "failed", full_extract)
|
||
|
||
self.assertEqual(cached, data)
|
||
self.assertEqual({}, usage)
|
||
self.assertIn("cache hit", trace)
|
||
self.assertIn(str(cache_path), trace)
|
||
full_extract.assert_not_called()
|
||
|
||
def test_pending_chapter_never_reuses_cache(self):
|
||
"""pending 即使存在形状完整的同名文件,也必须重新执行完整抽取。"""
|
||
fresh = {"outline": "新细纲", "entities": [], "hints": []}
|
||
full_extract = Mock(return_value=(fresh, {"prompt_tokens": 11}))
|
||
|
||
with tempfile.TemporaryDirectory() as tmp, patch.object(pll, "TMP", pathlib.Path(tmp)):
|
||
(pathlib.Path(tmp) / "7-9-scaffold.json").write_text(
|
||
'{"outline":"旧细纲","entities":[],"hints":[]}', encoding="utf-8")
|
||
data, usage, trace = pll.resolve_scaffold_payload(7, 9, "pending", full_extract)
|
||
|
||
self.assertEqual(fresh, data)
|
||
self.assertEqual({"prompt_tokens": 11}, usage)
|
||
self.assertIn("cache miss", trace)
|
||
self.assertIn("status=pending", trace)
|
||
full_extract.assert_called_once_with()
|
||
|
||
def test_missing_or_invalid_failed_cache_falls_back_to_full_extraction(self):
|
||
"""failed 缓存缺失、非严格对象或字段类型不合理时,都走完整抽取回退。"""
|
||
cases = {
|
||
"missing": None,
|
||
"root-list": '[]',
|
||
"bad-outline": '{"outline":[],"entities":[],"hints":[]}',
|
||
"bad-entities": '{"outline":"纲","entities":[1],"hints":[]}',
|
||
"bad-hints": '{"outline":"纲","entities":[],"hints":"线索"}',
|
||
"broken-json": '{"outline":',
|
||
}
|
||
for name, cache_text in cases.items():
|
||
with self.subTest(name=name), tempfile.TemporaryDirectory() as tmp, \
|
||
patch.object(pll, "TMP", pathlib.Path(tmp)):
|
||
if cache_text is not None:
|
||
(pathlib.Path(tmp) / "7-10-scaffold.json").write_text(cache_text, encoding="utf-8")
|
||
fresh = {"outline": f"新细纲-{name}", "entities": [], "hints": []}
|
||
full_extract = Mock(return_value=(fresh, {"completion_tokens": 7}))
|
||
|
||
data, usage, trace = pll.resolve_scaffold_payload(7, 10, "failed", full_extract)
|
||
|
||
self.assertEqual(fresh, data)
|
||
self.assertEqual({"completion_tokens": 7}, usage)
|
||
self.assertIn("cache miss", trace)
|
||
full_extract.assert_called_once_with()
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main(verbosity=2)
|