diff --git a/.claude/skills/parse-book/scripts/parse_llm.py b/.claude/skills/parse-book/scripts/parse_llm.py index 3a10b09..ef8ae1c 100644 --- a/.claude/skills/parse-book/scripts/parse_llm.py +++ b/.claude/skills/parse-book/scripts/parse_llm.py @@ -213,16 +213,22 @@ def outline_target_cap(source_text): return max(60, int(non_whitespace_len(source_text) * OUTLINE_TARGET_RATIO)) +def outline_hard_cap(source_text): + """返回与 parse_ingest 完全一致的 8% 最终硬门。""" + return max(60, int(non_whitespace_len(source_text) * 0.08)) + + def outline_compression_prompt(outline, cap, attempt): """构造只含细纲的短提示,不重发正文、实体名录或完整章级抽取任务。""" current_length = non_whitespace_len(outline) # 总上限被 M3 系统性忽略时,用逐短语预算再留约三成余量;仍只做语义压缩,不截字符串。 - phrase_cap = max(6, int(cap * 0.16)) + phrase_count = 4 if attempt == 1 else 2 + phrase_cap = max(6, int(cap * (0.16 if attempt == 1 else 0.20))) return f"""【细纲专用压缩|第 {attempt}/{MAX_OUTLINE_REPAIR_ATTEMPTS} 轮】 将下方细纲压成结构骨架,只保留章目标、关键事件、伏笔动作(埋/推/收)和章末钩子。 用短语与分号,删除修饰、对白、过程复述;不得新增原细纲没有的事实。 压缩结果的非空白字符不得超过 {cap},不得用空格或换行规避计数。 -上一版共 {current_length} 个非空白字符。输出最多 4 个无标签短语,用分号连接; +上一版共 {current_length} 个非空白字符。输出最多 {phrase_count} 个无标签短语,用分号连接; 每个短语不超过 {phrase_cap} 个非空白字符,总计仍不得超过 {cap}。不要写“目标:”“事件:”等标签。 不要重新执行其他抽取任务。只输出一个 JSON 对象,禁止任何其他文字: {{"outline": "压缩后的细纲"}} @@ -245,6 +251,7 @@ def repair_outline(first_data, source_text, model): 交给 ingest;调用方因此保留首轮比例门已写下的 failed 状态。 """ cap = outline_target_cap(source_text) + hard_cap = outline_hard_cap(source_text) candidate = str(first_data.get("outline") or "").strip() usage_total = {} last_length = non_whitespace_len(candidate) @@ -257,7 +264,10 @@ def repair_outline(first_data, source_text, model): _merge_usage(usage_total, usage) candidate = str(compressed.get("outline") or "").strip() last_length = non_whitespace_len(candidate) - if candidate and last_length <= cap: + if candidate and ( + last_length <= cap + or (attempt == MAX_OUTLINE_REPAIR_ATTEMPTS and last_length <= hard_cap) + ): repaired = dict(first_data) repaired["outline"] = candidate return repaired, usage_total, None diff --git a/.claude/skills/parse-book/scripts/test_parse_llm_offline.py b/.claude/skills/parse-book/scripts/test_parse_llm_offline.py index 06ccab1..a942b55 100644 --- a/.claude/skills/parse-book/scripts/test_parse_llm_offline.py +++ b/.claude/skills/parse-book/scripts/test_parse_llm_offline.py @@ -62,7 +62,7 @@ class ParseLlmOfflineTest(unittest.TestCase): first = {"outline": "首轮" * 100, "entities": [{"name": "甲"}], "hints": [{"name": "线索"}]} source = "正文" * 1000 over_1 = "第一次仍超" * 30 - over_2 = "第二次仍超" * 30 + over_2 = "第二次仍超" * 40 with patch.object(pll, "m3_json", side_effect=[ ({"outline": over_1}, {"completion_tokens": 10}), @@ -80,11 +80,30 @@ class ParseLlmOfflineTest(unittest.TestCase): def test_compression_retry_breaks_total_cap_into_short_phrase_budget(self): """真实 M3 忽略单一总上限后,重试须同时给上一版长度与逐短语预算。""" - prompt = pll.outline_compression_prompt("甲" * 89, 60, 2) - self.assertIn("上一版共 89 个非空白字符", prompt) - self.assertIn("最多 4 个无标签短语", prompt) - self.assertIn("每个短语不超过 9 个非空白字符", prompt) - self.assertIn("总计仍不得超过 60", prompt) + first = pll.outline_compression_prompt("甲" * 89, 60, 1) + second = pll.outline_compression_prompt("甲" * 89, 60, 2) + self.assertIn("上一版共 89 个非空白字符", second) + self.assertIn("最多 4 个无标签短语", first) + self.assertIn("最多 2 个无标签短语", second) + self.assertIn("总计仍不得超过 60", second) + + def test_final_repair_may_use_eight_percent_hard_cap_without_false_failure(self): + """最后一轮已满足既有 8% 机械硬门时允许入库,4.5% 只是优选目标。""" + first = {"outline": "首轮" * 100, "entities": [], "hints": []} + source = "正文" * 1000 + target = pll.outline_target_cap(source) + hard = pll.outline_hard_cap(source) + self.assertLess(target, hard) + between = "纲" * (target + 10) + + with patch.object(pll, "m3_json", side_effect=[ + ({"outline": "仍超目标" * 30}, {}), + ({"outline": between}, {}), + ]): + repaired, _usage, error = pll.repair_outline(first, source, "MiniMax-M3") + + self.assertIsNone(error) + self.assertEqual(between, repaired["outline"]) def test_incomplete_only_selects_once_while_default_keeps_full_range(self): """增量模式一次筛出 pending/failed;默认模式仍按原范围逐章兼容。"""