修复: 对齐细纲优选目标与机械硬门

This commit is contained in:
zizi 2026-07-20 20:08:31 +08:00
parent 078b819a06
commit f805b00dd1
2 changed files with 38 additions and 9 deletions

View File

@ -213,16 +213,22 @@ def outline_target_cap(source_text):
return max(60, int(non_whitespace_len(source_text) * OUTLINE_TARGET_RATIO))
def outline_hard_cap(source_text):
"""返回与 parse_ingest 完全一致的 8% 最终硬门。"""
return max(60, int(non_whitespace_len(source_text) * 0.08))
def outline_compression_prompt(outline, cap, attempt):
"""构造只含细纲的短提示,不重发正文、实体名录或完整章级抽取任务。"""
current_length = non_whitespace_len(outline)
# 总上限被 M3 系统性忽略时,用逐短语预算再留约三成余量;仍只做语义压缩,不截字符串。
phrase_cap = max(6, int(cap * 0.16))
phrase_count = 4 if attempt == 1 else 2
phrase_cap = max(6, int(cap * (0.16 if attempt == 1 else 0.20)))
return f"""【细纲专用压缩|第 {attempt}/{MAX_OUTLINE_REPAIR_ATTEMPTS} 轮】
将下方细纲压成结构骨架,只保留章目标、关键事件、伏笔动作(埋/推/收)和章末钩子。
用短语与分号,删除修饰、对白、过程复述;不得新增原细纲没有的事实。
压缩结果的非空白字符不得超过 {cap},不得用空格或换行规避计数。
上一版共 {current_length} 个非空白字符。输出最多 4 个无标签短语,用分号连接;
上一版共 {current_length} 个非空白字符。输出最多 {phrase_count} 个无标签短语,用分号连接;
每个短语不超过 {phrase_cap} 个非空白字符,总计仍不得超过 {cap}。不要写“目标:”“事件:”等标签。
不要重新执行其他抽取任务。只输出一个 JSON 对象,禁止任何其他文字:
{{"outline": "压缩后的细纲"}}
@ -245,6 +251,7 @@ def repair_outline(first_data, source_text, model):
交给 ingest;调用方因此保留首轮比例门已写下的 failed 状态。
"""
cap = outline_target_cap(source_text)
hard_cap = outline_hard_cap(source_text)
candidate = str(first_data.get("outline") or "").strip()
usage_total = {}
last_length = non_whitespace_len(candidate)
@ -257,7 +264,10 @@ def repair_outline(first_data, source_text, model):
_merge_usage(usage_total, usage)
candidate = str(compressed.get("outline") or "").strip()
last_length = non_whitespace_len(candidate)
if candidate and last_length <= cap:
if candidate and (
last_length <= cap
or (attempt == MAX_OUTLINE_REPAIR_ATTEMPTS and last_length <= hard_cap)
):
repaired = dict(first_data)
repaired["outline"] = candidate
return repaired, usage_total, None

View File

@ -62,7 +62,7 @@ class ParseLlmOfflineTest(unittest.TestCase):
first = {"outline": "首轮" * 100, "entities": [{"name": "甲"}], "hints": [{"name": "线索"}]}
source = "正文" * 1000
over_1 = "第一次仍超" * 30
over_2 = "第二次仍超" * 30
over_2 = "第二次仍超" * 40
with patch.object(pll, "m3_json", side_effect=[
({"outline": over_1}, {"completion_tokens": 10}),
@ -80,11 +80,30 @@ class ParseLlmOfflineTest(unittest.TestCase):
def test_compression_retry_breaks_total_cap_into_short_phrase_budget(self):
"""真实 M3 忽略单一总上限后,重试须同时给上一版长度与逐短语预算。"""
prompt = pll.outline_compression_prompt("甲" * 89, 60, 2)
self.assertIn("上一版共 89 个非空白字符", prompt)
self.assertIn("最多 4 个无标签短语", prompt)
self.assertIn("每个短语不超过 9 个非空白字符", prompt)
self.assertIn("总计仍不得超过 60", prompt)
first = pll.outline_compression_prompt("甲" * 89, 60, 1)
second = pll.outline_compression_prompt("甲" * 89, 60, 2)
self.assertIn("上一版共 89 个非空白字符", second)
self.assertIn("最多 4 个无标签短语", first)
self.assertIn("最多 2 个无标签短语", second)
self.assertIn("总计仍不得超过 60", second)
def test_final_repair_may_use_eight_percent_hard_cap_without_false_failure(self):
"""最后一轮已满足既有 8% 机械硬门时允许入库,4.5% 只是优选目标。"""
first = {"outline": "首轮" * 100, "entities": [], "hints": []}
source = "正文" * 1000
target = pll.outline_target_cap(source)
hard = pll.outline_hard_cap(source)
self.assertLess(target, hard)
between = "纲" * (target + 10)
with patch.object(pll, "m3_json", side_effect=[
({"outline": "仍超目标" * 30}, {}),
({"outline": between}, {}),
]):
repaired, _usage, error = pll.repair_outline(first, source, "MiniMax-M3")
self.assertIsNone(error)
self.assertEqual(between, repaired["outline"])
def test_incomplete_only_selects_once_while_default_keeps_full_range(self):
"""增量模式一次筛出 pending/failed;默认模式仍按原范围逐章兼容。"""