diff --git a/.claude/skills/parse-book/scripts/parse_llm.py b/.claude/skills/parse-book/scripts/parse_llm.py index 01e30fb..3a10b09 100644 --- a/.claude/skills/parse-book/scripts/parse_llm.py +++ b/.claude/skills/parse-book/scripts/parse_llm.py @@ -33,6 +33,8 @@ DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse- TENANT = 1 HERE = pathlib.Path(__file__).resolve().parent TMP = pathlib.Path("/tmp/muse-parse") +OUTLINE_TARGET_RATIO = 0.045 +MAX_OUTLINE_REPAIR_ATTEMPTS = 2 # ── 提示词资产(合同一律 load_contracts 动态渲染,禁手写——CONTRACTS 漂移冤案教训) ── @@ -40,6 +42,9 @@ IDENTITY = """你是知识抽取员(extractor),分析槽位的默认绑定件 元数据纪律:schema 有什么字段你就抽什么,schema 没有的不抽——字段合同就是抽取 checklist,不自造结构;归型走各型「判据」;归不进任何型的候选=枚举缺口,如实报不硬塞;每字段要有正文证据,置信度低标「?」。 通则:以正文为准,不脑补正文没写的;基础字段规范填;采纳正文≠确认知识。""" +OUTLINE_COMPRESS_IDENTITY = """你是细纲压缩员。只压缩给定细纲,不补写剧情,不重新抽取其他内容。 +必须严格遵守非空白字符上限,只输出指定 JSON 对象。""" + ENTITY_CRITERIA = ("实体型判据(脚手架级,只要 型/名称/一句话摘要):character=具名可指认的行动主体;" "location=有名字的地点/星球/设施;faction=组织/国家/军团/公司;" "power_system=力量体系/科技体系/修炼阶梯(体系本身,非招式);" @@ -195,6 +200,76 @@ def m3_json(prompt, model, need_keys, system=IDENTITY): raise RuntimeError(f"JSON 形状重试仍失败({used}): {err}") +def non_whitespace_len(text): + """统计机械比例门使用的非空白字符数,保证调用前预检与 ingest 判据同口径。""" + return len(re.sub(r"\s", "", str(text or ""))) + + +def outline_target_cap(source_text): + """给细纲压缩留出相对 8% 硬门的安全余量,目标固定在正文的 4.5%。 + + 60 字下限与 parse_ingest 的短章绝对豁免一致,避免感言、公告等极短章无解。 + """ + return max(60, int(non_whitespace_len(source_text) * OUTLINE_TARGET_RATIO)) + + +def outline_compression_prompt(outline, cap, attempt): + """构造只含细纲的短提示,不重发正文、实体名录或完整章级抽取任务。""" + current_length = non_whitespace_len(outline) + # 总上限被 M3 系统性忽略时,用逐短语预算再留约三成余量;仍只做语义压缩,不截字符串。 + phrase_cap = max(6, int(cap * 0.16)) + return f"""【细纲专用压缩|第 {attempt}/{MAX_OUTLINE_REPAIR_ATTEMPTS} 轮】 +将下方细纲压成结构骨架,只保留章目标、关键事件、伏笔动作(埋/推/收)和章末钩子。 +用短语与分号,删除修饰、对白、过程复述;不得新增原细纲没有的事实。 +压缩结果的非空白字符不得超过 {cap},不得用空格或换行规避计数。 +上一版共 {current_length} 个非空白字符。输出最多 4 个无标签短语,用分号连接; +每个短语不超过 {phrase_cap} 个非空白字符,总计仍不得超过 {cap}。不要写“目标:”“事件:”等标签。 +不要重新执行其他抽取任务。只输出一个 JSON 对象,禁止任何其他文字: +{{"outline": "压缩后的细纲"}} + +【待压缩细纲】 +{outline}""" + + +def _merge_usage(total, current): + """累加 token 数值项;忽略上游 usage 中不可相加的嵌套明细。""" + for key, value in (current or {}).items(): + if isinstance(value, (int, float)): + total[key] = total.get(key, 0) + value + + +def repair_outline(first_data, source_text, model): + """有限次数压缩首轮细纲;成功时仅替换 outline,其他首轮字段原样保留。 + + 返回 ``(合并数据或 None, usage, 错误或 None)``。超长结果绝不机械截断,也不会 + 交给 ingest;调用方因此保留首轮比例门已写下的 failed 状态。 + """ + cap = outline_target_cap(source_text) + candidate = str(first_data.get("outline") or "").strip() + usage_total = {} + last_length = non_whitespace_len(candidate) + last_error = None + for attempt in range(1, MAX_OUTLINE_REPAIR_ATTEMPTS + 1): + try: + compressed, usage = m3_json( + outline_compression_prompt(candidate, cap, attempt), model, ("outline",), + system=OUTLINE_COMPRESS_IDENTITY) + _merge_usage(usage_total, usage) + candidate = str(compressed.get("outline") or "").strip() + last_length = non_whitespace_len(candidate) + if candidate and last_length <= cap: + repaired = dict(first_data) + repaired["outline"] = candidate + return repaired, usage_total, None + last_error = "输出为空" if not candidate else f"压缩输出 {last_length} 字,目标不超过 {cap} 字" + except RuntimeError as exc: + # JSON/调用错误允许进入下一次有限重试;敏感全链耗尽仍由 SensitiveHardStop 向上硬停。 + last_error = f"调用失败:{exc}" + detail = last_error or f"压缩输出 {last_length} 字,目标不超过 {cap} 字" + return None, usage_total, (f"细纲专用压缩 {MAX_OUTLINE_REPAIR_ATTEMPTS} 轮仍失败:{detail};" + "未截断、未再次入库,保持 failed") + + def ingest(kind, work_id, key, payload, keyflag="--chapter-order"): """写临时文件 → parse_ingest 机械校验入库;返回 (是否成功, 输出文本)。""" TMP.mkdir(parents=True, exist_ok=True) @@ -206,6 +281,92 @@ def ingest(kind, work_id, key, payload, keyflag="--chapter-order"): return r.returncode == 0, (r.stdout + r.stderr).strip() +def ingest_scaffold_with_repair(work_id, chapter_order, first_data, source_text, model): + """首轮入库比例失败时只修细纲;返回入库结果、可追踪输出与压缩调用 usage。""" + ok, output = ingest("scaffold", work_id, chapter_order, first_data) + if ok or "细纲比例" not in output: + return ok, output, {} + repaired, usage, error = repair_outline(first_data, source_text, model) + if repaired is None: + return False, f"{output}\n{error}", usage + ok, repaired_output = ingest("scaffold", work_id, chapter_order, repaired) + return ok, repaired_output, usage + + +def _is_complete_scaffold_payload(data): + """缓存只接受完整章级对象;数组成员也必须是对象,拒绝容错修复后的模糊形状。""" + return ( + isinstance(data, dict) + and isinstance(data.get("outline"), str) + and bool(data["outline"].strip()) + and isinstance(data.get("entities"), list) + and all(isinstance(item, dict) for item in data["entities"]) + and isinstance(data.get("hints"), list) + and all(isinstance(item, dict) for item in data["hints"]) + ) + + +def resolve_scaffold_payload(work_id, chapter_order, scaffold_status, full_extract, trace_output=None): + """仅为 failed 章复用严格缓存;其他情况惰性调用正文完整抽取。 + + 返回 ``(载荷, usage, cache trace)``。缓存读取只用标准 JSON 解码,不走 LLM 输出的 + 容错提取;因此非法或残缺文件不会被误当成可复用首轮载荷。 + """ + cache_path = TMP / f"{work_id}-{chapter_order}-scaffold.json" + + def traced(message): + """先落追踪输出再做可能耗时或失败的完整抽取。""" + if trace_output is not None: + trace_output(message) + return message + + if scaffold_status != "failed": + trace = traced(f"cache miss: status={scaffold_status},不复用 {cache_path}") + data, usage = full_extract() + return data, usage, trace + try: + data = json.loads(cache_path.read_text(encoding="utf-8")) + except FileNotFoundError: + trace = traced(f"cache miss: 文件不存在 {cache_path}") + data, usage = full_extract() + return data, usage, trace + except (OSError, UnicodeError, json.JSONDecodeError) as exc: + trace = traced(f"cache miss: 无法严格解析 {cache_path}({type(exc).__name__})") + data, usage = full_extract() + return data, usage, trace + if not _is_complete_scaffold_payload(data): + trace = traced(f"cache miss: 对象或字段类型非法 {cache_path}") + fresh, usage = full_extract() + return fresh, usage, trace + trace = traced(f"cache hit: 复用首轮 failed 载荷 {cache_path}") + return data, {}, trace + + +def chapter_orders(conn, work_id, from_, to, incomplete_only): + """返回本轮目标章;增量模式在循环前一次筛出 pending/failed,避免扫描全部 done 章。""" + if not incomplete_only: + return list(range(from_, to + 1)) + rows = conn.execute( + """SELECT c.order_no + FROM muse_content_chapter c + JOIN example_parse_task t ON t.chapter_id=c.id AND t.tenant_id=c.tenant_id + WHERE c.tenant_id=%s AND c.work_id=%s AND c.order_no BETWEEN %s AND %s + AND c.deleted=FALSE + AND COALESCE(t.scaffold_status, 'pending') IN ('pending', 'failed') + ORDER BY c.order_no""", + (TENANT, work_id, from_, to)).fetchall() + return [row[0] for row in rows] + + +def initialize_tasks(work_id, from_, to, incomplete_only): + """首次全量运行建任务行;增量恢复复用已有任务表,避免再次逐章初始化。""" + if incomplete_only: + return + subprocess.run([sys.executable, str(HERE / "parse_ingest.py"), "init-tasks", + "--work-id", str(work_id), "--from", str(from_), "--to", str(to)], + capture_output=True, text=True) + + @click.group() def cli(): """M3 直调拆书(章级线索 → 窗级出卡)""" @@ -216,16 +377,19 @@ def cli(): @click.option("--from", "from_", type=int, required=True) @click.option("--to", type=int, required=True) @click.option("--model", default="MiniMax-M3", show_default=True) -def chapters(work_id, from_, to, model): +@click.option("--incomplete-only", is_flag=True, + help="只处理 pending/failed 章;循环前一次筛选,不逐章连接或打印 done 章") +def chapters(work_id, from_, to, model, incomplete_only): """章级 pass:逐章一次 M3(细纲+实体+范式候选线索)。正文只过这一遍。""" - # 任务行就绪(幂等) - subprocess.run([sys.executable, str(HERE / "parse_ingest.py"), "init-tasks", - "--work-id", str(work_id), "--from", str(from_), "--to", str(to)], - capture_output=True, text=True) + initialize_tasks(work_id, from_, to, incomplete_only) with psycopg.connect(DSN) as conn: title = conn.execute("SELECT title FROM muse_content_work WHERE id=%s", (work_id,)).fetchone()[0] + targets = chapter_orders(conn, work_id, from_, to, incomplete_only) + if incomplete_only: + click.echo(f"incomplete-only: 待处理 {len(targets)} 章") total_in = total_out = 0 - for ch in range(from_, to + 1): + cache_hits = cache_misses = 0 + for ch in targets: with psycopg.connect(DSN) as conn: row = conn.execute( """SELECT c.id, c.title, b.content_text, t.scaffold_status @@ -241,28 +405,33 @@ def chapters(work_id, from_, to, model): if s_st == "done": click.echo(f"#{ch} 脚手架已完成,跳过") continue - # 前文实体索引(紧凑:型/名称/摘要),判重用 - prev = [e for (ents,) in conn.execute( - """SELECT s.entities FROM example_parse_scaffold s - JOIN muse_content_chapter c ON c.id=s.chapter_id - WHERE s.tenant_id=%s AND s.work_id=%s AND c.order_no<%s AND s.deleted=FALSE""", - (TENANT, work_id, ch)).fetchall() for e in ents] try: - p1 = scaffold_prompt(title, ch, ch_title, text, prev) - data, usage = m3_json(p1, model, ("outline", "entities", "hints")) + def full_extract(): + """仅在缓存未命中时查询判重索引并执行正文完整抽取。""" + with psycopg.connect(DSN) as prev_conn: + prev = [e for (ents,) in prev_conn.execute( + """SELECT s.entities FROM example_parse_scaffold s + JOIN muse_content_chapter c ON c.id=s.chapter_id + WHERE s.tenant_id=%s AND s.work_id=%s AND c.order_no<%s AND s.deleted=FALSE""", + (TENANT, work_id, ch)).fetchall() for e in ents] + prompt = scaffold_prompt(title, ch, ch_title, text, prev) + return m3_json(prompt, model, ("outline", "entities", "hints")) + + if incomplete_only: + data, usage, cache_trace = resolve_scaffold_payload( + work_id, ch, s_st, full_extract, + trace_output=lambda trace: click.echo(f"#{ch} scaffold {trace}")) + if cache_trace.startswith("cache hit:"): + cache_hits += 1 + else: + cache_misses += 1 + else: + data, usage = full_extract() total_in += usage.get("prompt_tokens", 0) total_out += usage.get("completion_tokens", 0) - ok, out = ingest("scaffold", work_id, ch, data) - # 比例超标是 M3 高频病:带压缩指令重试 1 次(首轮实测 10/15 章超标) - if not ok and "细纲比例" in out: - cap = max(60, int(len(re.sub(r"\s", "", text)) * 0.05)) - data, usage = m3_json( - p1 + f"\n\n【重试】上次细纲 {len(data['outline'])} 字超标被退回。" - f"压缩到 {cap} 字以内:只留章目标/关键事件/伏笔动作/钩子,删掉一切修饰与过程描述。", - model, ("outline", "entities", "hints")) - total_in += usage.get("prompt_tokens", 0) - total_out += usage.get("completion_tokens", 0) - ok, out = ingest("scaffold", work_id, ch, data) + ok, out, repair_usage = ingest_scaffold_with_repair(work_id, ch, data, text, model) + total_in += repair_usage.get("prompt_tokens", 0) + total_out += repair_usage.get("completion_tokens", 0) click.echo(f" {out}") except SensitiveHardStop as e: # 降级链(主+2 备)全撞敏感——按创始人指令硬停该书解析并汇报,不跳过、不硬扛 @@ -279,6 +448,8 @@ def chapters(work_id, from_, to, model): return except RuntimeError as e: click.echo(f" #{ch} 章级 M3 失败: {e}") + if incomplete_only: + click.echo(f"scaffold cache: hit={cache_hits} miss={cache_misses}") click.echo(f"《{title}》{from_}–{to} 章级完成;token in={total_in:,} out={total_out:,}") diff --git a/.claude/skills/parse-book/scripts/test_parse_llm_offline.py b/.claude/skills/parse-book/scripts/test_parse_llm_offline.py new file mode 100644 index 0000000..06ccab1 --- /dev/null +++ b/.claude/skills/parse-book/scripts/test_parse_llm_offline.py @@ -0,0 +1,175 @@ +#!/usr/bin/env python3 +"""parse_llm 章级细纲修复的纯离线回归测试。 + +红线:不连接数据库,不调用网络或真实 LLM。测试只验证细纲专用压缩、失败状态流和 +incomplete-only 的目标章筛选,避免放量时用真实额度验证本可机械证明的行为。 +""" +import pathlib +import sys +import tempfile +import unittest +from unittest.mock import Mock, patch + + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) +import parse_llm as pll # noqa: E402 + + +class _RowsConnection: + """只实现目标章查询所需的最小假连接,并记录查询次数。""" + + def __init__(self, rows): + self.rows = rows + self.queries = [] + + def execute(self, sql, args): + self.queries.append((sql, args)) + return self + + def fetchall(self): + return self.rows + + +class ParseLlmOfflineTest(unittest.TestCase): + """覆盖章级比例失败后的最小修复合同。""" + + def test_outline_repair_only_replaces_outline_and_uses_short_prompt(self): + """专用调用只压细纲,首轮实体与线索对象必须原样保留。""" + entities = [{"type": "character", "name": "甲", "brief": "主角"}] + hints = [{"type": "craft", "name": "伏笔", "clue": "埋后收", "evidence": "本章"}] + first = {"outline": "旧细纲" * 80, "entities": entities, "hints": hints} + source = "正文" * 1000 + + with patch.object(pll, "m3_json", return_value=({"outline": "新纲" * 30}, {"prompt_tokens": 9})) as call: + repaired, usage, error = pll.repair_outline(first, source, "MiniMax-M3") + + self.assertIsNone(error) + self.assertEqual("新纲" * 30, repaired["outline"]) + self.assertIs(entities, repaired["entities"]) + self.assertIs(hints, repaired["hints"]) + self.assertEqual(9, usage["prompt_tokens"]) + prompt = call.call_args.args[0] + cap = pll.outline_target_cap(source) + self.assertLess(len(prompt), 1200) + self.assertNotIn(source, prompt) + self.assertNotIn("entities", prompt) + self.assertNotIn("hints", prompt) + self.assertIn(f"非空白字符不得超过 {cap}", prompt) + self.assertIn('{"outline": "压缩后的细纲"}', prompt) + + def test_outline_repair_retries_finitely_and_never_truncates_or_reingests_over_limit(self): + """两轮仍超目标时保持首轮失败,不截断、不用超长结果伪造 done。""" + first = {"outline": "首轮" * 100, "entities": [{"name": "甲"}], "hints": [{"name": "线索"}]} + source = "正文" * 1000 + over_1 = "第一次仍超" * 30 + over_2 = "第二次仍超" * 30 + + with patch.object(pll, "m3_json", side_effect=[ + ({"outline": over_1}, {"completion_tokens": 10}), + ({"outline": over_2}, {"completion_tokens": 11}), + ]) as llm_call, patch.object(pll, "ingest", return_value=(False, "细纲比例 9.0% 超标")) as ingest_call: + ok, output, usage = pll.ingest_scaffold_with_repair(7, 1286, first, source, "MiniMax-M3") + + self.assertFalse(ok) + self.assertEqual(2, llm_call.call_count) + self.assertEqual(1, ingest_call.call_count) + self.assertIn(f"压缩输出 {pll.non_whitespace_len(over_2)} 字", output) + self.assertIn("保持 failed", output) + self.assertNotIn(over_2[:pll.outline_target_cap(source)], str(ingest_call.call_args_list)) + self.assertEqual(21, usage["completion_tokens"]) + + def test_compression_retry_breaks_total_cap_into_short_phrase_budget(self): + """真实 M3 忽略单一总上限后,重试须同时给上一版长度与逐短语预算。""" + prompt = pll.outline_compression_prompt("甲" * 89, 60, 2) + self.assertIn("上一版共 89 个非空白字符", prompt) + self.assertIn("最多 4 个无标签短语", prompt) + self.assertIn("每个短语不超过 9 个非空白字符", prompt) + self.assertIn("总计仍不得超过 60", prompt) + + def test_incomplete_only_selects_once_while_default_keeps_full_range(self): + """增量模式一次筛出 pending/failed;默认模式仍按原范围逐章兼容。""" + conn = _RowsConnection([(3,), (8,), (13,)]) + self.assertEqual([3, 8, 13], pll.chapter_orders(conn, 7, 1, 20, incomplete_only=True)) + self.assertEqual(1, len(conn.queries)) + sql, args = conn.queries[0] + self.assertIn("scaffold_status, 'pending') IN ('pending', 'failed')", sql) + self.assertEqual((pll.TENANT, 7, 1, 20), args) + + untouched = _RowsConnection([]) + self.assertEqual(list(range(1, 21)), pll.chapter_orders(untouched, 7, 1, 20, incomplete_only=False)) + self.assertEqual([], untouched.queries) + + def test_incomplete_only_skips_full_range_task_initialization(self): + """任务表已存在的增量恢复不得再对全书逐行执行 init-tasks。""" + with patch.object(pll.subprocess, "run") as run: + pll.initialize_tasks(7, 1, 2250, incomplete_only=True) + run.assert_not_called() + + pll.initialize_tasks(7, 1, 2250, incomplete_only=False) + run.assert_called_once() + self.assertIn("init-tasks", run.call_args.args[0]) + + def test_failed_chapter_reuses_complete_strict_cache_without_full_extraction(self): + """failed 章命中完整首轮载荷时,直接复用且不得再次执行正文完整抽取。""" + cached = { + "outline": "仍然超长的首轮细纲", + "entities": [{"type": "character", "name": "甲", "brief": "主角"}], + "hints": [{"type": "craft", "name": "伏笔", "clue": "埋后收", "evidence": "本章"}], + } + full_extract = Mock(side_effect=AssertionError("cache hit 不应调用完整 m3_json 抽取")) + + with tempfile.TemporaryDirectory() as tmp, patch.object(pll, "TMP", pathlib.Path(tmp)): + cache_path = pathlib.Path(tmp) / "7-1286-scaffold.json" + cache_path.write_text(__import__("json").dumps(cached), encoding="utf-8") + data, usage, trace = pll.resolve_scaffold_payload(7, 1286, "failed", full_extract) + + self.assertEqual(cached, data) + self.assertEqual({}, usage) + self.assertIn("cache hit", trace) + self.assertIn(str(cache_path), trace) + full_extract.assert_not_called() + + def test_pending_chapter_never_reuses_cache(self): + """pending 即使存在形状完整的同名文件,也必须重新执行完整抽取。""" + fresh = {"outline": "新细纲", "entities": [], "hints": []} + full_extract = Mock(return_value=(fresh, {"prompt_tokens": 11})) + + with tempfile.TemporaryDirectory() as tmp, patch.object(pll, "TMP", pathlib.Path(tmp)): + (pathlib.Path(tmp) / "7-9-scaffold.json").write_text( + '{"outline":"旧细纲","entities":[],"hints":[]}', encoding="utf-8") + data, usage, trace = pll.resolve_scaffold_payload(7, 9, "pending", full_extract) + + self.assertEqual(fresh, data) + self.assertEqual({"prompt_tokens": 11}, usage) + self.assertIn("cache miss", trace) + self.assertIn("status=pending", trace) + full_extract.assert_called_once_with() + + def test_missing_or_invalid_failed_cache_falls_back_to_full_extraction(self): + """failed 缓存缺失、非严格对象或字段类型不合理时,都走完整抽取回退。""" + cases = { + "missing": None, + "root-list": '[]', + "bad-outline": '{"outline":[],"entities":[],"hints":[]}', + "bad-entities": '{"outline":"纲","entities":[1],"hints":[]}', + "bad-hints": '{"outline":"纲","entities":[],"hints":"线索"}', + "broken-json": '{"outline":', + } + for name, cache_text in cases.items(): + with self.subTest(name=name), tempfile.TemporaryDirectory() as tmp, \ + patch.object(pll, "TMP", pathlib.Path(tmp)): + if cache_text is not None: + (pathlib.Path(tmp) / "7-10-scaffold.json").write_text(cache_text, encoding="utf-8") + fresh = {"outline": f"新细纲-{name}", "entities": [], "hints": []} + full_extract = Mock(return_value=(fresh, {"completion_tokens": 7})) + + data, usage, trace = pll.resolve_scaffold_payload(7, 10, "failed", full_extract) + + self.assertEqual(fresh, data) + self.assertEqual({"completion_tokens": 7}, usage) + self.assertIn("cache miss", trace) + full_extract.assert_called_once_with() + + +if __name__ == "__main__": + unittest.main(verbosity=2)