#!/usr/bin/env python3 """clean-book-text Skill:高频水印全书规则收割(LLM 探测样例 → 代码全书扫净)。 M3 按窗探测对「每章重复的固定水印」只会报样例章(窗内看到≠逐章报全),残留由本脚本收割: - 种子=example_clean_log 中该书重复删除段(相同 removed_text 出现 ≥min-occur 次、长度 ≥min-len)—— 种子都是已过全部守卫的真垃圾,且 20+ 字逐字重复 3+ 次的段不可能是正文; - 全书逐章 replace 删除(含空白弹性),审计入库;20% 顶守卫保留。 """ import re import sys import click import psycopg DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse-example" "?keepalives=1&keepalives_idle=15&keepalives_interval=5&keepalives_count=3") TENANT, ACTOR = 1, "1" def flex_pattern(exact): """空白弹性正则(与 clean_apply 同构:文字逐字、空白段弹性)。""" parts = [p for p in re.split(r"\s+", exact) if p] return re.compile(r"\s+".join(re.escape(p) for p in parts)) if parts else None @click.command() @click.option("--work-id", type=int, required=True) @click.option("--batch", required=True, help="收割批次号(审计标注)") @click.option("--min-occur", type=int, default=3, show_default=True, help="种子最小重复次数") @click.option("--min-len", type=int, default=20, show_default=True, help="种子最小长度") @click.option("--seed", "manual_seeds", multiple=True, help="手动种子(替代审计自动发现;用于人工确认过的碎水印,如孤行网址)") @click.option("--dry-run", is_flag=True) def main(work_id, batch, min_occur, min_len, manual_seeds, dry_run): with psycopg.connect(DSN) as conn: # 种子:手动指定(人工确认的碎水印),或该书审计里的重复删除段(已过全部守卫的真垃圾) seeds = list(manual_seeds) or [r[0] for r in conn.execute( """SELECT removed_text FROM example_clean_log WHERE tenant_id=%s AND work_id=%s AND length(removed_text)>=%s GROUP BY removed_text HAVING count(*)>=%s ORDER BY count(*) DESC""", (TENANT, work_id, min_len, min_occur)).fetchall()] if not seeds: click.echo(f"work={work_id} 无重复水印种子") return click.echo(f"work={work_id} 种子 {len(seeds)} 个:") for s in seeds: click.echo(f" · {s[:50]!r}") rows = conn.execute( """SELECT c.order_no, c.id, b.id, b.content_text FROM muse_content_chapter c JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE WHERE c.tenant_id=%s AND c.work_id=%s AND c.deleted=FALSE ORDER BY c.order_no""", (TENANT, work_id)).fetchall() n_del = n_ch = chars = 0 for no, ch_id, blk_id, text in rows: new_text, audit = text, [] for seed in seeds: n = new_text.count(seed) how = "" if n: new_text = new_text.replace(seed, "") else: pat = flex_pattern(seed) if pat: new_text, n = pat.subn("", new_text) how = "(空白弹性)" if n: audit.append((seed, n, f"高频水印全书收割{how}")) if not audit: continue removed = len(text) - len(new_text) # 20% 顶守卫(绝对下限 120 字,与 clean_apply 一致) if removed > max(len(text) * 0.20, 120): click.echo(f" [拒] #{no}: 收割 {removed} 字超顶,整章回退") continue new_text = re.sub(r"(?:\s*\n\s*[**]{3,}\s*)+$", "\n", new_text) new_text = re.sub(r"\n{3,}", "\n\n", new_text) n_ch += 1 n_del += sum(a[1] for a in audit) chars += removed if dry_run: continue wc = len(re.sub(r"\s", "", new_text)) conn.execute("UPDATE muse_content_block SET content_text=%s, word_count=%s, updater=%s WHERE id=%s", (new_text, wc, ACTOR, blk_id)) for seed, n, note in audit: conn.execute( """INSERT INTO example_clean_log (work_id, chapter_id, batch, removed_text, occurrences, reason, model, creator, updater, tenant_id) VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)""", (work_id, ch_id, batch, seed, n, note, "rule-sweep", ACTOR, ACTOR, TENANT)) if not dry_run: conn.execute( """UPDATE muse_content_work w SET word_count=( SELECT COALESCE(sum(b.word_count),0) FROM muse_content_block b WHERE b.tenant_id=%s AND b.work_id=w.id AND b.deleted=FALSE), updater=%s WHERE w.id=%s""", (TENANT, ACTOR, work_id)) conn.commit() mode = "(dry-run)" if dry_run else "" click.echo(f"收割{mode}: {n_ch} 章 / {n_del} 处 / {chars:,} 字") if __name__ == "__main__": try: main() except psycopg.Error as e: click.echo(f"[db错误] {type(e).__name__}: {e}", err=True) sys.exit(1)