255 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""clean skill:精确匹配删除执行器(LLM 建议 ≠ 必删,守卫规则最终裁决)+ 审计入库。
输入 deletions JSON:[{chapter_order|chapter, exact, reason}](键名 ASCII,兼容中文键)。
匹配三级(M3 首窗实测的两类病对症):
1. 精确命中:exact 逐字在章内出现;
2. 空白弹性:文字逐字、连续空白段放宽为 \\s+(M3 会把原文 \\n\\n\\n 折叠成 \\n\\n);
3. 邻章容错:本章 1/2 级全失败时试 order±1 章(M3 在 10 万字窗内偶发章号偏移)。
三级全失败 → 拒绝(安全方向:宁漏删不误删)。
"""
import json
import pathlib
import re
import sys
import click
import psycopg
DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse-example"
"?keepalives=1&keepalives_idle=15&keepalives_interval=5&keepalives_count=3")
TENANT, ACTOR = 1, "1"
MIN_LEN, MAX_LEN = 4, 500
MAX_CH_RATIO = 0.20 # 单章累计删除上限(占章长比)
# 守卫4:垃圾行强特征——exact 的每一行必须命中其一,否则视为疑混正文。
# 首窗实测:盗版源会把广告插进正文句子中间,M3 会把「广告+被截断的正文」拼成一项报上来,
# 纯文本匹配防不住,必须按行做内容特征裁决(只用通用垃圾词,不含任何书内专名)。
GARBAGE_LINE_PAT = re.compile(
r"^\s*[((]?\s*ps\b|^\s*[((]?\s*附[::]"
r"|感谢|打赏|盟主|书友|书评|拜谢|恳求|打滚|鸣谢"
r"|(?:月|推荐|评价|催更|票)票|起点币|求(?:收藏|推荐|订阅|票|月票)|拉票|冲榜"
r"|龙套|上架|[aA]签|签约|加更|[一二三四五六七八九十]更|更新时间|码字|存稿|请假|感言"
r"|www|http|[\._]{2,}|→|♂|÷|一秒记住|记住本站|无弹窗|手机版|首发|最新章节|整理|转载"
r"|手打中|稍等片刻|重新刷新|获取最新" # 盗版站占位残留(1040-1060 验证窗实测)
r"|微信|公众号|书名[::]|想听到更多你们的声音|想收到更多你们的建议", # 作者导流/推书段(机战无限/星环实测)
re.I)
# 水印裁剪:盗版源把水印粘在正文行尾(如『“我草!”百镀一下“××”最新章节第一时间免费阅读。』),
# LLM 会把整行报上来——删除段若含此类明确水印且水印前有前缀(多为正文),只删水印部分。
WATERMARK_PAT = re.compile(r"(?:百镀一下|百度一下|一秒记住).{0,45}?(?:免费阅读。?|免费读!?|阅读。)")
def flex_pattern(exact):
"""空白弹性正则:非空白文字逐字 escape,连续空白段 → \\s+。
只放宽空白,文字仍须逐字一致——不会匹配到不同文字的正文。
"""
parts = [p for p in re.split(r"\s+", exact) if p]
if not parts:
return None
return re.compile(r"\s+".join(re.escape(p) for p in parts))
def try_remove(text, exact):
"""在 text 中删除 exact(精确→弹性两级)。
返回 (新text, 命中次数, 实删字数, 方式, 首次命中前文, 首次命中后文)。
前后文各取 ~60 字:质检报告用——人工确认删除段是独立垃圾而非正文一部分。
"""
n = text.count(exact)
if n:
i = text.find(exact)
before, after = text[max(0, i - 60):i], text[i + len(exact):i + len(exact) + 60]
return text.replace(exact, ""), n, len(exact) * n, "精确", before, after
pat = flex_pattern(exact)
if pat:
m = pat.search(text)
if m:
before, after = text[max(0, m.start() - 60):m.start()], text[m.end():m.end() + 60]
new, n = pat.subn("", text)
return new, n, len(text) - len(new), "空白弹性", before, after
return text, 0, 0, "", "", ""
@click.command()
@click.option("--work-id", type=int, required=True)
@click.option("--batch", required=True, help="清洗批次号(审计用)")
@click.option("--file", "file_", type=click.Path(exists=True), required=True)
@click.option("--model", default="MiniMax-M3", show_default=True, help="建议来源模型(审计标注)")
@click.option("--dry-run", is_flag=True, help="只对账不落库")
@click.option("--report-md", type=click.Path(), help="人读质检报告输出路径(每条删除带上下文对照)")
def main(work_id, batch, file_, model, dry_run, report_md):
raw = json.loads(pathlib.Path(file_).read_text())
if isinstance(raw, dict):
raw = raw.get("deletions") or raw.get("删除") or []
# 按章分组(键名 ASCII 为主,兼容中文键)
by_ch = {}
for d in raw:
no = d.get("chapter_order") or d.get("chapter") or d.get("章序")
exact = d.get("exact") or d.get("原文")
if no is None or not exact:
continue
by_ch.setdefault(int(no), []).append((exact, d.get("reason") or d.get("理由") or ""))
# 一次拉齐所有目标章及其邻章(±1),在内存统一匹配,减少 DB 往返
targets = sorted({n for no in by_ch for n in (no - 1, no, no + 1) if n >= 1})
stats = {"删除段": 0, "删除字数": 0, "拒绝": [], "弹性命中": 0, "邻章命中": 0}
with psycopg.connect(DSN) as conn:
# 书名水印豁免:盗版源每章插孤立书名行——孤行精确等于本书书名不可能是正文叙述
work_title = (conn.execute(
"SELECT title FROM muse_content_work WHERE tenant_id=%s AND id=%s",
(TENANT, work_id)).fetchone() or ("",))[0]
rows = conn.execute(
"""SELECT c.order_no, c.id, c.title, b.id, b.content_text
FROM muse_content_chapter c JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE
WHERE c.tenant_id=%s AND c.work_id=%s AND c.deleted=FALSE AND c.order_no = ANY(%s)""",
(TENANT, work_id, targets)).fetchall()
# chapters[no] = {ch_id, title, blk_id, text(可变), orig(原文), removed(累计), audit(审计行)}
chapters = {r[0]: {"ch_id": r[1], "title": r[2], "blk_id": r[3],
"text": r[4], "orig": r[4], "removed": 0, "audit": []} for r in rows}
for no, dels in sorted(by_ch.items()):
if no not in chapters:
stats["拒绝"].append(f"#{no}: 章不存在")
continue
for exact, reason in dels:
# 守卫1:长度界
if not (MIN_LEN <= len(exact) <= MAX_LEN):
stats["拒绝"].append(f"#{no}: 长度越界({len(exact)}) {exact[:24]!r}")
continue
# 守卫4:行级垃圾特征——无特征的行=疑混正文(孤立书名行水印豁免)
lines = [ln for ln in exact.split("\n") if ln.strip()]
plain = [ln for ln in lines
if not GARBAGE_LINE_PAT.search(ln)
and not (work_title and (ln.strip() == work_title
or f"《{work_title}》" in ln))]
if plain and len(lines) == 1:
stats["拒绝"].append(f"#{no}: 无垃圾特征疑正文 {exact[:24]!r}")
continue
if plain:
# 拆行降级:只删有垃圾特征的行,疑正文行跳过并显式报告(宁漏删不误删)
for ln in plain:
stats["拒绝"].append(f"#{no}: 拆行跳过疑正文行 {ln[:24]!r}")
units = [(ln, reason + "(拆行)") for ln in lines
if (GARBAGE_LINE_PAT.search(ln)
or (work_title and (ln.strip() == work_title
or f"《{work_title}》" in ln)))
and MIN_LEN <= len(ln) <= MAX_LEN]
else:
units = [(exact, reason)]
for u_exact, u_reason in units:
# 水印裁剪:水印前的前缀(多为正文)不删,exact 缩到水印起点
wm = WATERMARK_PAT.search(u_exact)
if wm and wm.start() > 0:
u_exact = u_exact[wm.start():]
u_reason += "(水印裁剪)"
# 本章两级匹配 → 邻章容错(章号偏移是 LLM 在长窗里的已知病)
hit_no = None
for cand in (no, no - 1, no + 1):
ch = chapters.get(cand)
if not ch:
continue
# 守卫2:不得含该章章题(按实际匹配章判断)
if ch["title"] and ch["title"] in u_exact:
stats["拒绝"].append(f"#{no}: 含章题 {u_exact[:24]!r}")
hit_no = -1 # 标记已裁决,不再试邻章
break
new, n, removed, how, before, after = try_remove(ch["text"], u_exact)
if n:
ch["text"] = new
ch["removed"] += removed
note = u_reason
if how == "空白弹性":
stats["弹性命中"] += 1
note += "(空白弹性)"
if cand != no:
stats["邻章命中"] += 1
note += f"(报#{no}实删#{cand})"
ch["audit"].append((u_exact, n, note, before, after))
stats["删除段"] += 1
stats["删除字数"] += removed
hit_no = cand
break
if hit_no is None:
stats["拒绝"].append(f"#{no}: 未命中 {u_exact[:24]!r}")
# 守卫3:单章删除比例硬顶——超顶整章回退不落库
dirty = []
for no, ch in sorted(chapters.items()):
if not ch["audit"]:
continue
# 20% 顶带绝对下限:短章(如百字残章)按比例几乎删不了任何垃圾,放宽到至少 120 字
if ch["removed"] > max(len(ch["orig"]) * MAX_CH_RATIO, 120):
stats["删除段"] -= len(ch["audit"])
stats["删除字数"] -= ch["removed"]
stats["拒绝"].append(f"#{no}: 累计删除 {ch['removed']} 字超章长20%上限,整章回退")
continue
# 章尾孤悬分隔线清理:删掉章尾 ps 段后,其前置星号线会孤悬(前50章实测32处
# 星号线全为 ps 前置分隔;场景分隔线后必有正文不会孤悬,故此规则天然安全)
ch["text"] = re.sub(r"(?:\s*\n\s*[**]{3,}\s*)+$", "\n", ch["text"])
ch["text"] = re.sub(r"\n{3,}", "\n\n", ch["text"]) # 删除后合并多余空行
dirty.append((no, ch))
if not dry_run:
for no, ch in dirty:
wc = len(re.sub(r"\s", "", ch["text"]))
conn.execute("UPDATE muse_content_block SET content_text=%s, word_count=%s, updater=%s WHERE id=%s",
(ch["text"], wc, ACTOR, ch["blk_id"]))
for exact, n, note, _, _ in ch["audit"]:
conn.execute(
"""INSERT INTO example_clean_log (work_id, chapter_id, batch, removed_text, occurrences,
reason, model, creator, updater, tenant_id)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)""",
(work_id, ch["ch_id"], batch, exact, n, note[:200], model, ACTOR, ACTOR, TENANT))
conn.execute(
"""UPDATE muse_content_work w SET word_count=(
SELECT COALESCE(sum(b.word_count),0) FROM muse_content_block b
WHERE b.tenant_id=%s AND b.work_id=w.id AND b.deleted=FALSE), updater=%s
WHERE w.id=%s""", (TENANT, ACTOR, work_id))
conn.commit()
mode = "(dry-run 未落库)" if dry_run else ""
click.echo(f"清洗执行{mode}: 删除 {stats['删除段']} 段 / {stats['删除字数']:,} 字"
f"(弹性命中 {stats['弹性命中']}、邻章命中 {stats['邻章命中']});拒绝 {len(stats['拒绝'])} 条")
for r in stats["拒绝"][:12]:
click.echo(f" [拒] {r}")
if dry_run:
# dry-run 打印完整删除明细:门①人工质检的数据源
click.echo("── 删除明细(质检用) ──")
for no, ch in dirty:
for exact, n, note, _, _ in ch["audit"]:
click.echo(f" [删] #{no} ×{n} {note or '(无理由)'}\n {exact!r}")
if report_md:
# 人读质检报告:每条删除带前后文对照——放行裁决的证据面
lines_md = [f"# 清洗质检报告(work={work_id} batch={batch}{mode})",
"",
f"拟删 **{stats['删除段']} 段 / {stats['删除字数']:,} 字**"
f"(弹性命中 {stats['弹性命中']}、邻章命中 {stats['邻章命中']});"
f"守卫拒绝 {len(stats['拒绝'])} 条(见文末)。",
"",
"> 看法:确认每条【删】段是独立垃圾、上下文(灰引文)是完整正文即可放行。",
"> 收尾规则(落库时自动执行,不逐条列出):被删 ps 段的前置星号分隔线若孤悬章尾",
"> 将连带清除;被清洗章的三连以上换行归一为一个空行(⏎ 符号即原文换行)。", ""]
seq = 0
for no, ch in dirty:
for exact, n, note, before, after in ch["audit"]:
seq += 1
ctx_b = before.replace("\n", " ⏎ ").strip() or "(章首)"
ctx_a = after.replace("\n", " ⏎ ").strip() or "(章尾)"
body = exact.replace("\n", " ⏎ ")
lines_md += [f"### {seq}. 第{no}章《{ch['title']}》 ×{n} | {note or '(无理由)'}",
f"> …{ctx_b}", f"", f"**【删】** `{body}`", f"", f"> {ctx_a}…", ""]
lines_md += ["## 守卫拒绝清单(未删,含保护的疑正文行)", ""]
lines_md += [f"- {r}" for r in stats["拒绝"]] or ["- 无"]
pathlib.Path(report_md).write_text("\n".join(lines_md))
click.echo(f"质检报告已写 {report_md}")
if __name__ == "__main__":
try:
main()
except psycopg.Error as e:
click.echo(f"[db错误] {type(e).__name__}: {e}", err=True)
sys.exit(1)