197 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""clean skill精确匹配删除执行器LLM 建议 ≠ 必删,守卫规则最终裁决)+ 审计入库。
输入 deletions JSON[{chapter_order|chapter, exact, reason}](键名 ASCII兼容中文键
匹配三级M3 首窗实测的两类病对症):
1. 精确命中exact 逐字在章内出现;
2. 空白弹性:文字逐字、连续空白段放宽为 \\s+M3 会把原文 \\n\\n\\n 折叠成 \\n\\n
3. 邻章容错:本章 1/2 级全失败时试 order±1 章M3 在 10 万字窗内偶发章号偏移)。
三级全失败 → 拒绝(安全方向:宁漏删不误删)。
"""
import json
import pathlib
import re
import sys
import click
import psycopg
DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse-example"
"?keepalives=1&keepalives_idle=15&keepalives_interval=5&keepalives_count=3")
TENANT, ACTOR = 1, "1"
MIN_LEN, MAX_LEN = 4, 500
MAX_CH_RATIO = 0.20 # 单章累计删除上限(占章长比)
# 守卫4垃圾行强特征——exact 的每一行必须命中其一,否则视为疑混正文。
# 首窗实测盗版源会把广告插进正文句子中间M3 会把「广告+被截断的正文」拼成一项报上来,
# 纯文本匹配防不住,必须按行做内容特征裁决(只用通用垃圾词,不含任何书内专名)。
GARBAGE_LINE_PAT = re.compile(
r"^\s*[(]?\s*ps\b|^\s*[(]?\s*附[:]"
r"|感谢|打赏|盟主|书友|书评|拜谢|恳求|打滚|鸣谢"
r"|(?:月|推荐|评价|催更|票)票|起点币|求(?:收藏|推荐|订阅|票|月票)|拉票|冲榜"
r"|龙套|上架|[aA]签|签约|加更|[一二三四五六七八九十]更|更新时间|码字|存稿|请假|感言"
r"|www|http|[\._]{2,}|→|♂|÷|一秒记住|记住本站|无弹窗|手机版|首发|最新章节|整理|转载",
re.I)
def flex_pattern(exact):
"""空白弹性正则:非空白文字逐字 escape连续空白段 → \\s+。
只放宽空白,文字仍须逐字一致——不会匹配到不同文字的正文。
"""
parts = [p for p in re.split(r"\s+", exact) if p]
if not parts:
return None
return re.compile(r"\s+".join(re.escape(p) for p in parts))
def try_remove(text, exact):
"""在 text 中删除 exact精确→弹性两级。返回 (新text, 命中次数, 实删字数, 方式)。"""
n = text.count(exact)
if n:
return text.replace(exact, ""), n, len(exact) * n, "精确"
pat = flex_pattern(exact)
if pat:
new, n = pat.subn("", text)
if n:
return new, n, len(text) - len(new), "空白弹性"
return text, 0, 0, ""
@click.command()
@click.option("--work-id", type=int, required=True)
@click.option("--batch", required=True, help="清洗批次号(审计用)")
@click.option("--file", "file_", type=click.Path(exists=True), required=True)
@click.option("--model", default="MiniMax-M3", show_default=True, help="建议来源模型(审计标注)")
@click.option("--dry-run", is_flag=True, help="只对账不落库")
def main(work_id, batch, file_, model, dry_run):
raw = json.loads(pathlib.Path(file_).read_text())
if isinstance(raw, dict):
raw = raw.get("deletions") or raw.get("删除") or []
# 按章分组(键名 ASCII 为主,兼容中文键)
by_ch = {}
for d in raw:
no = d.get("chapter_order") or d.get("chapter") or d.get("章序")
exact = d.get("exact") or d.get("原文")
if no is None or not exact:
continue
by_ch.setdefault(int(no), []).append((exact, d.get("reason") or d.get("理由") or ""))
# 一次拉齐所有目标章及其邻章±1在内存统一匹配减少 DB 往返
targets = sorted({n for no in by_ch for n in (no - 1, no, no + 1) if n >= 1})
stats = {"删除段": 0, "删除字数": 0, "拒绝": [], "弹性命中": 0, "邻章命中": 0}
with psycopg.connect(DSN) as conn:
rows = conn.execute(
"""SELECT c.order_no, c.id, c.title, b.id, b.content_text
FROM muse_content_chapter c JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE
WHERE c.tenant_id=%s AND c.work_id=%s AND c.deleted=FALSE AND c.order_no = ANY(%s)""",
(TENANT, work_id, targets)).fetchall()
# chapters[no] = {ch_id, title, blk_id, text(可变), orig(原文), removed(累计), audit(审计行)}
chapters = {r[0]: {"ch_id": r[1], "title": r[2], "blk_id": r[3],
"text": r[4], "orig": r[4], "removed": 0, "audit": []} for r in rows}
for no, dels in sorted(by_ch.items()):
if no not in chapters:
stats["拒绝"].append(f"#{no}: 章不存在")
continue
for exact, reason in dels:
# 守卫1长度界
if not (MIN_LEN <= len(exact) <= MAX_LEN):
stats["拒绝"].append(f"#{no}: 长度越界({len(exact)}) {exact[:24]!r}")
continue
# 守卫4行级垃圾特征——无特征的行=疑混正文
lines = [ln for ln in exact.split("\n") if ln.strip()]
plain = [ln for ln in lines if not GARBAGE_LINE_PAT.search(ln)]
if plain and len(lines) == 1:
stats["拒绝"].append(f"#{no}: 无垃圾特征疑正文 {exact[:24]!r}")
continue
if plain:
# 拆行降级:只删有垃圾特征的行,疑正文行跳过并显式报告(宁漏删不误删)
for ln in plain:
stats["拒绝"].append(f"#{no}: 拆行跳过疑正文行 {ln[:24]!r}")
units = [(ln, reason + "(拆行)") for ln in lines
if GARBAGE_LINE_PAT.search(ln) and MIN_LEN <= len(ln) <= MAX_LEN]
else:
units = [(exact, reason)]
for u_exact, u_reason in units:
# 本章两级匹配 → 邻章容错(章号偏移是 LLM 在长窗里的已知病)
hit_no = None
for cand in (no, no - 1, no + 1):
ch = chapters.get(cand)
if not ch:
continue
# 守卫2不得含该章章题按实际匹配章判断
if ch["title"] and ch["title"] in u_exact:
stats["拒绝"].append(f"#{no}: 含章题 {u_exact[:24]!r}")
hit_no = -1 # 标记已裁决,不再试邻章
break
new, n, removed, how = try_remove(ch["text"], u_exact)
if n:
ch["text"] = new
ch["removed"] += removed
note = u_reason
if how == "空白弹性":
stats["弹性命中"] += 1
note += "(空白弹性)"
if cand != no:
stats["邻章命中"] += 1
note += f"(报#{no}实删#{cand})"
ch["audit"].append((u_exact, n, note))
stats["删除段"] += 1
stats["删除字数"] += removed
hit_no = cand
break
if hit_no is None:
stats["拒绝"].append(f"#{no}: 未命中 {u_exact[:24]!r}")
# 守卫3单章删除比例硬顶——超顶整章回退不落库
dirty = []
for no, ch in sorted(chapters.items()):
if not ch["audit"]:
continue
if ch["removed"] > len(ch["orig"]) * MAX_CH_RATIO:
stats["删除段"] -= len(ch["audit"])
stats["删除字数"] -= ch["removed"]
stats["拒绝"].append(f"#{no}: 累计删除 {ch['removed']} 字超章长20%上限,整章回退")
continue
ch["text"] = re.sub(r"\n{3,}", "\n\n", ch["text"]) # 删除后合并多余空行
dirty.append((no, ch))
if not dry_run:
for no, ch in dirty:
wc = len(re.sub(r"\s", "", ch["text"]))
conn.execute("UPDATE muse_content_block SET content_text=%s, word_count=%s, updater=%s WHERE id=%s",
(ch["text"], wc, ACTOR, ch["blk_id"]))
for exact, n, note in ch["audit"]:
conn.execute(
"""INSERT INTO example_clean_log (work_id, chapter_id, batch, removed_text, occurrences,
reason, model, creator, updater, tenant_id)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)""",
(work_id, ch["ch_id"], batch, exact, n, note[:200], model, ACTOR, ACTOR, TENANT))
conn.execute(
"""UPDATE muse_content_work w SET word_count=(
SELECT COALESCE(sum(b.word_count),0) FROM muse_content_block b
WHERE b.tenant_id=%s AND b.work_id=w.id AND b.deleted=FALSE), updater=%s
WHERE w.id=%s""", (TENANT, ACTOR, work_id))
conn.commit()
mode = "dry-run 未落库)" if dry_run else ""
click.echo(f"清洗执行{mode}: 删除 {stats['删除段']} 段 / {stats['删除字数']:,}"
f"(弹性命中 {stats['弹性命中']}、邻章命中 {stats['邻章命中']});拒绝 {len(stats['拒绝'])}")
for r in stats["拒绝"][:12]:
click.echo(f" [拒] {r}")
if dry_run:
# dry-run 打印完整删除明细:门①人工质检的数据源
click.echo("── 删除明细(质检用) ──")
for no, ch in dirty:
for exact, n, note in ch["audit"]:
click.echo(f" [删] #{no} ×{n} {note or '(无理由)'}\n {exact!r}")
if __name__ == "__main__":
try:
main()
except psycopg.Error as e:
click.echo(f"[db错误] {type(e).__name__}: {e}", err=True)
sys.exit(1)