将角色与 Skill 从 .claude 迁入 .agent,移除 Claude CLI 运行时并接入固定 Opus 角色 profile、完整 schema、预算 deadline、raw 与回执证据链。 同步拆分 Skill 职责、复利 lesson、Gate 回放、Dashboard 人审入口、数据库登记和机械门禁;候选设计正文不包含在本提交中。
82 lines
4.0 KiB
Python
82 lines
4.0 KiB
Python
#!/usr/bin/env python3
|
||
"""B1 基础面验收报告:字数/分章/章节名/段落切分/残留垃圾扫描(纯规则,审查面)。"""
|
||
import re
|
||
import statistics
|
||
import sys
|
||
|
||
import click
|
||
import psycopg
|
||
|
||
from muse_db import connect
|
||
TENANT = 1
|
||
|
||
# 语义垃圾嫌疑模式(扫描计数用;真正删除走 clean-book-text 的 LLM 检测+代码执行)
|
||
JUNK_PATTERNS = {
|
||
"网址残留": re.compile(r'www\.|http|\.(?:com|net|cc|org|info)\b', re.I),
|
||
"一秒记住类": re.compile(r'一秒记住|天才.{0,3}记住|记住本站|首发域名|手机版阅读|无弹窗'),
|
||
"求票拉票": re.compile(r'求(?:月票|推荐票|订阅|收藏)|拉票|票票|投个|打赏|加更规则'),
|
||
"乱码符号": re.compile(r'[♂♀÷×]|&#\d+;|&[a-z]{2,6};'),
|
||
"作者话PS": re.compile(r'^PS[::]|^ps[::]|^附[::]|^作者的?话', re.M),
|
||
}
|
||
|
||
|
||
@click.command()
|
||
@click.option("--work-id", type=int, multiple=True, help="不给则全部")
|
||
@click.option("--sample-titles", default=5, show_default=True)
|
||
def main(work_id, sample_titles):
|
||
with connect() as conn:
|
||
works = conn.execute(
|
||
f"""SELECT id, title FROM muse_content_work WHERE tenant_id=%s AND deleted=FALSE
|
||
{'AND id = ANY(%s)' if work_id else ''} ORDER BY id""",
|
||
(TENANT, list(work_id)) if work_id else (TENANT,)).fetchall()
|
||
for wid, title in works:
|
||
rows = conn.execute(
|
||
"""SELECT c.order_no, c.title, b.content_text
|
||
FROM muse_content_chapter c JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE
|
||
WHERE c.tenant_id=%s AND c.work_id=%s AND c.deleted=FALSE ORDER BY c.order_no""",
|
||
(TENANT, wid)).fetchall()
|
||
if not rows:
|
||
continue
|
||
wc = [len(re.sub(r'\s', '', t)) for _, _, t in rows]
|
||
paras_per_ch, para_lens, fat_paras = [], [], 0
|
||
junk_hits = {k: 0 for k in JUNK_PATTERNS}
|
||
junk_samples = {}
|
||
for _, _, t in rows:
|
||
# 网文段落=非空行(一行一段是行业惯例;星环使命等源单换行分段,空行切分会误判粘连)
|
||
paras = [p for p in t.split("\n") if p.strip()]
|
||
paras_per_ch.append(len(paras))
|
||
for p in paras:
|
||
L = len(re.sub(r'\s', '', p))
|
||
para_lens.append(L)
|
||
if L > 2000:
|
||
fat_paras += 1
|
||
for k, pat in JUNK_PATTERNS.items():
|
||
ms = pat.findall(t)
|
||
if ms:
|
||
junk_hits[k] += len(ms)
|
||
if k not in junk_samples:
|
||
m = pat.search(t)
|
||
s = t[max(0, m.start() - 20):m.end() + 30].replace("\n", " ")
|
||
junk_samples[k] = s
|
||
print(f"═══ [{wid}] 《{title}》 ═══")
|
||
print(f" 章数 {len(rows)} | 总字数 {sum(wc):,} | 章字数 均{int(statistics.mean(wc))} 中位{int(statistics.median(wc))} 最短{min(wc)} 最长{max(wc)}")
|
||
shorts = sorted(zip(wc, (r[0] for r in rows), (r[1] for r in rows)))[:3]
|
||
print(f" 最短3章: " + " / ".join(f"#{o}《{t[:16]}》{w}字" for w, o, t in shorts))
|
||
print(f" 段落: 每章均{int(statistics.mean(paras_per_ch))}段 | 段均{int(statistics.mean(para_lens))}字 中位{int(statistics.median(para_lens))} | >2000字粘连段 {fat_paras}")
|
||
active = {k: v for k, v in junk_hits.items() if v}
|
||
print(f" 垃圾嫌疑: {active if active else '无命中'}")
|
||
for k, s in junk_samples.items():
|
||
print(f" [{k}] …{s}…")
|
||
heads = [f"#{r[0]}{r[1][:14]}" for r in rows[:sample_titles]]
|
||
tails = [f"#{r[0]}{r[1][:14]}" for r in rows[-3:]]
|
||
print(f" 章题样例: {' | '.join(heads)} … {' | '.join(tails)}")
|
||
print()
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
main()
|
||
except psycopg.Error as e:
|
||
click.echo(f"[db错误] {type(e).__name__}: {e}", err=True)
|
||
sys.exit(1)
|