339 lines
17 KiB
Python
339 lines
17 KiB
Python
#!/usr/bin/env python3
|
||
"""import skill:参考书/旧稿 txt 静态分章导入(B1/C8 共用,LLM 不参与)。
|
||
|
||
分章与修复规则、落库映射见同 skill 的 SKILL.md;写入约定见 db/表映射.md。
|
||
失败原样抛错不静默(公约)。
|
||
"""
|
||
import hashlib
|
||
import html
|
||
import json
|
||
import pathlib
|
||
import re
|
||
import sys
|
||
|
||
import click
|
||
import psycopg
|
||
from psycopg.types.json import Jsonb
|
||
|
||
DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse-example"
|
||
"?keepalives=1&keepalives_idle=15&keepalives_interval=5&keepalives_count=3")
|
||
# keepalive 防 Tailscale 半死连接(2026-07-13 实测:逐行插入两万次往返曾卡死 16 分钟)
|
||
TENANT, ACTOR, OWNER = 1, "1", 1 # 实验写入约定:系统主账号
|
||
|
||
CN_DIGITS = {"零": 0, "〇": 0, "一": 1, "二": 2, "两": 2, "三": 3, "四": 4,
|
||
"五": 5, "六": 6, "七": 7, "八": 8, "九": 9}
|
||
CN_UNITS = {"十": 10, "百": 100, "千": 1000}
|
||
|
||
# 章题格式(named group: vol=卷号, no=章号, title=题名)
|
||
PAT_VOL_CH = re.compile( # 第X卷 第Y章 标题(含无空格变体)——恒启用
|
||
r'^\s*第\s*(?P<vol>[零〇一二两三四五六七八九十百千0-9]+)\s*卷\s*'
|
||
r'第\s*(?P<no>[零〇一二两三四五六七八九十百千0-9]+)\s*[章回]\s*(?P<title>\S.*)?$')
|
||
PAT_CH = re.compile( # 第X章 标题——恒启用
|
||
r'^\s*第\s*(?P<no>[零〇一二两三四五六七八九十百千0-9]+)\s*[章回]\s*(?P<title>\S.*)?$')
|
||
PAT_VOL_ONLY = re.compile( # 纯卷行(有卷无章)——不开章
|
||
r'^\s*第\s*[零〇一二两三四五六七八九十百千0-9]+\s*卷\s*(?P<title>\S.*)?$')
|
||
PAT_NUM = re.compile(r'^\s*(?P<no>\d{3,4})\s+(?P<title>\S.*)$') # 001 标题——自适应
|
||
PAT_CN_BARE = re.compile( # 一百零二 标题——自适应(限长防误切正文)
|
||
r'^\s*(?P<no>[零〇一二两三四五六七八九十百千]{1,8})[ ]+(?P<title>\S.{0,28})$')
|
||
|
||
|
||
def cn2int(s: str):
|
||
"""中文数字→整数;混写/错写返回 None(调用方 fallback 前章+1)。"""
|
||
s = s.strip()
|
||
if s.isdigit():
|
||
return int(s)
|
||
total, section, num = 0, 0, 0
|
||
for ch in s:
|
||
if ch in CN_DIGITS:
|
||
num = CN_DIGITS[ch]
|
||
elif ch in CN_UNITS:
|
||
u = CN_UNITS[ch]
|
||
section += (num or 1) * u
|
||
num = 0
|
||
else:
|
||
return None
|
||
total = section + num
|
||
return total or None
|
||
|
||
|
||
def detect_adaptive(lines):
|
||
"""第一遍扫描:统计自适应格式命中数,≥50 次才启用(防普通正文误切)。"""
|
||
n_num = sum(1 for ln in lines if PAT_NUM.match(ln))
|
||
n_cn = sum(1 for ln in lines if PAT_CN_BARE.match(ln) and not PAT_CH.match(ln) and not PAT_VOL_ONLY.match(ln))
|
||
return n_num >= 50, n_cn >= 50
|
||
|
||
|
||
def match_title(line, use_num, use_cn):
|
||
"""按启用格式集识别章题行;返回 (卷号, 章号, 题名) 或 None。"""
|
||
m = PAT_VOL_CH.match(line)
|
||
if m:
|
||
return cn2int(m["vol"]), cn2int(m["no"]), (m["title"] or "").strip()
|
||
m = PAT_CH.match(line)
|
||
if m:
|
||
return None, cn2int(m["no"]), (m["title"] or "").strip()
|
||
if PAT_VOL_ONLY.match(line):
|
||
return "VOL_ONLY", None, None # 纯卷行标记
|
||
if use_num:
|
||
m = PAT_NUM.match(line)
|
||
if m:
|
||
return None, int(m["no"]), m["title"].strip()
|
||
if use_cn:
|
||
m = PAT_CN_BARE.match(line)
|
||
if m:
|
||
no = cn2int(m["no"])
|
||
if no is not None: # 数字非法则不认为是章题(正文行)
|
||
return None, no, m["title"].strip()
|
||
return None
|
||
|
||
|
||
def parse_book(path: pathlib.Path):
|
||
"""解析一本书 → (meta, chapters, stats)。修复规则见 SKILL.md。"""
|
||
raw = path.read_text(encoding="utf-8", errors="replace")
|
||
raw = html.unescape(raw) # 解 “ 等实体
|
||
lines = raw.split("\n")
|
||
|
||
# 头部 meta:# 《书名》 / # 作者: X / # 共 N 章
|
||
title = author = None
|
||
declared = None
|
||
for ln in lines[:8]:
|
||
m = re.match(r'^#\s*《(.+?)》', ln)
|
||
if m:
|
||
title = m.group(1)
|
||
m = re.match(r'^#?\s*书名[::]\s*(\S+)', ln)
|
||
if m:
|
||
title = title or m.group(1)
|
||
m = re.match(r'^#?\s*作者[::]\s*(\S+)', ln)
|
||
if m:
|
||
author = author or m.group(1)
|
||
m = re.match(r'^#\s*共\s*(\d+)\s*章', ln)
|
||
if m:
|
||
declared = int(m.group(1))
|
||
if not title or not author: # 文件名兜底:书名_作者.txt 或 书名(作者).txt
|
||
stem = path.stem
|
||
m = re.match(r'^(.+?)[((](.+?)[))]\s*$', stem)
|
||
if m:
|
||
title = title or m.group(1).strip()
|
||
author = author or m.group(2).strip()
|
||
else:
|
||
title = title or stem.split("_")[0]
|
||
author = author or (stem.split("_")[-1] if "_" in stem else None)
|
||
|
||
use_num, use_cn = detect_adaptive(lines)
|
||
|
||
# 第二遍:切章
|
||
chapters = [] # [{seq,vol,no,title,lines:[...]}]
|
||
seen_titles = {} # normalized 全题 → 首现章 idx(同题重现不开新章)
|
||
cur = None
|
||
stats = {"丢弃重复题行": 0, "纯卷行": 0, "章号解析失败": 0, "目录空章弹出": 0}
|
||
last_no = 0
|
||
for ln in lines:
|
||
hit = match_title(ln, use_num, use_cn)
|
||
if hit:
|
||
vol, no, t = hit
|
||
if vol == "VOL_ONLY":
|
||
stats["纯卷行"] += 1
|
||
continue
|
||
# 目录残留修复:上一题行至此无任何正文 → 那是目录行,弹出空章并注销其题名,
|
||
# 让后文的真章(同题)能正常开章(否则真章被判重、边界丢失)
|
||
if cur is not None and not any(l.strip() for l in cur["lines"]):
|
||
seen_titles.pop(cur["_norm"], None)
|
||
chapters.pop()
|
||
stats["目录空章弹出"] += 1
|
||
stats.setdefault("空题名样例", []).append(cur["title"][:20])
|
||
if len(stats["空题名样例"]) > 8:
|
||
stats["空题名样例"] = stats["空题名样例"][:8] + ["…"]
|
||
norm = re.sub(r'\s+', '', ln.strip())
|
||
if norm in seen_titles:
|
||
stats["丢弃重复题行"] += 1 # 盗版重贴:丢标题行,正文归当前章
|
||
continue
|
||
if no is None:
|
||
no = last_no + 1
|
||
stats["章号解析失败"] += 1
|
||
seen_titles[norm] = len(chapters)
|
||
cur = {"seq": len(chapters) + 1, "vol": vol, "no": no,
|
||
"title": t or f"第{no}章", "lines": [], "_norm": norm}
|
||
chapters.append(cur)
|
||
last_no = no
|
||
continue
|
||
if cur is not None:
|
||
cur["lines"].append(ln)
|
||
|
||
# 尾部残留截断(证据制,防误伤):
|
||
# 候选=尾部连续「章号 < 鲁棒峰值(90分位,防错打大号污染)一半」的段,且长度≤全书20%;
|
||
# 护栏1:段内末两章带完结标记(终章/全书完/大结局/(终)/(完))→ 是末卷重新计号的正文,保留;
|
||
# 护栏2:段内题文与前文章题文匹配率≥50% → 判为早期章节重贴残留,截断;否则保守保留。
|
||
if len(chapters) > 20:
|
||
nos = sorted(c["no"] for c in chapters)
|
||
robust_peak = nos[int(len(nos) * 0.9)]
|
||
run_start = None
|
||
for i in range(len(chapters) - 1, -1, -1):
|
||
if chapters[i]["no"] < robust_peak * 0.5:
|
||
run_start = i
|
||
else:
|
||
break
|
||
if run_start is not None and (len(chapters) - run_start) <= len(chapters) * 0.2:
|
||
run = chapters[run_start:]
|
||
end_marker = re.compile(r'终章|全书完|大结局|(终)|\(终\)|(完)|\(完\)|完本')
|
||
if any(end_marker.search(c["title"]) for c in run[-2:]):
|
||
stats["尾部低号段保留(带完结标记)"] = len(run)
|
||
else:
|
||
def tnorm(s):
|
||
return re.sub(r'[\s,。!?—…·、,.!?()()]+', '', s)
|
||
earlier = {tnorm(c["title"]) for c in chapters[:run_start]}
|
||
hits = sum(1 for c in run if tnorm(c["title"]) in earlier)
|
||
if hits >= len(run) * 0.5:
|
||
stats["尾部截断章"] = len(run)
|
||
stats["尾部截断起"] = run[0]["title"]
|
||
chapters = chapters[:run_start]
|
||
else:
|
||
stats["尾部低号段保留(题文不重)"] = len(run)
|
||
|
||
for c in chapters:
|
||
c["text"] = "\n".join(c["lines"]).strip()
|
||
del c["lines"]
|
||
c.pop("_norm", None)
|
||
dropped = [c["title"][:24] for c in chapters if not c["text"]]
|
||
if dropped: # 源文件孤题无正文(常见:盗版尾部只剩末章标题)——诚实报告
|
||
stats["无正文题名丢弃"] = dropped[:6] + (["…"] if len(dropped) > 6 else [])
|
||
chapters = [c for c in chapters if c["text"]]
|
||
for i, c in enumerate(chapters, 1):
|
||
c["seq"] = i
|
||
|
||
meta = {"title": title, "author": author, "declared": declared,
|
||
"file": path.name, "chars": len(raw),
|
||
"自适应格式": {"裸阿拉伯": use_num, "裸中文数字": use_cn}}
|
||
return meta, chapters, stats
|
||
|
||
|
||
def report(meta, chapters, stats):
|
||
"""对账表(审查面)。"""
|
||
print(f"《{meta['title']}》 作者:{meta['author'] or '?'} 源:{meta['file']}")
|
||
print(f" 声明章数:{meta['declared'] or '无'} 导入章数:{len(chapters)} 总字符:{meta['chars']:,}")
|
||
print(f" 自适应格式:{meta['自适应格式']} 修复统计:{stats}")
|
||
# 章号 vs 顺序号偏差(重号/跳号计数,纯对账不修正)
|
||
mismatch = sum(1 for c in chapters if c["no"] != c["seq"])
|
||
print(f" 章号≠顺序号: {mismatch} 章(断更补号/重号常见,仅供参考)")
|
||
for tag, c in [("首", chapters[0]), ("中", chapters[len(chapters) // 2]), ("末", chapters[-1])]:
|
||
first_line = next((l for l in c["text"].split("\n") if l.strip()), "")[:40]
|
||
print(f" [{tag}] #{c['seq']} 《{c['title'][:30]}》 {len(c['text'])}字 | {first_line}…")
|
||
|
||
|
||
def ensure_kbs(conn):
|
||
"""幂等 ensure 两个知识库行:私有参考书库 + 公共范式库。返回 (私有id, 公共id)。"""
|
||
ids = {}
|
||
for name, ktype, desc in [("参考书私有库", "user", "参考书全本原文(仅供拆书,不对作品侧开放)"),
|
||
("公共范式库", "global", "拆书产出的脱敏范式(管理员确认后可绑定)")]:
|
||
row = conn.execute(
|
||
"SELECT id FROM muse_knowledge_base WHERE tenant_id=%s AND name=%s AND deleted=FALSE",
|
||
(TENANT, name)).fetchone()
|
||
if row:
|
||
ids[name] = row[0]
|
||
else:
|
||
ids[name] = conn.execute(
|
||
"""INSERT INTO muse_knowledge_base (name, description, kb_type, owner_user_id, status,
|
||
creator, updater, tenant_id)
|
||
VALUES (%s,%s,%s,%s,'active',%s,%s,%s) RETURNING id""",
|
||
(name, desc, ktype, OWNER, ACTOR, ACTOR, TENANT)).fetchone()[0]
|
||
return ids["参考书私有库"], ids["公共范式库"]
|
||
|
||
|
||
def import_book(path: pathlib.Path, force: bool):
|
||
meta, chapters, stats = parse_book(path)
|
||
report(meta, chapters, stats)
|
||
if not chapters:
|
||
raise click.ClickException("解析出 0 章,拒绝入库")
|
||
file_hash = hashlib.sha256(path.read_bytes()).hexdigest()
|
||
command_id = f"import-{file_hash[:16]}"
|
||
total_words = sum(len(re.sub(r'\s', '', c['text'])) for c in chapters)
|
||
|
||
with psycopg.connect(DSN) as conn:
|
||
exist = conn.execute(
|
||
"SELECT id FROM muse_content_work WHERE tenant_id=%s AND title=%s AND deleted=FALSE",
|
||
(TENANT, meta["title"])).fetchone()
|
||
if exist and not force:
|
||
raise click.ClickException(f"作品《{meta['title']}》已存在(id={exist[0]}),重导请加 --force")
|
||
if exist and force: # 软删旧行(work/chapter/block/档案),审计可溯
|
||
wid = exist[0]
|
||
conn.execute("UPDATE muse_content_work SET deleted=TRUE, updater=%s WHERE id=%s", (ACTOR, wid))
|
||
conn.execute("UPDATE muse_content_chapter SET deleted=TRUE, updater=%s WHERE tenant_id=%s AND work_id=%s", (ACTOR, TENANT, wid))
|
||
conn.execute("UPDATE muse_content_block SET deleted=TRUE, updater=%s WHERE tenant_id=%s AND work_id=%s", (ACTOR, TENANT, wid))
|
||
conn.execute("UPDATE example_reference_work SET deleted=TRUE, updater=%s WHERE tenant_id=%s AND work_id=%s", (ACTOR, TENANT, wid))
|
||
print(f" --force: 旧作品 id={wid} 及章/块/档案已软删")
|
||
|
||
kb_private, _kb_public = ensure_kbs(conn)
|
||
|
||
work_id = conn.execute(
|
||
"""INSERT INTO muse_content_work (owner_user_id, title, description, genre, status,
|
||
import_status, parse_status, word_count, chapter_count, creator, updater, tenant_id)
|
||
VALUES (%s,%s,%s,'科幻','completed','imported','pending',%s,%s,%s,%s,%s) RETURNING id""",
|
||
(OWNER, meta["title"], f"参考书导入(拆书用);作者:{meta['author'] or '?'}",
|
||
total_words, len(chapters), ACTOR, ACTOR, TENANT)).fetchone()[0]
|
||
|
||
# 批量两阶段(executemany 走 pipeline,一书仅数次网络往返;此前逐行两万往返曾被半死连接卡死)
|
||
with conn.cursor() as cur:
|
||
cur.executemany(
|
||
"""INSERT INTO muse_content_chapter (work_id, title, order_no, status, outline_snapshot,
|
||
creator, updater, tenant_id)
|
||
VALUES (%s,%s,%s,'published',%s,%s,%s,%s)""",
|
||
[(work_id, c["title"][:200], c["seq"],
|
||
Jsonb({"解析章号": c["no"], "卷号": c["vol"]}), ACTOR, ACTOR, TENANT) for c in chapters])
|
||
id_map = dict(cur.execute(
|
||
"SELECT order_no, id FROM muse_content_chapter WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE",
|
||
(TENANT, work_id)).fetchall())
|
||
cur.executemany(
|
||
"""INSERT INTO muse_content_block (work_id, chapter_id, order_no, block_type, title,
|
||
content_text, word_count, creator, updater, tenant_id)
|
||
VALUES (%s,%s,1,'scene',%s,%s,%s,%s,%s,%s)""",
|
||
[(work_id, id_map[c["seq"]], c["title"][:500], c["text"],
|
||
len(re.sub(r'\s', '', c["text"])), ACTOR, ACTOR, TENANT) for c in chapters])
|
||
|
||
conn.execute(
|
||
"""INSERT INTO muse_knowledge_document (kb_id, title, file_name, file_size, mime_type, file_hash,
|
||
storage_ref, scan_status, parse_status, author, creator, updater, tenant_id)
|
||
VALUES (%s,%s,%s,%s,'text/plain',%s,%s,'completed','pending',%s,%s,%s,%s)""",
|
||
(kb_private, meta["title"], meta["file"], path.stat().st_size, file_hash,
|
||
str(path), meta["author"], ACTOR, ACTOR, TENANT))
|
||
|
||
conn.execute(
|
||
"""INSERT INTO example_reference_work (work_id, author, source_file, declared_chapter_count,
|
||
imported_chapter_count, char_count, parse_status, notes, creator, updater, tenant_id)
|
||
VALUES (%s,%s,%s,%s,%s,%s,'pending',%s,%s,%s,%s)""",
|
||
(work_id, meta["author"], meta["file"], meta["declared"], len(chapters), meta["chars"],
|
||
json.dumps({"修复统计": stats, "自适应格式": meta["自适应格式"]}, ensure_ascii=False),
|
||
ACTOR, ACTOR, TENANT))
|
||
|
||
snapshot = {"file": meta["file"], "declared": meta["declared"], "imported": len(chapters),
|
||
"chars": meta["chars"], "words": total_words, "修复统计": stats}
|
||
conn.execute(
|
||
"""INSERT INTO muse_content_import_task (work_id, owner_user_id, source_type, source_snapshot,
|
||
status, command_id, creator, updater, tenant_id)
|
||
VALUES (%s,%s,'txt',%s,'succeeded',%s,%s,%s,%s)
|
||
ON CONFLICT (tenant_id, command_id)
|
||
DO UPDATE SET work_id=EXCLUDED.work_id, source_snapshot=EXCLUDED.source_snapshot, updater=EXCLUDED.updater""",
|
||
(work_id, OWNER, Jsonb(snapshot), command_id, ACTOR, ACTOR, TENANT))
|
||
conn.commit()
|
||
print(f" ✅ 已入库 work_id={work_id}(章 {len(chapters)}、块 {len(chapters)}、档案 1、import_task {command_id})\n")
|
||
|
||
|
||
@click.command()
|
||
@click.argument("files", nargs=-1, required=True, type=click.Path(exists=True, path_type=pathlib.Path))
|
||
@click.option("--dry-run", is_flag=True, help="只解析打印对账,不落库")
|
||
@click.option("--force", is_flag=True, help="同名作品已存在时软删旧行重导")
|
||
def main(files, dry_run, force):
|
||
"""参考书/旧稿 txt 静态分章导入。"""
|
||
for p in files:
|
||
if dry_run:
|
||
meta, chapters, stats = parse_book(p)
|
||
report(meta, chapters, stats)
|
||
print()
|
||
else:
|
||
import_book(p, force)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
main()
|
||
except psycopg.Error as e:
|
||
click.echo(f"[db错误] {type(e).__name__}: {e}", err=True)
|
||
sys.exit(1)
|