339 lines
17 KiB
Python
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""import skill:参考书/旧稿 txt 静态分章导入(B1/C8 共用,LLM 不参与)。
分章与修复规则、落库映射见同 skill 的 SKILL.md;写入约定见 db/表映射.md。
失败原样抛错不静默(公约)。
"""
import hashlib
import html
import json
import pathlib
import re
import sys
import click
import psycopg
from psycopg.types.json import Jsonb
DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse-example"
"?keepalives=1&keepalives_idle=15&keepalives_interval=5&keepalives_count=3")
# keepalive 防 Tailscale 半死连接(2026-07-13 实测:逐行插入两万次往返曾卡死 16 分钟)
TENANT, ACTOR, OWNER = 1, "1", 1 # 实验写入约定:系统主账号
CN_DIGITS = {"零": 0, "〇": 0, "一": 1, "二": 2, "两": 2, "三": 3, "四": 4,
"五": 5, "六": 6, "七": 7, "八": 8, "九": 9}
CN_UNITS = {"十": 10, "百": 100, "千": 1000}
# 章题格式(named group: vol=卷号, no=章号, title=题名)
PAT_VOL_CH = re.compile( # 第X卷 第Y章 标题(含无空格变体)——恒启用
r'^\s*第\s*(?P<vol>[零〇一二两三四五六七八九十百千0-9]+)\s*卷\s*'
r'第\s*(?P<no>[零〇一二两三四五六七八九十百千0-9]+)\s*[章回]\s*(?P<title>\S.*)?$')
PAT_CH = re.compile( # 第X章 标题——恒启用
r'^\s*第\s*(?P<no>[零〇一二两三四五六七八九十百千0-9]+)\s*[章回]\s*(?P<title>\S.*)?$')
PAT_VOL_ONLY = re.compile( # 纯卷行(有卷无章)——不开章
r'^\s*第\s*[零〇一二两三四五六七八九十百千0-9]+\s*卷\s*(?P<title>\S.*)?$')
PAT_NUM = re.compile(r'^\s*(?P<no>\d{3,4})\s+(?P<title>\S.*)$') # 001 标题——自适应
PAT_CN_BARE = re.compile( # 一百零二 标题——自适应(限长防误切正文)
r'^\s*(?P<no>[零〇一二两三四五六七八九十百千]{1,8})[  ]+(?P<title>\S.{0,28})$')
def cn2int(s: str):
"""中文数字→整数;混写/错写返回 None(调用方 fallback 前章+1)。"""
s = s.strip()
if s.isdigit():
return int(s)
total, section, num = 0, 0, 0
for ch in s:
if ch in CN_DIGITS:
num = CN_DIGITS[ch]
elif ch in CN_UNITS:
u = CN_UNITS[ch]
section += (num or 1) * u
num = 0
else:
return None
total = section + num
return total or None
def detect_adaptive(lines):
"""第一遍扫描:统计自适应格式命中数,≥50 次才启用(防普通正文误切)。"""
n_num = sum(1 for ln in lines if PAT_NUM.match(ln))
n_cn = sum(1 for ln in lines if PAT_CN_BARE.match(ln) and not PAT_CH.match(ln) and not PAT_VOL_ONLY.match(ln))
return n_num >= 50, n_cn >= 50
def match_title(line, use_num, use_cn):
"""按启用格式集识别章题行;返回 (卷号, 章号, 题名) 或 None。"""
m = PAT_VOL_CH.match(line)
if m:
return cn2int(m["vol"]), cn2int(m["no"]), (m["title"] or "").strip()
m = PAT_CH.match(line)
if m:
return None, cn2int(m["no"]), (m["title"] or "").strip()
if PAT_VOL_ONLY.match(line):
return "VOL_ONLY", None, None # 纯卷行标记
if use_num:
m = PAT_NUM.match(line)
if m:
return None, int(m["no"]), m["title"].strip()
if use_cn:
m = PAT_CN_BARE.match(line)
if m:
no = cn2int(m["no"])
if no is not None: # 数字非法则不认为是章题(正文行)
return None, no, m["title"].strip()
return None
def parse_book(path: pathlib.Path):
"""解析一本书 → (meta, chapters, stats)。修复规则见 SKILL.md。"""
raw = path.read_text(encoding="utf-8", errors="replace")
raw = html.unescape(raw) # 解 &ldquo; 等实体
lines = raw.split("\n")
# 头部 meta:# 《书名》 / # 作者: X / # 共 N 章
title = author = None
declared = None
for ln in lines[:8]:
m = re.match(r'^#\s*《(.+?)》', ln)
if m:
title = m.group(1)
m = re.match(r'^#?\s*书名[::]\s*(\S+)', ln)
if m:
title = title or m.group(1)
m = re.match(r'^#?\s*作者[::]\s*(\S+)', ln)
if m:
author = author or m.group(1)
m = re.match(r'^#\s*共\s*(\d+)\s*章', ln)
if m:
declared = int(m.group(1))
if not title or not author: # 文件名兜底:书名_作者.txt 或 书名(作者).txt
stem = path.stem
m = re.match(r'^(.+?)[((](.+?)[))]\s*$', stem)
if m:
title = title or m.group(1).strip()
author = author or m.group(2).strip()
else:
title = title or stem.split("_")[0]
author = author or (stem.split("_")[-1] if "_" in stem else None)
use_num, use_cn = detect_adaptive(lines)
# 第二遍:切章
chapters = [] # [{seq,vol,no,title,lines:[...]}]
seen_titles = {} # normalized 全题 → 首现章 idx(同题重现不开新章)
cur = None
stats = {"丢弃重复题行": 0, "纯卷行": 0, "章号解析失败": 0, "目录空章弹出": 0}
last_no = 0
for ln in lines:
hit = match_title(ln, use_num, use_cn)
if hit:
vol, no, t = hit
if vol == "VOL_ONLY":
stats["纯卷行"] += 1
continue
# 目录残留修复:上一题行至此无任何正文 → 那是目录行,弹出空章并注销其题名,
# 让后文的真章(同题)能正常开章(否则真章被判重、边界丢失)
if cur is not None and not any(l.strip() for l in cur["lines"]):
seen_titles.pop(cur["_norm"], None)
chapters.pop()
stats["目录空章弹出"] += 1
stats.setdefault("空题名样例", []).append(cur["title"][:20])
if len(stats["空题名样例"]) > 8:
stats["空题名样例"] = stats["空题名样例"][:8] + ["…"]
norm = re.sub(r'\s+', '', ln.strip())
if norm in seen_titles:
stats["丢弃重复题行"] += 1 # 盗版重贴:丢标题行,正文归当前章
continue
if no is None:
no = last_no + 1
stats["章号解析失败"] += 1
seen_titles[norm] = len(chapters)
cur = {"seq": len(chapters) + 1, "vol": vol, "no": no,
"title": t or f"第{no}章", "lines": [], "_norm": norm}
chapters.append(cur)
last_no = no
continue
if cur is not None:
cur["lines"].append(ln)
# 尾部残留截断(证据制,防误伤):
# 候选=尾部连续「章号 < 鲁棒峰值(90分位,防错打大号污染)一半」的段,且长度≤全书20%;
# 护栏1:段内末两章带完结标记(终章/全书完/大结局/(终)/(完))→ 是末卷重新计号的正文,保留;
# 护栏2:段内题文与前文章题文匹配率≥50% → 判为早期章节重贴残留,截断;否则保守保留。
if len(chapters) > 20:
nos = sorted(c["no"] for c in chapters)
robust_peak = nos[int(len(nos) * 0.9)]
run_start = None
for i in range(len(chapters) - 1, -1, -1):
if chapters[i]["no"] < robust_peak * 0.5:
run_start = i
else:
break
if run_start is not None and (len(chapters) - run_start) <= len(chapters) * 0.2:
run = chapters[run_start:]
end_marker = re.compile(r'终章|全书完|大结局|(终)|\(终\)|(完)|\(完\)|完本')
if any(end_marker.search(c["title"]) for c in run[-2:]):
stats["尾部低号段保留(带完结标记)"] = len(run)
else:
def tnorm(s):
return re.sub(r'[\s,。!?—…·、,.!?()()]+', '', s)
earlier = {tnorm(c["title"]) for c in chapters[:run_start]}
hits = sum(1 for c in run if tnorm(c["title"]) in earlier)
if hits >= len(run) * 0.5:
stats["尾部截断章"] = len(run)
stats["尾部截断起"] = run[0]["title"]
chapters = chapters[:run_start]
else:
stats["尾部低号段保留(题文不重)"] = len(run)
for c in chapters:
c["text"] = "\n".join(c["lines"]).strip()
del c["lines"]
c.pop("_norm", None)
dropped = [c["title"][:24] for c in chapters if not c["text"]]
if dropped: # 源文件孤题无正文(常见:盗版尾部只剩末章标题)——诚实报告
stats["无正文题名丢弃"] = dropped[:6] + (["…"] if len(dropped) > 6 else [])
chapters = [c for c in chapters if c["text"]]
for i, c in enumerate(chapters, 1):
c["seq"] = i
meta = {"title": title, "author": author, "declared": declared,
"file": path.name, "chars": len(raw),
"自适应格式": {"裸阿拉伯": use_num, "裸中文数字": use_cn}}
return meta, chapters, stats
def report(meta, chapters, stats):
"""对账表(审查面)。"""
print(f"《{meta['title']}》 作者:{meta['author'] or '?'} 源:{meta['file']}")
print(f" 声明章数:{meta['declared'] or '无'} 导入章数:{len(chapters)} 总字符:{meta['chars']:,}")
print(f" 自适应格式:{meta['自适应格式']} 修复统计:{stats}")
# 章号 vs 顺序号偏差(重号/跳号计数,纯对账不修正)
mismatch = sum(1 for c in chapters if c["no"] != c["seq"])
print(f" 章号≠顺序号: {mismatch} 章(断更补号/重号常见,仅供参考)")
for tag, c in [("首", chapters[0]), ("中", chapters[len(chapters) // 2]), ("末", chapters[-1])]:
first_line = next((l for l in c["text"].split("\n") if l.strip()), "")[:40]
print(f" [{tag}] #{c['seq']} 《{c['title'][:30]}》 {len(c['text'])}字 | {first_line}…")
def ensure_kbs(conn):
"""幂等 ensure 两个知识库行:私有参考书库 + 公共范式库。返回 (私有id, 公共id)。"""
ids = {}
for name, ktype, desc in [("参考书私有库", "user", "参考书全本原文(仅供拆书,不对作品侧开放)"),
("公共范式库", "global", "拆书产出的脱敏范式(管理员确认后可绑定)")]:
row = conn.execute(
"SELECT id FROM muse_knowledge_base WHERE tenant_id=%s AND name=%s AND deleted=FALSE",
(TENANT, name)).fetchone()
if row:
ids[name] = row[0]
else:
ids[name] = conn.execute(
"""INSERT INTO muse_knowledge_base (name, description, kb_type, owner_user_id, status,
creator, updater, tenant_id)
VALUES (%s,%s,%s,%s,'active',%s,%s,%s) RETURNING id""",
(name, desc, ktype, OWNER, ACTOR, ACTOR, TENANT)).fetchone()[0]
return ids["参考书私有库"], ids["公共范式库"]
def import_book(path: pathlib.Path, force: bool):
meta, chapters, stats = parse_book(path)
report(meta, chapters, stats)
if not chapters:
raise click.ClickException("解析出 0 章,拒绝入库")
file_hash = hashlib.sha256(path.read_bytes()).hexdigest()
command_id = f"import-{file_hash[:16]}"
total_words = sum(len(re.sub(r'\s', '', c['text'])) for c in chapters)
with psycopg.connect(DSN) as conn:
exist = conn.execute(
"SELECT id FROM muse_content_work WHERE tenant_id=%s AND title=%s AND deleted=FALSE",
(TENANT, meta["title"])).fetchone()
if exist and not force:
raise click.ClickException(f"作品《{meta['title']}》已存在(id={exist[0]}),重导请加 --force")
if exist and force: # 软删旧行(work/chapter/block/档案),审计可溯
wid = exist[0]
conn.execute("UPDATE muse_content_work SET deleted=TRUE, updater=%s WHERE id=%s", (ACTOR, wid))
conn.execute("UPDATE muse_content_chapter SET deleted=TRUE, updater=%s WHERE tenant_id=%s AND work_id=%s", (ACTOR, TENANT, wid))
conn.execute("UPDATE muse_content_block SET deleted=TRUE, updater=%s WHERE tenant_id=%s AND work_id=%s", (ACTOR, TENANT, wid))
conn.execute("UPDATE example_reference_work SET deleted=TRUE, updater=%s WHERE tenant_id=%s AND work_id=%s", (ACTOR, TENANT, wid))
print(f" --force: 旧作品 id={wid} 及章/块/档案已软删")
kb_private, _kb_public = ensure_kbs(conn)
work_id = conn.execute(
"""INSERT INTO muse_content_work (owner_user_id, title, description, genre, status,
import_status, parse_status, word_count, chapter_count, creator, updater, tenant_id)
VALUES (%s,%s,%s,'科幻','completed','imported','pending',%s,%s,%s,%s,%s) RETURNING id""",
(OWNER, meta["title"], f"参考书导入(拆书用);作者:{meta['author'] or '?'}",
total_words, len(chapters), ACTOR, ACTOR, TENANT)).fetchone()[0]
# 批量两阶段(executemany 走 pipeline,一书仅数次网络往返;此前逐行两万往返曾被半死连接卡死)
with conn.cursor() as cur:
cur.executemany(
"""INSERT INTO muse_content_chapter (work_id, title, order_no, status, outline_snapshot,
creator, updater, tenant_id)
VALUES (%s,%s,%s,'published',%s,%s,%s,%s)""",
[(work_id, c["title"][:200], c["seq"],
Jsonb({"解析章号": c["no"], "卷号": c["vol"]}), ACTOR, ACTOR, TENANT) for c in chapters])
id_map = dict(cur.execute(
"SELECT order_no, id FROM muse_content_chapter WHERE tenant_id=%s AND work_id=%s AND deleted=FALSE",
(TENANT, work_id)).fetchall())
cur.executemany(
"""INSERT INTO muse_content_block (work_id, chapter_id, order_no, block_type, title,
content_text, word_count, creator, updater, tenant_id)
VALUES (%s,%s,1,'scene',%s,%s,%s,%s,%s,%s)""",
[(work_id, id_map[c["seq"]], c["title"][:500], c["text"],
len(re.sub(r'\s', '', c["text"])), ACTOR, ACTOR, TENANT) for c in chapters])
conn.execute(
"""INSERT INTO muse_knowledge_document (kb_id, title, file_name, file_size, mime_type, file_hash,
storage_ref, scan_status, parse_status, author, creator, updater, tenant_id)
VALUES (%s,%s,%s,%s,'text/plain',%s,%s,'completed','pending',%s,%s,%s,%s)""",
(kb_private, meta["title"], meta["file"], path.stat().st_size, file_hash,
str(path), meta["author"], ACTOR, ACTOR, TENANT))
conn.execute(
"""INSERT INTO example_reference_work (work_id, author, source_file, declared_chapter_count,
imported_chapter_count, char_count, parse_status, notes, creator, updater, tenant_id)
VALUES (%s,%s,%s,%s,%s,%s,'pending',%s,%s,%s,%s)""",
(work_id, meta["author"], meta["file"], meta["declared"], len(chapters), meta["chars"],
json.dumps({"修复统计": stats, "自适应格式": meta["自适应格式"]}, ensure_ascii=False),
ACTOR, ACTOR, TENANT))
snapshot = {"file": meta["file"], "declared": meta["declared"], "imported": len(chapters),
"chars": meta["chars"], "words": total_words, "修复统计": stats}
conn.execute(
"""INSERT INTO muse_content_import_task (work_id, owner_user_id, source_type, source_snapshot,
status, command_id, creator, updater, tenant_id)
VALUES (%s,%s,'txt',%s,'succeeded',%s,%s,%s,%s)
ON CONFLICT (tenant_id, command_id)
DO UPDATE SET work_id=EXCLUDED.work_id, source_snapshot=EXCLUDED.source_snapshot, updater=EXCLUDED.updater""",
(work_id, OWNER, Jsonb(snapshot), command_id, ACTOR, ACTOR, TENANT))
conn.commit()
print(f" ✅ 已入库 work_id={work_id}(章 {len(chapters)}、块 {len(chapters)}、档案 1、import_task {command_id})\n")
@click.command()
@click.argument("files", nargs=-1, required=True, type=click.Path(exists=True, path_type=pathlib.Path))
@click.option("--dry-run", is_flag=True, help="只解析打印对账,不落库")
@click.option("--force", is_flag=True, help="同名作品已存在时软删旧行重导")
def main(files, dry_run, force):
"""参考书/旧稿 txt 静态分章导入。"""
for p in files:
if dry_run:
meta, chapters, stats = parse_book(p)
report(meta, chapters, stats)
print()
else:
import_book(p, force)
if __name__ == "__main__":
try:
main()
except psycopg.Error as e:
click.echo(f"[db错误] {type(e).__name__}: {e}", err=True)
sys.exit(1)