#!/usr/bin/env python3 """存量升格卡 [窗N] → 真实章号迁移(升格卡改造 §6.1)——**只 --dry-run,全程零写库**。 背景:老升格卡的演变类字段(character.成长弧线 / character_relation.演变轨迹)里,条目带 运行时窗号前缀 [窗N]/[本窗]。窗号是切窗规则的运行产物(会变、脱离环境不可读),必须换成 永不变的真实章号(第X章)。本脚本把这些条目: - 实体型(character 等):对象化成里程碑 {章,台阶,周期} 汇入「演变历程」;成长弧线回归"未来计划"。 - 关系型(character_relation):保持「演变轨迹」字符串结构,只把 [窗N] 换真实章号前缀。 真实章号定位优先级(落实拍板#5「人工精确化」,不接受 12 章宽近似): 1) 内嵌章号——条目文字里模型已写"第420章/119章/Ch.21",正则直接抽,最准、零 LLM。 2) 前缀误标章号——[窗307-315] 里 307 远超该书窗号上限(work4=80/work8=116),是"第307章" 被误加了"窗"字,采信为章号、零 LLM。 3) 真窗号 [窗N](N≤上限)——读该窗源文、LLM(走 chat_governed 全局治理)定位到真实章。 4) [本窗]/无任何线索——定不了,打"待人工"标记,绝不用窗区间近似糊弄。 范围边界(诚实标注):本脚本只迁**演变类字段**(成长弧线/演变轨迹/演变历程/大事记/经历)。 item/power_system 等把升级线塞在当前态字段(流转计划/戏剧作用/跨体系换算…)的病根2,属于 "当前态字段污染",需 P0 提示词改造后**重抽**清理,非本机械迁移能正确归位——脚本对这类字段 只做「污染报告」提示、不动数据。 连接纪律(硬约束):短连接读→释放→LLM→(本脚本不写库)。绝不在 DB 事务里夹 LLM 调用。 本脚本无任何 UPDATE/INSERT muse_knowledge_draft——迁移落库由主代理 gate 后另行发起。 """ import json import pathlib import re import sys import click import psycopg # 复用 parse_upgrade 的守卫/工具(剥前缀、剥尾残、垃圾拦截、内嵌章号、生命周期推断、章号排序键、 # 窗源文加载、SOURCE_TYPE/TENANT/DSN)——迁移与抽取同一套清洗口径,不另立标准。 HERE = pathlib.Path(__file__).resolve().parent sys.path.insert(0, str(HERE)) sys.path.insert(0, str(HERE.parents[1] / "llm" / "scripts")) import parse_upgrade as pu # noqa: E402 from llm import chat_governed, extract_json # noqa: E402 DSN = pu.DSN # 带 keepalives(防 Tailscale 长空转掐断),与抽取同源 SOURCE_TYPE = pu.SOURCE_TYPE TENANT = pu.TENANT # 演变类字段白名单:只有这些字段的 [窗N] 条目属"演变台阶"、走章号迁移;其余字段的 [窗N] # 是当前态字段被污染,不在迁移范围(见文件头「范围边界」)。 EVOLUTION_FIELDS = {"成长弧线", "演变轨迹", "演变历程", "大事记", "经历"} RELATION_TYPE = pu.RELATION_TYPE # 条目行首 [窗…] 前缀内的原始标记('窗2' / '窗307-315' / '本窗' / '窗本窗') PREFIX_RE = re.compile(r"^\[([窗本][^\]]{0,12})\]") def parse_prefix(entry): """提取条目行首 [窗…] 前缀内容;无前缀返回 None。""" m = PREFIX_RE.match(str(entry).strip()) return m.group(1) if m else None def resolve_chapter(entry, win_max): """给单条演变条目定真实章号。返回 (章号, 来源, 待LLM窗号)。 章号:整数 / 区间字符串"307-315" / None(待LLM 或 待人工时为 None)。 来源 ∈ 内嵌 / 前缀误标章号 / 待LLM / 待人工。""" body = pu._strip_tail(pu._strip_prefix(entry)) inline = pu._extract_inline_chapter(body) if inline is not None: return inline, "内嵌", None # ① 最准:条目自带真实章号 pref = parse_prefix(entry) if pref: nums = re.findall(r"\d+", pref) if nums: lo = int(nums[0]) if lo > win_max: # ② 超窗号上限=章号被误标为窗号 ch = nums[0] if len(nums) == 1 else f"{nums[0]}-{nums[1]}" return ch, "前缀误标章号", None return None, "待LLM", lo # ③ 真窗号,需读源文 LLM 精确化 return None, "待人工", None # ④ [本窗]/无线索,定不了 def migrate_relation_entry(entry, ch, src): """关系型「演变轨迹」条目迁移:保持字符串,只把 [窗N] 换真实章号前缀(不对象化)。""" body = pu._strip_tail(pu._strip_prefix(entry)) if parse_prefix(entry) is None: return entry # 无 [窗N](已是真实章号或纯文本),原样不动 if src == "内嵌": return body # body 自带章号,剥掉窗前缀即干净 if ch is not None: return f"第{ch}章:{body}" # 误标章号/LLM 定位:补真实章号前缀 if src == "待LLM": return f"[待LLM] {body}" # 真窗号未精确化:留标记,别当定不了 return f"[待人工] {body}" # 定不了([本窗]/无线索):显式标记,不糊弄 def migrate_entity_entry(entry, ch, src): """实体型演变条目对象化成里程碑 {章,台阶,周期};降级保底不丢条。 章号定不了(src=待人工)打「待人工」标记供人工精确化;待LLM(未精确化)不标(开 --llm 会定到)。""" body = pu._strip_tail(pu._strip_prefix(entry)) m = {"章": ch, "台阶": body, "周期": pu._infer_lifecycle(body)} if ch is None and src == "待人工": m["待人工"] = True return m # ── 短连接读取(读完即释放,LLM/打印阶段不持连接)── def load_window_maxima(conn): """各作品窗号上限:{work_id: 最大窗号}——判「前缀数字是否超上限=章号误标」。""" rows = conn.execute( "SELECT work_id, max(window_no) FROM example_upgrade_window WHERE deleted=FALSE GROUP BY work_id" ).fetchall() return {w: int(mx) for w, mx in rows} def load_win_ranges(conn, work_id, win_nos): """读指定窗的章区间:{窗号: (from_ch, to_ch)}——给 LLM 精确化缩小定位范围。""" if not win_nos: return {} rows = conn.execute( """SELECT window_no, from_chapter, to_chapter FROM example_upgrade_window WHERE tenant_id=%s AND work_id=%s AND window_no = ANY(%s) AND deleted=FALSE""", (TENANT, work_id, list(win_nos))).fetchall() return {int(w): (a, b) for w, a, b in rows} def load_target_cards(conn, work_id, ids, limit): """读待迁卡:含演变类字段且该字段带 [窗/本窗] 前缀条目的 upgrade_book 卡。 返回 [(id, work_id, payload)]。""" sql = ["""SELECT id, work_id, draft_payload FROM muse_knowledge_draft WHERE tenant_id=%s AND source_type=%s AND deleted=FALSE""", ] args = [TENANT, SOURCE_TYPE] if work_id is not None: sql.append("AND work_id=%s"); args.append(work_id) if ids: sql.append("AND id = ANY(%s)"); args.append(ids) sql.append("ORDER BY id") if limit: sql.append(f"LIMIT {int(limit)}") rows = conn.execute("\n".join(sql), args).fetchall() out = [] for did, wid, payload in rows: fields = (payload or {}).get("字段") or {} # 只留"演变类字段里存在带窗前缀条目"的卡(含内嵌章号但无窗前缀的条目不算需迁) hit = any(k in EVOLUTION_FIELDS and isinstance(v, list) and any(parse_prefix(x) for x in v if isinstance(x, str)) for k, v in fields.items()) if ids or hit: # 显式点名的卡即使无命中也纳入(验收要看"无可迁"边界) out.append((did, wid, payload)) return out def scan_pollution(payload): """当前态字段污染报告:非演变类字段里带 [窗N] 的条目数(仅提示,不迁)。返回 {字段: 条数}。""" rep = {} for k, v in ((payload or {}).get("字段") or {}).items(): if k in EVOLUTION_FIELDS: continue vals = v if isinstance(v, list) else [v] n = sum(1 for x in vals if isinstance(x, str) and parse_prefix(x)) if n: rep[k] = n return rep # ── LLM 精确化(读窗源文 → chat_governed 批量定位;连接纪律:读源文短连接、LLM 无连接)── def llm_locate_prompt(title, a, b, text, entries): """构造"把若干本窗演变条目定位到真实章号"的 prompt。""" lines = "\n".join(f"{i + 1}. {e}" for i, e in enumerate(entries)) return f"""【功能指令(存量升格卡·演变条目真实章号精确化)】 下面是《{title}》第 {a}-{b} 章的正文,和若干条"确定发生在这段正文里、但没标注精确章号"的演变条目。 为每条在正文里找到它对应的**真实章号**(正文各章以「## 第N章」开头)。找不到确切章的返回 null,绝不猜。 【本窗正文(第 {a}-{b} 章)】 {text} 【待定位条目】 {lines} 【输出规则(只输出一个 JSON 对象)】 {{"定位": [{{"序号": 1, "章": 章号数字或null}}]}}""" def llm_precise(work_id, title, win_ranges, win_text, pending): """对 {窗号: [条目文本…]} 逐窗 LLM 定位,返回 {(窗号, 条目文本): 章号}。 win_text 预读的窗源文(连接已释放);LLM 走 chat_governed(全局额度治理,不自写降级)。""" located = {} for wno, entries in pending.items(): if wno not in win_ranges or wno not in win_text: continue a, b = win_ranges[wno] prompt = llm_locate_prompt(title, a, b, win_text[wno], entries) content, _usage, used = chat_governed(prompt, system=pu.IDENTITY) if used is None: # 全局降级链耗尽:本窗放弃精确化,条目留待人工 print(f" [窗{wno}] LLM 全链耗尽,该窗 {len(entries)} 条转待人工", file=sys.stderr) continue try: data = extract_json(content) except Exception as e: print(f" [窗{wno}] LLM 输出解析失败({str(e)[:60]}),该窗转待人工", file=sys.stderr) continue for r in [x for x in (data.get("定位") or []) if isinstance(x, dict)]: idx, ch = r.get("序号"), r.get("章") if isinstance(idx, int) and 1 <= idx <= len(entries) and isinstance(ch, int): located[(wno, entries[idx - 1])] = ch return located # ── dry-run 主流程 ── def plan_card(payload, win_max, located, llm_ran): """算单卡迁移计划(不写库)。返回 (字段级前后对照 dict, 来源计数 dict, 待人工数)。 located:{(窗号,条目文本): 真实章号} LLM 精确化结果(无则空 dict)。 llm_ran:该卡所属作品本次是否真跑了 LLM 精确化——决定真窗号条目未定到时算「待人工」还是仍留「待LLM」。""" t = payload.get("type", "") fields = payload.get("字段") or {} diffs, srcs, pending_manual = {}, {}, 0 milestones = [] # 实体型:各演变类字段汇入的里程碑 for k, v in fields.items(): if k not in EVOLUTION_FIELDS or not isinstance(v, list): continue new_rel = [] # 关系型:演变轨迹换章号后的字符串条目 for entry in v: if not isinstance(entry, str): continue if pu._is_garbage(entry): # 顺手清脏:结构垃圾条目丢弃 srcs["垃圾丢弃"] = srcs.get("垃圾丢弃", 0) + 1 continue ch, src, wno = resolve_chapter(entry, win_max) if src == "待LLM": # 真窗号条目:回填 LLM 精确化结果 body = pu._strip_tail(pu._strip_prefix(entry)) hit = located.get((wno, body)) if hit is not None: ch, src = hit, "LLM定位" elif llm_ran: ch, src = None, "待人工" # 跑了 LLM 仍定不到 → 真待人工 # 否则(未跑 LLM):保持 src="待LLM"、ch=None,不计入待人工(待精确化,非定不了) srcs[src] = srcs.get(src, 0) + 1 if src == "待人工": pending_manual += 1 if t == RELATION_TYPE: new_rel.append(migrate_relation_entry(entry, ch, src)) else: milestones.append(migrate_entity_entry(entry, ch, src)) if t == RELATION_TYPE and new_rel: # 去重(同文)+ 按章号排序(抽条目内嵌章号) seen, dedup = set(), [] for x in sorted(new_rel, key=lambda s: pu._chapter_sort_key(pu._extract_inline_chapter(s))): key = pu._strip_prefix(x).strip() if key and key not in seen: dedup.append(x); seen.add(key) diffs[k] = {"前": v, "后": dedup} if t != RELATION_TYPE and milestones: # 实体型:成长弧线等 → 演变历程(对象化、去重按台阶、排序按章号);原演变类字段清空(回归未来计划) seen, dedup = set(), [] for m in sorted(milestones, key=lambda m: pu._chapter_sort_key(m.get("章"))): key = m["台阶"].strip() if key and key not in seen: dedup.append(m); seen.add(key) old_evo = fields.get("演变历程") if isinstance(fields.get("演变历程"), list) else [] diffs["演变历程"] = {"前": old_evo, "后": dedup} for k in EVOLUTION_FIELDS - {"演变历程"}: if isinstance(fields.get(k), list) and fields[k]: diffs[k] = {"前": fields[k], "后": []} # 已发生台阶搬走,原字段清空 return diffs, srcs, pending_manual @click.command() @click.option("--work-id", type=int, help="限定作品(不给=全部 upgrade_book 作品)") @click.option("--ids", help="逗号分隔的卡 id 白名单(验收指定样例卡)") @click.option("--limit", type=int, default=0, help="最多处理卡数(0=不限)") @click.option("--llm", "use_llm", is_flag=True, help="对真窗号条目真调 LLM 读原文精确化(默认关=纯机械预览,真窗号条目暂标待LLM)") @click.option("--max-llm-windows", type=int, default=3, show_default=True, help="LLM 精确化最多读几个窗(控额度,验收演示用)") @click.option("--samples", type=int, default=5, show_default=True, help="打印几张卡的前后对照") def main(work_id, ids, limit, use_llm, max_llm_windows, samples): """存量 [窗N]→真实章号迁移 dry-run(零写库)。""" id_list = [int(x) for x in ids.split(",") if x.strip()] if ids else None # ── 短连接①:读窗号上限 + 待迁卡(读完即释放)── with psycopg.connect(DSN) as conn: win_max_map = load_window_maxima(conn) cards = load_target_cards(conn, work_id, id_list, limit) click.echo(f"待迁卡:{len(cards)} 张(演变类字段含 [窗N] 前缀条目)") if not cards: return # ── 无连接:首轮机械解析,收集各作品待 LLM 精确化的 (窗号→条目) ── pending_by_work = {} # work_id -> {窗号: set(条目body)} for did, wid, payload in cards: win_max = win_max_map.get(wid, 10 ** 9) for k, v in ((payload or {}).get("字段") or {}).items(): if k not in EVOLUTION_FIELDS or not isinstance(v, list): continue for entry in v: if not isinstance(entry, str) or pu._is_garbage(entry): continue _ch, src, wno = resolve_chapter(entry, win_max) if src == "待LLM": body = pu._strip_tail(pu._strip_prefix(entry)) pending_by_work.setdefault(wid, {}).setdefault(wno, set()).add(body) # ── LLM 精确化(可选):短连接②读窗源文→释放→无连接调 chat_governed ── located_by_work = {} # work_id -> {(窗号,body): 章号} if use_llm and pending_by_work: for wid, wmap in pending_by_work.items(): win_nos = sorted(wmap)[:max_llm_windows] # 控额度:只精确化前 N 个窗 with psycopg.connect(DSN) as conn: # 短连接读源文 title = conn.execute("SELECT title FROM muse_content_work WHERE id=%s", (wid,)).fetchone()[0] win_ranges = load_win_ranges(conn, wid, win_nos) win_text = {w: pu.load_window_text(conn, wid, *win_ranges[w]) for w in win_nos if w in win_ranges} # 连接已释放,此处纯 LLM(chat_governed 全局治理) pend = {w: sorted(wmap[w]) for w in win_nos} located_by_work[wid] = llm_precise(wid, title, win_ranges, win_text, pend) skipped = sum(len(wmap) for wmap in pending_by_work.values()) - \ sum(min(len(wmap), max_llm_windows) for wmap in pending_by_work.values()) if skipped: click.echo(f"(LLM 精确化受 --max-llm-windows={max_llm_windows} 限," f"{skipped} 个窗本次未精确化、其条目暂计待人工)") elif pending_by_work: n = sum(len(b) for wmap in pending_by_work.values() for b in wmap.values()) click.echo(f"(--llm 未开:{n} 条真窗号条目暂标『待LLM』,加 --llm 真调原文精确化)") # ── 无连接:出迁移计划 + 汇总统计 + 样例前后对照 ── total_src, total_manual, changed_cards = {}, 0, 0 printed = 0 for did, wid, payload in cards: win_max = win_max_map.get(wid, 10 ** 9) diffs, srcs, manual = plan_card(payload, win_max, located_by_work.get(wid, {}), use_llm and wid in located_by_work) pollution = scan_pollution(payload) for s, n in srcs.items(): total_src[s] = total_src.get(s, 0) + n total_manual += manual if diffs: changed_cards += 1 if printed < samples: printed += 1 name = payload.get("名称", "") t = payload.get("type", "") click.echo(f"\n{'=' * 70}\n卡#{did} [{t}] {name}(work={wid})") if not diffs: click.echo(" 演变类字段:无 [窗N] 条目可迁") for k, d in diffs.items(): click.echo(f" ── 字段「{k}」:{len(d['前'])} 条 → {len(d['后'])} 条") for x in list(d["前"])[:3]: click.echo(f" 前│ {str(x)[:110]}") for x in list(d["后"])[:3]: click.echo(f" 后│ {json.dumps(x, ensure_ascii=False)[:110] if isinstance(x, dict) else str(x)[:110]}") if pollution: click.echo(f" ⚠ 当前态字段 [窗N] 污染(非演变台阶,建议 P0 重抽清理、本脚本不迁):" + ",".join(f"{k}×{n}" for k, n in pollution.items())) # ── 总账 ── click.echo(f"\n{'=' * 70}\n【迁移 dry-run 总账】(零写库,落库由主代理 gate 后另行发起)") click.echo(f" 待迁卡 {len(cards)} 张,有实际迁移计划 {changed_cards} 张") click.echo(f" 条目章号来源分布:" + ",".join(f"{s}×{n}" for s, n in sorted(total_src.items()))) click.echo(f" 待人工条目(章号实在定不了):{total_manual} 条") if __name__ == "__main__": try: main() except (psycopg.Error, RuntimeError) as e: click.echo(f"[迁移错误] {type(e).__name__}: {e}", err=True) sys.exit(1)