muse-agent-example/.claude/skills/parse-book/scripts/migrate_upgrade_windows.py
zizi 178603568e feat(parse-book): 升格卡改造前向落地——实体成长「演变历程」骨架+真实章号里程碑+渐进披露+语义判重
schema(6型 item/power_system/faction/location/event/character):加「演变历程」(里程碑对象数组{章,台阶,周期},aiContext[detection,extraction]=续写不给防上万字过载、仅一致性检查与判重可见)+「演变概括」现状层+「前身/后继」串链;character「成长弧线」语义回归'未来计划'。已重跑 seed_schemas 灌字段合同(6型合同均含演变历程,亲验)。

parse_upgrade.py:里程碑对象合并去重按真实章号(不用运行时窗号)+撤销靠内部_win键(防同窗重跑重复追加)+当前态字段干净纪律(治升级线塞错字段病根)+observe/update提示词记进化台阶与登场→结局生命周期+语义判重(embed≥0.78召回+M3终判,同型自动并/跨型仅串链候选,--semantic-dedup默认关)。

migrate_upgrade_windows.py:存量[窗N]→真实章号迁移(全脚本只读零写库;人工精确化=无内嵌章号处再抽原文定章;全库563卡待迁/6616条待LLM/669内嵌自动精确)。read-context/detect SKILL.md 补渐进披露与演变连续性检查说明。待gate未执行:真迁移落库/开语义判重。设计+拍板见 docs/2026-07-16-升格卡改造设计.md。

创始人4项拍板:①里程碑结构化对象②五型全补③character一起改④存量人工精确化(+主代理定演变概括独立)。
2026-07-17 06:13:18 +08:00

369 lines
20 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""存量升格卡 [窗N] → 真实章号迁移(升格卡改造 §6.1)——**只 --dry-run,全程零写库**。
背景:老升格卡的演变类字段(character.成长弧线 / character_relation.演变轨迹)里,条目带
运行时窗号前缀 [窗N]/[本窗]。窗号是切窗规则的运行产物(会变、脱离环境不可读),必须换成
永不变的真实章号(第X章)。本脚本把这些条目:
- 实体型(character 等):对象化成里程碑 {章,台阶,周期} 汇入「演变历程」;成长弧线回归"未来计划"。
- 关系型(character_relation):保持「演变轨迹」字符串结构,只把 [窗N] 换真实章号前缀。
真实章号定位优先级(落实拍板#5「人工精确化」,不接受 12 章宽近似):
1) 内嵌章号——条目文字里模型已写"第420章/119章/Ch.21",正则直接抽,最准、零 LLM。
2) 前缀误标章号——[窗307-315] 里 307 远超该书窗号上限(work4=80/work8=116),是"第307章"
被误加了"窗"字,采信为章号、零 LLM。
3) 真窗号 [窗N](N≤上限)——读该窗源文、LLM(走 chat_governed 全局治理)定位到真实章。
4) [本窗]/无任何线索——定不了,打"待人工"标记,绝不用窗区间近似糊弄。
范围边界(诚实标注):本脚本只迁**演变类字段**(成长弧线/演变轨迹/演变历程/大事记/经历)。
item/power_system 等把升级线塞在当前态字段(流转计划/戏剧作用/跨体系换算…)的病根2,属于
"当前态字段污染",需 P0 提示词改造后**重抽**清理,非本机械迁移能正确归位——脚本对这类字段
只做「污染报告」提示、不动数据。
连接纪律(硬约束):短连接读→释放→LLM→(本脚本不写库)。绝不在 DB 事务里夹 LLM 调用。
本脚本无任何 UPDATE/INSERT muse_knowledge_draft——迁移落库由主代理 gate 后另行发起。
"""
import json
import pathlib
import re
import sys
import click
import psycopg
# 复用 parse_upgrade 的守卫/工具(剥前缀、剥尾残、垃圾拦截、内嵌章号、生命周期推断、章号排序键、
# 窗源文加载、SOURCE_TYPE/TENANT/DSN)——迁移与抽取同一套清洗口径,不另立标准。
HERE = pathlib.Path(__file__).resolve().parent
sys.path.insert(0, str(HERE))
sys.path.insert(0, str(HERE.parents[1] / "llm" / "scripts"))
import parse_upgrade as pu # noqa: E402
from llm import chat_governed, extract_json # noqa: E402
DSN = pu.DSN # 带 keepalives(防 Tailscale 长空转掐断),与抽取同源
SOURCE_TYPE = pu.SOURCE_TYPE
TENANT = pu.TENANT
# 演变类字段白名单:只有这些字段的 [窗N] 条目属"演变台阶"、走章号迁移;其余字段的 [窗N]
# 是当前态字段被污染,不在迁移范围(见文件头「范围边界」)。
EVOLUTION_FIELDS = {"成长弧线", "演变轨迹", "演变历程", "大事记", "经历"}
RELATION_TYPE = pu.RELATION_TYPE
# 条目行首 [窗…] 前缀内的原始标记('窗2' / '窗307-315' / '本窗' / '窗本窗')
PREFIX_RE = re.compile(r"^\[([窗本][^\]]{0,12})\]")
def parse_prefix(entry):
"""提取条目行首 [窗…] 前缀内容;无前缀返回 None。"""
m = PREFIX_RE.match(str(entry).strip())
return m.group(1) if m else None
def resolve_chapter(entry, win_max):
"""给单条演变条目定真实章号。返回 (章号, 来源, 待LLM窗号)。
章号:整数 / 区间字符串"307-315" / None(待LLM 或 待人工时为 None)。
来源 ∈ 内嵌 / 前缀误标章号 / 待LLM / 待人工。"""
body = pu._strip_tail(pu._strip_prefix(entry))
inline = pu._extract_inline_chapter(body)
if inline is not None:
return inline, "内嵌", None # ① 最准:条目自带真实章号
pref = parse_prefix(entry)
if pref:
nums = re.findall(r"\d+", pref)
if nums:
lo = int(nums[0])
if lo > win_max: # ② 超窗号上限=章号被误标为窗号
ch = nums[0] if len(nums) == 1 else f"{nums[0]}-{nums[1]}"
return ch, "前缀误标章号", None
return None, "待LLM", lo # ③ 真窗号,需读源文 LLM 精确化
return None, "待人工", None # ④ [本窗]/无线索,定不了
def migrate_relation_entry(entry, ch, src):
"""关系型「演变轨迹」条目迁移:保持字符串,只把 [窗N] 换真实章号前缀(不对象化)。"""
body = pu._strip_tail(pu._strip_prefix(entry))
if parse_prefix(entry) is None:
return entry # 无 [窗N](已是真实章号或纯文本),原样不动
if src == "内嵌":
return body # body 自带章号,剥掉窗前缀即干净
if ch is not None:
return f"第{ch}章:{body}" # 误标章号/LLM 定位:补真实章号前缀
if src == "待LLM":
return f"[待LLM] {body}" # 真窗号未精确化:留标记,别当定不了
return f"[待人工] {body}" # 定不了([本窗]/无线索):显式标记,不糊弄
def migrate_entity_entry(entry, ch, src):
"""实体型演变条目对象化成里程碑 {章,台阶,周期};降级保底不丢条。
章号定不了(src=待人工)打「待人工」标记供人工精确化;待LLM(未精确化)不标(开 --llm 会定到)。"""
body = pu._strip_tail(pu._strip_prefix(entry))
m = {"章": ch, "台阶": body, "周期": pu._infer_lifecycle(body)}
if ch is None and src == "待人工":
m["待人工"] = True
return m
# ── 短连接读取(读完即释放,LLM/打印阶段不持连接)──
def load_window_maxima(conn):
"""各作品窗号上限:{work_id: 最大窗号}——判「前缀数字是否超上限=章号误标」。"""
rows = conn.execute(
"SELECT work_id, max(window_no) FROM example_upgrade_window WHERE deleted=FALSE GROUP BY work_id"
).fetchall()
return {w: int(mx) for w, mx in rows}
def load_win_ranges(conn, work_id, win_nos):
"""读指定窗的章区间:{窗号: (from_ch, to_ch)}——给 LLM 精确化缩小定位范围。"""
if not win_nos:
return {}
rows = conn.execute(
"""SELECT window_no, from_chapter, to_chapter FROM example_upgrade_window
WHERE tenant_id=%s AND work_id=%s AND window_no = ANY(%s) AND deleted=FALSE""",
(TENANT, work_id, list(win_nos))).fetchall()
return {int(w): (a, b) for w, a, b in rows}
def load_target_cards(conn, work_id, ids, limit):
"""读待迁卡:含演变类字段且该字段带 [窗/本窗] 前缀条目的 upgrade_book 卡。
返回 [(id, work_id, payload)]。"""
sql = ["""SELECT id, work_id, draft_payload FROM muse_knowledge_draft
WHERE tenant_id=%s AND source_type=%s AND deleted=FALSE""", ]
args = [TENANT, SOURCE_TYPE]
if work_id is not None:
sql.append("AND work_id=%s"); args.append(work_id)
if ids:
sql.append("AND id = ANY(%s)"); args.append(ids)
sql.append("ORDER BY id")
if limit:
sql.append(f"LIMIT {int(limit)}")
rows = conn.execute("\n".join(sql), args).fetchall()
out = []
for did, wid, payload in rows:
fields = (payload or {}).get("字段") or {}
# 只留"演变类字段里存在带窗前缀条目"的卡(含内嵌章号但无窗前缀的条目不算需迁)
hit = any(k in EVOLUTION_FIELDS and isinstance(v, list)
and any(parse_prefix(x) for x in v if isinstance(x, str))
for k, v in fields.items())
if ids or hit: # 显式点名的卡即使无命中也纳入(验收要看"无可迁"边界)
out.append((did, wid, payload))
return out
def scan_pollution(payload):
"""当前态字段污染报告:非演变类字段里带 [窗N] 的条目数(仅提示,不迁)。返回 {字段: 条数}。"""
rep = {}
for k, v in ((payload or {}).get("字段") or {}).items():
if k in EVOLUTION_FIELDS:
continue
vals = v if isinstance(v, list) else [v]
n = sum(1 for x in vals if isinstance(x, str) and parse_prefix(x))
if n:
rep[k] = n
return rep
# ── LLM 精确化(读窗源文 → chat_governed 批量定位;连接纪律:读源文短连接、LLM 无连接)──
def llm_locate_prompt(title, a, b, text, entries):
"""构造"把若干本窗演变条目定位到真实章号"的 prompt。"""
lines = "\n".join(f"{i + 1}. {e}" for i, e in enumerate(entries))
return f"""【功能指令(存量升格卡·演变条目真实章号精确化)】
下面是《{title}》第 {a}-{b} 章的正文,和若干条"确定发生在这段正文里、但没标注精确章号"的演变条目。
为每条在正文里找到它对应的**真实章号**(正文各章以「## 第N章」开头)。找不到确切章的返回 null,绝不猜。
【本窗正文(第 {a}-{b} 章)】
{text}
【待定位条目】
{lines}
【输出规则(只输出一个 JSON 对象)】
{{"定位": [{{"序号": 1, "章": 章号数字或null}}]}}"""
def llm_precise(work_id, title, win_ranges, win_text, pending):
"""对 {窗号: [条目文本…]} 逐窗 LLM 定位,返回 {(窗号, 条目文本): 章号}。
win_text 预读的窗源文(连接已释放);LLM 走 chat_governed(全局额度治理,不自写降级)。"""
located = {}
for wno, entries in pending.items():
if wno not in win_ranges or wno not in win_text:
continue
a, b = win_ranges[wno]
prompt = llm_locate_prompt(title, a, b, win_text[wno], entries)
content, _usage, used = chat_governed(prompt, system=pu.IDENTITY)
if used is None: # 全局降级链耗尽:本窗放弃精确化,条目留待人工
print(f" [窗{wno}] LLM 全链耗尽,该窗 {len(entries)} 条转待人工", file=sys.stderr)
continue
try:
data = extract_json(content)
except Exception as e:
print(f" [窗{wno}] LLM 输出解析失败({str(e)[:60]}),该窗转待人工", file=sys.stderr)
continue
for r in [x for x in (data.get("定位") or []) if isinstance(x, dict)]:
idx, ch = r.get("序号"), r.get("章")
if isinstance(idx, int) and 1 <= idx <= len(entries) and isinstance(ch, int):
located[(wno, entries[idx - 1])] = ch
return located
# ── dry-run 主流程 ──
def plan_card(payload, win_max, located, llm_ran):
"""算单卡迁移计划(不写库)。返回 (字段级前后对照 dict, 来源计数 dict, 待人工数)。
located:{(窗号,条目文本): 真实章号} LLM 精确化结果(无则空 dict)。
llm_ran:该卡所属作品本次是否真跑了 LLM 精确化——决定真窗号条目未定到时算「待人工」还是仍留「待LLM」。"""
t = payload.get("type", "")
fields = payload.get("字段") or {}
diffs, srcs, pending_manual = {}, {}, 0
milestones = [] # 实体型:各演变类字段汇入的里程碑
for k, v in fields.items():
if k not in EVOLUTION_FIELDS or not isinstance(v, list):
continue
new_rel = [] # 关系型:演变轨迹换章号后的字符串条目
for entry in v:
if not isinstance(entry, str):
continue
if pu._is_garbage(entry): # 顺手清脏:结构垃圾条目丢弃
srcs["垃圾丢弃"] = srcs.get("垃圾丢弃", 0) + 1
continue
ch, src, wno = resolve_chapter(entry, win_max)
if src == "待LLM": # 真窗号条目:回填 LLM 精确化结果
body = pu._strip_tail(pu._strip_prefix(entry))
hit = located.get((wno, body))
if hit is not None:
ch, src = hit, "LLM定位"
elif llm_ran:
ch, src = None, "待人工" # 跑了 LLM 仍定不到 → 真待人工
# 否则(未跑 LLM):保持 src="待LLM"、ch=None,不计入待人工(待精确化,非定不了)
srcs[src] = srcs.get(src, 0) + 1
if src == "待人工":
pending_manual += 1
if t == RELATION_TYPE:
new_rel.append(migrate_relation_entry(entry, ch, src))
else:
milestones.append(migrate_entity_entry(entry, ch, src))
if t == RELATION_TYPE and new_rel:
# 去重(同文)+ 按章号排序(抽条目内嵌章号)
seen, dedup = set(), []
for x in sorted(new_rel, key=lambda s: pu._chapter_sort_key(pu._extract_inline_chapter(s))):
key = pu._strip_prefix(x).strip()
if key and key not in seen:
dedup.append(x); seen.add(key)
diffs[k] = {"前": v, "后": dedup}
if t != RELATION_TYPE and milestones:
# 实体型:成长弧线等 → 演变历程(对象化、去重按台阶、排序按章号);原演变类字段清空(回归未来计划)
seen, dedup = set(), []
for m in sorted(milestones, key=lambda m: pu._chapter_sort_key(m.get("章"))):
key = m["台阶"].strip()
if key and key not in seen:
dedup.append(m); seen.add(key)
old_evo = fields.get("演变历程") if isinstance(fields.get("演变历程"), list) else []
diffs["演变历程"] = {"前": old_evo, "后": dedup}
for k in EVOLUTION_FIELDS - {"演变历程"}:
if isinstance(fields.get(k), list) and fields[k]:
diffs[k] = {"前": fields[k], "后": []} # 已发生台阶搬走,原字段清空
return diffs, srcs, pending_manual
@click.command()
@click.option("--work-id", type=int, help="限定作品(不给=全部 upgrade_book 作品)")
@click.option("--ids", help="逗号分隔的卡 id 白名单(验收指定样例卡)")
@click.option("--limit", type=int, default=0, help="最多处理卡数(0=不限)")
@click.option("--llm", "use_llm", is_flag=True,
help="对真窗号条目真调 LLM 读原文精确化(默认关=纯机械预览,真窗号条目暂标待LLM)")
@click.option("--max-llm-windows", type=int, default=3, show_default=True,
help="LLM 精确化最多读几个窗(控额度,验收演示用)")
@click.option("--samples", type=int, default=5, show_default=True, help="打印几张卡的前后对照")
def main(work_id, ids, limit, use_llm, max_llm_windows, samples):
"""存量 [窗N]→真实章号迁移 dry-run(零写库)。"""
id_list = [int(x) for x in ids.split(",") if x.strip()] if ids else None
# ── 短连接①:读窗号上限 + 待迁卡(读完即释放)──
with psycopg.connect(DSN) as conn:
win_max_map = load_window_maxima(conn)
cards = load_target_cards(conn, work_id, id_list, limit)
click.echo(f"待迁卡:{len(cards)} 张(演变类字段含 [窗N] 前缀条目)")
if not cards:
return
# ── 无连接:首轮机械解析,收集各作品待 LLM 精确化的 (窗号→条目) ──
pending_by_work = {} # work_id -> {窗号: set(条目body)}
for did, wid, payload in cards:
win_max = win_max_map.get(wid, 10 ** 9)
for k, v in ((payload or {}).get("字段") or {}).items():
if k not in EVOLUTION_FIELDS or not isinstance(v, list):
continue
for entry in v:
if not isinstance(entry, str) or pu._is_garbage(entry):
continue
_ch, src, wno = resolve_chapter(entry, win_max)
if src == "待LLM":
body = pu._strip_tail(pu._strip_prefix(entry))
pending_by_work.setdefault(wid, {}).setdefault(wno, set()).add(body)
# ── LLM 精确化(可选):短连接②读窗源文→释放→无连接调 chat_governed ──
located_by_work = {} # work_id -> {(窗号,body): 章号}
if use_llm and pending_by_work:
for wid, wmap in pending_by_work.items():
win_nos = sorted(wmap)[:max_llm_windows] # 控额度:只精确化前 N 个窗
with psycopg.connect(DSN) as conn: # 短连接读源文
title = conn.execute("SELECT title FROM muse_content_work WHERE id=%s",
(wid,)).fetchone()[0]
win_ranges = load_win_ranges(conn, wid, win_nos)
win_text = {w: pu.load_window_text(conn, wid, *win_ranges[w])
for w in win_nos if w in win_ranges}
# 连接已释放,此处纯 LLM(chat_governed 全局治理)
pend = {w: sorted(wmap[w]) for w in win_nos}
located_by_work[wid] = llm_precise(wid, title, win_ranges, win_text, pend)
skipped = sum(len(wmap) for wmap in pending_by_work.values()) - \
sum(min(len(wmap), max_llm_windows) for wmap in pending_by_work.values())
if skipped:
click.echo(f"(LLM 精确化受 --max-llm-windows={max_llm_windows} 限,"
f"{skipped} 个窗本次未精确化、其条目暂计待人工)")
elif pending_by_work:
n = sum(len(b) for wmap in pending_by_work.values() for b in wmap.values())
click.echo(f"(--llm 未开:{n} 条真窗号条目暂标『待LLM』,加 --llm 真调原文精确化)")
# ── 无连接:出迁移计划 + 汇总统计 + 样例前后对照 ──
total_src, total_manual, changed_cards = {}, 0, 0
printed = 0
for did, wid, payload in cards:
win_max = win_max_map.get(wid, 10 ** 9)
diffs, srcs, manual = plan_card(payload, win_max, located_by_work.get(wid, {}),
use_llm and wid in located_by_work)
pollution = scan_pollution(payload)
for s, n in srcs.items():
total_src[s] = total_src.get(s, 0) + n
total_manual += manual
if diffs:
changed_cards += 1
if printed < samples:
printed += 1
name = payload.get("名称", "")
t = payload.get("type", "")
click.echo(f"\n{'=' * 70}\n卡#{did} [{t}] {name}(work={wid})")
if not diffs:
click.echo(" 演变类字段:无 [窗N] 条目可迁")
for k, d in diffs.items():
click.echo(f" ── 字段「{k}」:{len(d['前'])} 条 → {len(d['后'])} 条")
for x in list(d["前"])[:3]:
click.echo(f" 前│ {str(x)[:110]}")
for x in list(d["后"])[:3]:
click.echo(f" 后│ {json.dumps(x, ensure_ascii=False)[:110] if isinstance(x, dict) else str(x)[:110]}")
if pollution:
click.echo(f" ⚠ 当前态字段 [窗N] 污染(非演变台阶,建议 P0 重抽清理、本脚本不迁):"
+ ",".join(f"{k}×{n}" for k, n in pollution.items()))
# ── 总账 ──
click.echo(f"\n{'=' * 70}\n【迁移 dry-run 总账】(零写库,落库由主代理 gate 后另行发起)")
click.echo(f" 待迁卡 {len(cards)} 张,有实际迁移计划 {changed_cards} 张")
click.echo(f" 条目章号来源分布:" + ",".join(f"{s}×{n}" for s, n in sorted(total_src.items())))
click.echo(f" 待人工条目(章号实在定不了):{total_manual} 条")
if __name__ == "__main__":
try:
main()
except (psycopg.Error, RuntimeError) as e:
click.echo(f"[迁移错误] {type(e).__name__}: {e}", err=True)
sys.exit(1)