schema(6型 item/power_system/faction/location/event/character):加「演变历程」(里程碑对象数组{章,台阶,周期},aiContext[detection,extraction]=续写不给防上万字过载、仅一致性检查与判重可见)+「演变概括」现状层+「前身/后继」串链;character「成长弧线」语义回归'未来计划'。已重跑 seed_schemas 灌字段合同(6型合同均含演变历程,亲验)。
parse_upgrade.py:里程碑对象合并去重按真实章号(不用运行时窗号)+撤销靠内部_win键(防同窗重跑重复追加)+当前态字段干净纪律(治升级线塞错字段病根)+observe/update提示词记进化台阶与登场→结局生命周期+语义判重(embed≥0.78召回+M3终判,同型自动并/跨型仅串链候选,--semantic-dedup默认关)。
migrate_upgrade_windows.py:存量[窗N]→真实章号迁移(全脚本只读零写库;人工精确化=无内嵌章号处再抽原文定章;全库563卡待迁/6616条待LLM/669内嵌自动精确)。read-context/detect SKILL.md 补渐进披露与演变连续性检查说明。待gate未执行:真迁移落库/开语义判重。设计+拍板见 docs/2026-07-16-升格卡改造设计.md。
创始人4项拍板:①里程碑结构化对象②五型全补③character一起改④存量人工精确化(+主代理定演变概括独立)。
369 lines
20 KiB
Python
369 lines
20 KiB
Python
#!/usr/bin/env python3
|
||
"""存量升格卡 [窗N] → 真实章号迁移(升格卡改造 §6.1)——**只 --dry-run,全程零写库**。
|
||
|
||
背景:老升格卡的演变类字段(character.成长弧线 / character_relation.演变轨迹)里,条目带
|
||
运行时窗号前缀 [窗N]/[本窗]。窗号是切窗规则的运行产物(会变、脱离环境不可读),必须换成
|
||
永不变的真实章号(第X章)。本脚本把这些条目:
|
||
- 实体型(character 等):对象化成里程碑 {章,台阶,周期} 汇入「演变历程」;成长弧线回归"未来计划"。
|
||
- 关系型(character_relation):保持「演变轨迹」字符串结构,只把 [窗N] 换真实章号前缀。
|
||
|
||
真实章号定位优先级(落实拍板#5「人工精确化」,不接受 12 章宽近似):
|
||
1) 内嵌章号——条目文字里模型已写"第420章/119章/Ch.21",正则直接抽,最准、零 LLM。
|
||
2) 前缀误标章号——[窗307-315] 里 307 远超该书窗号上限(work4=80/work8=116),是"第307章"
|
||
被误加了"窗"字,采信为章号、零 LLM。
|
||
3) 真窗号 [窗N](N≤上限)——读该窗源文、LLM(走 chat_governed 全局治理)定位到真实章。
|
||
4) [本窗]/无任何线索——定不了,打"待人工"标记,绝不用窗区间近似糊弄。
|
||
|
||
范围边界(诚实标注):本脚本只迁**演变类字段**(成长弧线/演变轨迹/演变历程/大事记/经历)。
|
||
item/power_system 等把升级线塞在当前态字段(流转计划/戏剧作用/跨体系换算…)的病根2,属于
|
||
"当前态字段污染",需 P0 提示词改造后**重抽**清理,非本机械迁移能正确归位——脚本对这类字段
|
||
只做「污染报告」提示、不动数据。
|
||
|
||
连接纪律(硬约束):短连接读→释放→LLM→(本脚本不写库)。绝不在 DB 事务里夹 LLM 调用。
|
||
本脚本无任何 UPDATE/INSERT muse_knowledge_draft——迁移落库由主代理 gate 后另行发起。
|
||
"""
|
||
import json
|
||
import pathlib
|
||
import re
|
||
import sys
|
||
|
||
import click
|
||
import psycopg
|
||
|
||
# 复用 parse_upgrade 的守卫/工具(剥前缀、剥尾残、垃圾拦截、内嵌章号、生命周期推断、章号排序键、
|
||
# 窗源文加载、SOURCE_TYPE/TENANT/DSN)——迁移与抽取同一套清洗口径,不另立标准。
|
||
HERE = pathlib.Path(__file__).resolve().parent
|
||
sys.path.insert(0, str(HERE))
|
||
sys.path.insert(0, str(HERE.parents[1] / "llm" / "scripts"))
|
||
import parse_upgrade as pu # noqa: E402
|
||
from llm import chat_governed, extract_json # noqa: E402
|
||
|
||
DSN = pu.DSN # 带 keepalives(防 Tailscale 长空转掐断),与抽取同源
|
||
SOURCE_TYPE = pu.SOURCE_TYPE
|
||
TENANT = pu.TENANT
|
||
|
||
# 演变类字段白名单:只有这些字段的 [窗N] 条目属"演变台阶"、走章号迁移;其余字段的 [窗N]
|
||
# 是当前态字段被污染,不在迁移范围(见文件头「范围边界」)。
|
||
EVOLUTION_FIELDS = {"成长弧线", "演变轨迹", "演变历程", "大事记", "经历"}
|
||
RELATION_TYPE = pu.RELATION_TYPE
|
||
|
||
# 条目行首 [窗…] 前缀内的原始标记('窗2' / '窗307-315' / '本窗' / '窗本窗')
|
||
PREFIX_RE = re.compile(r"^\[([窗本][^\]]{0,12})\]")
|
||
|
||
|
||
def parse_prefix(entry):
|
||
"""提取条目行首 [窗…] 前缀内容;无前缀返回 None。"""
|
||
m = PREFIX_RE.match(str(entry).strip())
|
||
return m.group(1) if m else None
|
||
|
||
|
||
def resolve_chapter(entry, win_max):
|
||
"""给单条演变条目定真实章号。返回 (章号, 来源, 待LLM窗号)。
|
||
章号:整数 / 区间字符串"307-315" / None(待LLM 或 待人工时为 None)。
|
||
来源 ∈ 内嵌 / 前缀误标章号 / 待LLM / 待人工。"""
|
||
body = pu._strip_tail(pu._strip_prefix(entry))
|
||
inline = pu._extract_inline_chapter(body)
|
||
if inline is not None:
|
||
return inline, "内嵌", None # ① 最准:条目自带真实章号
|
||
pref = parse_prefix(entry)
|
||
if pref:
|
||
nums = re.findall(r"\d+", pref)
|
||
if nums:
|
||
lo = int(nums[0])
|
||
if lo > win_max: # ② 超窗号上限=章号被误标为窗号
|
||
ch = nums[0] if len(nums) == 1 else f"{nums[0]}-{nums[1]}"
|
||
return ch, "前缀误标章号", None
|
||
return None, "待LLM", lo # ③ 真窗号,需读源文 LLM 精确化
|
||
return None, "待人工", None # ④ [本窗]/无线索,定不了
|
||
|
||
|
||
def migrate_relation_entry(entry, ch, src):
|
||
"""关系型「演变轨迹」条目迁移:保持字符串,只把 [窗N] 换真实章号前缀(不对象化)。"""
|
||
body = pu._strip_tail(pu._strip_prefix(entry))
|
||
if parse_prefix(entry) is None:
|
||
return entry # 无 [窗N](已是真实章号或纯文本),原样不动
|
||
if src == "内嵌":
|
||
return body # body 自带章号,剥掉窗前缀即干净
|
||
if ch is not None:
|
||
return f"第{ch}章:{body}" # 误标章号/LLM 定位:补真实章号前缀
|
||
if src == "待LLM":
|
||
return f"[待LLM] {body}" # 真窗号未精确化:留标记,别当定不了
|
||
return f"[待人工] {body}" # 定不了([本窗]/无线索):显式标记,不糊弄
|
||
|
||
|
||
def migrate_entity_entry(entry, ch, src):
|
||
"""实体型演变条目对象化成里程碑 {章,台阶,周期};降级保底不丢条。
|
||
章号定不了(src=待人工)打「待人工」标记供人工精确化;待LLM(未精确化)不标(开 --llm 会定到)。"""
|
||
body = pu._strip_tail(pu._strip_prefix(entry))
|
||
m = {"章": ch, "台阶": body, "周期": pu._infer_lifecycle(body)}
|
||
if ch is None and src == "待人工":
|
||
m["待人工"] = True
|
||
return m
|
||
|
||
|
||
# ── 短连接读取(读完即释放,LLM/打印阶段不持连接)──
|
||
|
||
def load_window_maxima(conn):
|
||
"""各作品窗号上限:{work_id: 最大窗号}——判「前缀数字是否超上限=章号误标」。"""
|
||
rows = conn.execute(
|
||
"SELECT work_id, max(window_no) FROM example_upgrade_window WHERE deleted=FALSE GROUP BY work_id"
|
||
).fetchall()
|
||
return {w: int(mx) for w, mx in rows}
|
||
|
||
|
||
def load_win_ranges(conn, work_id, win_nos):
|
||
"""读指定窗的章区间:{窗号: (from_ch, to_ch)}——给 LLM 精确化缩小定位范围。"""
|
||
if not win_nos:
|
||
return {}
|
||
rows = conn.execute(
|
||
"""SELECT window_no, from_chapter, to_chapter FROM example_upgrade_window
|
||
WHERE tenant_id=%s AND work_id=%s AND window_no = ANY(%s) AND deleted=FALSE""",
|
||
(TENANT, work_id, list(win_nos))).fetchall()
|
||
return {int(w): (a, b) for w, a, b in rows}
|
||
|
||
|
||
def load_target_cards(conn, work_id, ids, limit):
|
||
"""读待迁卡:含演变类字段且该字段带 [窗/本窗] 前缀条目的 upgrade_book 卡。
|
||
返回 [(id, work_id, payload)]。"""
|
||
sql = ["""SELECT id, work_id, draft_payload FROM muse_knowledge_draft
|
||
WHERE tenant_id=%s AND source_type=%s AND deleted=FALSE""", ]
|
||
args = [TENANT, SOURCE_TYPE]
|
||
if work_id is not None:
|
||
sql.append("AND work_id=%s"); args.append(work_id)
|
||
if ids:
|
||
sql.append("AND id = ANY(%s)"); args.append(ids)
|
||
sql.append("ORDER BY id")
|
||
if limit:
|
||
sql.append(f"LIMIT {int(limit)}")
|
||
rows = conn.execute("\n".join(sql), args).fetchall()
|
||
out = []
|
||
for did, wid, payload in rows:
|
||
fields = (payload or {}).get("字段") or {}
|
||
# 只留"演变类字段里存在带窗前缀条目"的卡(含内嵌章号但无窗前缀的条目不算需迁)
|
||
hit = any(k in EVOLUTION_FIELDS and isinstance(v, list)
|
||
and any(parse_prefix(x) for x in v if isinstance(x, str))
|
||
for k, v in fields.items())
|
||
if ids or hit: # 显式点名的卡即使无命中也纳入(验收要看"无可迁"边界)
|
||
out.append((did, wid, payload))
|
||
return out
|
||
|
||
|
||
def scan_pollution(payload):
|
||
"""当前态字段污染报告:非演变类字段里带 [窗N] 的条目数(仅提示,不迁)。返回 {字段: 条数}。"""
|
||
rep = {}
|
||
for k, v in ((payload or {}).get("字段") or {}).items():
|
||
if k in EVOLUTION_FIELDS:
|
||
continue
|
||
vals = v if isinstance(v, list) else [v]
|
||
n = sum(1 for x in vals if isinstance(x, str) and parse_prefix(x))
|
||
if n:
|
||
rep[k] = n
|
||
return rep
|
||
|
||
|
||
# ── LLM 精确化(读窗源文 → chat_governed 批量定位;连接纪律:读源文短连接、LLM 无连接)──
|
||
|
||
def llm_locate_prompt(title, a, b, text, entries):
|
||
"""构造"把若干本窗演变条目定位到真实章号"的 prompt。"""
|
||
lines = "\n".join(f"{i + 1}. {e}" for i, e in enumerate(entries))
|
||
return f"""【功能指令(存量升格卡·演变条目真实章号精确化)】
|
||
下面是《{title}》第 {a}-{b} 章的正文,和若干条"确定发生在这段正文里、但没标注精确章号"的演变条目。
|
||
为每条在正文里找到它对应的**真实章号**(正文各章以「## 第N章」开头)。找不到确切章的返回 null,绝不猜。
|
||
|
||
【本窗正文(第 {a}-{b} 章)】
|
||
{text}
|
||
|
||
【待定位条目】
|
||
{lines}
|
||
|
||
【输出规则(只输出一个 JSON 对象)】
|
||
{{"定位": [{{"序号": 1, "章": 章号数字或null}}]}}"""
|
||
|
||
|
||
def llm_precise(work_id, title, win_ranges, win_text, pending):
|
||
"""对 {窗号: [条目文本…]} 逐窗 LLM 定位,返回 {(窗号, 条目文本): 章号}。
|
||
win_text 预读的窗源文(连接已释放);LLM 走 chat_governed(全局额度治理,不自写降级)。"""
|
||
located = {}
|
||
for wno, entries in pending.items():
|
||
if wno not in win_ranges or wno not in win_text:
|
||
continue
|
||
a, b = win_ranges[wno]
|
||
prompt = llm_locate_prompt(title, a, b, win_text[wno], entries)
|
||
content, _usage, used = chat_governed(prompt, system=pu.IDENTITY)
|
||
if used is None: # 全局降级链耗尽:本窗放弃精确化,条目留待人工
|
||
print(f" [窗{wno}] LLM 全链耗尽,该窗 {len(entries)} 条转待人工", file=sys.stderr)
|
||
continue
|
||
try:
|
||
data = extract_json(content)
|
||
except Exception as e:
|
||
print(f" [窗{wno}] LLM 输出解析失败({str(e)[:60]}),该窗转待人工", file=sys.stderr)
|
||
continue
|
||
for r in [x for x in (data.get("定位") or []) if isinstance(x, dict)]:
|
||
idx, ch = r.get("序号"), r.get("章")
|
||
if isinstance(idx, int) and 1 <= idx <= len(entries) and isinstance(ch, int):
|
||
located[(wno, entries[idx - 1])] = ch
|
||
return located
|
||
|
||
|
||
# ── dry-run 主流程 ──
|
||
|
||
def plan_card(payload, win_max, located, llm_ran):
|
||
"""算单卡迁移计划(不写库)。返回 (字段级前后对照 dict, 来源计数 dict, 待人工数)。
|
||
located:{(窗号,条目文本): 真实章号} LLM 精确化结果(无则空 dict)。
|
||
llm_ran:该卡所属作品本次是否真跑了 LLM 精确化——决定真窗号条目未定到时算「待人工」还是仍留「待LLM」。"""
|
||
t = payload.get("type", "")
|
||
fields = payload.get("字段") or {}
|
||
diffs, srcs, pending_manual = {}, {}, 0
|
||
milestones = [] # 实体型:各演变类字段汇入的里程碑
|
||
for k, v in fields.items():
|
||
if k not in EVOLUTION_FIELDS or not isinstance(v, list):
|
||
continue
|
||
new_rel = [] # 关系型:演变轨迹换章号后的字符串条目
|
||
for entry in v:
|
||
if not isinstance(entry, str):
|
||
continue
|
||
if pu._is_garbage(entry): # 顺手清脏:结构垃圾条目丢弃
|
||
srcs["垃圾丢弃"] = srcs.get("垃圾丢弃", 0) + 1
|
||
continue
|
||
ch, src, wno = resolve_chapter(entry, win_max)
|
||
if src == "待LLM": # 真窗号条目:回填 LLM 精确化结果
|
||
body = pu._strip_tail(pu._strip_prefix(entry))
|
||
hit = located.get((wno, body))
|
||
if hit is not None:
|
||
ch, src = hit, "LLM定位"
|
||
elif llm_ran:
|
||
ch, src = None, "待人工" # 跑了 LLM 仍定不到 → 真待人工
|
||
# 否则(未跑 LLM):保持 src="待LLM"、ch=None,不计入待人工(待精确化,非定不了)
|
||
srcs[src] = srcs.get(src, 0) + 1
|
||
if src == "待人工":
|
||
pending_manual += 1
|
||
if t == RELATION_TYPE:
|
||
new_rel.append(migrate_relation_entry(entry, ch, src))
|
||
else:
|
||
milestones.append(migrate_entity_entry(entry, ch, src))
|
||
if t == RELATION_TYPE and new_rel:
|
||
# 去重(同文)+ 按章号排序(抽条目内嵌章号)
|
||
seen, dedup = set(), []
|
||
for x in sorted(new_rel, key=lambda s: pu._chapter_sort_key(pu._extract_inline_chapter(s))):
|
||
key = pu._strip_prefix(x).strip()
|
||
if key and key not in seen:
|
||
dedup.append(x); seen.add(key)
|
||
diffs[k] = {"前": v, "后": dedup}
|
||
if t != RELATION_TYPE and milestones:
|
||
# 实体型:成长弧线等 → 演变历程(对象化、去重按台阶、排序按章号);原演变类字段清空(回归未来计划)
|
||
seen, dedup = set(), []
|
||
for m in sorted(milestones, key=lambda m: pu._chapter_sort_key(m.get("章"))):
|
||
key = m["台阶"].strip()
|
||
if key and key not in seen:
|
||
dedup.append(m); seen.add(key)
|
||
old_evo = fields.get("演变历程") if isinstance(fields.get("演变历程"), list) else []
|
||
diffs["演变历程"] = {"前": old_evo, "后": dedup}
|
||
for k in EVOLUTION_FIELDS - {"演变历程"}:
|
||
if isinstance(fields.get(k), list) and fields[k]:
|
||
diffs[k] = {"前": fields[k], "后": []} # 已发生台阶搬走,原字段清空
|
||
return diffs, srcs, pending_manual
|
||
|
||
|
||
@click.command()
|
||
@click.option("--work-id", type=int, help="限定作品(不给=全部 upgrade_book 作品)")
|
||
@click.option("--ids", help="逗号分隔的卡 id 白名单(验收指定样例卡)")
|
||
@click.option("--limit", type=int, default=0, help="最多处理卡数(0=不限)")
|
||
@click.option("--llm", "use_llm", is_flag=True,
|
||
help="对真窗号条目真调 LLM 读原文精确化(默认关=纯机械预览,真窗号条目暂标待LLM)")
|
||
@click.option("--max-llm-windows", type=int, default=3, show_default=True,
|
||
help="LLM 精确化最多读几个窗(控额度,验收演示用)")
|
||
@click.option("--samples", type=int, default=5, show_default=True, help="打印几张卡的前后对照")
|
||
def main(work_id, ids, limit, use_llm, max_llm_windows, samples):
|
||
"""存量 [窗N]→真实章号迁移 dry-run(零写库)。"""
|
||
id_list = [int(x) for x in ids.split(",") if x.strip()] if ids else None
|
||
|
||
# ── 短连接①:读窗号上限 + 待迁卡(读完即释放)──
|
||
with psycopg.connect(DSN) as conn:
|
||
win_max_map = load_window_maxima(conn)
|
||
cards = load_target_cards(conn, work_id, id_list, limit)
|
||
click.echo(f"待迁卡:{len(cards)} 张(演变类字段含 [窗N] 前缀条目)")
|
||
if not cards:
|
||
return
|
||
|
||
# ── 无连接:首轮机械解析,收集各作品待 LLM 精确化的 (窗号→条目) ──
|
||
pending_by_work = {} # work_id -> {窗号: set(条目body)}
|
||
for did, wid, payload in cards:
|
||
win_max = win_max_map.get(wid, 10 ** 9)
|
||
for k, v in ((payload or {}).get("字段") or {}).items():
|
||
if k not in EVOLUTION_FIELDS or not isinstance(v, list):
|
||
continue
|
||
for entry in v:
|
||
if not isinstance(entry, str) or pu._is_garbage(entry):
|
||
continue
|
||
_ch, src, wno = resolve_chapter(entry, win_max)
|
||
if src == "待LLM":
|
||
body = pu._strip_tail(pu._strip_prefix(entry))
|
||
pending_by_work.setdefault(wid, {}).setdefault(wno, set()).add(body)
|
||
|
||
# ── LLM 精确化(可选):短连接②读窗源文→释放→无连接调 chat_governed ──
|
||
located_by_work = {} # work_id -> {(窗号,body): 章号}
|
||
if use_llm and pending_by_work:
|
||
for wid, wmap in pending_by_work.items():
|
||
win_nos = sorted(wmap)[:max_llm_windows] # 控额度:只精确化前 N 个窗
|
||
with psycopg.connect(DSN) as conn: # 短连接读源文
|
||
title = conn.execute("SELECT title FROM muse_content_work WHERE id=%s",
|
||
(wid,)).fetchone()[0]
|
||
win_ranges = load_win_ranges(conn, wid, win_nos)
|
||
win_text = {w: pu.load_window_text(conn, wid, *win_ranges[w])
|
||
for w in win_nos if w in win_ranges}
|
||
# 连接已释放,此处纯 LLM(chat_governed 全局治理)
|
||
pend = {w: sorted(wmap[w]) for w in win_nos}
|
||
located_by_work[wid] = llm_precise(wid, title, win_ranges, win_text, pend)
|
||
skipped = sum(len(wmap) for wmap in pending_by_work.values()) - \
|
||
sum(min(len(wmap), max_llm_windows) for wmap in pending_by_work.values())
|
||
if skipped:
|
||
click.echo(f"(LLM 精确化受 --max-llm-windows={max_llm_windows} 限,"
|
||
f"{skipped} 个窗本次未精确化、其条目暂计待人工)")
|
||
elif pending_by_work:
|
||
n = sum(len(b) for wmap in pending_by_work.values() for b in wmap.values())
|
||
click.echo(f"(--llm 未开:{n} 条真窗号条目暂标『待LLM』,加 --llm 真调原文精确化)")
|
||
|
||
# ── 无连接:出迁移计划 + 汇总统计 + 样例前后对照 ──
|
||
total_src, total_manual, changed_cards = {}, 0, 0
|
||
printed = 0
|
||
for did, wid, payload in cards:
|
||
win_max = win_max_map.get(wid, 10 ** 9)
|
||
diffs, srcs, manual = plan_card(payload, win_max, located_by_work.get(wid, {}),
|
||
use_llm and wid in located_by_work)
|
||
pollution = scan_pollution(payload)
|
||
for s, n in srcs.items():
|
||
total_src[s] = total_src.get(s, 0) + n
|
||
total_manual += manual
|
||
if diffs:
|
||
changed_cards += 1
|
||
if printed < samples:
|
||
printed += 1
|
||
name = payload.get("名称", "")
|
||
t = payload.get("type", "")
|
||
click.echo(f"\n{'=' * 70}\n卡#{did} [{t}] {name}(work={wid})")
|
||
if not diffs:
|
||
click.echo(" 演变类字段:无 [窗N] 条目可迁")
|
||
for k, d in diffs.items():
|
||
click.echo(f" ── 字段「{k}」:{len(d['前'])} 条 → {len(d['后'])} 条")
|
||
for x in list(d["前"])[:3]:
|
||
click.echo(f" 前│ {str(x)[:110]}")
|
||
for x in list(d["后"])[:3]:
|
||
click.echo(f" 后│ {json.dumps(x, ensure_ascii=False)[:110] if isinstance(x, dict) else str(x)[:110]}")
|
||
if pollution:
|
||
click.echo(f" ⚠ 当前态字段 [窗N] 污染(非演变台阶,建议 P0 重抽清理、本脚本不迁):"
|
||
+ ",".join(f"{k}×{n}" for k, n in pollution.items()))
|
||
|
||
# ── 总账 ──
|
||
click.echo(f"\n{'=' * 70}\n【迁移 dry-run 总账】(零写库,落库由主代理 gate 后另行发起)")
|
||
click.echo(f" 待迁卡 {len(cards)} 张,有实际迁移计划 {changed_cards} 张")
|
||
click.echo(f" 条目章号来源分布:" + ",".join(f"{s}×{n}" for s, n in sorted(total_src.items())))
|
||
click.echo(f" 待人工条目(章号实在定不了):{total_manual} 条")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
main()
|
||
except (psycopg.Error, RuntimeError) as e:
|
||
click.echo(f"[迁移错误] {type(e).__name__}: {e}", err=True)
|
||
sys.exit(1)
|