一、技能重组(动作-对象命名) - 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为 clean-book-text/decide-candidate/write-next-chapter/access-database/ check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持) - agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名 二、先审后入创作闭环(本次核心) 正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交", DB 级兜底,编排层跳步即被硬拒。 - candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链 - fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量, 模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本 - projection_registry.py + example_projection_run(107):投影登记与恢复 - acceptance_state.py:接受前置实时状态重读 - lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格 - DDL 105:example_candidate 增 semantic_status/semantic_report_sha256 - write_canonical.accept:语义兜底+同事务合并增量+登记投影; run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链 - claude_runtime:兼容新 CLI modelUsage 信息字段 三、审查修复(独立子代理四维审查后) - 事实增量 propose→approve 翻态正道,不撞唯一键 - 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记 - 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理 测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。 创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
112 lines
5.3 KiB
Python
112 lines
5.3 KiB
Python
#!/usr/bin/env python3
|
||
"""顽固敏感章分段抢救(创始人 2026-07-15「定点处理遗留」)。
|
||
|
||
原理:内容安全拦截多为全文综合触发——把章正文对半拆成两段,各自独立走
|
||
scaffold 抽取(含 M3→M2.7→deepseek 降级链),机械合并两半结果后走原守卫入库。
|
||
每半的细纲上限=半章正文 5%,合并后自然满足全章 5% 上限(ingest 校验不变)。
|
||
任一半降级链仍全败 → 该章保持 failed,不硬磕(不无限烧额度)。
|
||
|
||
用法:salvage --work-id N 或 salvage --all(跑全部 failed 章)
|
||
"""
|
||
import json
|
||
import sys
|
||
|
||
import click
|
||
import psycopg
|
||
|
||
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parent))
|
||
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parents[2] / "call-content-model" / "scripts"))
|
||
from parse_llm import (m3_json, scaffold_prompt, ingest, SensitiveHardStop, # noqa: E402
|
||
DSN, TENANT)
|
||
|
||
MODEL = "MiniMax-M3"
|
||
|
||
|
||
def salvage_chapter(conn, work_id, title, ch, ch_title, text, prev):
|
||
"""对半拆分两段抽取→机械合并→守卫入库。返回 (是否成功, 说明)。
|
||
|
||
极短章(<1500 字)不拆半:细纲 60 字下限×2 段会超过正文本身(星环#97 实测
|
||
比例 113.7% 被守卫退回),整章单发重试一次。"""
|
||
# 感言/公告章(极短+作者口吻):无正文可拆,诚实标注直接入库,零 LLM(星环#97 实测 116 字感言)
|
||
plain = text.strip()
|
||
if len(plain) < 500 and any(w in plain for w in ("作者", "感言", "请假", "推荐票", "月票", "补更", "上架")):
|
||
ok, out = ingest("scaffold", work_id, ch,
|
||
{"outline": "(作者感言/公告章,无正文内容)", "entities": [], "hints": []})
|
||
return ok, "感言章标注入库" if ok else out[-120:]
|
||
if len(text) < 1500:
|
||
halves = [text]
|
||
else:
|
||
mid = len(text) // 2
|
||
# 对半点对齐段落边界(往后找最近换行,避免句子拦腰斩)
|
||
cut = text.find("\n", mid)
|
||
cut = cut if cut != -1 else mid
|
||
halves = [text[:cut], text[cut:]]
|
||
outlines, ents, hints = [], {}, []
|
||
n = len(halves)
|
||
for i, half in enumerate(halves, 1):
|
||
seg_note = f"(第{i}/{n}段)" if n > 1 else ""
|
||
p = scaffold_prompt(title, ch, f"{ch_title}{seg_note}", half, prev)
|
||
try:
|
||
data, _ = m3_json(p, MODEL, ("outline", "entities", "hints"))
|
||
except SensitiveHardStop:
|
||
return False, f"第{i}/2段降级链仍全败"
|
||
outlines.append(str(data.get("outline") or "").strip())
|
||
for e in data.get("entities") or []:
|
||
if isinstance(e, dict):
|
||
nm = (e.get("名称") or e.get("name") or "").strip()
|
||
if nm and nm not in ents:
|
||
ents[nm] = e
|
||
hints.extend(h for h in (data.get("hints") or []) if isinstance(h, dict))
|
||
outline = ";".join(o for o in outlines if o)
|
||
# 合并后超 8% 硬顶(降级模型偶超标,星环#1664 实测 11%):按「;」从尾机械砍段到 7% 留余量
|
||
import re as _re
|
||
wc = len(_re.sub(r"\s", "", text))
|
||
cap = max(60, int(wc * 0.07))
|
||
while len(_re.sub(r"\s", "", outline)) > cap and ";" in outline:
|
||
outline = outline.rsplit(";", 1)[0]
|
||
merged = {"outline": outline,
|
||
"entities": list(ents.values()), "hints": hints}
|
||
ok, out = ingest("scaffold", work_id, ch, merged)
|
||
return ok, out[-120:]
|
||
|
||
|
||
@click.command()
|
||
@click.option("--work-id", type=int, default=0, help="只跑指定书(0=全部)")
|
||
@click.option("--limit", type=int, default=0, help="最多抢救几章(0=不限)")
|
||
def main(work_id, limit):
|
||
with psycopg.connect(DSN) as conn:
|
||
cond = "AND t.work_id=%s" % work_id if work_id else ""
|
||
rows = conn.execute(f"""
|
||
SELECT t.work_id, w.title, c.order_no, c.title, b.content_text
|
||
FROM example_parse_task t
|
||
JOIN muse_content_chapter c ON c.id=t.chapter_id
|
||
JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE
|
||
JOIN muse_content_work w ON w.id=t.work_id
|
||
WHERE t.scaffold_status='failed' AND c.deleted=FALSE {cond}
|
||
ORDER BY t.work_id, c.order_no""").fetchall()
|
||
click.echo(f"待抢救 failed 章:{len(rows)}")
|
||
saved = failed = 0
|
||
for wid, title, ch, ch_title, text in rows[:limit or None]:
|
||
# 前文实体名清单(复用章级压缩口径)
|
||
with psycopg.connect(DSN) as conn:
|
||
prev = [e for (es,) in conn.execute(
|
||
"""SELECT s.entities FROM example_parse_scaffold s
|
||
JOIN muse_content_chapter c ON c.id=s.chapter_id
|
||
WHERE s.tenant_id=%s AND s.work_id=%s AND c.order_no<%s AND s.deleted=FALSE""",
|
||
(TENANT, wid, ch)).fetchall() for e in es]
|
||
try:
|
||
ok, note = salvage_chapter(conn, wid, title, ch, ch_title, text, prev)
|
||
except Exception as e: # 单章异常不挡后续章
|
||
ok, note = False, f"异常: {str(e)[:100]}"
|
||
if ok:
|
||
saved += 1
|
||
click.echo(f" ✓ 救回 {title}#{ch}")
|
||
else:
|
||
failed += 1
|
||
click.echo(f" ✗ 仍败 {title}#{ch}: {note}")
|
||
click.echo(f"=== 抢救收尾:救回{saved} / 仍败{failed} ===")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|