zizi b0bc7a8745 框架: 技能按动作-对象重组 + 先审后入创作闭环
一、技能重组(动作-对象命名)
- 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为
  clean-book-text/decide-candidate/write-next-chapter/access-database/
  check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持)
- agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名

二、先审后入创作闭环(本次核心)
正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交",
DB 级兜底,编排层跳步即被硬拒。
- candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链
- fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量,
  模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本
- projection_registry.py + example_projection_run(107):投影登记与恢复
- acceptance_state.py:接受前置实时状态重读
- lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格
- DDL 105:example_candidate 增 semantic_status/semantic_report_sha256
- write_canonical.accept:语义兜底+同事务合并增量+登记投影;
  run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链
- claude_runtime:兼容新 CLI modelUsage 信息字段

三、审查修复(独立子代理四维审查后)
- 事实增量 propose→approve 翻态正道,不撞唯一键
- 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记
- 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理

测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。
创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
2026-08-14 10:24:08 +08:00

112 lines
5.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""顽固敏感章分段抢救(创始人 2026-07-15「定点处理遗留」)。
原理:内容安全拦截多为全文综合触发——把章正文对半拆成两段,各自独立走
scaffold 抽取(含 M3→M2.7→deepseek 降级链),机械合并两半结果后走原守卫入库。
每半的细纲上限=半章正文 5%,合并后自然满足全章 5% 上限(ingest 校验不变)。
任一半降级链仍全败 → 该章保持 failed,不硬磕(不无限烧额度)。
用法:salvage --work-id N 或 salvage --all(跑全部 failed 章)
"""
import json
import sys
import click
import psycopg
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parent))
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parents[2] / "call-content-model" / "scripts"))
from parse_llm import (m3_json, scaffold_prompt, ingest, SensitiveHardStop, # noqa: E402
DSN, TENANT)
MODEL = "MiniMax-M3"
def salvage_chapter(conn, work_id, title, ch, ch_title, text, prev):
"""对半拆分两段抽取→机械合并→守卫入库。返回 (是否成功, 说明)。
极短章(<1500 字)不拆半:细纲 60 字下限×2 段会超过正文本身(星环#97 实测
比例 113.7% 被守卫退回),整章单发重试一次。"""
# 感言/公告章(极短+作者口吻):无正文可拆,诚实标注直接入库,零 LLM(星环#97 实测 116 字感言)
plain = text.strip()
if len(plain) < 500 and any(w in plain for w in ("作者", "感言", "请假", "推荐票", "月票", "补更", "上架")):
ok, out = ingest("scaffold", work_id, ch,
{"outline": "(作者感言/公告章,无正文内容)", "entities": [], "hints": []})
return ok, "感言章标注入库" if ok else out[-120:]
if len(text) < 1500:
halves = [text]
else:
mid = len(text) // 2
# 对半点对齐段落边界(往后找最近换行,避免句子拦腰斩)
cut = text.find("\n", mid)
cut = cut if cut != -1 else mid
halves = [text[:cut], text[cut:]]
outlines, ents, hints = [], {}, []
n = len(halves)
for i, half in enumerate(halves, 1):
seg_note = f"(第{i}/{n}段)" if n > 1 else ""
p = scaffold_prompt(title, ch, f"{ch_title}{seg_note}", half, prev)
try:
data, _ = m3_json(p, MODEL, ("outline", "entities", "hints"))
except SensitiveHardStop:
return False, f"第{i}/2段降级链仍全败"
outlines.append(str(data.get("outline") or "").strip())
for e in data.get("entities") or []:
if isinstance(e, dict):
nm = (e.get("名称") or e.get("name") or "").strip()
if nm and nm not in ents:
ents[nm] = e
hints.extend(h for h in (data.get("hints") or []) if isinstance(h, dict))
outline = ";".join(o for o in outlines if o)
# 合并后超 8% 硬顶(降级模型偶超标,星环#1664 实测 11%):按「;」从尾机械砍段到 7% 留余量
import re as _re
wc = len(_re.sub(r"\s", "", text))
cap = max(60, int(wc * 0.07))
while len(_re.sub(r"\s", "", outline)) > cap and ";" in outline:
outline = outline.rsplit(";", 1)[0]
merged = {"outline": outline,
"entities": list(ents.values()), "hints": hints}
ok, out = ingest("scaffold", work_id, ch, merged)
return ok, out[-120:]
@click.command()
@click.option("--work-id", type=int, default=0, help="只跑指定书(0=全部)")
@click.option("--limit", type=int, default=0, help="最多抢救几章(0=不限)")
def main(work_id, limit):
with psycopg.connect(DSN) as conn:
cond = "AND t.work_id=%s" % work_id if work_id else ""
rows = conn.execute(f"""
SELECT t.work_id, w.title, c.order_no, c.title, b.content_text
FROM example_parse_task t
JOIN muse_content_chapter c ON c.id=t.chapter_id
JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE
JOIN muse_content_work w ON w.id=t.work_id
WHERE t.scaffold_status='failed' AND c.deleted=FALSE {cond}
ORDER BY t.work_id, c.order_no""").fetchall()
click.echo(f"待抢救 failed 章:{len(rows)}")
saved = failed = 0
for wid, title, ch, ch_title, text in rows[:limit or None]:
# 前文实体名清单(复用章级压缩口径)
with psycopg.connect(DSN) as conn:
prev = [e for (es,) in conn.execute(
"""SELECT s.entities FROM example_parse_scaffold s
JOIN muse_content_chapter c ON c.id=s.chapter_id
WHERE s.tenant_id=%s AND s.work_id=%s AND c.order_no<%s AND s.deleted=FALSE""",
(TENANT, wid, ch)).fetchall() for e in es]
try:
ok, note = salvage_chapter(conn, wid, title, ch, ch_title, text, prev)
except Exception as e: # 单章异常不挡后续章
ok, note = False, f"异常: {str(e)[:100]}"
if ok:
saved += 1
click.echo(f" ✓ 救回 {title}#{ch}")
else:
failed += 1
click.echo(f" ✗ 仍败 {title}#{ch}: {note}")
click.echo(f"=== 抢救收尾:救回{saved} / 仍败{failed} ===")
if __name__ == "__main__":
main()