一、技能重组(动作-对象命名) - 旧目录 clean/confirm/continuation/db/detect/embed/… 重组为 clean-book-text/decide-candidate/write-next-chapter/access-database/ check-content-consistency/embed-knowledge/…(git 识别为 rename,内容保持) - agents/*.md、AGENTS.md/CLAUDE.md 收编、example_skill 登记表同步新名 二、先审后入创作闭环(本次核心) 正文接受从"机械门一过就写正典"改为"机械门+语义审查双通过+用户批准+单事务原子提交", DB 级兜底,编排层跳步即被硬拒。 - candidate_cas.py + example_candidate_cas(109):持久化 CAS 状态链 - fact_delta.py + example_fact_delta/example_fact_ledger(106):结构化事实增量, 模型只提六型闭集增量+正文证据引文,仅用户批准的增量随正文同事务入账本 - projection_registry.py + example_projection_run(107):投影登记与恢复 - acceptance_state.py:接受前置实时状态重读 - lesson_registry.py + example_lesson(108):经验升格链,禁止自动升格 - DDL 105:example_candidate 增 semantic_status/semantic_report_sha256 - write_canonical.accept:语义兜底+同事务合并增量+登记投影; run_writer_pipeline/persist_writer_run/run_writer_semantic_detector/step2 接入全链 - claude_runtime:兼容新 CLI modelUsage 信息字段 三、审查修复(独立子代理四维审查后) - 事实增量 propose→approve 翻态正道,不撞唯一键 - 冻结配置探针重刷(CLI 2.1.211→2.1.231 漂移),profileSha256/adapterVersion 再登记 - 可视化合同悬空路径/五六空间矛盾、 SoT 旧技能名漂移、行尾空白清理 测试:离线 65 套 + 真实库集成 5 套(CAS/接受故障注入/事实增量/投影/经验升格)+ 回放 79 项全绿。 创作内容(docs/design、生成正文 artifacts)按"框架与创作分开"未入本提交。
420 lines
21 KiB
Python
420 lines
21 KiB
Python
#!/usr/bin/env python3
|
||
"""call-content-model Skill:New-API 统一调用入口(默认 MiniMax-M3)。
|
||
|
||
管线内所有内容生产型 LLM 调用(清洗探测/拆书抽取)必须经此入口:
|
||
- trust_env=False(本机代理环境变量会劫持内网直连,教训固化);
|
||
- 超时 + 指数退避重试;<think> 剥离;JSON 三级容错提取(json-repair 兜底);
|
||
- 每次调用向 stderr 打印 token 用量与耗时(成本审计),stdout 只出内容。
|
||
"""
|
||
import json
|
||
import pathlib
|
||
import re
|
||
import sys
|
||
import time
|
||
|
||
import click
|
||
import psycopg
|
||
import requests
|
||
|
||
BASE = "http://100.64.0.8:3000"
|
||
# New-API 普通令牌(仓库政策允许明文;严禁换管理令牌打 /v1)
|
||
TOKEN = "sk-DyVqO3lDmEvQZ3PqGpbNaaaHZHhbh0xaHRIiynhYSmVlLHl2"
|
||
DEFAULT_MODEL = "MiniMax-M3"
|
||
|
||
# ── 额度治理常量(B: 把散在各调用方的降级链上收到 chat_governed 统一治理)──
|
||
# 额度账本 DSN(内网 Tailscale,凭据明文入仓为既定政策;带 keepalives 防长空转被掐)
|
||
QUOTA_DSN = ("postgresql://root:f6710e2d0294eb1c10e26a805a64bc54@100.64.0.8:5433/muse-example"
|
||
"?keepalives=1&keepalives_idle=15&keepalives_interval=5&keepalives_count=3")
|
||
MINIMAX_MODELS = {"MiniMax-M3", "MiniMax-M2.7"} # 计入每窗预算的模型
|
||
BUDGET_CHAIN = ["MiniMax-M3", "MiniMax-M2.7", "glm-5.2", "deepseek-v4-flash"] # 全局统一降级链
|
||
WINDOW_BUDGET_USD = 24.0 # 每窗 MiniMax 花费上限(创始人 2026-07-18 提额 $10→$24),超则切 glm-5.2→deepseek
|
||
WINDOW_CALL_CAP = 6000 # 每窗全模型调用上限(创始人 2026-07-18 提额 4000→6000),达则自动睡到下一窗续跑
|
||
# 上游实测上限:请求前主动裁剪,避免依赖不同渠道含混甚至错误的 HTTP 400 文案再猜测重发。
|
||
# M3 / deepseek 未观察到该限制,故不在表内、不主动裁剪。
|
||
MODEL_MAX_TOKENS = {
|
||
"MiniMax-M2.7": 196608,
|
||
"glm-5.2": 12000,
|
||
}
|
||
# 费率兜底(model_ratio, completion_ratio, cache_ratio),与 New-API /api/pricing 一致(2026-07-16 快照)
|
||
PRICING_FALLBACK = {
|
||
"MiniMax-M3": (0.15, 4.0, 0.2),
|
||
"MiniMax-M2.7": (0.15, 4.0, 0.2),
|
||
"glm-5.2": (0.5634, 3.5, 0.25),
|
||
"deepseek-v4-flash": (0.07, 2.0, 0.071428571429),
|
||
}
|
||
|
||
|
||
class SensitiveError(Exception):
|
||
"""上游内容安全拦截(响应体含 sensitive,如 new_sensitive 1026)。
|
||
|
||
同模型退避重试必再触发(放量实测每敏感章空烧 3 次),故不在此退避,
|
||
立即抛给上层走模型降级链(创始人 2026-07-14:M3→MiniMax-M2.7→deepseek-v4-flash)。"""
|
||
|
||
|
||
class PlanQuotaExhausted(Exception):
|
||
"""上游模型渠道的 Token Plan 已耗尽。
|
||
|
||
该错误在同一额度窗内重试不会恢复,必须立即交给治理层熔断当前模型;它与普通限流 429
|
||
不同,普通 429 仍保留指数退避重试。"""
|
||
|
||
|
||
def _default_persist_call(event):
|
||
"""按需加载运行证据持久化器,避免离线调用被迫连库。"""
|
||
evidence_scripts = (
|
||
pathlib.Path(__file__).resolve().parents[2]
|
||
/ "record-run-evidence"
|
||
/ "scripts"
|
||
)
|
||
if str(evidence_scripts) not in sys.path:
|
||
sys.path.insert(0, str(evidence_scripts))
|
||
from persist_llm_call import persist_call
|
||
return persist_call(event)
|
||
|
||
|
||
def chat(prompt, model=DEFAULT_MODEL, max_tokens=512000, temperature=0.2,
|
||
retries=2, timeout=900, system=None, top_p=None, *, run_id=None,
|
||
caller=None, requested_model_id=None, persist_call=None):
|
||
"""单轮对话,返回 (content, usage)。网络错/5xx/普通 429 指数退避重试。
|
||
|
||
content 已剥离 <think>…</think>(推理模型可能把思考混进正文)。
|
||
system:身份段与任务材料分离(角色遵从更稳、身份段利于上游缓存)。
|
||
top_p:随 temperature 分化实验用(M 家族官方推荐 1.0/0.95,eval A/B 后定版)。
|
||
max_tokens 默认 512000;仅对有实测硬上限的 M2.7/GLM 请求前主动裁剪。
|
||
预扣费机制备忘:New-API 按 max_tokens 预扣(512k 预扣 $0.15375/次,网关已验证接受该值;
|
||
结算按实际用量,余额充足时预扣不产生额外成本)——**余额须 ≥ 并发路数 × $0.154**,
|
||
否则触发 403「预扣费额度失败」(2026-07-15 余额见底实测坐实此机制)。
|
||
"""
|
||
if persist_call is None and (run_id or caller):
|
||
persist_call = _default_persist_call
|
||
if persist_call is not None and not callable(persist_call):
|
||
raise TypeError("persist_call 必须是可调用对象")
|
||
s = requests.Session()
|
||
s.trust_env = False # 本机代理 env 会劫持内网直连
|
||
messages = ([{"role": "system", "content": system}] if system else []) \
|
||
+ [{"role": "user", "content": prompt}]
|
||
model_cap = MODEL_MAX_TOKENS.get(model)
|
||
effective_max_tokens = min(max_tokens, model_cap) if model_cap is not None else max_tokens
|
||
if effective_max_tokens != max_tokens:
|
||
print(f"[llm] {model} max_tokens={max_tokens} 主动裁为模型上限 {effective_max_tokens}",
|
||
file=sys.stderr)
|
||
payload = {
|
||
"model": model,
|
||
"messages": messages,
|
||
"max_tokens": effective_max_tokens,
|
||
"temperature": temperature,
|
||
}
|
||
if top_p is not None:
|
||
payload["top_p"] = top_p
|
||
prompt_raw = json.dumps({"messages": messages, **payload},
|
||
ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
||
requested_model_id = requested_model_id or model
|
||
last_err = None
|
||
for attempt in range(retries + 1):
|
||
try:
|
||
t0 = time.time()
|
||
r = s.post(f"{BASE}/v1/chat/completions",
|
||
headers={"Authorization": f"Bearer {TOKEN}"},
|
||
json=payload, timeout=timeout)
|
||
# Token Plan 耗尽不是瞬时限流:同模型退避只会白等 8/16 秒,立即交治理层按窗熔断。
|
||
if r.status_code == 429 and "Token Plan 用量上限" in r.text:
|
||
raise PlanQuotaExhausted(f"Token Plan 已耗尽 HTTP 429: {r.text[:200]}")
|
||
# 内容安全拦截:同模型退避重试必再敏感,立即抛 SensitiveError 交上层
|
||
# 降级换模型,不在此浪费退避(否则一敏感章空烧 3 次,实测占放量请求 23%)
|
||
if r.status_code >= 500 and "sensitive" in r.text.lower():
|
||
raise SensitiveError(f"内容安全拦截 HTTP {r.status_code}: {r.text[:150]}")
|
||
# 429/5xx 属于可重试的服务端瞬时问题
|
||
if r.status_code in (429,) or r.status_code >= 500:
|
||
last_err = f"HTTP {r.status_code}: {r.text[:200]}"
|
||
raise requests.RequestException(last_err)
|
||
r.raise_for_status()
|
||
data = r.json()
|
||
content = data["choices"][0]["message"]["content"] or ""
|
||
content = re.sub(r"<think>.*?</think>", "", content, flags=re.S).strip()
|
||
usage = data.get("usage", {})
|
||
# 缓存命中数(OpenAI 式 prompt_tokens_details.cached_tokens)——验证前缀缓存是否生效、省了多少
|
||
cached = (usage.get("prompt_tokens_details") or {}).get("cached_tokens", 0)
|
||
print(f"[llm] {model} in={usage.get('prompt_tokens', '?')} "
|
||
f"cached={cached} out={usage.get('completion_tokens', '?')} "
|
||
f"耗时{time.time() - t0:.0f}s finish={data['choices'][0].get('finish_reason')}",
|
||
file=sys.stderr)
|
||
if persist_call is not None:
|
||
persist_call({
|
||
"window_key": window_key(_now()),
|
||
"run_id": run_id,
|
||
"caller": caller or "",
|
||
"requested_model_id": requested_model_id,
|
||
"actual_model_id": model,
|
||
"usage": usage,
|
||
"cost_usd": cost_usd(model, usage),
|
||
"stop_reason": data["choices"][0].get("finish_reason"),
|
||
"duration_ms": max(0, int(round((time.time() - t0) * 1000))),
|
||
"prompt": prompt_raw,
|
||
"response": json.dumps(data, ensure_ascii=False, sort_keys=True,
|
||
separators=(",", ":"), default=str),
|
||
"role": caller,
|
||
})
|
||
return content, usage
|
||
except (requests.RequestException, KeyError, json.JSONDecodeError) as e:
|
||
last_err = str(e)
|
||
if attempt < retries:
|
||
wait = 8 * (2 ** attempt)
|
||
print(f"[llm] 第{attempt + 1}次失败({last_err[:120]}),{wait}s 后重试",
|
||
file=sys.stderr)
|
||
time.sleep(wait)
|
||
raise RuntimeError(f"LLM 调用重试耗尽: {last_err}")
|
||
|
||
|
||
def extract_json(text):
|
||
"""JSON 三级容错提取:直接解析 → 首尾括号截取 → json-repair 兜底。
|
||
|
||
opus 试拆实测过两类 JSON 病(中文引号、缺逗号)——任何模型都可能犯,统一在此兜住。
|
||
"""
|
||
try:
|
||
return json.loads(text)
|
||
except json.JSONDecodeError:
|
||
pass
|
||
# 剥 markdown 代码围栏后按最外层大括号/中括号截取
|
||
t = re.sub(r"^```(?:json)?\s*|\s*```$", "", text.strip(), flags=re.M)
|
||
for a, b in (("{", "}"), ("[", "]")):
|
||
i, j = t.find(a), t.rfind(b)
|
||
if i != -1 and j > i:
|
||
frag = t[i:j + 1]
|
||
try:
|
||
return json.loads(frag)
|
||
except json.JSONDecodeError:
|
||
import json_repair
|
||
return json_repair.loads(frag)
|
||
import json_repair
|
||
return json_repair.loads(t)
|
||
|
||
|
||
# ══ 额度治理层(chat_governed)══
|
||
# WHY 上收:此前每个调用方各写一套模型降级链(parse_llm/parse_outline/review_cards 各一份,
|
||
# 链名/顺序还不一致),既无法全局限预算、也无法跨进程共享"这一窗烧了多少/调了多少次"。
|
||
# 统一到 chat_governed 后:一本共享账本按 5 小时窗计钱计次,MiniMax 超 $24/窗自动切非 MiniMax 链,
|
||
# 全窗调用达 6000 次自动睡到下一窗续跑——降级策略只此一处,调用方只管拿结果。
|
||
|
||
_PRICING_CACHE = None
|
||
# 仅保存当前额度窗内已确认 Token Plan 耗尽的模型。进程重启会自然重探;跨窗也会清空重探。
|
||
_PLAN_QUOTA_OPEN = {}
|
||
|
||
|
||
def _plan_quota_open_models(wk):
|
||
"""返回当前窗已熔断模型集合,并清除其他窗口的陈旧状态。"""
|
||
stale = [key for key in _PLAN_QUOTA_OPEN if key != wk]
|
||
for key in stale:
|
||
del _PLAN_QUOTA_OPEN[key]
|
||
return _PLAN_QUOTA_OPEN.setdefault(wk, set())
|
||
|
||
|
||
def get_pricing():
|
||
"""返回 {model: (model_ratio, completion_ratio, cache_ratio)}。
|
||
进程内只拉一次 /api/pricing;拉取失败或字段异常时回退硬编码,绝不因定价接口抖动崩管线。"""
|
||
global _PRICING_CACHE
|
||
if _PRICING_CACHE is not None:
|
||
return _PRICING_CACHE
|
||
# WHY 先复制兜底再逐字段覆盖:定价接口只是"锦上添花",任何一环出问题都必须能退回硬编码,
|
||
# 让成本核算继续跑;只认能解析成正数的字段,脏数据/0/负数一律不覆盖(算废预算比抖动更危险)。
|
||
merged = {m: list(r) for m, r in PRICING_FALLBACK.items()}
|
||
try:
|
||
s = requests.Session()
|
||
s.trust_env = False # 与 chat 同源:本机代理 env 会劫持内网直连
|
||
r = s.get(f"{BASE}/api/pricing", timeout=10)
|
||
r.raise_for_status()
|
||
by_name = {row.get("model_name"): row
|
||
for row in (r.json().get("data") or []) if isinstance(row, dict)}
|
||
for m in merged:
|
||
row = by_name.get(m)
|
||
if not row:
|
||
continue
|
||
for idx, key in enumerate(("model_ratio", "completion_ratio", "cache_ratio")):
|
||
try:
|
||
v = float(row.get(key))
|
||
except (TypeError, ValueError):
|
||
continue # 字段缺失/非数:保留兜底值
|
||
if v > 0:
|
||
merged[m][idx] = v
|
||
except Exception as e:
|
||
# 网络/HTTP/JSON 任何异常:整体回退硬编码(不吃半拉子覆盖的脏账)
|
||
print(f"[llm] /api/pricing 拉取失败({type(e).__name__}),用兜底费率", file=sys.stderr)
|
||
_PRICING_CACHE = {m: tuple(r) for m, r in PRICING_FALLBACK.items()}
|
||
return _PRICING_CACHE
|
||
_PRICING_CACHE = {m: tuple(r) for m, r in merged.items()}
|
||
return _PRICING_CACHE
|
||
|
||
|
||
def cost_usd(model, usage):
|
||
"""按 New-API 口径算单次调用美元成本($1 = 500000 配额单位)。
|
||
cost = model_ratio × ((prompt-cached) + cached×cache_ratio + completion×completion_ratio) / 500000"""
|
||
pricing = get_pricing()
|
||
if model in pricing:
|
||
model_ratio, completion_ratio, cache_ratio = pricing[model]
|
||
else:
|
||
# 未知模型宁高估勿漏计(漏计会让预算穿底),用 M3 费率兜底并告警
|
||
model_ratio, completion_ratio, cache_ratio = pricing["MiniMax-M3"]
|
||
print(f"[llm] cost_usd 未知模型 {model},用 MiniMax-M3 费率兜底计价", file=sys.stderr)
|
||
prompt = usage.get("prompt_tokens", 0) or 0
|
||
completion = usage.get("completion_tokens", 0) or 0
|
||
cached = (usage.get("prompt_tokens_details") or {}).get("cached_tokens", 0) or 0
|
||
billable = (prompt - cached) + cached * cache_ratio + completion * completion_ratio
|
||
return model_ratio * billable / 500000
|
||
|
||
|
||
def _now():
|
||
from datetime import datetime
|
||
return datetime.now() # 单独封装便于单测打桩
|
||
|
||
|
||
def window_key(dt):
|
||
"""把时刻归到所属窗口边界键。窗口起点 0/5/10/15/20 点,末窗 20-24=4h。"""
|
||
wh = (dt.hour // 5) * 5 # 0..4→0,5..9→5,10..14→10,15..19→15,20..23→20
|
||
return f"{dt:%Y-%m-%d}T{wh:02d}"
|
||
|
||
|
||
def seconds_to_next_window(dt):
|
||
"""距下一窗边界的秒数(<5→05:00,<10→10:00,<15→15:00,<20→20:00,否则次日00:00)。"""
|
||
from datetime import timedelta
|
||
h = dt.hour
|
||
if h < 5:
|
||
boundary = dt.replace(hour=5, minute=0, second=0, microsecond=0)
|
||
elif h < 10:
|
||
boundary = dt.replace(hour=10, minute=0, second=0, microsecond=0)
|
||
elif h < 15:
|
||
boundary = dt.replace(hour=15, minute=0, second=0, microsecond=0)
|
||
elif h < 20:
|
||
boundary = dt.replace(hour=20, minute=0, second=0, microsecond=0)
|
||
else:
|
||
boundary = (dt + timedelta(days=1)).replace(hour=0, minute=0, second=0, microsecond=0)
|
||
# 至少 1 秒:边界精确命中时避免 0/负导致空睡后原地打转
|
||
return max(1, int((boundary - dt).total_seconds()))
|
||
|
||
|
||
def _read_window(wk):
|
||
"""读某窗账本,返回 (minimax_usd:float, total_calls:int);无行返回 (0.0,0)。短连接即关。"""
|
||
# WHY 短连接:LLM/sleep 期间绝不持 DB 连接(Tailscale 长事务空转会被掐断),读完立刻释放
|
||
with psycopg.connect(QUOTA_DSN) as c:
|
||
row = c.execute(
|
||
"SELECT minimax_usd, total_calls FROM example_llm_quota WHERE window_key=%s",
|
||
(wk,)).fetchone()
|
||
if not row:
|
||
return 0.0, 0
|
||
return float(row[0]), int(row[1])
|
||
|
||
|
||
def _bump_window(wk, add_usd):
|
||
"""原子累加:该窗 minimax_usd += add_usd、total_calls += 1,返回累加后的 (usd,calls)。
|
||
单语句 upsert,多分片共用一本账靠 PG 行锁串行化。短连接即关。"""
|
||
with psycopg.connect(QUOTA_DSN) as c:
|
||
row = c.execute(
|
||
"""INSERT INTO example_llm_quota (window_key, minimax_usd, total_calls, updated_at)
|
||
VALUES (%s, %s, 1, now())
|
||
ON CONFLICT (window_key) DO UPDATE
|
||
SET minimax_usd = example_llm_quota.minimax_usd + EXCLUDED.minimax_usd,
|
||
total_calls = example_llm_quota.total_calls + 1, updated_at = now()
|
||
RETURNING minimax_usd, total_calls""",
|
||
(wk, add_usd)).fetchone()
|
||
return float(row[0]), int(row[1]) # psycopg 返回 Decimal,转 float
|
||
|
||
|
||
def chat_governed(prompt, model=DEFAULT_MODEL, system=None, max_tokens=512000,
|
||
temperature=0.2, top_p=None, *, run_id=None, caller=None,
|
||
persist_call=None):
|
||
"""全局额度治理下的对话入口,返回 (content, usage, actual_model)。
|
||
契约:成功→三元组;全链耗尽(所有模型敏感/不可用)→(None,None,None)。
|
||
model 参数仅作兼容保留:实际用哪个模型由全局额度策略决定,不由调用方指定。
|
||
策略(每次调用前):
|
||
1) 读本窗账本;本窗 total_calls ≥ WINDOW_CALL_CAP → 打日志、睡到下一窗边界(不持DB连接)、重读续跑;
|
||
2) 本窗 minimax_usd ≥ WINDOW_BUDGET_USD → 降级链去掉 MiniMax 前缀(只剩 glm-5.2→deepseek),否则用全链;
|
||
3) 跳过本窗已确认 Token Plan 耗尽的模型;其余模型沿链调用,敏感/不可用时换下一个;成功即止;
|
||
4) 成功后 _bump_window(本窗, MiniMax模型才计成本否则0),返回三元组;全链失败返回 (None,None,None)。"""
|
||
from datetime import timedelta
|
||
while True:
|
||
wk = window_key(_now())
|
||
usd, calls = _read_window(wk) # 短连接读完即释放,下面 LLM/sleep 阶段不持连接
|
||
# 1) 调用数达上限:睡到下一窗边界再重来(睡眠期间不持任何 DB 连接)
|
||
if calls >= WINDOW_CALL_CAP:
|
||
now2 = _now()
|
||
secs = seconds_to_next_window(now2)
|
||
# 目标窗边界:secs 经 int() 截断可能落在边界前 <1s(如 04:59:59),+1s 归整到整分,
|
||
# 否则 %H 会把 04:59:59 显示成上一整点"04"、误导成非法边界(窗边界只有 00/05/10/15/20)
|
||
wake = (now2 + timedelta(seconds=secs + 1)).replace(second=0, microsecond=0)
|
||
print(f"[llm] 本窗 {wk} 已达 {calls} 次调用上限(≥{WINDOW_CALL_CAP}),"
|
||
f"睡 {secs // 60} 分钟到下一窗 {wake:%H:%M} 续跑", file=sys.stderr)
|
||
time.sleep(secs)
|
||
continue # 醒来重读账本:跨过窗边界后是新窗,calls 归 0
|
||
# 2) 预算耗尽:本窗改用非 MiniMax 链;否则用全链
|
||
if usd >= WINDOW_BUDGET_USD:
|
||
chain = [m for m in BUDGET_CHAIN if m not in MINIMAX_MODELS]
|
||
print(f"[llm] 本窗 {wk} MiniMax 花费 ${usd:.4f} 已达预算上限 ${WINDOW_BUDGET_USD},"
|
||
f"本窗改用非 MiniMax 链 {chain}", file=sys.stderr)
|
||
else:
|
||
chain = list(BUDGET_CHAIN)
|
||
plan_quota_open = _plan_quota_open_models(wk)
|
||
skipped = [m for m in chain if m in plan_quota_open]
|
||
if skipped:
|
||
print(f"[llm] 本窗 {wk} 跳过 Token Plan 已耗尽模型 {skipped}", file=sys.stderr)
|
||
chain = [m for m in chain if m not in plan_quota_open]
|
||
# 3) 沿链逐个模型调用;撞敏感/不可用换下一个
|
||
for m in chain:
|
||
try:
|
||
content, usage = chat(
|
||
prompt,
|
||
model=m,
|
||
system=system,
|
||
max_tokens=max_tokens,
|
||
temperature=temperature,
|
||
top_p=top_p,
|
||
run_id=run_id,
|
||
caller=caller,
|
||
requested_model_id=model,
|
||
persist_call=persist_call,
|
||
)
|
||
except PlanQuotaExhausted as e:
|
||
plan_quota_open.add(m)
|
||
print(f"[llm] 治理链 {m} Token Plan 本窗耗尽,立即熔断并降级下一个:{str(e)[:80]}",
|
||
file=sys.stderr)
|
||
continue
|
||
except (SensitiveError, RuntimeError) as e:
|
||
print(f"[llm] 治理链 {m} 失败({type(e).__name__}: {str(e)[:80]}),降级下一个",
|
||
file=sys.stderr)
|
||
continue
|
||
# 4) 成功记账:只有 MiniMax 计入 $24/窗 预算,其余模型成本计 0(只占调用数)
|
||
add = cost_usd(m, usage) if m in MINIMAX_MODELS else 0.0
|
||
_bump_window(wk, add) # 全新短连接原子累加,写完即释放
|
||
return content, usage, m
|
||
# 全链走完仍无成功:交上层处置(拆书硬停 / 判重保守 keep / 审核标 blocked)
|
||
return None, None, None
|
||
|
||
|
||
@click.group()
|
||
def cli():
|
||
pass
|
||
|
||
|
||
@cli.command("chat")
|
||
@click.option("--prompt-file", type=click.Path(exists=True), required=True)
|
||
@click.option("--model", default=DEFAULT_MODEL, show_default=True)
|
||
@click.option("--max-tokens", type=int, default=32000, show_default=True)
|
||
@click.option("--temperature", type=float, default=0.2, show_default=True)
|
||
@click.option("--out", type=click.Path(), help="内容写文件(不给则打 stdout)")
|
||
@click.option("--extract-json", "extract_", is_flag=True, help="容错提取 JSON 后输出")
|
||
def chat_cmd(prompt_file, model, max_tokens, temperature, out, extract_):
|
||
prompt = pathlib.Path(prompt_file).read_text()
|
||
content, _ = chat(prompt, model=model, max_tokens=max_tokens, temperature=temperature,
|
||
run_id=None, caller="llm-cli")
|
||
if extract_:
|
||
content = json.dumps(extract_json(content), ensure_ascii=False, indent=1)
|
||
if out:
|
||
pathlib.Path(out).write_text(content)
|
||
click.echo(f"已写 {out}({len(content):,} 字符)", err=True)
|
||
else:
|
||
click.echo(content)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
cli()
|
||
except RuntimeError as e:
|
||
click.echo(f"[llm错误] {e}", err=True)
|
||
sys.exit(1)
|