lili 7dffe6b2c1 feat(cheap-worker): 便宜档预算闸 <¥10 fail-closed 接线 + 成本落盘(M1 U4)
开 CircuitBreakerMiddleware 的 ¥ 累进硬闸(enable_rmb_gate=True)+ 设便宜档专属 ¥10 硬
上限(区别 tier2 ¥3);on_model_call 发起前预估、越线 fail-closed 抛 Tier2CircuitBreak
(budget);取价不可达由 _ensure_pricing 降级次数闸(既有行为,不静默超支/阻断)。run-summary
补 costRmb(breaker.spent_rmb 同 new-api quota 折价口径)/ rmbGate(active|degraded)/
tokens(usage_sum)。

验证:test_budget_gate.py 6/6(¥10 fail-closed / 正常不拦 / 边界 / 降级放行 / 开关旁路 /
¥10 与 tier2 隔离);真跑一款打地鼠成本落盘 costRmb=1.1373<¥10、rmbGate=active、tokens 462642。

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-27 02:01:20 -07:00

209 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
cheap_studio.py — 便宜档 ReAct 主编排(spike 核心;对照源:Node gen.mjs 主循环 + tier2 studio.py:run_studio)。
组装:内置 Agent + ReActConfig + 外层有界 resume 循环(照 studio.py:404-458)。done 由 cheap_toolkit 的 finish
工具触发(工具内复核 check+build 绿);finish 收敛后收口 stage → smoke → ensure_play_spec → 循环外九门 play。
熔断(CircuitBreakerMiddleware)/ trace(Tier2TraceMiddleware)/ 历史压缩(ContextConfig)经 import 复用 tier2,零重写。
为什么要外层 resume:AgentScope 2.0.2 内置 ReAct 在"模型产出无 tool_call 的纯文本回合"即退出(看一次门没绿就停)。
外层在 agent 停下后,若【没真 finish】且【门没绿】且【还有预算】就带反馈再 reply()——跨 reply 保留 memory = 原地续修。
这一圈正对应 Node gen.mjs 的 while 主循环。
CLI:cheap-worker/.venv/bin/python cheap-worker/cheap_studio.py --id cheap-run1 --brief "一个简单的点击得分小游戏"
"""
import json
import sys
import time
from pathlib import Path
# 跨包 import 兜底 + key 注入(_bootstrap 模块级把 tier2/gen-worker 加进 sys.path)。
sys.path.insert(0, str(Path(__file__).resolve().parent)) # → cheap-worker/(CLI 直跑兼容)
import _bootstrap # noqa: E402,F401
from cheap_roles import build_system_prompt # noqa: E402
from cheap_toolkit import CheapSession, build_toolkit # noqa: E402
import cheap_run # noqa: E402
# tier2 框架层(import config 触发代理旁路 + 加载 agentscope,必须在裸 import agentscope/openai 之前)。
from worker import config # noqa: E402
from worker.middleware import ( # noqa: E402
CircuitBreakerMiddleware, Tier2TraceMiddleware, Tier2CircuitBreak,
)
# AgentScope(此时 agentscope 已由 config import 链加载、代理旁路已装)。
from agentscope.agent import Agent, ReActConfig # noqa: E402
from agentscope.message import UserMsg # noqa: E402
from agentscope.state import AgentState # noqa: E402
from agentscope.permission import PermissionContext, PermissionMode # noqa: E402
def _rec(msg: str) -> None:
"""进度日志走 stderr(保 CLI stdout 末行可作机读 JSON)。"""
print(f"[cheap_studio] {msg}", file=sys.stderr, flush=True)
def _bypass_state() -> AgentState:
"""无人值守跳过工具 ASK(否则 FunctionTool 默认 check_permissions 返回 ASK 卡死;对 studio.py:_bypass_state)。"""
return AgentState(permission_context=PermissionContext(mode=PermissionMode.BYPASS))
def _user_msg(text: str) -> UserMsg:
"""2.0.1+ UserMsg 必须带 name(对 studio.py:user_msg)。"""
return UserMsg(name="user", content=text)
def _verdict_brief(v) -> dict:
"""从九门 verdict.json 抽 {pass, failedGates}。"""
if not isinstance(v, dict):
return {"pass": None, "failedGates": None}
guards = v.get("guards") or {}
failed = [k for k, g in guards.items() if isinstance(g, dict) and g.get("pass") is False]
return {"pass": v.get("pass"), "failedGates": failed}
async def run_studio(game_id, brief, *, max_iters=40, max_resumes=6, max_tokens=16000,
port=4320, cdp_port=9222, run_gates=True):
"""跑便宜档一局生成(scaffold → ReAct 写 src/ → done 门 → 收口 stage+smoke+九门)。返回 run-summary dict。
run_gates=True(默认,spike 单局):收口跑全套 stage→smoke→ensure_play_spec→九门 play。
run_gates=False(generation-only,给 compare_node 对照用):收口只到 stage+smoke 为止,**不**自动产 play-spec、
**不**跑内部九门 play —— 由调用方在生成与 play 之间注入金标 spec 再单独 play,把驱动器从对照变量里摘掉
(Codex C1:run_studio 内部已 play,对照需把生成与 play 拆开)。
"""
t0 = time.time()
_rec(f"model={_bootstrap.SPIKE_MODEL} id={game_id} brief=「{brief}」max_iters={max_iters} max_resumes={max_resumes}")
sc = cheap_run.scaffold(game_id)
if not sc["ok"]:
return {"ok": False, "gameId": game_id, "fail": "scaffold 失败:" + sc["output"]}
_rec(f"scaffold amgen-{game_id}(clone _template + SAA 信封)")
session = CheapSession(game_id=game_id)
toolkit = build_toolkit(session)
model = _bootstrap.build_cheap_model(max_tokens=max_tokens)
# 熔断:便宜档开 ¥ 累进硬闸(M1 U4 / KTD5)+ 设便宜档专属 ¥10 硬上限(区别 tier2 的 ¥3 genconfig 默认)。
# on_model_call 在每次裸模型调用前按「已花 + 本次预估」判、越线即 fail-closed 抛 Tier2CircuitBreak(budget);
# 取价不可达时 _ensure_pricing 降级为次数闸(既有行为,不静默超支、不静默阻断)。步数/超时仍用 tier2 默认。
breaker = CircuitBreakerMiddleware(enable_rmb_gate=True, rmb_hard_limit=10.0)
tracer = Tier2TraceMiddleware(trace_id=game_id)
writer = Agent(
name="cheap-writer",
system_prompt=build_system_prompt(game_id),
model=model,
toolkit=toolkit,
middlewares=[tracer, breaker], # trace 外层、熔断内层(列表序=洋葱序)
state=_bypass_state(),
react_config=ReActConfig(max_iters=max_iters),
context_config=config.build_context_config(), # 历史压缩
)
kick = (f"请按这个 brief 造一款游戏:「{brief}」。先 read_file 读手册(.agents/skills/littlejs-game-dev.md)"
"和你的起点 game-logic.js 再动手;核心玩法实现完、check 与 build 都绿了就立即 finish。")
breaker_tripped = None
attempts = 0
try:
for attempt in range(max_resumes + 1):
attempts = attempt + 1
_rec(f"resume attempt {attempts}/{max_resumes + 1} → writer.reply …")
await writer.reply(_user_msg(kick))
# ① 真 finish → 收敛退出
if session.finished is not None:
_rec("finish 已接受(check+build 绿)→ 收敛")
break
# ② 门已绿但没 finish → 踹一脚让它 finish
c_ok = bool(session.last_check and session.last_check.get("ok"))
b_ok = bool(session.last_build and session.last_build.get("ok"))
if c_ok and b_ok:
_rec("门绿但未 finish → 踹 finish")
kick = "check 与 build 都已 PASS。现在直接调 finish 交付(summary 一句话),不要再改。"
continue
# ③ resume 预算耗尽 → 停
if attempt >= max_resumes:
_rec("resume 预算耗尽,停")
break
# ④ 门没绿 + agent 自停 → 带失败反馈再 reply(跨 reply memory 保留=原地续修)
fb = "尚未跑出绿 check/build。"
if session.last_check and not session.last_check.get("ok"):
fb = "上次 check 失败:\n" + (session.last_check.get("output") or "")[:1500]
elif session.last_build and not session.last_build.get("ok"):
fb = "上次 build 失败:\n" + (session.last_build.get("output") or "")[:1500]
_rec(f"门没绿 + agent 自停 → 带反馈再 reply(attempt {attempts})")
kick = (f"你刚才停下了,但 check/build 还没全绿——不要放弃。{fb}\n"
"请据失败 write_file 针对性修,再 check → build,直到都绿再 finish。先修最关键的错。")
except Tier2CircuitBreak as e:
breaker_tripped = {"kind": e.kind, "reason": e.reason}
_rec(f"熔断 CircuitBreak:kind={e.kind} reason={e.reason}")
except Exception as e: # noqa: BLE001 顶层兜底,落 breaker 信息便于诊断
breaker_tripped = {"kind": None, "reason": f"{type(e).__name__}: {e}"}
_rec(f"未捕获异常:{type(e).__name__}: {e}")
# ── 收口:finish 后 stage → smoke → ensure_play_spec → 循环外九门 play ──
finished = session.finished is not None
staged = False
smoke_ok = None
verdict = None
if finished:
st = cheap_run.stage(game_id)
staged = st["ok"]
_rec(f"stage {'OK' if staged else 'FAIL: ' + st['output']}")
if staged:
sm = cheap_run.smoke(game_id, port=port, cdp_port=cdp_port)
smoke_ok = sm["ok"]
_rec(f"smoke {'PASS' if smoke_ok else 'FAIL'}(抓 state 供 play-spec)")
if run_gates:
ps = cheap_run.ensure_play_spec(game_id, sm.get("state"))
_rec(f"play-spec {'已产 driver=' + ps.get('driverType', '?') if ps.get('wrote') else ps.get('reason', '?')}")
pr = cheap_run.play(game_id, port=port, cdp_port=cdp_port)
verdict = pr["verdict"]
vb = _verdict_brief(verdict)
_rec(f"九门 play pass={vb['pass']} failedGates={vb['failedGates']}")
else:
_rec("generation-only(run_gates=False):跳过 ensure_play_spec + 九门 play,交对照方注入金标 spec 后单独 play")
tok_in, tok_out = model.usage_sum() # 成本落盘(R4):token 台账始终可取(RecordingOpenAIChatModel.records 累计)。
summary = {
"ok": finished,
"gameId": game_id,
"brief": brief,
"model": _bootstrap.SPIKE_MODEL,
"finished": finished,
"attempts": attempts,
"check": {"ok": bool(session.last_check and session.last_check.get("ok"))} if session.last_check else None,
"build": {"ok": bool(session.last_build and session.last_build.get("ok"))} if session.last_build else None,
"staged": staged,
"smoke": {"ok": smoke_ok} if smoke_ok is not None else None,
"verdict": _verdict_brief(verdict) if verdict is not None else None,
"breaker": breaker_tripped,
# 成本落盘(R4):¥ 走 breaker 同口径(new-api quota 折价 cost.compute);取价不可达时 spent_rmb=0 且
# rmbGate=degraded 标识(_ensure_pricing 已打降级日志可查)——不静默超支、不静默阻断。
"costRmb": round(breaker.spent_rmb, 4),
"rmbGate": "active" if breaker._rmb_gate_active else "degraded",
"tokens": {"in": tok_in, "out": tok_out, "total": tok_in + tok_out},
"doneSummary": session.finished.get("summary") if session.finished else None,
"wallSec": round(time.time() - t0, 1),
}
ev_dir = cheap_run.game_dir(game_id) / "evidence"
ev_dir.mkdir(parents=True, exist_ok=True)
(ev_dir / "run-summary.json").write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
_rec(f"run-summary 落盘 → {ev_dir / 'run-summary.json'} wallSec={summary['wallSec']}")
# 注:n=5 收敛环批跑(RunRecord/batch_run)是 plan deferred;spike 单局 run-summary 已含等价字段。
return summary
if __name__ == "__main__":
import argparse
import asyncio
ap = argparse.ArgumentParser(description="便宜档 ReAct 主编排(spike 单局)")
ap.add_argument("--id", default="cheap-run1")
ap.add_argument("--brief", default="一个简单的点击得分小游戏")
ap.add_argument("--max-iters", type=int, default=40)
ap.add_argument("--max-resumes", type=int, default=6)
ap.add_argument("--max-tokens", type=int, default=16000)
a = ap.parse_args()
result = asyncio.run(run_studio(a.id, a.brief, max_iters=a.max_iters,
max_resumes=a.max_resumes, max_tokens=a.max_tokens))
# stdout 末行单行 JSON(U6 对照 shell-out 据此解析)。
print(json.dumps(result, ensure_ascii=False))
sys.exit(0 if result.get("ok") else 1)