lili 8d851ea716
Some checks failed
contract-gates / contract-gates (push) Has been cancelled
docs-gate / docs-gate (push) Has been cancelled
feat(cheap-gen): W-AXIS-V2 波2——取证契约退役+提示全加载链清洗,四门∧测试员为唯一验收链
opus 执行位交付、主会话四步验收过(三套测试亲跑/红线 diff 亲读/冒烟截图亲眼):

- 验收消费:check 新增源码级插件调用存在断言(_pluginCallPresenceErrors 词法层,
  接替 F_wiring 真接线护栏);_forensicsView「必须实现」拦截摘除,降级为可选调试
  接口(handleTap 输入契约与 ctx.log 日志纪律保留);gate_judge 放行措辞对齐四门
- play-spec/驱动器退役(逐件裁定 10 件):gen.mjs done 门自动产 play-spec 摘除、
  cheap_gates/cheap_studio/cheap_modify/service_driver 调用摘除、_read_driver_type
  改读 evidence/playtest/playtest.json;golden-spec 三工具(auto_vs_golden/bake_off/
  compare_node)显式封存;ensure_play_spec 两实现保留一个版本窗口(只摘调用不删码)
- prompt 全加载链清洗:cheap-system.md v1.6.3→1.7.0(删 _forensicsView 自动验收
  契约红线与全部 targets/occupied 语义,五法句/品类指路同步),cheap_roles 内置
  回退逐字节同步(test_roles 12/12 证 parity),modify 切片经运行时切块自动同步;
  5 品类 skills+littlejs 手册+recipes×2+模板 8 件+README 6 件教学清洗,真设计
  价值(单击 casual/可解性质/双层反馈/latch)保留
- 批账:hard_genre_batch/xtheme 行加 playtest 段,_row_total_cost 修 shadow 下
  测试员成本漏算(v2 取 max 防双计)

验收(主会话亲验):pytest cheap-worker 512 绿 + tier2 141 绿 + node 49/49;
25 局基线新旧口径重放对照——真坏 5 局全拦(floor 拦 2/测试员拦 3)、无一例旧拦
新放,翻案 2 局均为破案坐实的仪器误杀;n=3 冒烟(mode=v2)全链绿:w2v2-sim
harness verdict.pass=True 而测试员 reject(机械放行测试员逮住,盘上双证)、
w2v2-sim-fix 现象反馈两轮回喂→fix2 收敛(配方显示修成,局内截图亲眼);
rg 零残留(模板残留均为「验收不读它」的降级调试接口合规形态);真跑 ¥8.4。

已知余项:repairFeedback 有产地、Service 续修环消费者未接(测试员 reject 不入
RepairMiddleware,续修判据仍四门)——是否接入随波3 校准数据一并裁;v1 配置级
回退窗口关闭为创始人知悉项,genconfig 默认 mode 仍 shadow、切 v2 归创始人。

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-10 09:21:04 -07:00

307 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
auto_vs_golden.py — M2 auto-vs-golden delta 门(Node 退役授权第三条 · 按品类 · 零 LLM)。
【封存(W-AXIS-V2 波2,2026-07-10)】对照范式建立在 ensure_play_spec 自动 spec 与 tap-targets 驱动器上;
两者已随取证契约退出便宜档验收链(验收权威 = 四门投影 ∧ 测试 agent 真玩)。其使命——Node 退役授权的
「自动 spec 不比金标驱得差」——已随 Node 路退役完结,不改挂测试员(范式不同,无对照意义)。保留代码
供历史复跑;生产链与批跑不再调用,play-spec 生成器窗口期结束后随其一并清理。
对同一款生成产物,在同一份 staged src/ 上先用生产自动 spec(ensure_play_spec)、后用金标 spec
(inject_golden)各驱动一次九门,逐门比过门率。方向单边(承 plan① M2「关键门不得低于金标基线」):
查的是自动 spec 是否比金标**驱得差**(delta<0 退化),不是查自动比金标松——后者对断言同质的
tap-targets 不成立(M1 已把自动 spec 断言加厚到与金标逐字段一致)。
这道门在 tap-targets 覆盖面内鉴别力有限(delta≈0 几必然),诚实定性为「不退化确认 + 逐款驱动器
推断不失效」,非决定性独立闸;Node 退役授权的强证据是 M1 绝对达标 + 002 两路等价(见 U3 三条齐)。
U1 逐门 delta 判据(本段,纯逻辑可单测);U2 同游戏双驱动 play 真跑;U3 三条齐退役授权判定。
"""
import asyncio
import json
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent)) # → cheap-worker/
import _bootstrap # noqa: E402,F401 跨包 sys.path + 代理旁路 + key 兜底(须在 import cheap_studio 前)
import bake_off # noqa: E402 复用 _clamp_conc 并发夹取口径
import cheap_run # noqa: E402 smoke / ensure_play_spec / play / wg1_game_dir / game_dir
import cheap_studio # noqa: E402 run_studio(run_gates=False)gen-only
import compare_node # noqa: E402 inject_golden / golden_spec_path / gates_from_verdict / src_shape / 端口池
_RESULTS_DIR = Path(__file__).resolve().parent / "results"
_DEFAULT_GENRES = ["click-score", "whack-mole", "shop-serve"]
_clamp_conc = bake_off._clamp_conc # 并发夹取 [1,15] 复用 M1 达标门口径(R4 上限纪律)
# 九门里唯一对驱动器敏感的三门(未驱动时 E_live/H_progress 降 advisory、G_input 无输入 skip):
# 「这游戏到底能不能玩」的判据,自动低于金标即驱动更差,容差 0。
_KEY_GATES = frozenset({"E_live", "H_progress", "G_input"})
_DOUBLE_LOW = 0.5 # 关键门 auto 与 golden 过门率都 <此 → 两路同挂、delta=0 假象 → inconclusive(红)。
_EPS = 1e-9
def _rate(runs: list, side: str, gate: str) -> float:
"""该侧(autoGates/goldenGates)某门在 N 款里的过门率。"""
total = len(runs)
if total == 0:
return 0.0
return round(sum(1 for r in runs if (r.get(side) or {}).get(gate)) / total, 3)
def judge_genre_delta(runs: list) -> dict:
"""单品类逐门 delta 判据。
runs = [{"autoGates": {门: pass_bool}, "goldenGates": {门: pass_bool}}, ...]。
逐门 delta = auto 过门率 - golden 过门率;关键门容差 0(delta≥0)、其余门容差 -1/N(单款 flake)。
double-low 守卫:关键门两路都 <0.5 → inconclusive(红),不被 delta=0 误判 aligned。
status:aligned(不退化)/ regressed(自动驱得差)/ inconclusive(关键门双低)/ insufficient(空样本)。
"""
total = len(runs)
if total == 0:
return {"total": 0, "perGate": {}, "regressedGates": [], "inconclusiveGates": [],
"status": "insufficient", "meets": False}
# 门集 = 各款 autoGates 键的并集(稳健于个别款缺门)。
gate_names = sorted({g for r in runs for g in (r.get("autoGates") or {})})
nonkey_tol = -1.0 / total
per_gate, inconclusive, regressed = {}, [], []
for g in gate_names:
auto_rate = _rate(runs, "autoGates", g)
golden_rate = _rate(runs, "goldenGates", g)
delta = round(auto_rate - golden_rate, 3)
per_gate[g] = {"autoRate": auto_rate, "goldenRate": golden_rate, "delta": delta}
is_key = g in _KEY_GATES
# double-low:关键门两路同低 → 红、不计入 aligned(不论 delta)。
if is_key and auto_rate < _DOUBLE_LOW and golden_rate < _DOUBLE_LOW:
inconclusive.append(g)
continue
tol = 0.0 if is_key else nonkey_tol
if delta < tol - _EPS:
regressed.append(g)
if inconclusive:
status = "inconclusive"
elif regressed:
status = "regressed"
else:
status = "aligned"
return {"total": total, "perGate": per_gate, "regressedGates": regressed,
"inconclusiveGates": inconclusive, "status": status, "meets": status == "aligned"}
def aggregate_delta(genre_runs: dict, required_genres: list) -> dict:
"""按品类聚合 auto-vs-golden delta 判据。
整体 meets = 所有 required 品类都有样本 且 逐品类 aligned(不退化、无关键门双低)。
任一缺样本(missing)/ regressed / inconclusive → 整体 meets=False,卡死品类不被平均掩盖。
"""
per_genre = {k: judge_genre_delta(genre_runs.get(k, [])) for k in required_genres}
missing = [k for k in required_genres if per_genre[k]["total"] == 0]
regressed = [k for k in required_genres if per_genre[k]["status"] == "regressed"]
inconclusive = [k for k in required_genres if per_genre[k]["status"] == "inconclusive"]
meets = (not missing) and (not regressed) and (not inconclusive)
return {
"perGenre": per_genre,
"requiredGenres": required_genres,
"missingGenres": missing,
"regressedGenres": regressed,
"inconclusiveGenres": inconclusive,
"overallMeets": meets,
"meets": meets,
"note": "auto-vs-golden delta 门(同款双驱动 · 关键门容差 0 · 零 LLM);"
"覆盖面内鉴别力有限、定性为不退化确认,退役强证据=M1 达标+002 等价(见 U3)",
}
# ───────────────────────── U3 Node 退役授权三条齐判定(纯逻辑 · 零真跑 · 只读既有报告)─────────────────────────
def _read_json(path):
"""容错读 JSON 报告:文件缺失 / 解析失败 → None(由调用方判「前置证据缺失」、不静默当通过)。"""
p = Path(path)
if not p.exists():
return None
try:
return json.loads(p.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError):
return None
def retire_authorization(m1_path, compare_path, avg_path, required_genres=None) -> dict:
"""Node 退役授权:三个硬前置全绿才出 authorized。
① M1 ≥80% 达标 = M1 收口报告 overallMeets is True(bake-off-M1-final.json)。
② 002 对照等价 = 002 对照报告 summary.equivalentGenres ⊇ required_genres(compare-multi-002-final.json)。
③ M2 auto-vs-golden = U2 报告 meets is True(auto-vs-golden-*.json)。
任一 False / 报告缺失 / 字段类型不符 → authorized=False、missing 列明欠项(不静默当通过)。
纯函数、零 LLM、零真跑(只读既有落盘报告)。
"""
req = list(required_genres) if required_genres else list(_DEFAULT_GENRES)
missing = []
# ① M1 达标。
m1 = _read_json(m1_path)
if m1 is None:
m1_met = False
missing.append("M1 达标:前置证据缺失(报告文件缺失或不可解析)")
elif m1.get("overallMeets") is not True:
m1_met = False
missing.append("M1 达标:overallMeets 非 True(≥80% 达标门未绿)")
else:
m1_met = True
# ② 002 对照等价。
cmp = _read_json(compare_path)
equiv = ((cmp or {}).get("summary") or {}).get("equivalentGenres")
if cmp is None:
parity_ok = False
missing.append("对照等价:前置证据缺失(报告文件缺失或不可解析)")
elif not isinstance(equiv, list) or not set(req).issubset(set(equiv)):
parity_ok = False
lack = [g for g in req if g not in (equiv or [])]
missing.append(f"对照等价:002 equivalentGenres 未覆盖全部品类(缺 {lack})")
else:
parity_ok = True
# ③ M2 auto-vs-golden。
avg = _read_json(avg_path)
if avg is None:
avg_ok = False
missing.append("auto-vs-golden:前置证据缺失(报告文件缺失或不可解析)")
elif avg.get("meets") is not True:
avg_ok = False
missing.append("auto-vs-golden:meets 非 True(逐门 delta 门未绿)")
else:
avg_ok = True
authorized = m1_met and parity_ok and avg_ok
return {
"authorized": authorized,
"conditions": {"m1Met": m1_met, "parityEquivalent": parity_ok, "autoVsGoldenAligned": avg_ok},
"requiredGenres": req,
"missing": missing,
"note": "Node 退役授权三条齐(M1 达标 ∧ 002 对照等价 ∧ auto-vs-golden 不退化);"
"authorized=授权切默认路由的条件齐备,真切路由/D12 扣退在 M3。本机口径,生产复验在 M3。",
}
# ───────────────────────── U2 同游戏双驱动 play 编排(真跑)─────────────────────────
def run_pair_dual_sync(genre: dict, brief: str, gid: str, port: int = 4320, cdp: int = 9222) -> dict:
"""一款 auto-vs-golden:同一份 staged src/ 上先自动 spec、后金标各 play 一次,取两组逐门 verdict。
同步整对(由 run_batch 经 to_thread 并发调度,每对一线程独立端口);隔离生成方差(同款产物 play 两次,
只换驱动器)。编排序:gen-only → smoke 取 state → 断言 staged 无残留 spec → ensure_play_spec(自动 spec)
→ play 取 auto_gates → inject_golden 强制覆写金标 → play 取 golden_gates。
"""
# ① gen-only:run_gates=False 跳过内部 ensure_play_spec + play(cheap_studio.py:153)。
asyncio.run(cheap_studio.run_studio(gid, brief, run_gates=False, port=port, cdp_port=cdp))
# ② smoke → state(供 ensure_play_spec 据 _forensicsView 形态推 driver)。
sm = cheap_run.smoke(gid, port=port, cdp_port=cdp)
state = (sm or {}).get("state")
# ③ 断言 staged 无残留 play-spec.json(ensure_play_spec 是「已存在不覆盖」语义 cheap_run.py:253,
# 残留旧 spec 会被静默复用、污染 auto_gates 且无报错;把隐式依赖变显式前置)。
spec_path = cheap_run.wg1_game_dir(gid) / "play-spec.json"
if spec_path.exists():
raise RuntimeError(f"staged 残留 play-spec.json,auto 路会静默复用旧 spec:{spec_path}")
cheap_run.ensure_play_spec(gid, state)
auto_play = cheap_run.play(gid, port=port, cdp_port=cdp)
auto_gates = compare_node.gates_from_verdict((auto_play or {}).get("verdict"))
# ④ inject 金标强制覆写 → play(此时 auto_gates 已捕获、不被覆盖)。
compare_node.inject_golden(gid, compare_node.golden_spec_path(genre))
golden_play = cheap_run.play(gid, port=port, cdp_port=cdp)
golden_gates = compare_node.gates_from_verdict((golden_play or {}).get("verdict"))
real = compare_node.src_shape(cheap_run.game_dir(gid)).get("realSrcMultifile", False)
return {"gid": gid, "autoGates": auto_gates, "goldenGates": golden_gates, "realSrc": real}
def _next_report_path() -> Path:
"""报告分批次文件、不覆盖历史。"""
_RESULTS_DIR.mkdir(exist_ok=True)
idx = compare_node._next_index([p.stem for p in _RESULTS_DIR.glob("auto-vs-golden-*.json")])
return _RESULTS_DIR / f"auto-vs-golden-{idx}.json"
async def run_batch(genre_keys: list, n: int, conc: int = 1, offset: int = 0,
base_port: int = 4320, base_cdp: int = 9222) -> dict:
"""三品类 × n 同游戏双驱动批跑:逐款 run_pair_dual_sync,按品类聚合逐门 delta 判据。
前台进程内有界并发:端口池(每槽独立 port/cdp,复用 compare_node._port_pool)+ 信号量限并发(≤conc),
每款经 to_thread 独立线程跑。禁后台子代理 / monitor / 自我唤醒重试(R4)。
"""
conc = _clamp_conc(conc)
n = max(1, int(n))
pool = asyncio.Queue()
for pp in compare_node._port_pool(conc, base_port, base_cdp):
pool.put_nowait(pp)
valid, genre_runs = [], {}
for key in genre_keys:
genre = compare_node._GENRE_BY_KEY.get(key)
if genre is None:
print(f"[skip] 未知品类 {key}", file=sys.stderr)
continue
brief = compare_node.load_brief(genre)
if not brief:
print(f"[skip] {key} 无 brief(base {genre['base']} run-summary 缺失)", file=sys.stderr)
continue
valid.append((key, genre, brief))
genre_runs[key] = []
async def one(key, genre, brief, k):
port, cdp = await pool.get()
gid = f"avg-{key}-{offset + k}"
try:
print(f"[{key} {k + 1}/{n}] port={port}/{cdp} 双驱动(自动 spec vs 金标)…", file=sys.stderr)
r = await asyncio.to_thread(run_pair_dual_sync, genre, brief, gid, port, cdp)
genre_runs[key].append(r)
ag, gg = r["autoGates"], r["goldenGates"]
print(f"[{key} {k + 1}/{n}] auto={compare_node._pass_rate(ag)} golden={compare_node._pass_rate(gg)}",
file=sys.stderr)
except Exception as e: # noqa: BLE001 单款失败不拖垮整批
print(f"[{key} {k + 1}/{n}] 异常:{type(e).__name__}: {e}", file=sys.stderr)
finally:
pool.put_nowait((port, cdp))
tasks = [one(key, genre, brief, k) for (key, genre, brief) in valid for k in range(n)]
print(f"[auto-vs-golden] {len(valid)} 品类 × n={n} = {len(tasks)} 款,conc={conc}", file=sys.stderr)
await asyncio.gather(*tasks)
report = aggregate_delta(genre_runs, [k for (k, _, _) in valid])
report["runs"] = genre_runs
out = _next_report_path()
out.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
print(f"[report] → {out}", file=sys.stderr)
_print_report(report)
return report
def _print_report(report: dict) -> None:
print("\n=== auto-vs-golden delta 门(同款双驱动 · 关键门容差 0 · 零 LLM)===", file=sys.stderr)
for k, j in report["perGenre"].items():
extra = ""
if j.get("regressedGates"):
extra = f" 退化门={j['regressedGates']}"
elif j.get("inconclusiveGates"):
extra = f" 双低门={j['inconclusiveGates']}"
print(f"[{k}] {j['status']} (n={j['total']}){extra}", file=sys.stderr)
flag = "不退化 ✅" if report["overallMeets"] else "退化/未达 ❌"
print(f">>> 整体:{flag} 缺样本:{report['missingGenres']} 退化:{report['regressedGenres']} "
f"双低:{report['inconclusiveGenres']}", file=sys.stderr)
print(f">>> 注:{report['note']}\n", file=sys.stderr)
if __name__ == "__main__":
import argparse
ap = argparse.ArgumentParser(description="auto-vs-golden delta 门(同款双驱动 · 按品类 · 零 LLM)")
ap.add_argument("--genres", default="click-score,whack-mole,shop-serve", help="逗号分隔品类键")
ap.add_argument("--n", type=int, default=7, help="每品类款数")
ap.add_argument("--conc", type=int, default=1, help="并发款数(端口池+线程,≤15;本机建议 ≤4)")
ap.add_argument("--offset", type=int, default=0, help="gameId 起始索引(补跑用)")
a = ap.parse_args()
keys = [k.strip() for k in a.genres.split(",") if k.strip()]
rep = asyncio.run(run_batch(keys, a.n, conc=a.conc, offset=a.offset))
sys.exit(0 if rep["overallMeets"] else 1)