接 sim-business 设计指导到生成 agent(cheap_roles+prompt.mjs 双源,先设计后写码+MVP-first);新建纯 LLM 丰富度验证 agent cheap_verify(非阻塞·不进 verdict·零 code-presence 断言,红线落地);bake_off 加 richness 列;cheap_roles 三级回落配置热取(C2a 加载器);cheap_studio trace 接线(C1b)。富游戏 M1 三品类各 5/5=100% 过九门达标。测试 test_roles 12 / test_cheap_verify 11 / test_bake_off 17 / test_cheap_trace 4 全绿。 Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
189 lines
8.1 KiB
Python
189 lines
8.1 KiB
Python
"""
|
|
test_bake_off.py — 便宜档 ≥80% 达标门聚合/判定逻辑单测(M1 U3,mock verdict、不真跑)。
|
|
|
|
守的不变量:
|
|
· 质量口径:达标率分母 = 收敛款(finished=True);编排未收敛(finished=False)剔出、单列 unconverged。
|
|
· 按品类聚合 passed/converged、阈值 0.8(含边界 4/5=0.8 算达标)。
|
|
· 假绿防护:某品类全挂、其余全过 → 整体未达标(不被平均成总体达标)。
|
|
· 覆盖不足:核心品类无收敛样本 → missing、整体未达标。
|
|
· 未收敛剔除:未收敛款不拉低质量达标率,但 rawPassRate 并报(贴近生产真实交付率)。
|
|
· 并发上限夹取 ≤15。
|
|
· 判定零 LLM:judge/aggregate 纯函数,无任何模型调用、同输入恒同输出。
|
|
|
|
跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_bake_off.py
|
|
"""
|
|
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
|
|
import bake_off as B # noqa: E402
|
|
|
|
|
|
def _p(n):
|
|
"""n 款过九门(收敛 + verdict.pass)。"""
|
|
return [{"passed": True, "finished": True} for _ in range(n)]
|
|
|
|
|
|
def _f(n):
|
|
"""n 款九门质量挂(收敛但 verdict.pass=False)。"""
|
|
return [{"passed": False, "finished": True} for _ in range(n)]
|
|
|
|
|
|
def _u(n):
|
|
"""n 款编排未收敛(finished=False,无 verdict)。"""
|
|
return [{"passed": False, "finished": False} for _ in range(n)]
|
|
|
|
|
|
# ───────────────────────── 单品类判定 ─────────────────────────
|
|
|
|
def test_judge_genre_meets_full():
|
|
j = B.judge_genre(_p(5))
|
|
assert j["meets"] is True and j["passRate"] == 1.0 and j["status"] == "meets"
|
|
|
|
|
|
def test_judge_genre_boundary_080():
|
|
"""4/5 = 0.8 恰达线 → 达标。"""
|
|
j = B.judge_genre(_p(4) + _f(1))
|
|
assert j["passRate"] == 0.8 and j["meets"] is True
|
|
|
|
|
|
def test_judge_genre_below():
|
|
"""3/5 = 0.6 < 0.8 → below。"""
|
|
j = B.judge_genre(_p(3) + _f(2))
|
|
assert j["passRate"] == 0.6 and j["meets"] is False and j["status"] == "below"
|
|
|
|
|
|
def test_judge_genre_insufficient():
|
|
j = B.judge_genre([])
|
|
assert j["status"] == "insufficient" and j["meets"] is False and j["total"] == 0
|
|
|
|
|
|
def test_judge_genre_unconverged_excluded_from_denominator():
|
|
"""质量口径核心:11 过 + 1 质量挂 + 2 未收敛 = 质量 11/12=0.917 达标;原始 11/14=0.786。
|
|
创始人 2026-06-27 定:未收敛剔出分母、归 M2,达标门只衡量九门质量地板。"""
|
|
j = B.judge_genre(_p(11) + _f(1) + _u(2))
|
|
assert j["converged"] == 12 and j["unconverged"] == 2 and j["total"] == 14
|
|
assert j["passRate"] == 0.917 and j["meets"] is True
|
|
assert j["rawPassRate"] == 0.786 # 含未收敛、贴近生产真实交付率、不作判据
|
|
|
|
|
|
def test_judge_genre_all_unconverged_is_insufficient():
|
|
"""全未收敛 → 无收敛样本可判质量 → insufficient(不误判达标)。"""
|
|
j = B.judge_genre(_u(5))
|
|
assert j["status"] == "insufficient" and j["meets"] is False and j["converged"] == 0
|
|
|
|
|
|
# ───────────────────────── 整体聚合 ─────────────────────────
|
|
|
|
def test_aggregate_all_meet():
|
|
gp = {"a": _p(5), "b": _p(4) + _f(1), "c": _p(5)}
|
|
r = B.aggregate(gp, ["a", "b", "c"])
|
|
assert r["overallMeets"] is True and not r["belowGenres"] and not r["missingGenres"]
|
|
|
|
|
|
def test_aggregate_one_below_not_averaged():
|
|
"""假绿防护:c 全挂(0/5),a/b 全过 → 整体未达标(不被 a/b 平均掩盖)。"""
|
|
gp = {"a": _p(5), "b": _p(5), "c": _f(5)}
|
|
r = B.aggregate(gp, ["a", "b", "c"])
|
|
assert r["overallMeets"] is False and "c" in r["belowGenres"]
|
|
|
|
|
|
def test_aggregate_missing_genre_is_insufficient():
|
|
"""覆盖不足:c 缺样本 → missing、整体未达标(即便 a/b 全过)。"""
|
|
gp = {"a": _p(5), "b": _p(5)}
|
|
r = B.aggregate(gp, ["a", "b", "c"])
|
|
assert r["overallMeets"] is False and "c" in r["missingGenres"]
|
|
|
|
|
|
def test_aggregate_partial_below_boundary():
|
|
"""b 恰 0.8 达线、a/c 全过 → 整体达标(边界不误判)。"""
|
|
gp = {"a": _p(5), "b": _p(4) + _f(1), "c": _p(5)}
|
|
r = B.aggregate(gp, ["a", "b", "c"])
|
|
assert r["overallMeets"] is True
|
|
|
|
|
|
def test_aggregate_unconverged_total_surfaced():
|
|
"""编排未收敛跨品类汇总 unconvergedTotal、单列归 M2,不影响质量达标判定。"""
|
|
gp = {"a": _p(5) + _u(1), "b": _p(5), "c": _p(5) + _u(2)}
|
|
r = B.aggregate(gp, ["a", "b", "c"])
|
|
assert r["overallMeets"] is True and r["unconvergedTotal"] == 3 and r["sampleTotal"] == 18
|
|
|
|
|
|
# ───────────────────────── 丰富度分布(U-B1 · additive · 与达标正交)─────────────────────────
|
|
|
|
def _pr(n, score):
|
|
"""n 款过九门 + richness 评分 score(非降级)。"""
|
|
return [{"passed": True, "finished": True,
|
|
"richness": {"score": score, "max": 8, "degraded": False}} for _ in range(n)]
|
|
|
|
|
|
def _pd(n):
|
|
"""n 款过九门 + richness 降级(LLM 评分失败/超时,score=None)。"""
|
|
return [{"passed": True, "finished": True,
|
|
"richness": {"score": None, "max": 8, "degraded": True}} for _ in range(n)]
|
|
|
|
|
|
def test_richness_dist_basic():
|
|
"""丰富度分布:非降级款收集 score 算均分/分布,降级款单列计数(纯报告)。"""
|
|
d = B.richness_dist(_pr(2, 6) + _pr(1, 8) + _pd(1))
|
|
assert d["scoredCount"] == 3 and d["degradedCount"] == 1
|
|
assert d["meanScore"] == round((6 + 6 + 8) / 3, 2) and d["scores"] == [6, 6, 8] and d["max"] == 8
|
|
|
|
|
|
def test_richness_dist_all_degraded_or_missing():
|
|
"""全降级 / 无 richness 字段(对照路/老产物)→ meanScore=None、scoredCount=0,不报错。"""
|
|
d = B.richness_dist(_pd(2) + _p(1)) # _p 无 richness 字段
|
|
assert d["scoredCount"] == 0 and d["meanScore"] is None and d["degradedCount"] == 2
|
|
|
|
|
|
def test_aggregate_richness_additive_does_not_change_meets():
|
|
"""红线:richness 是 additive 报告 —— 加 richness 后 overallMeets / perGenre 达标判定与无 richness 时完全一致。"""
|
|
bare = {"a": _p(4) + _f(1)} # 无 richness
|
|
rich = {"a": _pr(4, 7) + _f(1)} # 4 过门款带 richness=7、1 质量挂
|
|
rb = B.aggregate(bare, ["a"])
|
|
rr = B.aggregate(rich, ["a"])
|
|
# 达标判定完全不受 richness 影响(判据只看 verdict.pass,与 richness 正交)
|
|
assert rr["overallMeets"] == rb["overallMeets"] is True
|
|
assert rr["perGenre"]["a"]["passRate"] == rb["perGenre"]["a"]["passRate"] == 0.8
|
|
assert rr["perGenre"]["a"]["meets"] == rb["perGenre"]["a"]["meets"] is True
|
|
# richness 分布 additive 出现,不混进达标判据
|
|
assert rr["richnessByGenre"]["a"]["meanScore"] == 7.0
|
|
assert rb["richnessByGenre"]["a"]["scoredCount"] == 0 # 无 richness 款 → 空分布,不报错
|
|
|
|
|
|
# ───────────────────────── 并发上限 / 零 LLM ─────────────────────────
|
|
|
|
def test_clamp_conc_upper_cap():
|
|
assert B._clamp_conc(100) == 15 # 不可超 15
|
|
assert B._clamp_conc(16) == 15
|
|
|
|
|
|
def test_clamp_conc_lower_and_bad():
|
|
assert B._clamp_conc(0) == 1
|
|
assert B._clamp_conc(3) == 3
|
|
assert B._clamp_conc("x") == 1 # 脏值兜底
|
|
|
|
|
|
def test_judge_is_deterministic_zero_llm():
|
|
"""判定全程纯函数(吃 run dict 出确定结论),不含任何模型调用 —— 同输入恒同输出。"""
|
|
gp = {"a": _p(4) + _f(1)}
|
|
r1 = B.aggregate(gp, ["a"])
|
|
r2 = B.aggregate(gp, ["a"])
|
|
assert r1["perGenre"]["a"] == r2["perGenre"]["a"]
|
|
assert r1["perGenre"]["a"]["passRate"] == 0.8 and r1["overallMeets"] is True
|
|
|
|
|
|
if __name__ == "__main__":
|
|
_fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
|
|
_failed = 0
|
|
for _fn in _fns:
|
|
try:
|
|
_fn()
|
|
print(f" PASS {_fn.__name__}")
|
|
except Exception as e: # noqa: BLE001
|
|
_failed += 1
|
|
print(f" FAIL {_fn.__name__}: {type(e).__name__}: {e}")
|
|
print(f"\n{len(_fns) - _failed}/{len(_fns)} passed")
|
|
sys.exit(1 if _failed else 0)
|