games-development-ai/cheap-worker/tests/test_cheap_verify.py
lili caadfe4c06 feat(cheap): TRPG 品类 rubric(5条·分母独立)+ 金标 play-spec(W-GENRE 品类件④)
品类扩展 rubric(质量模型 SoT §4 规范二·五六):cheap_verify.GENRE_CHECKLISTS['trpg'] 5 条
(掷骰过程可见 L2 / 风险回报取舍 L2 / 成长进判定公式 L2 / 难度爬坡登顶 L3 / 冒险战报炫耀 L4),
每条 0/1 标层 + 一对正反例;仍是喂 LLM judge 的评分尺数据、纯 LLM 非阻塞(红线:不写代码校验
/不进九门/不进 verdict/不进脚手架)。同一次评分 additive 追加品类条目,**分母独立小计**落
richness.genre{key,hits,score,max,groups},通用 11 条 groups/score 分母不混(规范二);
接线 = run_studio 据 scaffold_template 查 GENRE_BY_TEMPLATE 透传 genre,未登记/None 行为
与既有完全一致(无品类路 judge prompt 逐字节不变,锚定纪律)。

金标 play-spec:fixtures/golden-specs/trpg.play-spec.json(tap-targets occupied 反应族
+ score increased + expectLatch,formalize 件② p11c-tpl 已验的 occupied spec;不用货币
递减门——SoT §3 裁定二门级判据不跨档);test_golden_specs 期望集同步纳 trpg(schema/
契约依赖/无坐标绑定 20/20 绿)。

金标复验(规范三,真跑 M3):正例 _template-trpg 骨架 品类 5/5、通用 8/11;薄反例(黑箱
点卡+无骰无取舍无成长无终点)品类 0/5、通用 0/11——正反例拉满、评分尺有鉴别力;同一正例
品类路 vs 基线路通用 11 条漂移 = 0(≤±1,加品类段不扰动通用判定)。单测 +5(条目形状/
独立分母/位置兜底偏移/无品类向后兼容/judge prompt additive)= test_cheap_verify 19/19 绿。

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-02 22:46:45 -07:00

258 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
test_cheap_verify.py — U-A2 便宜档「丰富度」LLM 验证 agent 单测(rubric v2:11 条 + 三分组小计)。
两层:
① parse_judge_output 纯函数(解析 LLM 输出 → 结构化评分 + 三分组小计 + degraded 兜底)——无 agentscope、零网络,
plain python 即可跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_cheap_verify.py
② verify_richness 非阻塞契约(fake model 注入,零真网络)——验证「LLM 失败/超时/解析失败/空 src → degraded、
绝不抛、绝不阻断」。需 agentscope.message(venv 有),缺则自动跳过。
rubric v2(质量模型 SoT §3.3):通用底座 11 条,分层 L2×6 / L3×3 / L4×2;
逐条 0/1 标层 + L2/L3/L4 三分组小计;score 保留为向后兼容字段(bake_off 读),非单一总分。
跑:cheap-worker/.venv/bin/python -m pytest cheap-worker/tests/test_cheap_verify.py -q
"""
import asyncio
import json
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
import cheap_verify # noqa: E402
try:
import agentscope.message # noqa: E402,F401
_HAS_AS = True
except Exception: # noqa: BLE001 无 agentscope 时 ② 类测试跳过(纯函数 ① 不受影响)
_HAS_AS = False
def _full_checks(hits):
"""据布尔列表造 checks(name 用 cheap_verify 的标准名);hits 短于 11 条则 zip 取最短(余条判 False)。"""
return [{"name": n, "hit": h, "why": "理由"} for (n, _l, _m), h in zip(cheap_verify.RICHNESS_CHECKLIST, hits)]
# 通用底座 v2 层索引(据 RICHNESS_CHECKLIST 顺序):L2={0,1,3,4,5,7} L3={2,6,8} L4={9,10}
def test_checklist_shape_v2():
"""底座 = 11 条,分层 L2×6 / L3×3 / L4×2(分组归属对齐 §3.3 第 1 条)。"""
cl = cheap_verify.RICHNESS_CHECKLIST
assert len(cl) == 11 and cheap_verify._MAX == 11
layers = [layer for _n, layer, _m in cl]
assert layers.count("L2") == 6 and layers.count("L3") == 3 and layers.count("L4") == 2
# 原 8 条 name/顺序锁定(锚定纪律:评分尺变更不得动原条目)
assert [n for n, _l, _m in cl][:8] == [
"即时反馈", "可见成长", "下一个解锁", "30秒爽点", "数值滚雪球", "情感锚", "放置回归", "音反馈"]
# ───────────────────────── ① parse_judge_output 纯函数(无 agentscope/无网络)─────────────────────────
def test_parse_valid_full():
"""合法 JSON(喂前 8 条命中态)→ 命中数正确、非 degraded、hits 带 layer、groups 三分组小计落位。"""
text = json.dumps({"checks": _full_checks([True, True, False, True, False, False, False, True]),
"notes": "还行"}, ensure_ascii=False)
r = cheap_verify.parse_judge_output(text)
assert r["degraded"] is False
assert r["score"] == 4 and r["max"] == 11
assert len(r["hits"]) == 11 and r["hits"][0]["name"] == "即时反馈" and r["hits"][0]["layer"] == "L2"
# 前 8 条命中态 [T,T,F,T,F,F,F,T]:L2(0,1,3,4,5,7)=4;L3(2,6,8=F,F,未喂F)=0;L4(9,10 未喂)=0
assert r["groups"] == {"L2": {"score": 4, "max": 6}, "L3": {"score": 0, "max": 3}, "L4": {"score": 0, "max": 2}}
assert r["notes"] == "还行"
def test_parse_11_full_and_groups():
"""11 条全命中 → score=11、三分组小计满分(L2 6/6 · L3 3/3 · L4 2/2)。"""
r = cheap_verify.parse_judge_output(json.dumps({"checks": _full_checks([True] * 11), "notes": "满"}))
assert r["degraded"] is False and r["score"] == 11 and r["max"] == 11
assert r["groups"] == {"L2": {"score": 6, "max": 6}, "L3": {"score": 3, "max": 3}, "L4": {"score": 2, "max": 2}}
def test_parse_groups_by_layer():
"""按层聚合正确:命中态 [T,T,F,T,F,F,F,T,F,T,T] → L2=4 L3=0 L4=2、score=6。"""
pat = [True, True, False, True, False, False, False, True, False, True, True]
r = cheap_verify.parse_judge_output(json.dumps({"checks": _full_checks(pat)}))
assert r["score"] == 6
assert r["groups"]["L2"] == {"score": 4, "max": 6}
assert r["groups"]["L3"] == {"score": 0, "max": 3}
assert r["groups"]["L4"] == {"score": 2, "max": 2}
def test_parse_fenced_json():
"""markdown 围栏 + 前后赘语 → json_repair 兜得住、正确解析。"""
body = json.dumps({"checks": _full_checks([True] * 11), "notes": "满"}, ensure_ascii=False)
text = "这是我的评分:\n```json\n" + body + "\n```\n以上。"
r = cheap_verify.parse_judge_output(text)
assert r["degraded"] is False and r["score"] == 11
def test_parse_garbage_degraded():
"""完全非 JSON → degraded(score=None, groups=None),不抛。"""
r = cheap_verify.parse_judge_output("完全不是 JSON 的一段话")
assert r["degraded"] is True and r["score"] is None and r["max"] == 11 and r["groups"] is None
def test_parse_none_and_empty_degraded():
"""None / 空串 → degraded。"""
assert cheap_verify.parse_judge_output(None)["degraded"] is True
assert cheap_verify.parse_judge_output("")["degraded"] is True
assert cheap_verify.parse_judge_output(" ")["degraded"] is True
def test_parse_non_dict_degraded():
"""JSON 数组(非对象)→ degraded。"""
assert cheap_verify.parse_judge_output("[1,2,3]")["degraded"] is True
def test_parse_missing_checks_degraded():
"""有 JSON 对象但缺 checks 数组 → degraded(不静默判 0 分)。"""
assert cheap_verify.parse_judge_output(json.dumps({"notes": "x"}))["degraded"] is True
assert cheap_verify.parse_judge_output(json.dumps({"checks": []}))["degraded"] is True
def test_parse_coerce_and_positional():
"""hit 用字符串/数字 + name 对不上时按位置兜底(容忍模型改名)。"""
checks = [{"hit": "true", "why": "a"}, {"hit": 1}, {"hit": "否"}] + [{"hit": False}] * 5
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}))
assert r["degraded"] is False
assert r["score"] == 2 # 前两条 "true"/1 命中,第三条「否」不命中,其余 False
assert r["max"] == 11 and r["hits"][0]["layer"] == "L2" # 位置兜底也带 checklist 权威 layer
# ───────────────────────── ② verify_richness 非阻塞契约(fake model,零真网络)─────────────────────────
class _FakeResp:
"""mock ChatResponse:content 用 dict 文本块(_extract_text 兼容 dict 块)。"""
def __init__(self, text):
self.content = [{"type": "text", "text": text}]
class _FakeModel:
"""async 可调用 mock LLM:返回固定文本 / 抛异常 / 睡眠(测超时)。usage_sum 仿 RecordingModel。"""
def __init__(self, text="", raise_exc=None, sleep=0.0):
self._text, self._raise, self._sleep = text, raise_exc, sleep
async def __call__(self, messages, **kw):
if self._sleep:
await asyncio.sleep(self._sleep)
if self._raise:
raise self._raise
return _FakeResp(self._text)
def usage_sum(self):
return (11, 22)
_SRC = "// game-logic.js\nfunction createGame(){ /* 进货 库存 解锁 playSfx 飘字 结算 remix */ }"
def _skip_no_as(name):
print(f" SKIP {name}(无 agentscope)")
def test_verify_happy_path():
"""fake model 返回合法 JSON → 结构化评分 + 三分组小计 + judgeTokens;非 degraded。"""
if not _HAS_AS:
return _skip_no_as("test_verify_happy_path")
body = json.dumps({"checks": _full_checks([True] * 5 + [False] * 6), "notes": "不错"}, ensure_ascii=False)
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC, model=_FakeModel(text=body)))
assert r["degraded"] is False and r["score"] == 5
# 前 5 条命中 [T,T,T,T,T] 均属 L2 索引{0,1,3,4}? idx0,1L2 idx2L3 idx3,4L2 → L2=4 L3=1
assert r["groups"]["L2"]["score"] == 4 and r["groups"]["L3"]["score"] == 1 and r["groups"]["L4"]["score"] == 0
assert r["judgeTokens"]["total"] == 33
def test_verify_model_raises_degraded():
"""LLM 抛异常 → degraded(绝不抛、绝不阻断)。"""
if not _HAS_AS:
return _skip_no_as("test_verify_model_raises_degraded")
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC, model=_FakeModel(raise_exc=RuntimeError("boom"))))
assert r["degraded"] is True and r["score"] is None
def test_verify_timeout_degraded():
"""LLM 超时 → degraded(不卡死收口)。"""
if not _HAS_AS:
return _skip_no_as("test_verify_timeout_degraded")
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC,
model=_FakeModel(text="{}", sleep=0.3), timeout=0.05))
assert r["degraded"] is True
def test_verify_empty_sources_skips_llm():
"""空 src → 不调 LLM 直接 degraded(model 给个会抛的,证明它没被调用)。"""
if not _HAS_AS:
return _skip_no_as("test_verify_empty_sources_skips_llm")
r = asyncio.run(cheap_verify.verify_richness("x", sources=" ",
model=_FakeModel(raise_exc=RuntimeError("不该被调用"))))
assert r["degraded"] is True
assert "跳过" in r.get("reason", "") # 走的是「空 src 跳过」路径,而非 LLM 抛异常路径
# ── W-GENRE 品类件④:品类扩展 rubric(分母独立小计)────────────────────────────
def test_genre_checklist_trpg_shape():
"""trpg 品类扩展 ≥4 条、形状同通用底座 (名, 层标, 含义含正反例)、层标合法(§4 规范二)。"""
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
assert len(gc) >= 4, f"品类扩展应 ≥4 条,得 {len(gc)}"
for name, layer, meaning in gc:
assert isinstance(name, str) and name, "标准名非空"
assert layer in ("L2", "L3", "L4"), f"层标非法:{layer}"
assert "正:" in meaning and "反:" in meaning, f"「{name}」含义应带一对正反例(规范三锚定)"
assert cheap_verify.GENRE_BY_TEMPLATE["_template-trpg"] == "trpg"
def test_parse_with_genre_checklist_independent_denominator():
"""品类段独立分母:通用 score/groups 不被品类条目污染;richness.genre 形状齐全。"""
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
checks = [{"name": n, "hit": True, "why": "w"} for (n, _l, _m) in cheap_verify.RICHNESS_CHECKLIST]
checks += [{"name": n, "hit": (i != len(gc) - 1), "why": "w"} for i, (n, _l, _m) in enumerate(gc)]
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks, "notes": "n"}),
genre_checklist=gc, genre_key="trpg")
assert r["score"] == 11 and r["max"] == 11, "通用分母被品类条目污染"
g = r["genre"]
assert g["key"] == "trpg" and g["max"] == len(gc) and g["score"] == len(gc) - 1
total = sum(v["max"] for v in g["groups"].values())
assert total == len(gc), "品类分组小计分母应 = 品类条目数"
def test_parse_genre_positional_fallback_offset():
"""品类段位置兜底:name 全对不上时按「通用在前、品类紧随」的偏移对位。"""
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
n = 11 + len(gc)
checks = [{"name": f"未知{i}", "hit": True, "why": ""} for i in range(n)]
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}),
genre_checklist=gc, genre_key="trpg")
assert r["score"] == 11 and r["genre"]["score"] == len(gc)
def test_parse_no_genre_backcompat_no_genre_key():
"""不传品类:输出形状与既有完全一致(无 genre 键——老消费面零感知)。"""
checks = [{"name": n, "hit": True, "why": "w"} for (n, _l, _m) in cheap_verify.RICHNESS_CHECKLIST]
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}))
assert "genre" not in r and r["score"] == 11
def test_build_judge_user_genre_block_additive():
"""user prompt:无品类段落基线不变;有品类只 additive 加品类清单段 + 条数契约(11+N)。"""
base = cheap_verify._build_judge_user("SRC", "B")
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
withg = cheap_verify._build_judge_user("SRC", "B", genre_key="trpg", genre_checklist=gc)
assert "品类扩展清单" not in base and "共 11 条" in base
assert "trpg 品类扩展清单" in withg and f"共 {11 + len(gc)} 条" in withg
for name, _l, _m in gc:
assert name in withg, f"品类条目「{name}」应出现在 judge prompt"
if __name__ == "__main__":
_fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
_failed = 0
for _fn in _fns:
try:
_fn()
print(f" PASS {_fn.__name__}")
except AssertionError as e:
_failed += 1
print(f" FAIL {_fn.__name__}: {e}")
print(f"\n{len(_fns) - _failed}/{len(_fns)} passed")
sys.exit(1 if _failed else 0)