games-development-ai/cheap-worker/tests/test_cheap_verify.py

290 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
test_cheap_verify.py — U-A2 便宜档「丰富度」LLM 验证 agent 单测(rubric v2:11 条 + 三分组小计)。
两层:
① parse_judge_output 纯函数(解析 LLM 输出 → 结构化评分 + 三分组小计 + degraded 兜底)——无 agentscope、零网络,
plain python 即可跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_cheap_verify.py
② verify_richness 非阻塞契约(fake model 注入,零真网络)——验证「LLM 失败/超时/解析失败/空 src → degraded、
绝不抛、绝不阻断」。需 agentscope.message(venv 有),缺则自动跳过。
rubric v2(质量模型 SoT §3.3):通用底座 11 条,分层 L2×6 / L3×3 / L4×2;
逐条 0/1 标层 + L2/L3/L4 三分组小计;score 保留为向后兼容字段(bake_off 读),非单一总分。
跑:cheap-worker/.venv/bin/python -m pytest cheap-worker/tests/test_cheap_verify.py -q
"""
import asyncio
import json
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
import cheap_verify # noqa: E402
try:
import agentscope.message # noqa: E402,F401
_HAS_AS = True
except Exception: # noqa: BLE001 无 agentscope 时 ② 类测试跳过(纯函数 ① 不受影响)
_HAS_AS = False
def _full_checks(hits):
"""据布尔列表造 checks(name 用 cheap_verify 的标准名);hits 短于 11 条则 zip 取最短(余条判 False)。"""
return [{"name": n, "hit": h, "why": "理由"} for (n, _l, _m), h in zip(cheap_verify.RICHNESS_CHECKLIST, hits)]
# 通用底座层索引(据 RICHNESS_CHECKLIST 顺序):L2={0,1,3,4,5,7,11} L3={2,6,8} L4={9,10}
def test_checklist_shape_v2():
"""底座 = 12 条(2026-07-03 ⑨升第 12 条「核心操作非无脑」),分层 L2×7 / L3×3 / L4×2(质量 SoT §4 规范一 12 条口径)。"""
cl = cheap_verify.RICHNESS_CHECKLIST
assert len(cl) == 12 and cheap_verify._MAX == 12
layers = [layer for _n, layer, _m in cl]
assert layers.count("L2") == 7 and layers.count("L3") == 3 and layers.count("L4") == 2
# 原 8 条 name/顺序锁定(锚定纪律:评分尺变更不得动原条目)
assert [n for n, _l, _m in cl][:8] == [
"即时反馈", "可见成长", "下一个解锁", "30秒爽点", "数值滚雪球", "情感锚", "放置回归", "音反馈"]
# 第 12 条钉名+钉位(追加于末尾、层 L2;锚定纪律=前 11 条一字不动)
assert cl[11][0] == "核心操作非无脑" and cl[11][1] == "L2"
# ───────────────────────── ① parse_judge_output 纯函数(无 agentscope/无网络)─────────────────────────
def test_parse_valid_full():
"""合法 JSON(喂前 8 条命中态)→ 命中数正确、非 degraded、hits 带 layer、groups 三分组小计落位。"""
text = json.dumps({"checks": _full_checks([True, True, False, True, False, False, False, True]),
"notes": "还行"}, ensure_ascii=False)
r = cheap_verify.parse_judge_output(text)
assert r["degraded"] is False
assert r["score"] == 4 and r["max"] == 12
assert len(r["hits"]) == 12 and r["hits"][0]["name"] == "即时反馈" and r["hits"][0]["layer"] == "L2"
# 前 8 条命中态 [T,T,F,T,F,F,F,T]:L2(0,1,3,4,5,7=4 中,11 未喂 F)=4;L3(2,6,8=F,F,未喂F)=0;L4(9,10 未喂)=0
assert r["groups"] == {"L2": {"score": 4, "max": 7}, "L3": {"score": 0, "max": 3}, "L4": {"score": 0, "max": 2}}
assert r["notes"] == "还行"
def test_parse_11_full_and_groups():
"""12 条全命中 → score=12、三分组小计满分(L2 7/7 · L3 3/3 · L4 2/2)。"""
r = cheap_verify.parse_judge_output(json.dumps({"checks": _full_checks([True] * 12), "notes": "满"}))
assert r["degraded"] is False and r["score"] == 12 and r["max"] == 12
assert r["groups"] == {"L2": {"score": 7, "max": 7}, "L3": {"score": 3, "max": 3}, "L4": {"score": 2, "max": 2}}
def test_parse_groups_by_layer():
"""按层聚合正确:命中态 [T,T,F,T,F,F,F,T,F,T,T] → L2=4 L3=0 L4=2、score=6。"""
pat = [True, True, False, True, False, False, False, True, False, True, True]
r = cheap_verify.parse_judge_output(json.dumps({"checks": _full_checks(pat)}))
assert r["score"] == 6
assert r["groups"]["L2"] == {"score": 4, "max": 7}
assert r["groups"]["L3"] == {"score": 0, "max": 3}
assert r["groups"]["L4"] == {"score": 2, "max": 2}
def test_parse_fenced_json():
"""markdown 围栏 + 前后赘语 → json_repair 兜得住、正确解析。"""
body = json.dumps({"checks": _full_checks([True] * 12), "notes": "满"}, ensure_ascii=False)
text = "这是我的评分:\n```json\n" + body + "\n```\n以上。"
r = cheap_verify.parse_judge_output(text)
assert r["degraded"] is False and r["score"] == 12
def test_parse_garbage_degraded():
"""完全非 JSON → degraded(score=None, groups=None),不抛。"""
r = cheap_verify.parse_judge_output("完全不是 JSON 的一段话")
assert r["degraded"] is True and r["score"] is None and r["max"] == 12 and r["groups"] is None
def test_parse_none_and_empty_degraded():
"""None / 空串 → degraded。"""
assert cheap_verify.parse_judge_output(None)["degraded"] is True
assert cheap_verify.parse_judge_output("")["degraded"] is True
assert cheap_verify.parse_judge_output(" ")["degraded"] is True
def test_parse_non_dict_degraded():
"""JSON 数组(非对象)→ degraded。"""
assert cheap_verify.parse_judge_output("[1,2,3]")["degraded"] is True
def test_parse_missing_checks_degraded():
"""有 JSON 对象但缺 checks 数组 → degraded(不静默判 0 分)。"""
assert cheap_verify.parse_judge_output(json.dumps({"notes": "x"}))["degraded"] is True
assert cheap_verify.parse_judge_output(json.dumps({"checks": []}))["degraded"] is True
def test_parse_coerce_and_positional():
"""hit 用字符串/数字 + name 对不上时按位置兜底(容忍模型改名)。"""
checks = [{"hit": "true", "why": "a"}, {"hit": 1}, {"hit": "否"}] + [{"hit": False}] * 5
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}))
assert r["degraded"] is False
assert r["score"] == 2 # 前两条 "true"/1 命中,第三条「否」不命中,其余 False
assert r["max"] == 12 and r["hits"][0]["layer"] == "L2" # 位置兜底也带 checklist 权威 layer
# ───────────────────────── ② verify_richness 非阻塞契约(fake model,零真网络)─────────────────────────
class _FakeResp:
"""mock ChatResponse:content 用 dict 文本块(_extract_text 兼容 dict 块)。"""
def __init__(self, text):
self.content = [{"type": "text", "text": text}]
class _FakeModel:
"""async 可调用 mock LLM:返回固定文本 / 抛异常 / 睡眠(测超时)。usage_sum 仿 RecordingModel。"""
def __init__(self, text="", raise_exc=None, sleep=0.0):
self._text, self._raise, self._sleep = text, raise_exc, sleep
async def __call__(self, messages, **kw):
if self._sleep:
await asyncio.sleep(self._sleep)
if self._raise:
raise self._raise
return _FakeResp(self._text)
def usage_sum(self):
return (11, 22)
_SRC = "// game-logic.js\nfunction createGame(){ /* 进货 库存 解锁 playSfx 飘字 结算 remix */ }"
def _skip_no_as(name):
print(f" SKIP {name}(无 agentscope)")
def test_verify_happy_path():
"""fake model 返回合法 JSON → 结构化评分 + 三分组小计 + judgeTokens;非 degraded。"""
if not _HAS_AS:
return _skip_no_as("test_verify_happy_path")
body = json.dumps({"checks": _full_checks([True] * 5 + [False] * 6), "notes": "不错"}, ensure_ascii=False)
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC, model=_FakeModel(text=body)))
assert r["degraded"] is False and r["score"] == 5
# 前 5 条命中 [T,T,T,T,T] 均属 L2 索引{0,1,3,4}? idx0,1L2 idx2L3 idx3,4L2 → L2=4 L3=1
assert r["groups"]["L2"]["score"] == 4 and r["groups"]["L3"]["score"] == 1 and r["groups"]["L4"]["score"] == 0
assert r["judgeTokens"]["total"] == 33
def test_verify_model_raises_degraded():
"""LLM 抛异常 → degraded(绝不抛、绝不阻断)。"""
if not _HAS_AS:
return _skip_no_as("test_verify_model_raises_degraded")
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC, model=_FakeModel(raise_exc=RuntimeError("boom"))))
assert r["degraded"] is True and r["score"] is None
def test_verify_timeout_degraded():
"""LLM 超时 → degraded(不卡死收口)。"""
if not _HAS_AS:
return _skip_no_as("test_verify_timeout_degraded")
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC,
model=_FakeModel(text="{}", sleep=0.3), timeout=0.05))
assert r["degraded"] is True
def test_verify_empty_sources_skips_llm():
"""空 src → 不调 LLM 直接 degraded(model 给个会抛的,证明它没被调用)。"""
if not _HAS_AS:
return _skip_no_as("test_verify_empty_sources_skips_llm")
r = asyncio.run(cheap_verify.verify_richness("x", sources=" ",
model=_FakeModel(raise_exc=RuntimeError("不该被调用"))))
assert r["degraded"] is True
assert "跳过" in r.get("reason", "") # 走的是「空 src 跳过」路径,而非 LLM 抛异常路径
# ── W-GENRE 品类件④:品类扩展 rubric(分母独立小计)────────────────────────────
def test_genre_checklist_trpg_shape():
"""trpg 品类扩展(fixture 单源)≥4 条、形状同通用底座 (名, 层标, 含义含正反例)、层标合法(§4 规范二)。"""
gc = cheap_verify.load_genre_checklist("trpg")
assert len(gc) >= 4, f"品类扩展应 ≥4 条,得 {len(gc)}"
for name, layer, meaning in gc:
assert isinstance(name, str) and name, "标准名非空"
assert layer in ("L2", "L3", "L4"), f"层标非法:{layer}"
assert "正:" in meaning and "反:" in meaning, f"「{name}」含义应带一对正反例(规范三锚定)"
# 条目名/顺序锁定 = 内联时代原文(caadfe4c);迁出 fixture 后锚定不动(锚定纪律,同通用 8 条锁定)
assert [n for n, _l, _m in gc] == ["掷骰过程可见", "风险回报取舍", "成长进判定公式", "难度爬坡登顶", "冒险战报炫耀"]
assert cheap_verify.GENRE_BY_TEMPLATE["_template-trpg"] == "trpg"
def test_inline_genre_checklists_retired():
"""rubric 挂点归一不变量:内联 GENRE_CHECKLISTS 已退役,品类条目只剩 fixtures/genre-rubrics/ 一路(防两路复活)。"""
assert not hasattr(cheap_verify, "GENRE_CHECKLISTS"), "内联品类字典应已退役(单一数据源=fixture)"
def test_genre_checklist_heritage_wired():
"""非遗接线(rubric 挂点归一):fixture 单源可加载、条目名锁定、_template-feiyi 登记品类键 heritage。
品类键收敛口径 = 'heritage'(rubric/GENRE_BY_TEMPLATE 统一);模板目录 _template-feiyi、金标
spec _genre='heritage-craft' 是同一品类的资产名/金标名,映射见 heritage.json 的 _note。
"""
gc = cheap_verify.load_genre_checklist("heritage")
assert len(gc) >= 4, f"品类扩展应 ≥4 条,得 {len(gc)}"
for name, layer, meaning in gc:
assert isinstance(name, str) and name and layer in ("L2", "L3", "L4")
assert isinstance(meaning, str) and meaning
# 条目名/顺序锁定 = heritage.rubric.json 时代原文(39cc2c09);规整改名(entries→items)后锚定不动
assert [n for n, _l, _m in gc] == ["工序链成立", "节律张力", "技艺成长可见", "作品可展示", "文化氛围一致"]
assert cheap_verify.GENRE_BY_TEMPLATE["_template-feiyi"] == "heritage"
# fixture 形状:每条带一对正反例(金标锚材料;与 puzzle 形状门同语义)
p = Path(cheap_verify.__file__).resolve().parent / "fixtures" / "genre-rubrics" / "heritage.json"
data = json.loads(p.read_text(encoding="utf-8"))
assert data.get("genre") == "heritage"
for it in data["items"]:
assert isinstance(it.get("positive"), str) and it["positive"], f"条目 {it.get('name')} 缺正例"
assert isinstance(it.get("negative"), str) and it["negative"], f"条目 {it.get('name')} 缺反例"
def test_parse_with_genre_checklist_independent_denominator():
"""品类段独立分母:通用 score/groups 不被品类条目污染;richness.genre 形状齐全。"""
gc = cheap_verify.load_genre_checklist("trpg")
checks = [{"name": n, "hit": True, "why": "w"} for (n, _l, _m) in cheap_verify.RICHNESS_CHECKLIST]
checks += [{"name": n, "hit": (i != len(gc) - 1), "why": "w"} for i, (n, _l, _m) in enumerate(gc)]
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks, "notes": "n"}),
genre_checklist=gc, genre_key="trpg")
assert r["score"] == 12 and r["max"] == 12, "通用分母被品类条目污染"
g = r["genre"]
assert g["key"] == "trpg" and g["max"] == len(gc) and g["score"] == len(gc) - 1
total = sum(v["max"] for v in g["groups"].values())
assert total == len(gc), "品类分组小计分母应 = 品类条目数"
def test_parse_genre_positional_fallback_offset():
"""品类段位置兜底:name 全对不上时按「通用在前、品类紧随」的偏移对位。"""
gc = cheap_verify.load_genre_checklist("trpg")
n = len(cheap_verify.RICHNESS_CHECKLIST) + len(gc)
checks = [{"name": f"未知{i}", "hit": True, "why": ""} for i in range(n)]
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}),
genre_checklist=gc, genre_key="trpg")
assert r["score"] == 12 and r["genre"]["score"] == len(gc)
def test_parse_no_genre_backcompat_no_genre_key():
"""不传品类:输出形状与既有完全一致(无 genre 键——老消费面零感知)。"""
checks = [{"name": n, "hit": True, "why": "w"} for (n, _l, _m) in cheap_verify.RICHNESS_CHECKLIST]
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}))
assert "genre" not in r and r["score"] == 12
def test_build_judge_user_genre_block_additive():
"""user prompt:无品类段落基线不变;有品类只 additive 加品类清单段 + 条数契约(11+N)。"""
base = cheap_verify._build_judge_user("SRC", "B")
gc = cheap_verify.load_genre_checklist("trpg")
withg = cheap_verify._build_judge_user("SRC", "B", genre_key="trpg", genre_checklist=gc)
assert "品类扩展清单" not in base and "共 12 条" in base
assert "trpg 品类扩展清单" in withg and f"共 {12 + len(gc)} 条" in withg
for name, _l, _m in gc:
assert name in withg, f"品类条目「{name}」应出现在 judge prompt"
if __name__ == "__main__":
_fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
_failed = 0
for _fn in _fns:
try:
_fn()
print(f" PASS {_fn.__name__}")
except AssertionError as e:
_failed += 1
print(f" FAIL {_fn.__name__}: {e}")
print(f"\n{len(_fns) - _failed}/{len(_fns)} passed")
sys.exit(1 if _failed else 0)