品类扩展 rubric(质量模型 SoT §4 规范二·五六):cheap_verify.GENRE_CHECKLISTS['trpg'] 5 条
(掷骰过程可见 L2 / 风险回报取舍 L2 / 成长进判定公式 L2 / 难度爬坡登顶 L3 / 冒险战报炫耀 L4),
每条 0/1 标层 + 一对正反例;仍是喂 LLM judge 的评分尺数据、纯 LLM 非阻塞(红线:不写代码校验
/不进九门/不进 verdict/不进脚手架)。同一次评分 additive 追加品类条目,**分母独立小计**落
richness.genre{key,hits,score,max,groups},通用 11 条 groups/score 分母不混(规范二);
接线 = run_studio 据 scaffold_template 查 GENRE_BY_TEMPLATE 透传 genre,未登记/None 行为
与既有完全一致(无品类路 judge prompt 逐字节不变,锚定纪律)。
金标 play-spec:fixtures/golden-specs/trpg.play-spec.json(tap-targets occupied 反应族
+ score increased + expectLatch,formalize 件② p11c-tpl 已验的 occupied spec;不用货币
递减门——SoT §3 裁定二门级判据不跨档);test_golden_specs 期望集同步纳 trpg(schema/
契约依赖/无坐标绑定 20/20 绿)。
金标复验(规范三,真跑 M3):正例 _template-trpg 骨架 品类 5/5、通用 8/11;薄反例(黑箱
点卡+无骰无取舍无成长无终点)品类 0/5、通用 0/11——正反例拉满、评分尺有鉴别力;同一正例
品类路 vs 基线路通用 11 条漂移 = 0(≤±1,加品类段不扰动通用判定)。单测 +5(条目形状/
独立分母/位置兜底偏移/无品类向后兼容/judge prompt additive)= test_cheap_verify 19/19 绿。
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
258 lines
13 KiB
Python
258 lines
13 KiB
Python
"""
|
||
test_cheap_verify.py — U-A2 便宜档「丰富度」LLM 验证 agent 单测(rubric v2:11 条 + 三分组小计)。
|
||
|
||
两层:
|
||
① parse_judge_output 纯函数(解析 LLM 输出 → 结构化评分 + 三分组小计 + degraded 兜底)——无 agentscope、零网络,
|
||
plain python 即可跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_cheap_verify.py
|
||
② verify_richness 非阻塞契约(fake model 注入,零真网络)——验证「LLM 失败/超时/解析失败/空 src → degraded、
|
||
绝不抛、绝不阻断」。需 agentscope.message(venv 有),缺则自动跳过。
|
||
|
||
rubric v2(质量模型 SoT §3.3):通用底座 11 条,分层 L2×6 / L3×3 / L4×2;
|
||
逐条 0/1 标层 + L2/L3/L4 三分组小计;score 保留为向后兼容字段(bake_off 读),非单一总分。
|
||
|
||
跑:cheap-worker/.venv/bin/python -m pytest cheap-worker/tests/test_cheap_verify.py -q
|
||
"""
|
||
|
||
import asyncio
|
||
import json
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
|
||
import cheap_verify # noqa: E402
|
||
|
||
try:
|
||
import agentscope.message # noqa: E402,F401
|
||
_HAS_AS = True
|
||
except Exception: # noqa: BLE001 无 agentscope 时 ② 类测试跳过(纯函数 ① 不受影响)
|
||
_HAS_AS = False
|
||
|
||
|
||
def _full_checks(hits):
|
||
"""据布尔列表造 checks(name 用 cheap_verify 的标准名);hits 短于 11 条则 zip 取最短(余条判 False)。"""
|
||
return [{"name": n, "hit": h, "why": "理由"} for (n, _l, _m), h in zip(cheap_verify.RICHNESS_CHECKLIST, hits)]
|
||
|
||
|
||
# 通用底座 v2 层索引(据 RICHNESS_CHECKLIST 顺序):L2={0,1,3,4,5,7} L3={2,6,8} L4={9,10}
|
||
def test_checklist_shape_v2():
|
||
"""底座 = 11 条,分层 L2×6 / L3×3 / L4×2(分组归属对齐 §3.3 第 1 条)。"""
|
||
cl = cheap_verify.RICHNESS_CHECKLIST
|
||
assert len(cl) == 11 and cheap_verify._MAX == 11
|
||
layers = [layer for _n, layer, _m in cl]
|
||
assert layers.count("L2") == 6 and layers.count("L3") == 3 and layers.count("L4") == 2
|
||
# 原 8 条 name/顺序锁定(锚定纪律:评分尺变更不得动原条目)
|
||
assert [n for n, _l, _m in cl][:8] == [
|
||
"即时反馈", "可见成长", "下一个解锁", "30秒爽点", "数值滚雪球", "情感锚", "放置回归", "音反馈"]
|
||
|
||
|
||
# ───────────────────────── ① parse_judge_output 纯函数(无 agentscope/无网络)─────────────────────────
|
||
|
||
def test_parse_valid_full():
|
||
"""合法 JSON(喂前 8 条命中态)→ 命中数正确、非 degraded、hits 带 layer、groups 三分组小计落位。"""
|
||
text = json.dumps({"checks": _full_checks([True, True, False, True, False, False, False, True]),
|
||
"notes": "还行"}, ensure_ascii=False)
|
||
r = cheap_verify.parse_judge_output(text)
|
||
assert r["degraded"] is False
|
||
assert r["score"] == 4 and r["max"] == 11
|
||
assert len(r["hits"]) == 11 and r["hits"][0]["name"] == "即时反馈" and r["hits"][0]["layer"] == "L2"
|
||
# 前 8 条命中态 [T,T,F,T,F,F,F,T]:L2(0,1,3,4,5,7)=4;L3(2,6,8=F,F,未喂F)=0;L4(9,10 未喂)=0
|
||
assert r["groups"] == {"L2": {"score": 4, "max": 6}, "L3": {"score": 0, "max": 3}, "L4": {"score": 0, "max": 2}}
|
||
assert r["notes"] == "还行"
|
||
|
||
|
||
def test_parse_11_full_and_groups():
|
||
"""11 条全命中 → score=11、三分组小计满分(L2 6/6 · L3 3/3 · L4 2/2)。"""
|
||
r = cheap_verify.parse_judge_output(json.dumps({"checks": _full_checks([True] * 11), "notes": "满"}))
|
||
assert r["degraded"] is False and r["score"] == 11 and r["max"] == 11
|
||
assert r["groups"] == {"L2": {"score": 6, "max": 6}, "L3": {"score": 3, "max": 3}, "L4": {"score": 2, "max": 2}}
|
||
|
||
|
||
def test_parse_groups_by_layer():
|
||
"""按层聚合正确:命中态 [T,T,F,T,F,F,F,T,F,T,T] → L2=4 L3=0 L4=2、score=6。"""
|
||
pat = [True, True, False, True, False, False, False, True, False, True, True]
|
||
r = cheap_verify.parse_judge_output(json.dumps({"checks": _full_checks(pat)}))
|
||
assert r["score"] == 6
|
||
assert r["groups"]["L2"] == {"score": 4, "max": 6}
|
||
assert r["groups"]["L3"] == {"score": 0, "max": 3}
|
||
assert r["groups"]["L4"] == {"score": 2, "max": 2}
|
||
|
||
|
||
def test_parse_fenced_json():
|
||
"""markdown 围栏 + 前后赘语 → json_repair 兜得住、正确解析。"""
|
||
body = json.dumps({"checks": _full_checks([True] * 11), "notes": "满"}, ensure_ascii=False)
|
||
text = "这是我的评分:\n```json\n" + body + "\n```\n以上。"
|
||
r = cheap_verify.parse_judge_output(text)
|
||
assert r["degraded"] is False and r["score"] == 11
|
||
|
||
|
||
def test_parse_garbage_degraded():
|
||
"""完全非 JSON → degraded(score=None, groups=None),不抛。"""
|
||
r = cheap_verify.parse_judge_output("完全不是 JSON 的一段话")
|
||
assert r["degraded"] is True and r["score"] is None and r["max"] == 11 and r["groups"] is None
|
||
|
||
|
||
def test_parse_none_and_empty_degraded():
|
||
"""None / 空串 → degraded。"""
|
||
assert cheap_verify.parse_judge_output(None)["degraded"] is True
|
||
assert cheap_verify.parse_judge_output("")["degraded"] is True
|
||
assert cheap_verify.parse_judge_output(" ")["degraded"] is True
|
||
|
||
|
||
def test_parse_non_dict_degraded():
|
||
"""JSON 数组(非对象)→ degraded。"""
|
||
assert cheap_verify.parse_judge_output("[1,2,3]")["degraded"] is True
|
||
|
||
|
||
def test_parse_missing_checks_degraded():
|
||
"""有 JSON 对象但缺 checks 数组 → degraded(不静默判 0 分)。"""
|
||
assert cheap_verify.parse_judge_output(json.dumps({"notes": "x"}))["degraded"] is True
|
||
assert cheap_verify.parse_judge_output(json.dumps({"checks": []}))["degraded"] is True
|
||
|
||
|
||
def test_parse_coerce_and_positional():
|
||
"""hit 用字符串/数字 + name 对不上时按位置兜底(容忍模型改名)。"""
|
||
checks = [{"hit": "true", "why": "a"}, {"hit": 1}, {"hit": "否"}] + [{"hit": False}] * 5
|
||
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}))
|
||
assert r["degraded"] is False
|
||
assert r["score"] == 2 # 前两条 "true"/1 命中,第三条「否」不命中,其余 False
|
||
assert r["max"] == 11 and r["hits"][0]["layer"] == "L2" # 位置兜底也带 checklist 权威 layer
|
||
|
||
|
||
# ───────────────────────── ② verify_richness 非阻塞契约(fake model,零真网络)─────────────────────────
|
||
|
||
class _FakeResp:
|
||
"""mock ChatResponse:content 用 dict 文本块(_extract_text 兼容 dict 块)。"""
|
||
def __init__(self, text):
|
||
self.content = [{"type": "text", "text": text}]
|
||
|
||
|
||
class _FakeModel:
|
||
"""async 可调用 mock LLM:返回固定文本 / 抛异常 / 睡眠(测超时)。usage_sum 仿 RecordingModel。"""
|
||
def __init__(self, text="", raise_exc=None, sleep=0.0):
|
||
self._text, self._raise, self._sleep = text, raise_exc, sleep
|
||
|
||
async def __call__(self, messages, **kw):
|
||
if self._sleep:
|
||
await asyncio.sleep(self._sleep)
|
||
if self._raise:
|
||
raise self._raise
|
||
return _FakeResp(self._text)
|
||
|
||
def usage_sum(self):
|
||
return (11, 22)
|
||
|
||
|
||
_SRC = "// game-logic.js\nfunction createGame(){ /* 进货 库存 解锁 playSfx 飘字 结算 remix */ }"
|
||
|
||
|
||
def _skip_no_as(name):
|
||
print(f" SKIP {name}(无 agentscope)")
|
||
|
||
|
||
def test_verify_happy_path():
|
||
"""fake model 返回合法 JSON → 结构化评分 + 三分组小计 + judgeTokens;非 degraded。"""
|
||
if not _HAS_AS:
|
||
return _skip_no_as("test_verify_happy_path")
|
||
body = json.dumps({"checks": _full_checks([True] * 5 + [False] * 6), "notes": "不错"}, ensure_ascii=False)
|
||
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC, model=_FakeModel(text=body)))
|
||
assert r["degraded"] is False and r["score"] == 5
|
||
# 前 5 条命中 [T,T,T,T,T] 均属 L2 索引{0,1,3,4}? idx0,1L2 idx2L3 idx3,4L2 → L2=4 L3=1
|
||
assert r["groups"]["L2"]["score"] == 4 and r["groups"]["L3"]["score"] == 1 and r["groups"]["L4"]["score"] == 0
|
||
assert r["judgeTokens"]["total"] == 33
|
||
|
||
|
||
def test_verify_model_raises_degraded():
|
||
"""LLM 抛异常 → degraded(绝不抛、绝不阻断)。"""
|
||
if not _HAS_AS:
|
||
return _skip_no_as("test_verify_model_raises_degraded")
|
||
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC, model=_FakeModel(raise_exc=RuntimeError("boom"))))
|
||
assert r["degraded"] is True and r["score"] is None
|
||
|
||
|
||
def test_verify_timeout_degraded():
|
||
"""LLM 超时 → degraded(不卡死收口)。"""
|
||
if not _HAS_AS:
|
||
return _skip_no_as("test_verify_timeout_degraded")
|
||
r = asyncio.run(cheap_verify.verify_richness("x", sources=_SRC,
|
||
model=_FakeModel(text="{}", sleep=0.3), timeout=0.05))
|
||
assert r["degraded"] is True
|
||
|
||
|
||
def test_verify_empty_sources_skips_llm():
|
||
"""空 src → 不调 LLM 直接 degraded(model 给个会抛的,证明它没被调用)。"""
|
||
if not _HAS_AS:
|
||
return _skip_no_as("test_verify_empty_sources_skips_llm")
|
||
r = asyncio.run(cheap_verify.verify_richness("x", sources=" ",
|
||
model=_FakeModel(raise_exc=RuntimeError("不该被调用"))))
|
||
assert r["degraded"] is True
|
||
assert "跳过" in r.get("reason", "") # 走的是「空 src 跳过」路径,而非 LLM 抛异常路径
|
||
|
||
|
||
# ── W-GENRE 品类件④:品类扩展 rubric(分母独立小计)────────────────────────────
|
||
|
||
def test_genre_checklist_trpg_shape():
|
||
"""trpg 品类扩展 ≥4 条、形状同通用底座 (名, 层标, 含义含正反例)、层标合法(§4 规范二)。"""
|
||
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
|
||
assert len(gc) >= 4, f"品类扩展应 ≥4 条,得 {len(gc)}"
|
||
for name, layer, meaning in gc:
|
||
assert isinstance(name, str) and name, "标准名非空"
|
||
assert layer in ("L2", "L3", "L4"), f"层标非法:{layer}"
|
||
assert "正:" in meaning and "反:" in meaning, f"「{name}」含义应带一对正反例(规范三锚定)"
|
||
assert cheap_verify.GENRE_BY_TEMPLATE["_template-trpg"] == "trpg"
|
||
|
||
|
||
def test_parse_with_genre_checklist_independent_denominator():
|
||
"""品类段独立分母:通用 score/groups 不被品类条目污染;richness.genre 形状齐全。"""
|
||
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
|
||
checks = [{"name": n, "hit": True, "why": "w"} for (n, _l, _m) in cheap_verify.RICHNESS_CHECKLIST]
|
||
checks += [{"name": n, "hit": (i != len(gc) - 1), "why": "w"} for i, (n, _l, _m) in enumerate(gc)]
|
||
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks, "notes": "n"}),
|
||
genre_checklist=gc, genre_key="trpg")
|
||
assert r["score"] == 11 and r["max"] == 11, "通用分母被品类条目污染"
|
||
g = r["genre"]
|
||
assert g["key"] == "trpg" and g["max"] == len(gc) and g["score"] == len(gc) - 1
|
||
total = sum(v["max"] for v in g["groups"].values())
|
||
assert total == len(gc), "品类分组小计分母应 = 品类条目数"
|
||
|
||
|
||
def test_parse_genre_positional_fallback_offset():
|
||
"""品类段位置兜底:name 全对不上时按「通用在前、品类紧随」的偏移对位。"""
|
||
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
|
||
n = 11 + len(gc)
|
||
checks = [{"name": f"未知{i}", "hit": True, "why": ""} for i in range(n)]
|
||
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}),
|
||
genre_checklist=gc, genre_key="trpg")
|
||
assert r["score"] == 11 and r["genre"]["score"] == len(gc)
|
||
|
||
|
||
def test_parse_no_genre_backcompat_no_genre_key():
|
||
"""不传品类:输出形状与既有完全一致(无 genre 键——老消费面零感知)。"""
|
||
checks = [{"name": n, "hit": True, "why": "w"} for (n, _l, _m) in cheap_verify.RICHNESS_CHECKLIST]
|
||
r = cheap_verify.parse_judge_output(json.dumps({"checks": checks}))
|
||
assert "genre" not in r and r["score"] == 11
|
||
|
||
|
||
def test_build_judge_user_genre_block_additive():
|
||
"""user prompt:无品类段落基线不变;有品类只 additive 加品类清单段 + 条数契约(11+N)。"""
|
||
base = cheap_verify._build_judge_user("SRC", "B")
|
||
gc = cheap_verify.GENRE_CHECKLISTS["trpg"]
|
||
withg = cheap_verify._build_judge_user("SRC", "B", genre_key="trpg", genre_checklist=gc)
|
||
assert "品类扩展清单" not in base and "共 11 条" in base
|
||
assert "trpg 品类扩展清单" in withg and f"共 {11 + len(gc)} 条" in withg
|
||
for name, _l, _m in gc:
|
||
assert name in withg, f"品类条目「{name}」应出现在 judge prompt"
|
||
|
||
|
||
if __name__ == "__main__":
|
||
_fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
|
||
_failed = 0
|
||
for _fn in _fns:
|
||
try:
|
||
_fn()
|
||
print(f" PASS {_fn.__name__}")
|
||
except AssertionError as e:
|
||
_failed += 1
|
||
print(f" FAIL {_fn.__name__}: {e}")
|
||
print(f"\n{len(_fns) - _failed}/{len(_fns)} passed")
|
||
sys.exit(1 if _failed else 0)
|