games-development-ai/cheap-worker/tests/test_acceptance_v2.py
lili cbfd4d871b
Some checks failed
contract-gates / contract-gates (push) Has been cancelled
docs-gate / docs-gate (push) Has been cancelled
feat(acceptance): 闭合 playtest v3 与 A+ 可信消费链
固化 Match-3 生产者、视觉、音频与双 Judge 证据闭包。

将《山海行纪》r1.1 绑定新的不可变 release,并以生产预检现场核验 bundle、Registry/2 和 25 项 Writer 快照。

同步地图1平衡锁值、跨游戏回归修复、验收契约与 SoT 证据。
2026-07-28 20:16:13 -07:00

406 lines
21 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
test_acceptance_v2.py — W-AXIS-V2 波1 验收编排单测(四门投影 / run_playtest 二掷 / run_acceptance mode 三态 /
修复反馈现象分离 / fail-closed 各路径)。全部零真网络零起服(fake roll 注入 + tmp 证据目录 + monkeypatch)。
§3 八件的覆盖分工:第 1-5 件(坐标尺 / 落点回显 / 同点硬提示 / 反早退 / 坐标协议)是 playtest.cdp.cjs 的浏览器
DOM 行为,由 10 局考卷真跑覆盖(见 spikes/playtest-agent/README 真相表);本单测覆盖可 Python 单测的部分——
第 6 件(firstPlay/预算字段流转)、第 7 件(二掷编排)、第 8 件(图像通道 fail-closed),及编排器 mode 三态、
投影函数、fail-closed 各路径(超时 / 运行错误 / 未裁决)。
跑:cheap-worker/.venv/bin/python -m pytest cheap-worker/tests/test_acceptance_v2.py -q
"""
import asyncio
import json
import sys
import tempfile
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
import cheap_verify # noqa: E402
# ══════════════════════════════════════════════════════════════════════════════
# ① project_floor 四门投影(floor 唯一产地;不复用 harness verdict.pass)
# ══════════════════════════════════════════════════════════════════════════════
def _guards(a=True, b=True, c=True, d=True, extra=None):
g = {"A_boot": {"pass": a}, "B_uncaught": {"pass": b}, "C_frame": {"pass": c}, "D_render": {"pass": d}}
if extra:
g.update(extra)
return {"guards": g}
def test_project_floor_all_green():
f = cheap_verify.project_floor(_guards())
assert f["pass"] is True
assert f["gates"] == {"A": True, "B": True, "C": True, "D": True}
def test_project_floor_one_gate_fail():
assert cheap_verify.project_floor(_guards(c=False))["pass"] is False
assert cheap_verify.project_floor(_guards(a=False))["gates"]["A"] is False
def test_project_floor_missing_gate_fail_closed():
# 只有 A_boot,B/C/D 缺 → fail-closed(缺门即 False)
f = cheap_verify.project_floor({"guards": {"A_boot": {"pass": True}}})
assert f["pass"] is False
assert f["gates"]["A"] is True and f["gates"]["B"] is False
def test_project_floor_empty():
assert cheap_verify.project_floor(None)["pass"] is False
assert cheap_verify.project_floor({})["pass"] is False
assert cheap_verify.project_floor({"guards": {}})["pass"] is False
def test_project_floor_not_reuse_verdict_pass():
# 即使 harness verdict.pass=True,只要四门不全,floor 就不放行(绝不复用 verdict.pass)。
v = {"pass": True, "guards": {"A_boot": {"pass": True}, "E_live": {"pass": True}}}
assert cheap_verify.project_floor(v)["pass"] is False
def test_project_floor_ignores_eghif_gates():
# E/G/H/I/F 挂但四门全绿 → floor 放行(契约门降观测,不入地板)。
v = _guards(extra={"E_live": {"pass": False}, "G_input": {"pass": False}, "H_progress": {"pass": False}})
assert cheap_verify.project_floor(v)["pass"] is True
# ══════════════════════════════════════════════════════════════════════════════
# ② run_playtest 二掷编排(§3 第7件)+ 图像 fail-closed(§3 第8件)——fake roll 注入
# ══════════════════════════════════════════════════════════════════════════════
def _roll(pass_=None, degraded=False, image_blind=False, reason=None):
r = {"gid": "t", "roll": None, "pass": pass_, "canSee": not image_blind, "problems": [],
"feedback": "", "summary": "s", "steps": 5, "tapPoints": 3, "tokens": {"in": 1000, "out": 100},
"firstPlay": {"playableAtMs": 100, "firstFeedbackMs": 100, "loopClosed": bool(pass_)}}
if degraded:
r["degraded"] = True
r["reason"] = reason or "degraded"
if image_blind:
r["degraded"] = True
r["imageBlind"] = True
r["reason"] = "图像故障"
return r
def _patch_rolls(monkeypatch, rolls, cost=0.1):
"""按 roll 序号返回预设结果(rolls=[roll1, roll2, ...]);成本固定,绕网关取价。"""
seq = list(rolls)
def fake(game_id, brief_file, *, roll, **kw):
out = dict(seq[roll - 1] if roll - 1 < len(seq) else seq[-1])
out["roll"] = roll
return out
monkeypatch.setattr(cheap_verify, "_run_playtest_roll_sync", fake)
monkeypatch.setattr(cheap_verify, "_playtest_roll_cost", lambda mn, r: cost)
def _run_pt(monkeypatch, rolls, cost=0.1, **kw):
_patch_rolls(monkeypatch, rolls, cost=cost)
ev = Path(tempfile.mkdtemp())
return asyncio.run(cheap_verify.run_playtest("t", brief="b", evidence_dir=ev, **kw)), ev
def test_playtest_pass_single_roll(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=True)], second_roll=True)
assert r["accepted"] is True and r["rollCount"] == 1 # pass 一掷即过,不二掷
def test_playtest_fail_then_pass_second_roll(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=False), _roll(pass_=True)], second_roll=True)
assert r["accepted"] is True and r["rollCount"] == 2 # 二掷翻案(治坐标失手假阴)
def test_playtest_fail_both_rolls(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=False), _roll(pass_=False)], second_roll=True)
assert r["accepted"] is False and r["rollCount"] == 2 and r["verdict"] == "reject" # 两掷同 fail
def test_playtest_degraded_no_second_roll(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(degraded=True), _roll(pass_=True)], second_roll=True)
assert r["degraded"] is True and r["accepted"] is False and r["rollCount"] == 1 # degraded 不二掷
def test_playtest_image_blind_fail_closed(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(image_blind=True)], second_roll=True)
assert r["degraded"] is True and r["accepted"] is False and r["imageBlind"] is True # §3 第8件
def test_playtest_second_roll_disabled(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=False), _roll(pass_=True)], second_roll=False)
assert r["accepted"] is False and r["rollCount"] == 1 # 关二掷 → 单掷 fail
def test_playtest_cost_cap_no_second_roll(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=False), _roll(pass_=True)], cost=2.0,
second_roll=True, cost_cap_rmb=1.5)
assert r["rollCount"] == 1 and r["accepted"] is False # roll-1 成本已撞上限,不二掷
def test_playtest_cost_accumulates(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=False), _roll(pass_=False)], cost=0.3, second_roll=True)
assert abs(r["costRmb"] - 0.6) < 1e-6 # 两掷成本累加
def test_playtest_firstplay_passthrough(monkeypatch):
r, _ = _run_pt(monkeypatch, [_roll(pass_=True)], second_roll=True)
assert r["firstPlay"]["loopClosed"] is True and r["firstPlay"]["playableAtMs"] == 100 # §5 三字段透传
def test_playtest_persist_json(monkeypatch):
_, ev = _run_pt(monkeypatch, [_roll(pass_=True)], second_roll=True)
assert (ev / "playtest" / "playtest.json").is_file() # 真相层落盘(§6.13 亲眼验收看它)
# ══════════════════════════════════════════════════════════════════════════════
# ③ run_acceptance mode 三态(v2 / shadow / v1)
# ══════════════════════════════════════════════════════════════════════════════
async def _fake_pt_accept(game_id, **kw):
return {"accepted": True, "verdict": "accept", "pass": True, "degraded": False, "problems": [],
"feedback": "", "summary": "ok", "rollCount": 1, "costRmb": 0.1, "model": "M",
"firstPlay": {"playableAtMs": 100, "firstFeedbackMs": 100, "loopClosed": True}}
async def _fake_pt_reject(game_id, **kw):
return {"accepted": False, "verdict": "reject", "pass": False, "degraded": False,
"problems": ["第2步点(100,200)后画面未变;推测:事件未绑定"], "feedback": "修", "summary": "no",
"rollCount": 2, "costRmb": 0.2, "model": "M",
"firstPlay": {"playableAtMs": None, "firstFeedbackMs": None, "loopClosed": False}}
async def _fake_pt_degraded(game_id, **kw):
"""模拟测试员图像通道故障,验证它被归到验收仪器而非游戏缺陷。"""
return {"accepted": False, "verdict": "reject", "pass": False, "degraded": True,
"reason": "图像通道故障", "problems": [], "feedback": "", "summary": "image-blind",
"rollCount": 1, "costRmb": 0.1, "model": "M",
"firstPlay": {"playableAtMs": None, "firstFeedbackMs": None, "loopClosed": False}}
async def _fake_apply_judge(summary, **kw):
summary["accepted"] = True
summary["ok"] = True
summary["judge"] = {"ran": True, "verdict": "accept", "blocking": True}
return summary
def _acc(monkeypatch, verdict, mode, pt=_fake_pt_accept, judge=None, summary=None):
monkeypatch.setattr(cheap_verify, "run_playtest", pt)
if judge is not None:
monkeypatch.setattr(cheap_verify, "apply_gameplay_judge", judge)
monkeypatch.setattr(cheap_verify, "_writeback_verdict_json", lambda *a, **k: None)
ev = str(Path(tempfile.mkdtemp()))
return asyncio.run(cheap_verify.run_acceptance(summary or {}, game_id="t", brief="b", verdict=verdict,
mode=mode, evidence_dir=ev))
def test_run_acceptance_v2_accept(monkeypatch):
# 模拟 CLI/Service 在验收前已经按旧 E_live 写入 gameplay 失败;最终 accepted 必须覆盖成成功语义。
legacy = {"failureLayer": {"layer": "gameplay", "reason": "旧 E_live 失败", "failedGates": ["E_live"]}}
verdict = _guards(extra={"E_live": {"pass": False}})
s = _acc(monkeypatch, verdict, "v2", pt=_fake_pt_accept, summary=legacy)
assert s["accepted"] is True and s["ok"] is True and s["acceptanceVersion"] == "v2"
assert s["floor"]["pass"] is True and s["playtest"]["accepted"] is True
assert s["failureLayer"] == {"layer": "none", "reason": None, "failedGates": []}
assert "repairFeedback" not in s # accepted 无失败现象,不拼装续修反馈(E2E 曾误拼自相矛盾文案)
def test_run_acceptance_v2_reject_by_playtest(monkeypatch):
verdict = _guards(extra={"E_live": {"pass": False}, "H_progress": {"pass": False}})
s = _acc(monkeypatch, verdict, "v2", pt=_fake_pt_reject)
assert s["accepted"] is False # 四门过但测试员拒
assert s["judge"]["rejectClasses"] == ["playtest_reject"]
assert s["failureLayer"]["layer"] == "gameplay"
assert "第2步点(100,200)后画面未变" in s["failureLayer"]["reason"]
assert "no" not in s["failureLayer"]["reason"] # 有客观现象时不得混入测试员摘要中的判断性文本
assert "推测" not in s["failureLayer"]["reason"]
assert "E_live" not in s["failureLayer"]["reason"] and "H_progress" not in s["failureLayer"]["reason"]
assert "repairFeedback" in s and "第2步点(100,200)后画面未变" in s["repairFeedback"]
def test_run_acceptance_v2_floor_fail_skips_playtest(monkeypatch):
called = {"n": 0}
async def spy(game_id, **kw):
called["n"] += 1
return {"accepted": True}
s = _acc(monkeypatch, _guards(c=False), "v2", pt=spy)
assert called["n"] == 0 and s["accepted"] is False # 四门不过不跑测试员(省 ¥)
assert s["playtest"]["reason"].startswith("四门")
assert s["failureLayer"]["layer"] == "mechanical"
assert s["failureLayer"]["failedGates"] == ["C_frame"]
def test_run_acceptance_v2_tester_degraded_attribution(monkeypatch):
"""四门全绿但测试员评不出时,归因必须是 tester_degraded,不得指责玩法。"""
s = _acc(monkeypatch, _guards(extra={"E_live": {"pass": False}}), "v2", pt=_fake_pt_degraded)
assert s["accepted"] is False
assert s["failureLayer"]["layer"] == "tester_degraded"
assert "图像通道故障" in s["failureLayer"]["reason"]
assert s["failureLayer"]["failedGates"] == []
def test_v2_failure_attribution_missing_floor_fails_closed_as_mechanical():
"""floor 证据缺失时按四门全失败归因,不得越过机械地板指责玩法。"""
s = cheap_verify.apply_acceptance_failure_attribution({
"acceptanceVersion": "v2", "accepted": False,
"playtest": {"accepted": False, "degraded": False, "problems": ["点击后无变化"]},
})
assert s["failureLayer"]["layer"] == "mechanical"
assert s["failureLayer"]["failedGates"] == ["A_boot", "B_uncaught", "C_frame", "D_render"]
def test_run_acceptance_v2_exception_after_green_floor_is_tester_degraded(monkeypatch):
"""四门全绿后测试员调用抛错,顶层 fail-closed 必须归验收器降级。"""
async def broken_playtest(game_id, **kw):
raise RuntimeError("测试员崩溃")
s = _acc(monkeypatch, _guards(), "v2", pt=broken_playtest)
assert s["accepted"] is False and s["ok"] is False
assert s["failureLayer"]["layer"] == "tester_degraded"
assert "测试员崩溃" in s["failureLayer"]["reason"]
def test_run_acceptance_v2_writeback_exception_overrides_existing_playtest(monkeypatch):
"""测试员已接受但证据写回失败时,最终失败属于验收器,不得误记为玩法失败。"""
monkeypatch.setattr(cheap_verify, "run_playtest", _fake_pt_accept)
def broken_writeback(*args, **kwargs):
raise OSError("verdict 写回失败")
monkeypatch.setattr(cheap_verify, "_writeback_verdict_json", broken_writeback)
ev = str(Path(tempfile.mkdtemp()))
s = asyncio.run(cheap_verify.run_acceptance(
{}, game_id="t", brief="b", verdict=_guards(), mode="v2", evidence_dir=ev))
assert s["accepted"] is False and s["ok"] is False
assert s["playtest"]["accepted"] is True # 保留测试员原始证据
assert s["failureLayer"]["layer"] == "tester_degraded"
assert "verdict 写回失败" in s["failureLayer"]["reason"]
def test_run_acceptance_shadow_takes_old_verdict(monkeypatch):
# shadow:accepted 取旧口径(apply_gameplay_judge accept),测试员照跑照落 + shadowV2Accepted 对照。
legacy = {"failureLayer": {"layer": "driver_contract", "reason": "旧口径归因", "failedGates": ["H_progress"]}}
s = _acc(monkeypatch, _guards(), "shadow", pt=_fake_pt_reject, judge=_fake_apply_judge, summary=legacy)
assert s["accepted"] is True # 旧口径判 accept
assert s["shadowV2Accepted"] is False # v2 影子(测试员拒)——新旧口径对照
assert s["playtest"]["accepted"] is False # 测试员照跑照落证据
assert s["failureLayer"] == legacy["failureLayer"] # shadow 保持旧验收归因,避免灰度口径漂移
def test_run_acceptance_v1_uses_apply_judge(monkeypatch):
called = {"n": 0}
async def spy(*a, **k):
called["n"] += 1
return {}
legacy = {"failureLayer": {"layer": "none", "reason": "旧口径未见失败门", "failedGates": []}}
s = _acc(monkeypatch, _guards(), "v1", pt=spy, judge=_fake_apply_judge, summary=legacy)
assert s["accepted"] is True and called["n"] == 0 # v1 不跑测试员,走旧判定器
assert s["acceptanceVersion"] == "v1" and s["floor"]["pass"] is True
assert s["failureLayer"] == legacy["failureLayer"] # v1 回退态继续沿用旧归因
# ══════════════════════════════════════════════════════════════════════════════
# ④ 修复反馈现象/推测强制分离(续修只引现象段)
# ══════════════════════════════════════════════════════════════════════════════
def test_repair_feedback_strips_speculation():
floor = {"pass": True, "gates": {"A": True, "B": True, "C": True, "D": True}}
pt = {"degraded": False, "summary": "no",
"problems": ["第2步点(100,200)后画面未变;推测:事件未绑定", "推测:可能是渲染 bug"]}
fb = cheap_verify._build_repair_feedback(floor, pt)
assert "第2步点(100,200)后画面未变" in fb # 现象保留
assert "事件未绑定" not in fb # 分句里的推测剔除
assert "渲染 bug" not in fb # 整条推测剔除
def test_repair_feedback_degraded_no_repair():
fb = cheap_verify._build_repair_feedback({"gates": {}}, {"degraded": True, "reason": "图像故障"})
assert "不派续修" in fb and "图像故障" in fb
# ══════════════════════════════════════════════════════════════════════════════
# ⑤ _run_playtest_roll_sync 退出码语义(fail-closed 各路径)+ 配置/辅助
# ══════════════════════════════════════════════════════════════════════════════
class _FakeProc:
def __init__(self, stdout, returncode):
self.stdout, self.stderr, self.returncode = stdout, "", returncode
def _roll_sync(monkeypatch, proc_or_exc):
monkeypatch.setattr(cheap_verify, "_playtest_env", lambda: {})
if isinstance(proc_or_exc, Exception):
def r(*a, **k):
raise proc_or_exc
else:
def r(*a, **k):
return proc_or_exc
monkeypatch.setattr(cheap_verify.subprocess, "run", r)
return cheap_verify._run_playtest_roll_sync(
"t", "/tmp/b", model_name="M", roll=1, seed=1, steps_base=14, steps_max=24,
evidence_dir="/tmp", port=1, cdp_port=2, timeout=10)
def test_roll_sync_image_blind_exit3(monkeypatch):
v = json.dumps({"gid": "t", "pass": False, "canSee": False, "imageBlind": True})
r = _roll_sync(monkeypatch, _FakeProc(v, 3))
assert r["degraded"] is True and r["imageBlind"] is True # 退出码 3 → 图像故障 fail-closed
def test_roll_sync_runner_error_exit2(monkeypatch):
r = _roll_sync(monkeypatch, _FakeProc("", 2))
assert r["degraded"] is True # 退出码 2 / 未产裁决 → degraded
def test_roll_sync_normal_pass(monkeypatch):
v = json.dumps({"gid": "t", "pass": True, "canSee": True})
r = _roll_sync(monkeypatch, _FakeProc(v, 0))
assert r["pass"] is True and not r.get("degraded") # 退出码 0 正常裁决
def test_roll_sync_timeout(monkeypatch):
r = _roll_sync(monkeypatch, cheap_verify.subprocess.TimeoutExpired("cmd", 10))
assert r["degraded"] is True and "超时" in r["reason"] # 子进程超时 → fail-closed
def test_roll_sync_no_json(monkeypatch):
r = _roll_sync(monkeypatch, _FakeProc("some non-json noise", 0))
assert r["degraded"] is True and "未产出裁决" in r["reason"] # 无裁决 JSON → fail-closed
def test_acceptance_cfg_defaults():
cfg = cheap_verify._acceptance_cfg()
assert cfg["mode"] in ("v2", "shadow", "v1")
assert isinstance(cfg["steps_base"], int) and isinstance(cfg["steps_max"], int)
assert cfg["steps_max"] >= cfg["steps_base"] and cfg["cost_cap_rmb"] > 0
def test_derive_playtest_ports_stable():
p1 = cheap_verify._derive_playtest_ports("hard-puzzle-r1")
p2 = cheap_verify._derive_playtest_ports("hard-puzzle-r1")
assert p1 == p2 and 4998 <= p1[0] < 5100 and 9331 <= p1[1] < 9431 # 同 id 恒同端口(可复现)
def test_derive_playtest_ports_skips_chrome_unsafe():
# 2026-07-10 考卷实锤:sb2 曾派生到 5060(SIP,Chrome ERR_UNSAFE_PORT 拒加载)→ 假 boot-timeout。
# 派生必须跳过 unsafe 清单;全 hash 空间扫一遍保证无一落入。
assert cheap_verify._derive_playtest_ports("hard-sim-business-r2")[0] not in cheap_verify._CHROME_UNSAFE_PORTS
for h in range(100):
port = 4998 + h
if port in cheap_verify._CHROME_UNSAFE_PORTS:
port += 2
assert port not in cheap_verify._CHROME_UNSAFE_PORTS
def test_roll_brief_shape():
b = cheap_verify._roll_brief({"roll": 1, "seed": 42, "pass": True, "steps": 5,
"summary": "s", "tokens": {"in": 1, "out": 2}})
assert b["roll"] == 1 and b["pass"] is True and b["tokens"] == {"in": 1, "out": 2}
if __name__ == "__main__":
import pytest
sys.exit(pytest.main([__file__, "-q"]))