games-development-ai/cheap-worker/tests/test_hard_genre_batch.py
lili cbfd4d871b
Some checks failed
contract-gates / contract-gates (push) Has been cancelled
docs-gate / docs-gate (push) Has been cancelled
feat(acceptance): 闭合 playtest v3 与 A+ 可信消费链
固化 Match-3 生产者、视觉、音频与双 Judge 证据闭包。

将《山海行纪》r1.1 绑定新的不可变 release,并以生产预检现场核验 bundle、Registry/2 和 25 项 Writer 快照。

同步地图1平衡锁值、跨游戏回归修复、验收契约与 SoT 证据。
2026-07-28 20:16:13 -07:00

97 lines
4.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""hard_genre_batch 聚合结果命名测试。"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from hard_genre_batch import _aggregate_rows, _row_total_cost, _session_result_path, _v3_batch_metrics # noqa: E402
def test_session_result_path_follows_jsonl_stem():
"""自定义 JSONL 必须拥有独立聚合前缀,不能覆盖默认 wax 基线。"""
base = Path("/tmp/results/wax2-baseline.jsonl")
assert _session_result_path(base, 1, 1) == Path("/tmp/results/wax2-baseline-r1-1.json")
assert _session_result_path(base, 2, 4) == Path("/tmp/results/wax2-baseline-r2-5.json")
def test_session_result_path_keeps_default_name():
"""默认 wax-baseline 命名保持兼容。"""
base = Path("/tmp/results/wax-baseline.jsonl")
assert _session_result_path(base, 1, 1) == Path("/tmp/results/wax-baseline-r1-1.json")
def test_aggregate_rows_keeps_existing_rows_on_resume(tmp_path):
"""断点重入没有新行时,聚合仍从 JSONL 恢复既有结果,不能被空数组覆盖。"""
p = tmp_path / "wax2-baseline.jsonl"
p.write_text(
'{"gid":"wax2-narrative-r1","genre":"narrative","round":1}\n'
'{"gid":"wax2-trpg-r1","genre":"trpg","round":1}\n',
encoding="utf-8",
)
rows = _aggregate_rows(p, 1, 1)
assert [row["gid"] for row in rows] == ["wax2-narrative-r1", "wax2-trpg-r1"]
def test_aggregate_rows_merges_genres_but_filters_rounds(tmp_path):
"""分品类进程共享同一轮次聚合时保留全部品类,同时排除范围外轮次。"""
p = tmp_path / "wax2-baseline.jsonl"
p.write_text(
'{"gid":"wax2-narrative-r1","genre":"narrative","round":1}\n'
'{"gid":"wax2-trpg-r1","genre":"trpg","round":1}\n'
'{"gid":"wax2-puzzle-r2","genre":"puzzle","round":2}\n',
encoding="utf-8",
)
rows = _aggregate_rows(p, 1, 1)
assert {row["genre"] for row in rows} == {"narrative", "trpg"}
def test_v3_metrics_use_factual_shadow_decision_and_keep_repair_history():
"""shadow compatibility.accepted=false 不能污染质量分子;repair 前后事实必须同时可审计。"""
first = {"schemaVersion": "playtest/3", "outcome": "reject", "decision": {"accepted": False}}
final = {
"schemaVersion": "playtest/3", "outcome": "accept",
"acceptanceMode": "v3_shadow", "costRmb": 0.3, "sourceRollId": "roll-2",
"merge": {"mergeCandidate": "accept", "rescuedByRoll": 2},
"decision": {"accepted": True, "publishFrozen": True, "rescuedByRoll": 2,
"parentChainCostRmb": 1.7},
"firstPlay": {"loopClosed": True},
"proofObligations": [
{"id": "narrative.ending-reached", "required": True, "status": "satisfied"},
],
"compatibility": {
"accepted": False, "ok": False, "acceptanceVersion": "v3_shadow",
"publishFrozen": True, "schemaVersion": "playtest/3", "sourceRollId": "roll-2",
},
}
metrics = _v3_batch_metrics({"acceptanceV3": final, "acceptanceV3FirstPass": first})
assert metrics["accepted"] is True and metrics["firstPassAccepted"] is False
assert metrics["repairAttempted"] is True and metrics["acceptedAfterRepair"] is True
assert metrics["rescuedByRoll"] == 2 and metrics["publishFrozen"] is True
assert metrics["proofComplete"] is True
assert set(final["compatibility"]) == {
"accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId",
}
def test_frozen_v2_metrics_are_historical_observation_only():
"""playtest/2 只保留历史 proofComplete 观测,明确冻结且不伪装为 active v3。"""
frozen = {
"schemaVersion": "playtest/2", "outcome": "accept",
"acceptanceMode": "historical_replay", "costRmb": 0.1,
"decision": {"accepted": True, "publishFrozen": True, "parentChainCostRmb": 0.1},
"compatibility": {"playtest": {"proofComplete": True}},
}
metrics = _v3_batch_metrics({"acceptanceV3": frozen})
assert metrics["acceptanceVersion"] == "historical_replay"
assert metrics["publishFrozen"] is True
assert metrics["proofComplete"] is True
def test_v3_total_cost_uses_parent_chain_without_double_counting():
row = {"acceptanceVersion": "v3", "costRmb": 1.0, "acceptanceCostRmb": 0.3,
"parentChainCostRmb": 1.7}
assert _row_total_cost(row) == 1.7