lili cbfd4d871b
Some checks failed
contract-gates / contract-gates (push) Has been cancelled
docs-gate / docs-gate (push) Has been cancelled
feat(acceptance): 闭合 playtest v3 与 A+ 可信消费链
固化 Match-3 生产者、视觉、音频与双 Judge 证据闭包。

将《山海行纪》r1.1 绑定新的不可变 release,并以生产预检现场核验 bundle、Registry/2 和 25 项 Writer 快照。

同步地图1平衡锁值、跨游戏回归修复、验收契约与 SoT 证据。
2026-07-28 20:16:13 -07:00

1151 lines
56 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""test_result_out.py — §6.1 result-out 三字段组装单测(M3a U1)。
守的不变量(KTD2 BLOCKER 修正,镜像 SaaGraphDispatcher.buildCallbackReqVO + service.py build_callback_payload):
· gameConfig = 最小占位 Map(templateId/title/theme/engineDriven),绝不塞源工程;
· engineBundle = bundle.iife.js 全文(含 __GameBundle)= 运行时承重产物;
· 成功门 = active v3 canonical accept,或真正 legacy 验收通过;两路都要求 engineBundle 含 __GameBundle;
· sourceProject = best-effort 可选(profile 派生不出即省略,不伪造、不阻断成功);
· traceId 取 job.traceId 或 job_id;纯函数无网络无 LLM、同输入恒同输出。
跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_result_out.py
"""
import copy
import hashlib
import json
import sys
import tempfile
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
import result_out as R # noqa: E402
import worker_service as W # noqa: E402
_GOOD_BUNDLE = "var __GameBundle=(function(){return{bootGameHost(){}}})();"
def _tmp() -> Path:
return Path(tempfile.mkdtemp(prefix="ro-test-"))
def _make_game_dir(tmp: Path, *, bundle=_GOOD_BUNDLE, src_files=None) -> Path:
"""造 amgen-<id>/ 形态产物目录(bundle.iife.js + src/)。"""
gd = tmp / "amgen-t1"
(gd / "src").mkdir(parents=True)
if bundle is not None:
(gd / "bundle.iife.js").write_text(bundle, encoding="utf-8")
for name, content in (src_files or {"game-logic.js": "// logic\n", "core.js": "// core\n"}).items():
(gd / "src" / name).write_text(content, encoding="utf-8")
return gd
def test_succeeded_shape():
"""成功路:verdict.pass ∧ bundle 含 __GameBundle → status=succeeded + gameConfig 占位 + engineBundle 承重。"""
gd = _make_game_dir(_tmp())
job = {"job_id": "trace-abc", "traceId": "trace-abc", "templateId": "generic", "brief": "点点乐"}
summary = {"ok": True, "finished": True, "verdict": {"pass": True, "failedGates": []}, "costRmb": 0.05}
out = R.build_result_out(job, summary, gd)
assert out["traceId"] == "trace-abc"
assert out["status"] == "succeeded"
assert out["templateId"] == "generic"
# gameConfig 占位,绝不含源工程(BLOCKER 修正)
assert out["gameConfig"]["templateId"] == "generic"
assert out["gameConfig"]["theme"] == "generic"
assert out["gameConfig"]["engineDriven"] is True
assert "files" not in out["gameConfig"] and "entry" not in out["gameConfig"]
# engineBundle 承重
assert "__GameBundle" in out["engineBundle"]
assert out.get("failureReason") is None
def test_failed_when_gates_not_pass():
"""verdict.pass=False → status=failed + 七值 failureReason,不带 engineBundle。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "t2", "templateId": "generic", "brief": "x"}
summary = {"finished": True, "verdict": {"pass": False, "failedGates": ["E_live"]}}
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
assert out.get("engineBundle") is None
assert out["failureReason"] in R._FAILURE_REASONS # 对齐后端 FailureReasonEnum 真七值
def test_failed_when_engine_bundle_missing___gamebundle():
"""即使 verdict.pass,bundle 无 __GameBundle → status=failed(承重门:engineBundle 含 __GameBundle)。"""
gd = _make_game_dir(_tmp(), bundle="var notTheGlobal=1;")
job = {"traceId": "t3", "templateId": "generic", "brief": "x"}
summary = {"finished": True, "verdict": {"pass": True, "failedGates": []}}
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
assert out.get("engineBundle") is None
def test_failed_breaker_maps_llm_error():
"""预算熔断 / 内部真因不在七值内 → failureReason=llm_error(契约#6 catch-all;枚举无 budget 专项值)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "t4", "templateId": "generic", "brief": "x"}
summary = {"finished": False, "breaker": True, "verdict": {"pass": None, "failedGates": None}}
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
assert out["failureReason"] == "llm_error"
def test_explicit_failure_reason_wins():
"""显式 failure_reason(在七值内)优先于推断。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "t5", "templateId": "generic", "brief": "x"}
summary = {"finished": True, "verdict": {"pass": False, "failedGates": ["A_boot"]}}
out = R.build_result_out(job, summary, gd, failure_reason="timeout")
assert out["status"] == "failed"
assert out["failureReason"] == "timeout"
# ── WU2 §3.9:额度耗尽干净失败 + one-api 惯例分辨 ──
def test_quota_exhausted_in_enum():
"""quota_exhausted 已进 failureReason 枚举(跨契约共享枚举同批新增,与 aigc.yaml/dify-workflow-io.json 一致)。"""
assert "quota_exhausted" in R._FAILURE_REASONS
def test_summary_failure_reason_quota_maps_through():
"""WU2 §3.9:summary.failureReason=quota_exhausted(Service 分辨→driver 透传)→ 落 failureReason=quota_exhausted。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "tq", "templateId": "generic", "brief": "x"}
summary = {"finished": False, "verdict": {"pass": False, "failedGates": None},
"failureReason": "quota_exhausted"}
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
assert out["failureReason"] == "quota_exhausted"
def test_summary_failure_reason_ignored_when_not_in_enum():
"""summary.failureReason 非枚举值 → 不透传,回落 catch-all llm_error(防脏值污染契约)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "tq2", "templateId": "generic", "brief": "x"}
summary = {"finished": False, "verdict": {"pass": False}, "failureReason": "some_garbage"}
out = R.build_result_out(job, summary, gd)
assert out["failureReason"] == "llm_error"
def test_explicit_failure_reason_wins_over_summary():
"""显式入参优先级高于 summary.failureReason。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "tq3", "templateId": "generic", "brief": "x"}
summary = {"verdict": {"pass": False}, "failureReason": "quota_exhausted"}
out = R.build_result_out(job, summary, gd, failure_reason="timeout")
assert out["failureReason"] == "timeout"
def test_classify_newapi_status_codes():
"""状态码分辨:402/403(one-api 惯例额度码)→ quota_exhausted;裸 401 / 500 / None(无额度关键词)→ llm_error 可重试。"""
assert R.classify_newapi_failure_reason(402) == "quota_exhausted"
assert R.classify_newapi_failure_reason(403) == "quota_exhausted"
assert R.classify_newapi_failure_reason(401) == "llm_error" # 裸 401 无 message = 真凭据失效
assert R.classify_newapi_failure_reason(500) == "llm_error"
assert R.classify_newapi_failure_reason(None) == "llm_error"
def test_classify_newapi_real_quota_exhausted_401():
"""2026-07-08 真机耗尽探测坐实:new-api 对额度耗尽返 HTTP 401 + 「该令牌额度已用尽 …RemainQuota = 0」。
401 与凭据失效状态码重叠,故必须按 message 分辨——耗尽判 quota_exhausted(干净失败),真凭据失效仍 llm_error(可重试)。
这是关键回归:旧代码 401→llm_error 会把耗尽误判成可重试 → 反复重试到 watchdog 超时。"""
real = "[sk-air***0AS] 该令牌额度已用尽 !token.UnlimitedQuota && token.RemainQuota = 0 (request id: 20260708...)"
assert R.classify_newapi_failure_reason(401, real) == "quota_exhausted"
assert R.classify_newapi_failure_reason(401, "无效的令牌,请检查") == "llm_error" # 真凭据失效仍可重试、不塌成耗尽
def test_classify_newapi_keyword_fallback():
"""message 关键词分辨额度耗尽(状态码取不到 / 被框架包装时同样生效,优先于状态码)。"""
assert R.classify_newapi_failure_reason(None, "error: insufficient user quota") == "quota_exhausted"
assert R.classify_newapi_failure_reason(None, "当前分组余额不足,请充值") == "quota_exhausted"
assert R.classify_newapi_failure_reason(None, "connection reset by peer") == "llm_error"
def test_newapi_error_status_text_extraction():
"""从带 status_code/response 的异常对象抽 (status, text);抽不到 status 返 None、文本兜底 str(exc)。"""
class _Err(Exception):
status_code = 402
message = "insufficient user quota"
st, txt = R.newapi_error_status_text(_Err("boom"))
assert st == 402 and "insufficient" in txt
class _Resp:
status_code = 403
class _Wrapped(Exception):
response = _Resp()
st2, _ = R.newapi_error_status_text(_Wrapped("x"))
assert st2 == 403
st3, txt3 = R.newapi_error_status_text(ValueError("plain error"))
assert st3 is None and "plain error" in txt3
def test_traceid_fallback_to_job_id():
"""traceId 缺 → 用 job_id(贯穿键同值)。"""
gd = _make_game_dir(_tmp())
job = {"job_id": "jid-9", "templateId": "generic", "brief": "x"}
summary = {"verdict": {"pass": True}}
out = R.build_result_out(job, summary, gd)
assert out["traceId"] == "jid-9"
def test_cost_goes_into_trace_not_toplevel():
"""成本落 trace.cost.totalRmb——顶层不得有 costRmb(DifyCallbackReqVO 无此字段、后端严格拒未知)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "t6", "templateId": "generic", "brief": "x"}
summary = {"verdict": {"pass": True}, "costRmb": 0.07}
out = R.build_result_out(job, summary, gd)
assert "costRmb" not in out, "顶层不得有 costRmb(会被后端 ObjectMapper 拒)"
assert out["trace"]["cost"]["totalRmb"] == 0.07
def test_result_out_keys_within_vo_known_set():
"""result-out 顶层键严格 ⊆ DifyCallbackReqVO 已知字段集(后端拒未知字段的硬约束)。"""
vo_known = {"traceId", "status", "templateId", "gameConfig", "assets",
"engineBundle", "failureReason", "qualityScore", "sourceProject", "trace"}
gd = _make_game_dir(_tmp())
job = {"traceId": "t9", "templateId": "generic", "brief": "x"}
out_ok = R.build_result_out(job, {"verdict": {"pass": True}, "costRmb": 0.07}, gd,
source_project='{"schemaVersion":"2.0"}')
out_fail = R.build_result_out(job, {"verdict": {"pass": False}}, gd)
assert set(out_ok) <= vo_known, f"成功路含未知字段:{set(out_ok) - vo_known}"
assert set(out_fail) <= vo_known, f"失败路含未知字段:{set(out_fail) - vo_known}"
def test_build_source_project_valid_2_0():
"""profile 可派生 + src/ 有文件 → sourceProject 是合法 2.0(必填齐 + sourceHash 64-hex)。"""
import json
import re
gd = _make_game_dir(_tmp(), src_files={"game-logic.js": "// l\n", "core.js": "// c\n"})
profile = {"tickModel": "frame", "inputModel": "tap-targets", "progressModel": "score"}
sp_str = R.build_source_project(gd, profile, entry="entry.js")
assert sp_str is not None
sp = json.loads(sp_str)
assert sp["schemaVersion"] == "2.0"
assert sp["globalName"] == "__GameBundle"
assert sp["entry"] == "entry.js"
assert sp["profile"] == profile
assert "src/game-logic.js" in sp["files"] and "src/core.js" in sp["files"]
assert re.fullmatch(r"[a-f0-9]{64}", sp["sourceHash"])
def test_build_source_project_carries_trusted_scaffold_without_changing_source_hash():
"""scaffoldTemplate 是 additive 血缘;sourceHash 冻结为只覆盖 files,不能暗改既有寻址语义。"""
gd = _make_game_dir(_tmp())
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
without_route = json.loads(R.build_source_project(gd, profile))
brief = "点击开始完成一局"
brief_hash = hashlib.sha256(brief.encode("utf-8")).hexdigest()
with_route = json.loads(R.build_source_project(
gd, profile, scaffold_template="_template-story",
origin_brief=brief, origin_brief_hash=brief_hash))
assert with_route["scaffoldTemplate"] == "_template-story"
assert with_route["originBrief"] == brief and with_route["originBriefHash"] == brief_hash
assert with_route["sourceHash"] == without_route["sourceHash"]
assert with_route["files"] == without_route["files"]
def test_build_source_project_rejects_unknown_scaffold_lineage():
"""任意模板名不能进入受信 SourceProject 血缘。"""
gd = _make_game_dir(_tmp())
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
assert R.build_source_project(gd, profile, scaffold_template="_template-generic") is None
def test_build_source_project_rejects_empty_or_forged_origin_brief():
"""active v3 原题面必须非空且正文与 sha256 自洽,不能只带一个格式正确的 hash。"""
gd = _make_game_dir(_tmp())
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
assert R.build_source_project(gd, profile, scaffold_template="_template-story",
origin_brief="", origin_brief_hash=hashlib.sha256(b"").hexdigest()) is None
assert R.build_source_project(gd, profile, scaffold_template="_template-story",
origin_brief="题面甲", origin_brief_hash="a" * 64) is None
def test_source_project_contract_samples_cover_route_whitelist():
"""新增正负样本必须由 Draft-07 schema 一正一拒,防 enum 只写文档未进机器门。"""
from jsonschema import Draft7Validator
root = Path(__file__).resolve().parents[2]
schema = json.loads((root / "contracts/agent-loop/source-project.schema.json").read_text(encoding="utf-8"))
valid = json.loads((root / "contracts/agent-loop/samples/source-project/valid"
/ "01-v2-with-scaffold-template.json").read_text(encoding="utf-8"))
invalid = json.loads((root / "contracts/agent-loop/samples/source-project/invalid"
/ "01-unknown-scaffold-template.json").read_text(encoding="utf-8"))
missing_origin_hash = json.loads((root / "contracts/agent-loop/samples/source-project/invalid"
/ "02-origin-brief-missing-hash.json").read_text(encoding="utf-8"))
validator = Draft7Validator(schema)
assert list(validator.iter_errors(valid)) == []
assert list(validator.iter_errors(invalid)), "未知 scaffoldTemplate 必须被 schema 拒绝"
assert list(validator.iter_errors(missing_origin_hash)), "originBrief/hash 必须成对出现"
def test_build_source_project_omitted_when_profile_incomplete():
"""profile 派生不出(None / 三枚举不全)→ 返 None(省略、绝不伪造)。"""
gd = _make_game_dir(_tmp())
assert R.build_source_project(gd, None, entry="entry.js") is None
assert R.build_source_project(gd, {"tickModel": "frame"}, entry="entry.js") is None
def test_build_source_project_omitted_when_no_src():
"""src/ 无文件 → 返 None。"""
empty = _tmp() / "amgen-e"
(empty / "src").mkdir(parents=True)
profile = {"tickModel": "frame", "inputModel": "tap-targets", "progressModel": "score"}
assert R.build_source_project(empty, profile, entry="entry.js") is None
def test_result_out_source_project_additive():
"""build_result_out 收到 source_project(JSON str)→ payload 带;缺省不带(additive)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "t7", "templateId": "generic", "brief": "x"}
summary = {"verdict": {"pass": True}}
out_with = R.build_result_out(job, summary, gd, source_project='{"schemaVersion":"2.0"}')
assert out_with["sourceProject"] == '{"schemaVersion":"2.0"}'
out_without = R.build_result_out(job, summary, gd)
assert "sourceProject" not in out_without
def test_deterministic_zero_llm():
"""纯函数:同输入恒同输出(无网络无 LLM)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "t8", "templateId": "generic", "brief": "稳定"}
summary = {"verdict": {"pass": True}, "costRmb": 0.01}
assert R.build_result_out(job, summary, gd) == R.build_result_out(job, summary, gd)
# ──────────────────────────────────────────────────────────────────────────────
# M3b U2:trace 9d 七项 + D11 首局维(sevenGateVerdict/gatespec)+ repairs 口径 + None 容错
# 守的不变量(镜像 wg1 _extract_trace + 后端 ReadinessScorer 读取口径):
# · 必填七项 camelCase:pass/repairs/wallS/models/attempts/gameId/stage;
# · sevenGateVerdict 只取 {pass,guards}(去 url/ts 噪声,guards.H_progress 保嵌套含 .pass);
# · gatespec 键名必须是 `driver`(落 driverType 则后端 hasDriver=false、firstPlay 永中性);
# · repairs = max(0, attempts-1)(首轮 attempts=1→repairs=0);
# · 成功/失败路都抽 trace;verdictFull/driverType=None 时省略 sevenGateVerdict/gatespec(不塞 None、不崩)。
# ──────────────────────────────────────────────────────────────────────────────
# U1 富化后的成功路 summary(合成;含完整 verdictFull 带 url/ts 噪声 + driverType + stage + models)。
_SUMMARY_SUCCESS = {
"ok": True, "finished": True, "gameId": "amgen-t1", "attempts": 1,
"verdict": {"pass": True, "failedGates": []}, # brief(既有键)
"verdictFull": { # U1 新增:完整 verdict(带噪声 + guards.H_progress 嵌套)
"pass": True, "url": "http://localhost:4320/?x", "ts": 1700000000,
"guards": {
"A_boot": {"pass": True},
"H_progress": {"pass": True, "checks": ["score+"], "latch": True, "state0": 0, "state1": 3},
},
},
"driverType": "tap-targets",
"models": {"code": "MiniMax-M3"},
"stage": "play",
"wallSec": 102.3,
"costRmb": 0.0431,
}
def test_trace_seven_required_fields_and_d11_dims_success():
"""成功路 trace:七项齐 + sevenGateVerdict.guards.H_progress.pass + gatespec.driver;噪声 url/ts 已去。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "u2-ok", "templateId": "generic", "brief": "点击得分小游戏"}
out = R.build_result_out(job, _SUMMARY_SUCCESS, gd)
tr = out["trace"]
# 必填七项(camelCase 逐字镜像 _extract_trace)
assert tr["pass"] is True
assert tr["repairs"] == 0 # attempts=1 → max(0,0)=0(首轮 0)
assert tr["wallS"] == 102.3
assert tr["models"] == {"code": "MiniMax-M3"}
assert tr["attempts"] == 1
assert tr["gameId"] == "amgen-t1"
assert tr["stage"] == "play"
# D11 首局维:sevenGateVerdict 只 {pass,guards},guards.H_progress 嵌套含 .pass(后端 guardPass 读它)
assert set(tr["sevenGateVerdict"].keys()) == {"pass", "guards"}
assert "url" not in tr["sevenGateVerdict"] and "ts" not in tr["sevenGateVerdict"]
assert tr["sevenGateVerdict"]["guards"]["H_progress"]["pass"] is True
# gatespec 键名必须是 driver(落 driverType 会让后端 hasDriver=false)
assert tr["gatespec"] == {"driver": "tap-targets"}
assert "driverType" not in tr["gatespec"]
# cost 维持现状
assert tr["cost"]["totalRmb"] == 0.0431
def test_trace_repairs_equals_max0_attempts_minus_1():
"""repairs = max(0, attempts-1):attempts=1→0、attempts=3→2(对齐 wg1 len(attempts)-1)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "u2-rep", "templateId": "generic", "brief": "x"}
s1 = dict(_SUMMARY_SUCCESS, attempts=1)
assert R.build_result_out(job, s1, gd)["trace"]["repairs"] == 0
s3 = dict(_SUMMARY_SUCCESS, attempts=3)
assert R.build_result_out(job, s3, gd)["trace"]["repairs"] == 2
def test_trace_failed_path_still_has_seven_required():
"""失败路(verdictFull/driverType=None)trace 仍含七项(repairs/attempts/stage/wallS 排障价值)。"""
gd = _make_game_dir(_tmp(), bundle="var notTheGlobal=1;") # 无 __GameBundle → status=failed
job = {"traceId": "u2-fail", "templateId": "generic", "brief": "x"}
summary = {
"ok": False, "finished": False, "gameId": "amgen-t2", "attempts": 6,
"verdict": None, "verdictFull": None, "driverType": None,
"models": {"code": "MiniMax-M3"}, "stage": "code", "wallSec": 50.1, "costRmb": 0.02,
}
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
tr = out["trace"]
for k in ("pass", "repairs", "wallS", "models", "attempts", "gameId", "stage"):
assert k in tr, f"失败路 trace 缺必填项 {k}"
assert tr["repairs"] == 5 # attempts=6 → 5
assert tr["stage"] == "code"
assert tr["pass"] is None # play 没跑
def test_trace_none_tolerance_omits_d11_dims():
"""verdictFull/driverType=None(对照路)→ 省略 sevenGateVerdict/gatespec(不塞 None、不崩)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "u2-none", "templateId": "generic", "brief": "x"}
summary = {
"gameId": "amgen-t3", "attempts": 2, "verdict": None,
"verdictFull": None, "driverType": None,
"models": {"code": "MiniMax-M3"}, "stage": "smoke", "wallSec": 10.0,
}
out = R.build_result_out(job, summary, gd)
tr = out["trace"]
assert "sevenGateVerdict" not in tr # verdictFull=None → 省略
assert "gatespec" not in tr # driverType=None → 省略
assert tr["repairs"] == 1 # attempts=2 → 1
def test_trace_keys_camelcase_match_readiness_scorer():
"""trace 键名逐字镜像 ReadinessScorer 读取口径(拼错会被后端静默接收 → 假绿)。"""
gd = _make_game_dir(_tmp())
job = {"traceId": "u2-keys", "templateId": "generic", "brief": "x"}
tr = R.build_result_out(job, _SUMMARY_SUCCESS, gd)["trace"]
# ReadinessScorer 读取的确切键(camelCase):trace.pass / gatespec.driver /
# sevenGateVerdict.guards.H_progress.pass / repairs / cost.totalRmb
assert "pass" in tr and "gatespec" in tr and "sevenGateVerdict" in tr
assert "wallS" in tr and "gameId" in tr # 非 wall_s / game_id(snake 会被后端读不到)
assert "driver" in tr["gatespec"]
assert "H_progress" in tr["sevenGateVerdict"]["guards"]
# ──────────────────────────────────────────────────────────────────────────────
# 生成验收 v3 消费层:decision/outcome 唯一写权、shadow 冻结、trace.playtest 轻量投影
# ──────────────────────────────────────────────────────────────────────────────
_V2_HISTORY_SAMPLE_DIR = (Path(__file__).resolve().parents[2] / "contracts" / "play-loop"
/ "samples" / "playtest-evidence" / "valid")
_V2_HISTORY_SAMPLE_BY_OUTCOME = {
"accept": "01-accepted-narrative.json",
"reject": "02-rejected-mechanical.json",
"tester_error": "03-tester-error-action-protocol.json",
"inconclusive": "04-inconclusive-insufficient-proof.json",
}
def _v3_fixture_hash(label: str) -> str:
"""为 A/B/consensus/目标解析 fixture 生成稳定 SHA-256。"""
return hashlib.sha256(label.encode("utf-8")).hexdigest()
def _upgrade_v3_action(action: dict, index: int) -> dict:
"""把历史坐标动作升级为 playtest/3 的结构化选择和确定性解析事实。"""
upgraded = copy.deepcopy(action)
normalized = upgraded["normalized"]
action_type = normalized["type"]
target_set_hash = _v3_fixture_hash(f"target-set-{index}")
if action_type == "tap":
target = {"id": f"r{index:02d}"}
selection = {
"schemaVersion": "ActorSelection/1", "type": "tap",
"targetSetHash": target_set_hash, "target": target,
}
points = [{"role": "target", "target": target,
"point": {"x": normalized["x"], "y": normalized["y"]}}]
elif action_type == "key":
selection = {"schemaVersion": "ActorSelection/1", "type": "key", "key": normalized["key"]}
points = None
elif action_type == "wait":
selection = {"schemaVersion": "ActorSelection/1", "type": "wait", "ms": normalized["ms"]}
points = None
else:
raise AssertionError(f"v3 result fixture 暂不支持动作:{action_type}")
upgraded["request"] = {
"raw": json.dumps(selection, ensure_ascii=False, sort_keys=True),
"parsed": copy.deepcopy(selection),
}
upgraded["retryCount"] = 0
upgraded["protocolErrors"] = []
if points is None:
upgraded["targetResolution"] = {"status": "not_applicable"}
else:
upgraded["targetResolution"] = {
"status": "resolved", "targetSetRef": f"targets/action-{index}.json",
"targetSetHash": target_set_hash,
"sourceFrameRef": upgraded["preFrameRef"]["path"],
"sourceFrameHash": upgraded["preFrameRef"]["hash"],
"guideManifestRef": f"guides/action-{index}.json",
"guideManifestHash": _v3_fixture_hash(f"guide-{index}"),
"resolverVersion": "1.0.0", "selection": copy.deepcopy(selection),
"resolvedPoints": points, "error": None,
}
return upgraded
def _upgrade_history_result_to_v3(v2: dict) -> dict:
"""把冻结 v2 DECIDED 样本升级成 active/shadow 测试使用的 playtest/3 对象。"""
v3 = copy.deepcopy(v2)
v3.pop("_note", None)
v3["schemaVersion"] = "playtest/3"
v3["environment"]["virtualTime"]["waitMaxMs"] = 600
if isinstance(v3.get("actor"), dict):
v3["actor"]["contextProjectionVersion"] = "ActorView/3"
if v3.get("outcome") == "tester_error":
# 失败 selection 只属于 attempt audit,不得伪装为正式 actionRecord。
v3["actions"] = []
else:
v3["actions"] = [_upgrade_v3_action(action, index)
for index, action in enumerate(v3.get("actions") or [], start=1)]
legacy_judge = v2.get("judge") if isinstance(v2.get("judge"), dict) else None
source_roll_id = None
if legacy_judge is not None and v3.get("rolls"):
source_roll_id = str(v3["rolls"][-1]["rollId"])
package_hash = str(legacy_judge["packageHash"])
result_a = _v3_fixture_hash(f"{source_roll_id}-judge-a-result")
result_b = _v3_fixture_hash(f"{source_roll_id}-judge-b-result")
consensus_hash = _v3_fixture_hash(f"{source_roll_id}-consensus")
for roll in v3["rolls"]:
roll_id = str(roll["rollId"])
roll.update({
"proofPackageHash": _v3_fixture_hash(f"{roll_id}-proof"),
"judgePackageHash": package_hash,
"judgeARef": {
"rawRef": f"{roll_id}/judge-a/raw.json", "rawHash": _v3_fixture_hash(f"{roll_id}-a-raw"),
"parsedRef": f"{roll_id}/judge-a/parsed.json", "parsedHash": _v3_fixture_hash(f"{roll_id}-a-parsed"),
"normalizedRef": f"{roll_id}/judge-a/normalized.json", "resultHash": result_a,
},
"judgeAHash": _v3_fixture_hash(f"{roll_id}-a-manifest"),
"judgeBRef": {
"rawRef": f"{roll_id}/judge-b/raw.json", "rawHash": _v3_fixture_hash(f"{roll_id}-b-raw"),
"parsedRef": f"{roll_id}/judge-b/parsed.json", "parsedHash": _v3_fixture_hash(f"{roll_id}-b-parsed"),
"normalizedRef": f"{roll_id}/judge-b/normalized.json", "resultHash": result_b,
},
"judgeBHash": _v3_fixture_hash(f"{roll_id}-b-manifest"),
"judgeConsensusRef": f"{roll_id}/consensus.json",
"judgeConsensusHash": consensus_hash,
"costReservationRef": f"{roll_id}/cost-reservation.json",
"costReservationHash": _v3_fixture_hash(f"{roll_id}-reservation"),
})
v3["judge"] = {
"schemaVersion": "JudgeConsensus/1", "sourceRollId": source_roll_id,
"consensusRef": f"{source_roll_id}/consensus.json", "consensusHash": consensus_hash,
"packageHash": package_hash, "strategyVersion": "1.0.0",
"decision": legacy_judge["decision"], "degraded": bool(legacy_judge.get("degraded")),
"sameModel": True,
"independenceChecks": {
"differentLogicalClient": True, "differentSession": True, "differentPolicySeed": True,
"differentPromptArtifact": True, "differentEvidenceDir": True,
"differentRawFile": True, "sameJudgePackage": True,
"sameImageManifest": True, "sameModel": True,
},
"judgeAResultHash": result_a, "judgeBResultHash": result_b,
"obligationResults": copy.deepcopy(legacy_judge["obligationResults"]),
"conflicts": list(legacy_judge.get("contradictions") or []),
"failureSignature": None,
}
else:
v3["judge"] = None
# 旧 action-protocol tester_error 只有失败 attempt;正式 actionRecord 不应收录它。
if v3.get("outcome") == "tester_error":
v3["actions"], v3["rolls"], v3["events"] = [], [], []
v3["costRmb"] = 0
v3["decision"]["costRmb"] = 0
v3["sourceRollId"] = source_roll_id
accepted = v3["outcome"] == "accept"
v3["compatibility"] = {
"accepted": accepted and v3["acceptanceMode"] == "v3",
"ok": accepted and v3["acceptanceMode"] == "v3",
"acceptanceVersion": v3["acceptanceMode"],
"publishFrozen": bool(v3["decision"]["publishFrozen"]),
"schemaVersion": "playtest/3", "sourceRollId": source_roll_id,
}
return v3
def _set_v3_mode(v3: dict, mode: str) -> None:
"""在完整 canonical 样本上同步切换 active/shadow,保持所有镜像字段逐字一致。"""
accepted = v3["outcome"] == "accept"
frozen = mode == "v3_shadow" or not accepted
v3["acceptanceMode"] = mode
v3["decision"]["shadowAccepted"] = accepted if mode == "v3_shadow" else None
v3["decision"]["publishFrozen"] = frozen
compatibility = v3["compatibility"]
compatibility["accepted"] = accepted and mode == "v3"
compatibility["ok"] = accepted and mode == "v3"
compatibility["acceptanceVersion"] = mode
compatibility["publishFrozen"] = frozen
def _v3_result(outcome="accept", *, mode="v3", accepted=None, frozen=None, proof_complete=None):
"""把冻结样本升级为 v3;旧文件只作历史素材,active fixture 只返回 playtest/3。"""
path = _V2_HISTORY_SAMPLE_DIR / _V2_HISTORY_SAMPLE_BY_OUTCOME[outcome]
v3 = _upgrade_history_result_to_v3(json.loads(path.read_text(encoding="utf-8")))
_set_v3_mode(v3, mode)
if accepted is not None:
v3["decision"]["accepted"] = accepted
if frozen is not None:
v3["decision"]["publishFrozen"] = frozen
v3["compatibility"]["publishFrozen"] = frozen
if proof_complete is not None:
for obligation in v3["proofObligations"]:
if obligation.get("required") is True:
obligation["status"] = "satisfied" if proof_complete else "missing"
return v3
def _v3_summary(outcome="accept", **kwargs):
"""生产形态:完整结果在 acceptanceV3,顶层只镜像六字段 compatibility。"""
v3 = _v3_result(outcome, **kwargs)
compatibility = v3["compatibility"]
summary = {"verdict": {"pass": True}, "floor": copy.deepcopy(v3["floor"])}
for key in ("accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId"):
summary[key] = copy.deepcopy(compatibility[key])
summary["acceptanceV3"] = v3
summary["publishFrozen"] = v3["decision"]["publishFrozen"]
return summary
def _bind_v3_run(summary: dict, gd: Path, *, game_id: str = "t1", brief: str = "x") -> dict:
"""构造真实发布边界:当前 job、run-summary、staged artifact 与 release bundle 四者同源。"""
v3 = summary.get("acceptanceV3")
assert isinstance(v3, dict), "生产绑定夹具必须使用 acceptanceV3 包装态"
staged = gd.parent / "_wg1-gen" / game_id
staged.mkdir(parents=True, exist_ok=True)
release_bytes = (gd / "bundle.iife.js").read_bytes()
(staged / "bundle.iife.js").write_bytes(release_bytes)
staged_src = staged / "src"
staged_src.mkdir(parents=True, exist_ok=True)
for source_file in sorted((gd / "src").glob("*.js")):
(staged_src / source_file.name).write_bytes(source_file.read_bytes())
v3["gameId"] = game_id
v3["briefHash"] = hashlib.sha256(brief.encode("utf-8")).hexdigest()
trace_id = f"trace-{game_id}"
v3["taskBindingHash"] = hashlib.sha256(trace_id.encode("utf-8")).hexdigest()
v3["artifactHash"] = R.cheap_verify._artifact_hash_v3(game_id, staged)
v3["environment"]["buildRef"] = v3["artifactHash"]
summary["gameId"] = game_id
summary["acceptanceRunId"] = v3["runId"]
summary["acceptanceRequestHash"] = v3["acceptanceRequestHash"]
summary["acceptanceTaskBindingHash"] = v3["taskBindingHash"]
summary["acceptanceArtifactHash"] = v3["artifactHash"]
summary["acceptanceBriefHash"] = v3["briefHash"]
return {"traceId": trace_id, "gameId": game_id, "brief": brief}
def _v3_source_project(summary: dict, gd: Path, origin_brief: str) -> str:
"""从验收同源 staged/src 组 active v3 SourceProject 测试夹具。"""
v3 = summary["acceptanceV3"]
staged = gd.parent / "_wg1-gen" / v3["gameId"]
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
source_project = R.build_source_project(
staged, profile, scaffold_template=v3["templateRoute"], origin_brief=origin_brief,
origin_brief_hash=hashlib.sha256(origin_brief.encode("utf-8")).hexdigest())
assert source_project is not None
return source_project
def test_v3_accept_projects_trusted_trace_and_succeeds():
"""active v3 accept + 完整硬证才成功;trace 只保留后端消费所需轻量字段。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
out = R.build_result_out(job, summary, gd,
source_project=_v3_source_project(summary, gd, "x"))
assert out["status"] == "succeeded"
playtest = out["trace"]["playtest"]
assert playtest["schemaVersion"] == "playtest/3"
assert playtest["acceptanceMode"] == "v3" and playtest["outcome"] == "accept"
assert playtest["accepted"] is True and playtest["publishFrozen"] is False
assert playtest["compatibilityAccepted"] is True and playtest["compatibilityOk"] is True
assert playtest["floorPass"] is True and playtest["finalPostguardPass"] is True
assert playtest["writeAuthority"] is True and all(playtest["finalChecks"].values())
assert playtest["blockingProblems"] == [] and playtest["contradictions"] == []
assert playtest["mergeConflicts"] == [] and playtest["proofComplete"] is True
assert all(playtest["evidenceRefs"][key] for key in ("actions", "frames", "events"))
assert playtest["firstPlay"]["loopClosed"] is True and playtest["firstPlay"]["proofRefs"]
assert out["trace"]["acceptanceState"] == "active"
identity = out["trace"]["playtest"]
assert {key: identity[key] for key in (
"runId", "gameId", "briefHash", "artifactHash", "taskBindingHash", "acceptanceRequestHash"
)} == {key: summary["acceptanceV3"][key] for key in (
"runId", "gameId", "briefHash", "artifactHash", "taskBindingHash", "acceptanceRequestHash"
)}
assert identity["engineBundleHash"] == hashlib.sha256(out["engineBundle"].encode("utf-8")).hexdigest()
assert set(summary["acceptanceV3"]["compatibility"]) == {
"accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId",
}
def test_v3_shadow_accept_is_frozen_even_when_other_fields_are_green():
"""shadow decision 可为 accept,但 publishFrozen 是硬闸,不能携包进入 succeeded。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept", mode="v3_shadow")
job = _bind_v3_run(summary, gd)
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
assert out["trace"]["playtest"]["outcome"] == "accept"
assert out["trace"]["playtest"]["accepted"] is True
assert out["trace"]["playtest"]["publishFrozen"] is True
assert out["trace"]["acceptanceState"] == "shadow"
assert "engineBundle" not in out
def test_v3_non_accept_outcomes_always_fail_despite_legacy_accept_true():
"""旧平铺 accepted 被污染为 true 也不能覆盖 v3 四态 decision。"""
gd = _make_game_dir(_tmp())
for outcome in ("reject", "inconclusive", "tester_error"):
summary = _v3_summary(outcome)
summary["accepted"] = True
summary["ok"] = True
job = _bind_v3_run(summary, gd)
job["traceId"] = f"v3-{outcome}"
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed", outcome
assert out["trace"]["playtest"]["outcome"] == outcome
assert out["trace"]["playtest"]["accepted"] is False
def _assert_v3_failed(summary: dict, trace_id: str) -> None:
"""统一断言 v3 单点变异不能携带产物进入 succeeded。"""
gd = _make_game_dir(_tmp())
job = _bind_v3_run(summary, gd)
job["traceId"] = trace_id
out = R.build_result_out(job, summary, gd)
assert out["status"] == "failed"
assert "engineBundle" not in out
assert out["trace"]["acceptanceState"] in ("downgrade_detected", "shadow", "frozen")
def test_v3_contradiction_1_top_outcome_fails_closed():
"""顶层 outcome 与 decision/compatibility 不一致,canonical 语义校验必须拦截。"""
summary = _v3_summary("accept")
summary["acceptanceV3"]["outcome"] = "reject"
_assert_v3_failed(summary, "v3-bad-outcome")
def test_v3_contradiction_2_floor_fails_closed():
"""accept 却让机械四门 floor 失败,不能靠旧 verdict.pass=true 放行。"""
summary = _v3_summary("accept")
summary["acceptanceV3"]["floor"]["pass"] = False
_assert_v3_failed(summary, "v3-bad-floor")
def test_v3_contradiction_3_final_postguard_fails_closed():
"""finalPostguard 任一检查、写权或矛盾项异常都必须 fail-closed。"""
variants = []
bad_check = _v3_summary("accept")
bad_check["acceptanceV3"]["finalPostguard"]["checks"]["floorPass"] = False
variants.append(bad_check)
bad_writer = _v3_summary("accept")
bad_writer["acceptanceV3"]["finalPostguard"]["writeAuthority"] = False
variants.append(bad_writer)
bad_problem = _v3_summary("accept")
bad_problem["acceptanceV3"]["finalPostguard"]["blockingProblems"] = ["仍有阻塞问题"]
variants.append(bad_problem)
for index, summary in enumerate(variants):
_assert_v3_failed(summary, f"v3-bad-final-{index}")
def test_v3_contradiction_4_merge_fails_closed():
"""mergeCandidate 被改写或仍有两掷冲突时,最终 accept 不成立。"""
bad_candidate = _v3_summary("accept")
bad_candidate["acceptanceV3"]["merge"]["mergeCandidate"] = "reject"
bad_conflict = _v3_summary("accept")
bad_conflict["acceptanceV3"]["merge"]["conflicts"] = ["两掷证据冲突"]
_assert_v3_failed(bad_candidate, "v3-bad-merge-candidate")
_assert_v3_failed(bad_conflict, "v3-bad-merge-conflict")
def test_v3_contradiction_5_judge_fails_closed():
"""独立 Judge 拒绝或降级时,兼容投影再绿也不能发布。"""
bad_decision = _v3_summary("accept")
bad_decision["acceptanceV3"]["judge"]["decision"] = "reject"
bad_degraded = _v3_summary("accept")
bad_degraded["acceptanceV3"]["judge"]["degraded"] = True
_assert_v3_failed(bad_decision, "v3-bad-judge-decision")
_assert_v3_failed(bad_degraded, "v3-bad-judge-degraded")
def test_v3_contradiction_6_compatibility_fails_closed():
"""compatibility 的 mode、冻结位或 accepted/ok 与权威 decision 矛盾时不得放行。"""
for key, value in (("ok", False), ("accepted", False), ("publishFrozen", True),
("acceptanceVersion", "v3_shadow")):
summary = _v3_summary("accept")
summary["acceptanceV3"]["compatibility"][key] = value
_assert_v3_failed(summary, f"v3-bad-compat-{key}")
def test_v3_contradiction_7_proof_refs_loop_fails_closed():
"""required proof、三类引用或首局闭环任一缺失,都不能被模型 accept 文本替代。"""
bad_proof = _v3_summary("accept")
bad_proof["acceptanceV3"]["proofObligations"][0]["status"] = "failed"
bad_refs = _v3_summary("accept")
bad_refs["acceptanceV3"]["proofObligations"][0]["actionRefs"] = []
bad_loop = _v3_summary("accept")
bad_loop["acceptanceV3"]["firstPlay"]["loopClosed"] = False
for index, summary in enumerate((bad_proof, bad_refs, bad_loop)):
_assert_v3_failed(summary, f"v3-bad-proof-{index}")
def test_v3_flat_projection_contradiction_fails_closed():
"""完整封存对象虽有效,run-summary 平铺 accepted 被污染也必须失败。"""
summary = _v3_summary("accept")
summary["accepted"] = False
_assert_v3_failed(summary, "v3-bad-flat")
def test_v3_declared_without_acceptance_payload_is_downgrade_detected():
"""任一 v3 声明已出现却缺 acceptanceV3 时,不得回落 legacy verdict/accepted。"""
claims = [
{"acceptanceVersion": "v3"},
{"acceptanceVersion": "historical_replay"},
{"acceptanceVersion": "v999"},
{"acceptanceVersion": {"corrupt": True}},
{"publishFrozen": False},
{"trace": {"playtest": {"schemaVersion": "playtest/2", "outcome": "accept"}}},
]
gd = _make_game_dir(_tmp())
for index, claim in enumerate(claims):
summary = {"verdict": {"pass": True}, "accepted": True, "ok": True, **claim}
out = R.build_result_out({"traceId": f"v3-downgrade-{index}", "brief": "x"}, summary, gd)
assert out["status"] == "failed"
assert out["trace"]["acceptanceState"] == "downgrade_detected"
assert out["trace"]["publishFrozen"] is True
assert "playtest" not in out["trace"]
def test_expected_v3_blocks_fully_stripped_summary_from_legacy_fallback():
"""生产边界已冻结 v3 时,即使 summary 所有 marker 都被剥掉,legacy pass=true 也不能放行。"""
gd = _make_game_dir(_tmp())
summary = {"gameId": "t1", "verdict": {"pass": True}, "accepted": True, "ok": True}
out = R.build_result_out(
{"traceId": "v3-fully-stripped", "gameId": "t1", "brief": "x"}, summary, gd,
expected_acceptance_mode="v3")
assert out["status"] == "failed"
assert out["trace"]["acceptanceState"] == "downgrade_detected"
assert out["trace"]["publishFrozen"] is True
assert "engineBundle" not in out
def test_legacy_v1_stays_compatible_but_sealed_v2_is_observation_only():
"""无协议的 v1 存量仍兼容;冻结 playtest/2 可观察但不能作为 active 发布授权。"""
gd = _make_game_dir(_tmp())
legacy = {"verdict": {"pass": True}, "accepted": True, "playtest": {"accepted": True}}
legacy_out = R.build_result_out({"traceId": "legacy-v1", "brief": "x"}, legacy, gd)
assert legacy_out["status"] == "succeeded"
assert "acceptanceState" not in legacy_out["trace"]
frozen_v2 = json.loads(
(_V2_HISTORY_SAMPLE_DIR / _V2_HISTORY_SAMPLE_BY_OUTCOME["accept"]).read_text(encoding="utf-8")
)
observed = R.build_result_out({"traceId": "historical-v2", "brief": "x"}, frozen_v2, gd)
assert observed["status"] == "failed"
assert observed["trace"]["playtest"]["schemaVersion"] == "playtest/2"
assert "engineBundle" not in observed
def test_v3_unknown_schema_and_outcome_fail_closed():
"""未知 schema/outcome 不能回落顶层 accepted 或旧四门判定。"""
gd = _make_game_dir(_tmp())
bad_schema = _v3_summary("accept")
bad_schema["acceptanceV3"]["schemaVersion"] = "playtest/999"
bad_outcome = _v3_summary("accept")
bad_outcome["acceptanceV3"]["decision"]["outcome"] = "maybe"
schema_job = _bind_v3_run(bad_schema, gd)
outcome_job = _bind_v3_run(bad_outcome, gd)
assert R.build_result_out(schema_job, bad_schema, gd)["status"] == "failed"
assert R.build_result_out(outcome_job, bad_outcome, gd)["status"] == "failed"
def test_v3_failure_ignores_stale_legacy_failure_reason():
"""v3 outcome 已成为真相后,旧 worker 残留的合法枚举也不能把玩法失败误归为 unsafe_prompt。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("reject", mode="v3")
summary["failureReason"] = "unsafe_prompt"
job = _bind_v3_run(summary, gd)
out = R.build_result_out(job, summary, gd,
failure_reason="unsafe_prompt")
assert out["status"] == "failed"
assert out["failureReason"] == "llm_error"
assert out["trace"]["playtest"]["outcome"] == "reject"
def test_historical_replay_marker_never_falls_back_to_legacy_pass():
"""历史回放即使携带旧 pass/accepted=true 且 bundle 完整,也只能降级阻断,永不发布。"""
gd = _make_game_dir(_tmp())
summary = {
"verdict": {"pass": True}, "accepted": True, "ok": True,
"acceptanceVersion": "historical_replay",
}
out = R.build_result_out({"traceId": "historical-no-publish", "brief": "x"}, summary, gd)
assert out["status"] == "failed"
assert out["trace"]["acceptanceState"] == "downgrade_detected"
assert out["trace"]["publishFrozen"] is True
assert "engineBundle" not in out
def test_v3_direct_sealed_result_is_observable_but_never_publishable():
"""裸 decision.json 可离线观察,但缺当前 run-summary 身份绑定,不能直接授权发布。"""
gd = _make_game_dir(_tmp())
out = R.build_result_out({"traceId": "v3-direct", "gameId": "t1", "brief": "x"},
_v3_result("accept"), gd)
assert out["status"] == "failed"
assert out["trace"]["playtest"]["schemaVersion"] == "playtest/3"
def test_v3_old_accept_cannot_publish_for_different_job():
"""同一封存 accept 被塞给另一 gameId 时必须 fail-closed。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
job["gameId"] = "t2"
assert R.build_result_out(job, summary, gd)["status"] == "failed"
def test_v3_accept_cannot_publish_for_different_create_brief():
"""create 当前 brief 与封存 briefHash 不同,旧 accept 不得跨题面重放。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
job["brief"] = "另一道题"
assert R.build_result_out(job, summary, gd)["status"] == "failed"
def test_v3_modify_keeps_original_brief_hash_binding():
"""modify job.brief 是修改指令;原题面 briefHash 由封存验收继承,不能按 create 规则误拒。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd, brief="原游戏题面")
job.update({"brief": "把速度调快", "modifyMode": "deterministic", "baseVersionId": "base-1"})
source_project = _v3_source_project(summary, gd, "原游戏题面")
assert R.build_result_out(job, summary, gd, source_project=source_project)["status"] == "succeeded"
def test_v3_modify_rejects_missing_empty_or_cross_brief_source_origin():
"""modify 不得用缺失/空/另一题面的 SourceProject 原题面授权 active v3 发布。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd, brief="原游戏题面")
job.update({"brief": "把速度调快", "modifyMode": "deterministic", "baseVersionId": "base-1"})
assert R.build_result_out(job, summary, gd, source_project=None)["status"] == "failed"
valid = json.loads(_v3_source_project(summary, gd, "原游戏题面"))
for brief in ("", "另一款游戏题面"):
forged = dict(valid)
forged["originBrief"] = brief
forged["originBriefHash"] = hashlib.sha256(brief.encode("utf-8")).hexdigest()
out = R.build_result_out(job, summary, gd, source_project=json.dumps(forged, ensure_ascii=False))
assert out["status"] == "failed"
def test_v3_regenerate_outer_assertion_failure_blocks_canonical_accept():
"""玩法验收虽 accept,模块重生成 no-op/越界的外层失败也必须阻断发布。"""
gd = _make_game_dir(_tmp())
accepted_summary = _v3_summary("accept")
job = _bind_v3_run(accepted_summary, gd, brief="原游戏题面")
job.update({"brief": "改玩法", "modifyMode": "regenerate-module", "baseVersionId": "base-1"})
failed_result = {
"status": "failed",
"stage": "play",
"summary": accepted_summary,
"verdict": accepted_summary.get("verdictFull"),
"error": "断言①失败:目标文件未变",
}
wrapped = W._regen_summary(failed_result, "t1")
out = R.build_result_out(job, wrapped, gd, expected_acceptance_mode="v3")
assert wrapped["ok"] is False, "外层三断言失败必须覆盖旧 run-summary.ok=true"
assert out["status"] == "failed"
assert "engineBundle" not in out
def test_v3_run_summary_identity_mismatch_fails_closed():
"""acceptanceRunId/requestHash 等平铺身份必须与本次 sealed decision 逐字段一致。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
summary["acceptanceRunId"] = "old-run"
assert R.build_result_out(job, summary, gd)["status"] == "failed"
def test_v3_accept_cannot_replay_same_content_across_tasks():
"""作品、题面与产物完全相同,只要当前 traceId 改变,旧封存裁决也不得授权新任务。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
replay = {**job, "traceId": "another-task-trace"}
source_project = _v3_source_project(summary, gd, "x")
assert R.build_result_out(replay, summary, gd, source_project=source_project)["status"] == "failed"
def test_v3_accept_cannot_publish_foreign_release_bundle():
"""发布目录 bundle 在验收后被替换,即使仍含 __GameBundle 也不得回调成功。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
(gd / "bundle.iife.js").write_text("var __GameBundle={foreign:true};", encoding="utf-8")
assert R.build_result_out(job, summary, gd)["status"] == "failed"
def test_v3_accept_cannot_publish_drifted_staged_artifact():
"""staged 目录任一普通文件漂移都会改变 artifactHash,不能只比 bundle 文件。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
staged = gd.parent / "_wg1-gen" / "t1"
(staged / "late-change.txt").write_text("drift", encoding="utf-8")
assert R.build_result_out(job, summary, gd)["status"] == "failed"
def test_v3_snapshot_rejects_bundle_rewrite_then_restore(monkeypatch):
"""快照若读到恶意中间态,即使文件随后恢复,内存快照 hash 仍不匹配旧验收,必须拒绝。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
source_project = _v3_source_project(summary, gd, "x")
staged_bundle = gd.parent / "_wg1-gen" / "t1" / "bundle.iife.js"
accepted_bytes = staged_bundle.read_bytes()
original_capture = R._capture_artifact_snapshot
def capture_evil_then_restore(root):
staged_bundle.write_text("var __GameBundle={evil:true};", encoding="utf-8")
try:
return original_capture(root)
finally:
staged_bundle.write_bytes(accepted_bytes)
monkeypatch.setattr(R, "_capture_artifact_snapshot", capture_evil_then_restore)
out = R.build_result_out(job, summary, gd, source_project=source_project)
assert out["status"] == "failed"
assert "engineBundle" not in out
def test_v3_snapshot_rejects_artifact_file_count_over_hard_cap(monkeypatch):
"""发布快照文件数超过与 Node 一致的 4096 硬帽时,必须 fail-closed。"""
assert R.artifact_snapshot.MAX_ARTIFACT_FILES == 4096
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
# 标准 staged 树已有 bundle + 两个 src 文件;缩小测试帽即可覆盖第 N+1 个文件前拒绝的路径。
monkeypatch.setattr(R.artifact_snapshot, "MAX_ARTIFACT_FILES", 2)
out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x"))
assert out["status"] == "failed"
assert "engineBundle" not in out
def test_v3_snapshot_rejects_artifact_bytes_over_hard_cap(monkeypatch):
"""发布快照总字节超过与 Node 一致的 128 MiB 硬帽时,必须在发布前拒绝。"""
assert R.artifact_snapshot.MAX_ARTIFACT_BYTES == 128 * 1024 * 1024
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
monkeypatch.setattr(R.artifact_snapshot, "MAX_ARTIFACT_BYTES", 1)
out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x"))
assert out["status"] == "failed"
assert "engineBundle" not in out
def test_v3_source_project_rejects_release_src_drift_and_accepts_staged_src():
"""v3 源血缘只认验收同源 staged/src;release/src 的后验改写不得落库。"""
gd = _make_game_dir(_tmp(), src_files={"game-logic.js": "// accepted logic\n",
"core.js": "export const SPEED=1;\n"})
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
staged = gd.parent / "_wg1-gen" / "t1"
profile = {"tickModel": "realtime", "inputModel": "discrete-choice", "progressModel": "metric"}
# 验收后仅漂移 release 源,bundle 与 staged artifact 仍保持不变。
(gd / "src" / "core.js").write_text("export const SPEED=999; // 未验收源\n", encoding="utf-8")
drifted = R.build_source_project(
gd, profile, scaffold_template="_template-story", origin_brief="x",
origin_brief_hash=hashlib.sha256(b"x").hexdigest())
rejected = R.build_result_out(job, summary, gd, source_project=drifted,
expected_acceptance_mode="v3")
assert rejected["status"] == "failed", "active v3 的源血缘属于跨请求可信边界,漂移时必须阻断"
assert "sourceProject" not in rejected, "release/src 后验漂移不得随成功回调落库"
accepted = R.build_source_project(
staged, profile, scaffold_template="_template-story", origin_brief="x",
origin_brief_hash=hashlib.sha256(b"x").hexdigest())
published = R.build_result_out(job, summary, gd, source_project=accepted,
expected_acceptance_mode="v3")
assert published["status"] == "succeeded"
assert json.loads(published["sourceProject"])["files"]["src/core.js"] == "export const SPEED=1;\n"
def test_v3_failed_result_never_carries_source_project():
"""v3 最终发布绑定失败时,即使 SourceProject 来自 staged,也不得落成可继承血缘。"""
gd = _make_game_dir(_tmp())
summary = _v3_summary("accept")
job = _bind_v3_run(summary, gd)
staged = gd.parent / "_wg1-gen" / "t1"
profile = {"tickModel": "realtime", "inputModel": "discrete-choice", "progressModel": "metric"}
source_project = R.build_source_project(
staged, profile, scaffold_template="_template-story", origin_brief="x",
origin_brief_hash=hashlib.sha256(b"x").hexdigest())
(gd / "bundle.iife.js").write_text("var __GameBundle={foreign:true};", encoding="utf-8")
out = R.build_result_out(job, summary, gd, source_project=source_project,
expected_acceptance_mode="v3")
assert out["status"] == "failed"
assert "sourceProject" not in out
if __name__ == "__main__":
_fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
_failed = 0
for _fn in _fns:
try:
_fn()
print(f" PASS {_fn.__name__}")
except Exception as e: # noqa: BLE001
_failed += 1
print(f" FAIL {_fn.__name__}: {type(e).__name__}: {e}")
print(f"\n{len(_fns) - _failed}/{len(_fns)} passed")
sys.exit(1 if _failed else 0)