固化 Match-3 生产者、视觉、音频与双 Judge 证据闭包。 将《山海行纪》r1.1 绑定新的不可变 release,并以生产预检现场核验 bundle、Registry/2 和 25 项 Writer 快照。 同步地图1平衡锁值、跨游戏回归修复、验收契约与 SoT 证据。
1151 lines
56 KiB
Python
1151 lines
56 KiB
Python
"""test_result_out.py — §6.1 result-out 三字段组装单测(M3a U1)。
|
||
|
||
守的不变量(KTD2 BLOCKER 修正,镜像 SaaGraphDispatcher.buildCallbackReqVO + service.py build_callback_payload):
|
||
· gameConfig = 最小占位 Map(templateId/title/theme/engineDriven),绝不塞源工程;
|
||
· engineBundle = bundle.iife.js 全文(含 __GameBundle)= 运行时承重产物;
|
||
· 成功门 = active v3 canonical accept,或真正 legacy 验收通过;两路都要求 engineBundle 含 __GameBundle;
|
||
· sourceProject = best-effort 可选(profile 派生不出即省略,不伪造、不阻断成功);
|
||
· traceId 取 job.traceId 或 job_id;纯函数无网络无 LLM、同输入恒同输出。
|
||
|
||
跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_result_out.py
|
||
"""
|
||
|
||
import copy
|
||
import hashlib
|
||
import json
|
||
import sys
|
||
import tempfile
|
||
from pathlib import Path
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/
|
||
import result_out as R # noqa: E402
|
||
import worker_service as W # noqa: E402
|
||
|
||
_GOOD_BUNDLE = "var __GameBundle=(function(){return{bootGameHost(){}}})();"
|
||
|
||
|
||
def _tmp() -> Path:
|
||
return Path(tempfile.mkdtemp(prefix="ro-test-"))
|
||
|
||
|
||
def _make_game_dir(tmp: Path, *, bundle=_GOOD_BUNDLE, src_files=None) -> Path:
|
||
"""造 amgen-<id>/ 形态产物目录(bundle.iife.js + src/)。"""
|
||
gd = tmp / "amgen-t1"
|
||
(gd / "src").mkdir(parents=True)
|
||
if bundle is not None:
|
||
(gd / "bundle.iife.js").write_text(bundle, encoding="utf-8")
|
||
for name, content in (src_files or {"game-logic.js": "// logic\n", "core.js": "// core\n"}).items():
|
||
(gd / "src" / name).write_text(content, encoding="utf-8")
|
||
return gd
|
||
|
||
|
||
def test_succeeded_shape():
|
||
"""成功路:verdict.pass ∧ bundle 含 __GameBundle → status=succeeded + gameConfig 占位 + engineBundle 承重。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"job_id": "trace-abc", "traceId": "trace-abc", "templateId": "generic", "brief": "点点乐"}
|
||
summary = {"ok": True, "finished": True, "verdict": {"pass": True, "failedGates": []}, "costRmb": 0.05}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert out["traceId"] == "trace-abc"
|
||
assert out["status"] == "succeeded"
|
||
assert out["templateId"] == "generic"
|
||
# gameConfig 占位,绝不含源工程(BLOCKER 修正)
|
||
assert out["gameConfig"]["templateId"] == "generic"
|
||
assert out["gameConfig"]["theme"] == "generic"
|
||
assert out["gameConfig"]["engineDriven"] is True
|
||
assert "files" not in out["gameConfig"] and "entry" not in out["gameConfig"]
|
||
# engineBundle 承重
|
||
assert "__GameBundle" in out["engineBundle"]
|
||
assert out.get("failureReason") is None
|
||
|
||
|
||
def test_failed_when_gates_not_pass():
|
||
"""verdict.pass=False → status=failed + 七值 failureReason,不带 engineBundle。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t2", "templateId": "generic", "brief": "x"}
|
||
summary = {"finished": True, "verdict": {"pass": False, "failedGates": ["E_live"]}}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert out["status"] == "failed"
|
||
assert out.get("engineBundle") is None
|
||
assert out["failureReason"] in R._FAILURE_REASONS # 对齐后端 FailureReasonEnum 真七值
|
||
|
||
|
||
def test_failed_when_engine_bundle_missing___gamebundle():
|
||
"""即使 verdict.pass,bundle 无 __GameBundle → status=failed(承重门:engineBundle 含 __GameBundle)。"""
|
||
gd = _make_game_dir(_tmp(), bundle="var notTheGlobal=1;")
|
||
job = {"traceId": "t3", "templateId": "generic", "brief": "x"}
|
||
summary = {"finished": True, "verdict": {"pass": True, "failedGates": []}}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert out["status"] == "failed"
|
||
assert out.get("engineBundle") is None
|
||
|
||
|
||
def test_failed_breaker_maps_llm_error():
|
||
"""预算熔断 / 内部真因不在七值内 → failureReason=llm_error(契约#6 catch-all;枚举无 budget 专项值)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t4", "templateId": "generic", "brief": "x"}
|
||
summary = {"finished": False, "breaker": True, "verdict": {"pass": None, "failedGates": None}}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert out["status"] == "failed"
|
||
assert out["failureReason"] == "llm_error"
|
||
|
||
|
||
def test_explicit_failure_reason_wins():
|
||
"""显式 failure_reason(在七值内)优先于推断。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t5", "templateId": "generic", "brief": "x"}
|
||
summary = {"finished": True, "verdict": {"pass": False, "failedGates": ["A_boot"]}}
|
||
|
||
out = R.build_result_out(job, summary, gd, failure_reason="timeout")
|
||
|
||
assert out["status"] == "failed"
|
||
assert out["failureReason"] == "timeout"
|
||
|
||
|
||
# ── WU2 §3.9:额度耗尽干净失败 + one-api 惯例分辨 ──
|
||
|
||
def test_quota_exhausted_in_enum():
|
||
"""quota_exhausted 已进 failureReason 枚举(跨契约共享枚举同批新增,与 aigc.yaml/dify-workflow-io.json 一致)。"""
|
||
assert "quota_exhausted" in R._FAILURE_REASONS
|
||
|
||
|
||
def test_summary_failure_reason_quota_maps_through():
|
||
"""WU2 §3.9:summary.failureReason=quota_exhausted(Service 分辨→driver 透传)→ 落 failureReason=quota_exhausted。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "tq", "templateId": "generic", "brief": "x"}
|
||
summary = {"finished": False, "verdict": {"pass": False, "failedGates": None},
|
||
"failureReason": "quota_exhausted"}
|
||
out = R.build_result_out(job, summary, gd)
|
||
assert out["status"] == "failed"
|
||
assert out["failureReason"] == "quota_exhausted"
|
||
|
||
|
||
def test_summary_failure_reason_ignored_when_not_in_enum():
|
||
"""summary.failureReason 非枚举值 → 不透传,回落 catch-all llm_error(防脏值污染契约)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "tq2", "templateId": "generic", "brief": "x"}
|
||
summary = {"finished": False, "verdict": {"pass": False}, "failureReason": "some_garbage"}
|
||
out = R.build_result_out(job, summary, gd)
|
||
assert out["failureReason"] == "llm_error"
|
||
|
||
|
||
def test_explicit_failure_reason_wins_over_summary():
|
||
"""显式入参优先级高于 summary.failureReason。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "tq3", "templateId": "generic", "brief": "x"}
|
||
summary = {"verdict": {"pass": False}, "failureReason": "quota_exhausted"}
|
||
out = R.build_result_out(job, summary, gd, failure_reason="timeout")
|
||
assert out["failureReason"] == "timeout"
|
||
|
||
|
||
def test_classify_newapi_status_codes():
|
||
"""状态码分辨:402/403(one-api 惯例额度码)→ quota_exhausted;裸 401 / 500 / None(无额度关键词)→ llm_error 可重试。"""
|
||
assert R.classify_newapi_failure_reason(402) == "quota_exhausted"
|
||
assert R.classify_newapi_failure_reason(403) == "quota_exhausted"
|
||
assert R.classify_newapi_failure_reason(401) == "llm_error" # 裸 401 无 message = 真凭据失效
|
||
assert R.classify_newapi_failure_reason(500) == "llm_error"
|
||
assert R.classify_newapi_failure_reason(None) == "llm_error"
|
||
|
||
|
||
def test_classify_newapi_real_quota_exhausted_401():
|
||
"""2026-07-08 真机耗尽探测坐实:new-api 对额度耗尽返 HTTP 401 + 「该令牌额度已用尽 …RemainQuota = 0」。
|
||
401 与凭据失效状态码重叠,故必须按 message 分辨——耗尽判 quota_exhausted(干净失败),真凭据失效仍 llm_error(可重试)。
|
||
这是关键回归:旧代码 401→llm_error 会把耗尽误判成可重试 → 反复重试到 watchdog 超时。"""
|
||
real = "[sk-air***0AS] 该令牌额度已用尽 !token.UnlimitedQuota && token.RemainQuota = 0 (request id: 20260708...)"
|
||
assert R.classify_newapi_failure_reason(401, real) == "quota_exhausted"
|
||
assert R.classify_newapi_failure_reason(401, "无效的令牌,请检查") == "llm_error" # 真凭据失效仍可重试、不塌成耗尽
|
||
|
||
|
||
def test_classify_newapi_keyword_fallback():
|
||
"""message 关键词分辨额度耗尽(状态码取不到 / 被框架包装时同样生效,优先于状态码)。"""
|
||
assert R.classify_newapi_failure_reason(None, "error: insufficient user quota") == "quota_exhausted"
|
||
assert R.classify_newapi_failure_reason(None, "当前分组余额不足,请充值") == "quota_exhausted"
|
||
assert R.classify_newapi_failure_reason(None, "connection reset by peer") == "llm_error"
|
||
|
||
|
||
def test_newapi_error_status_text_extraction():
|
||
"""从带 status_code/response 的异常对象抽 (status, text);抽不到 status 返 None、文本兜底 str(exc)。"""
|
||
class _Err(Exception):
|
||
status_code = 402
|
||
message = "insufficient user quota"
|
||
st, txt = R.newapi_error_status_text(_Err("boom"))
|
||
assert st == 402 and "insufficient" in txt
|
||
|
||
class _Resp:
|
||
status_code = 403
|
||
class _Wrapped(Exception):
|
||
response = _Resp()
|
||
st2, _ = R.newapi_error_status_text(_Wrapped("x"))
|
||
assert st2 == 403
|
||
|
||
st3, txt3 = R.newapi_error_status_text(ValueError("plain error"))
|
||
assert st3 is None and "plain error" in txt3
|
||
|
||
|
||
def test_traceid_fallback_to_job_id():
|
||
"""traceId 缺 → 用 job_id(贯穿键同值)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"job_id": "jid-9", "templateId": "generic", "brief": "x"}
|
||
summary = {"verdict": {"pass": True}}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert out["traceId"] == "jid-9"
|
||
|
||
|
||
def test_cost_goes_into_trace_not_toplevel():
|
||
"""成本落 trace.cost.totalRmb——顶层不得有 costRmb(DifyCallbackReqVO 无此字段、后端严格拒未知)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t6", "templateId": "generic", "brief": "x"}
|
||
summary = {"verdict": {"pass": True}, "costRmb": 0.07}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert "costRmb" not in out, "顶层不得有 costRmb(会被后端 ObjectMapper 拒)"
|
||
assert out["trace"]["cost"]["totalRmb"] == 0.07
|
||
|
||
|
||
def test_result_out_keys_within_vo_known_set():
|
||
"""result-out 顶层键严格 ⊆ DifyCallbackReqVO 已知字段集(后端拒未知字段的硬约束)。"""
|
||
vo_known = {"traceId", "status", "templateId", "gameConfig", "assets",
|
||
"engineBundle", "failureReason", "qualityScore", "sourceProject", "trace"}
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t9", "templateId": "generic", "brief": "x"}
|
||
out_ok = R.build_result_out(job, {"verdict": {"pass": True}, "costRmb": 0.07}, gd,
|
||
source_project='{"schemaVersion":"2.0"}')
|
||
out_fail = R.build_result_out(job, {"verdict": {"pass": False}}, gd)
|
||
assert set(out_ok) <= vo_known, f"成功路含未知字段:{set(out_ok) - vo_known}"
|
||
assert set(out_fail) <= vo_known, f"失败路含未知字段:{set(out_fail) - vo_known}"
|
||
|
||
|
||
def test_build_source_project_valid_2_0():
|
||
"""profile 可派生 + src/ 有文件 → sourceProject 是合法 2.0(必填齐 + sourceHash 64-hex)。"""
|
||
import json
|
||
import re
|
||
gd = _make_game_dir(_tmp(), src_files={"game-logic.js": "// l\n", "core.js": "// c\n"})
|
||
profile = {"tickModel": "frame", "inputModel": "tap-targets", "progressModel": "score"}
|
||
|
||
sp_str = R.build_source_project(gd, profile, entry="entry.js")
|
||
|
||
assert sp_str is not None
|
||
sp = json.loads(sp_str)
|
||
assert sp["schemaVersion"] == "2.0"
|
||
assert sp["globalName"] == "__GameBundle"
|
||
assert sp["entry"] == "entry.js"
|
||
assert sp["profile"] == profile
|
||
assert "src/game-logic.js" in sp["files"] and "src/core.js" in sp["files"]
|
||
assert re.fullmatch(r"[a-f0-9]{64}", sp["sourceHash"])
|
||
|
||
|
||
def test_build_source_project_carries_trusted_scaffold_without_changing_source_hash():
|
||
"""scaffoldTemplate 是 additive 血缘;sourceHash 冻结为只覆盖 files,不能暗改既有寻址语义。"""
|
||
gd = _make_game_dir(_tmp())
|
||
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
|
||
without_route = json.loads(R.build_source_project(gd, profile))
|
||
brief = "点击开始完成一局"
|
||
brief_hash = hashlib.sha256(brief.encode("utf-8")).hexdigest()
|
||
with_route = json.loads(R.build_source_project(
|
||
gd, profile, scaffold_template="_template-story",
|
||
origin_brief=brief, origin_brief_hash=brief_hash))
|
||
assert with_route["scaffoldTemplate"] == "_template-story"
|
||
assert with_route["originBrief"] == brief and with_route["originBriefHash"] == brief_hash
|
||
assert with_route["sourceHash"] == without_route["sourceHash"]
|
||
assert with_route["files"] == without_route["files"]
|
||
|
||
|
||
def test_build_source_project_rejects_unknown_scaffold_lineage():
|
||
"""任意模板名不能进入受信 SourceProject 血缘。"""
|
||
gd = _make_game_dir(_tmp())
|
||
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
|
||
assert R.build_source_project(gd, profile, scaffold_template="_template-generic") is None
|
||
|
||
|
||
def test_build_source_project_rejects_empty_or_forged_origin_brief():
|
||
"""active v3 原题面必须非空且正文与 sha256 自洽,不能只带一个格式正确的 hash。"""
|
||
gd = _make_game_dir(_tmp())
|
||
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
|
||
assert R.build_source_project(gd, profile, scaffold_template="_template-story",
|
||
origin_brief="", origin_brief_hash=hashlib.sha256(b"").hexdigest()) is None
|
||
assert R.build_source_project(gd, profile, scaffold_template="_template-story",
|
||
origin_brief="题面甲", origin_brief_hash="a" * 64) is None
|
||
|
||
|
||
def test_source_project_contract_samples_cover_route_whitelist():
|
||
"""新增正负样本必须由 Draft-07 schema 一正一拒,防 enum 只写文档未进机器门。"""
|
||
from jsonschema import Draft7Validator
|
||
|
||
root = Path(__file__).resolve().parents[2]
|
||
schema = json.loads((root / "contracts/agent-loop/source-project.schema.json").read_text(encoding="utf-8"))
|
||
valid = json.loads((root / "contracts/agent-loop/samples/source-project/valid"
|
||
/ "01-v2-with-scaffold-template.json").read_text(encoding="utf-8"))
|
||
invalid = json.loads((root / "contracts/agent-loop/samples/source-project/invalid"
|
||
/ "01-unknown-scaffold-template.json").read_text(encoding="utf-8"))
|
||
missing_origin_hash = json.loads((root / "contracts/agent-loop/samples/source-project/invalid"
|
||
/ "02-origin-brief-missing-hash.json").read_text(encoding="utf-8"))
|
||
validator = Draft7Validator(schema)
|
||
assert list(validator.iter_errors(valid)) == []
|
||
assert list(validator.iter_errors(invalid)), "未知 scaffoldTemplate 必须被 schema 拒绝"
|
||
assert list(validator.iter_errors(missing_origin_hash)), "originBrief/hash 必须成对出现"
|
||
|
||
|
||
def test_build_source_project_omitted_when_profile_incomplete():
|
||
"""profile 派生不出(None / 三枚举不全)→ 返 None(省略、绝不伪造)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
assert R.build_source_project(gd, None, entry="entry.js") is None
|
||
assert R.build_source_project(gd, {"tickModel": "frame"}, entry="entry.js") is None
|
||
|
||
|
||
def test_build_source_project_omitted_when_no_src():
|
||
"""src/ 无文件 → 返 None。"""
|
||
empty = _tmp() / "amgen-e"
|
||
(empty / "src").mkdir(parents=True)
|
||
profile = {"tickModel": "frame", "inputModel": "tap-targets", "progressModel": "score"}
|
||
assert R.build_source_project(empty, profile, entry="entry.js") is None
|
||
|
||
|
||
def test_result_out_source_project_additive():
|
||
"""build_result_out 收到 source_project(JSON str)→ payload 带;缺省不带(additive)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t7", "templateId": "generic", "brief": "x"}
|
||
summary = {"verdict": {"pass": True}}
|
||
|
||
out_with = R.build_result_out(job, summary, gd, source_project='{"schemaVersion":"2.0"}')
|
||
assert out_with["sourceProject"] == '{"schemaVersion":"2.0"}'
|
||
|
||
out_without = R.build_result_out(job, summary, gd)
|
||
assert "sourceProject" not in out_without
|
||
|
||
|
||
def test_deterministic_zero_llm():
|
||
"""纯函数:同输入恒同输出(无网络无 LLM)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "t8", "templateId": "generic", "brief": "稳定"}
|
||
summary = {"verdict": {"pass": True}, "costRmb": 0.01}
|
||
assert R.build_result_out(job, summary, gd) == R.build_result_out(job, summary, gd)
|
||
|
||
|
||
# ──────────────────────────────────────────────────────────────────────────────
|
||
# M3b U2:trace 9d 七项 + D11 首局维(sevenGateVerdict/gatespec)+ repairs 口径 + None 容错
|
||
# 守的不变量(镜像 wg1 _extract_trace + 后端 ReadinessScorer 读取口径):
|
||
# · 必填七项 camelCase:pass/repairs/wallS/models/attempts/gameId/stage;
|
||
# · sevenGateVerdict 只取 {pass,guards}(去 url/ts 噪声,guards.H_progress 保嵌套含 .pass);
|
||
# · gatespec 键名必须是 `driver`(落 driverType 则后端 hasDriver=false、firstPlay 永中性);
|
||
# · repairs = max(0, attempts-1)(首轮 attempts=1→repairs=0);
|
||
# · 成功/失败路都抽 trace;verdictFull/driverType=None 时省略 sevenGateVerdict/gatespec(不塞 None、不崩)。
|
||
# ──────────────────────────────────────────────────────────────────────────────
|
||
|
||
# U1 富化后的成功路 summary(合成;含完整 verdictFull 带 url/ts 噪声 + driverType + stage + models)。
|
||
_SUMMARY_SUCCESS = {
|
||
"ok": True, "finished": True, "gameId": "amgen-t1", "attempts": 1,
|
||
"verdict": {"pass": True, "failedGates": []}, # brief(既有键)
|
||
"verdictFull": { # U1 新增:完整 verdict(带噪声 + guards.H_progress 嵌套)
|
||
"pass": True, "url": "http://localhost:4320/?x", "ts": 1700000000,
|
||
"guards": {
|
||
"A_boot": {"pass": True},
|
||
"H_progress": {"pass": True, "checks": ["score+"], "latch": True, "state0": 0, "state1": 3},
|
||
},
|
||
},
|
||
"driverType": "tap-targets",
|
||
"models": {"code": "MiniMax-M3"},
|
||
"stage": "play",
|
||
"wallSec": 102.3,
|
||
"costRmb": 0.0431,
|
||
}
|
||
|
||
|
||
def test_trace_seven_required_fields_and_d11_dims_success():
|
||
"""成功路 trace:七项齐 + sevenGateVerdict.guards.H_progress.pass + gatespec.driver;噪声 url/ts 已去。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "u2-ok", "templateId": "generic", "brief": "点击得分小游戏"}
|
||
|
||
out = R.build_result_out(job, _SUMMARY_SUCCESS, gd)
|
||
tr = out["trace"]
|
||
|
||
# 必填七项(camelCase 逐字镜像 _extract_trace)
|
||
assert tr["pass"] is True
|
||
assert tr["repairs"] == 0 # attempts=1 → max(0,0)=0(首轮 0)
|
||
assert tr["wallS"] == 102.3
|
||
assert tr["models"] == {"code": "MiniMax-M3"}
|
||
assert tr["attempts"] == 1
|
||
assert tr["gameId"] == "amgen-t1"
|
||
assert tr["stage"] == "play"
|
||
# D11 首局维:sevenGateVerdict 只 {pass,guards},guards.H_progress 嵌套含 .pass(后端 guardPass 读它)
|
||
assert set(tr["sevenGateVerdict"].keys()) == {"pass", "guards"}
|
||
assert "url" not in tr["sevenGateVerdict"] and "ts" not in tr["sevenGateVerdict"]
|
||
assert tr["sevenGateVerdict"]["guards"]["H_progress"]["pass"] is True
|
||
# gatespec 键名必须是 driver(落 driverType 会让后端 hasDriver=false)
|
||
assert tr["gatespec"] == {"driver": "tap-targets"}
|
||
assert "driverType" not in tr["gatespec"]
|
||
# cost 维持现状
|
||
assert tr["cost"]["totalRmb"] == 0.0431
|
||
|
||
|
||
def test_trace_repairs_equals_max0_attempts_minus_1():
|
||
"""repairs = max(0, attempts-1):attempts=1→0、attempts=3→2(对齐 wg1 len(attempts)-1)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "u2-rep", "templateId": "generic", "brief": "x"}
|
||
|
||
s1 = dict(_SUMMARY_SUCCESS, attempts=1)
|
||
assert R.build_result_out(job, s1, gd)["trace"]["repairs"] == 0
|
||
|
||
s3 = dict(_SUMMARY_SUCCESS, attempts=3)
|
||
assert R.build_result_out(job, s3, gd)["trace"]["repairs"] == 2
|
||
|
||
|
||
def test_trace_failed_path_still_has_seven_required():
|
||
"""失败路(verdictFull/driverType=None)trace 仍含七项(repairs/attempts/stage/wallS 排障价值)。"""
|
||
gd = _make_game_dir(_tmp(), bundle="var notTheGlobal=1;") # 无 __GameBundle → status=failed
|
||
job = {"traceId": "u2-fail", "templateId": "generic", "brief": "x"}
|
||
summary = {
|
||
"ok": False, "finished": False, "gameId": "amgen-t2", "attempts": 6,
|
||
"verdict": None, "verdictFull": None, "driverType": None,
|
||
"models": {"code": "MiniMax-M3"}, "stage": "code", "wallSec": 50.1, "costRmb": 0.02,
|
||
}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
assert out["status"] == "failed"
|
||
tr = out["trace"]
|
||
for k in ("pass", "repairs", "wallS", "models", "attempts", "gameId", "stage"):
|
||
assert k in tr, f"失败路 trace 缺必填项 {k}"
|
||
assert tr["repairs"] == 5 # attempts=6 → 5
|
||
assert tr["stage"] == "code"
|
||
assert tr["pass"] is None # play 没跑
|
||
|
||
|
||
def test_trace_none_tolerance_omits_d11_dims():
|
||
"""verdictFull/driverType=None(对照路)→ 省略 sevenGateVerdict/gatespec(不塞 None、不崩)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "u2-none", "templateId": "generic", "brief": "x"}
|
||
summary = {
|
||
"gameId": "amgen-t3", "attempts": 2, "verdict": None,
|
||
"verdictFull": None, "driverType": None,
|
||
"models": {"code": "MiniMax-M3"}, "stage": "smoke", "wallSec": 10.0,
|
||
}
|
||
|
||
out = R.build_result_out(job, summary, gd)
|
||
tr = out["trace"]
|
||
assert "sevenGateVerdict" not in tr # verdictFull=None → 省略
|
||
assert "gatespec" not in tr # driverType=None → 省略
|
||
assert tr["repairs"] == 1 # attempts=2 → 1
|
||
|
||
|
||
def test_trace_keys_camelcase_match_readiness_scorer():
|
||
"""trace 键名逐字镜像 ReadinessScorer 读取口径(拼错会被后端静默接收 → 假绿)。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = {"traceId": "u2-keys", "templateId": "generic", "brief": "x"}
|
||
tr = R.build_result_out(job, _SUMMARY_SUCCESS, gd)["trace"]
|
||
# ReadinessScorer 读取的确切键(camelCase):trace.pass / gatespec.driver /
|
||
# sevenGateVerdict.guards.H_progress.pass / repairs / cost.totalRmb
|
||
assert "pass" in tr and "gatespec" in tr and "sevenGateVerdict" in tr
|
||
assert "wallS" in tr and "gameId" in tr # 非 wall_s / game_id(snake 会被后端读不到)
|
||
assert "driver" in tr["gatespec"]
|
||
assert "H_progress" in tr["sevenGateVerdict"]["guards"]
|
||
|
||
|
||
# ──────────────────────────────────────────────────────────────────────────────
|
||
# 生成验收 v3 消费层:decision/outcome 唯一写权、shadow 冻结、trace.playtest 轻量投影
|
||
# ──────────────────────────────────────────────────────────────────────────────
|
||
|
||
_V2_HISTORY_SAMPLE_DIR = (Path(__file__).resolve().parents[2] / "contracts" / "play-loop"
|
||
/ "samples" / "playtest-evidence" / "valid")
|
||
_V2_HISTORY_SAMPLE_BY_OUTCOME = {
|
||
"accept": "01-accepted-narrative.json",
|
||
"reject": "02-rejected-mechanical.json",
|
||
"tester_error": "03-tester-error-action-protocol.json",
|
||
"inconclusive": "04-inconclusive-insufficient-proof.json",
|
||
}
|
||
|
||
|
||
def _v3_fixture_hash(label: str) -> str:
|
||
"""为 A/B/consensus/目标解析 fixture 生成稳定 SHA-256。"""
|
||
return hashlib.sha256(label.encode("utf-8")).hexdigest()
|
||
|
||
|
||
def _upgrade_v3_action(action: dict, index: int) -> dict:
|
||
"""把历史坐标动作升级为 playtest/3 的结构化选择和确定性解析事实。"""
|
||
upgraded = copy.deepcopy(action)
|
||
normalized = upgraded["normalized"]
|
||
action_type = normalized["type"]
|
||
target_set_hash = _v3_fixture_hash(f"target-set-{index}")
|
||
if action_type == "tap":
|
||
target = {"id": f"r{index:02d}"}
|
||
selection = {
|
||
"schemaVersion": "ActorSelection/1", "type": "tap",
|
||
"targetSetHash": target_set_hash, "target": target,
|
||
}
|
||
points = [{"role": "target", "target": target,
|
||
"point": {"x": normalized["x"], "y": normalized["y"]}}]
|
||
elif action_type == "key":
|
||
selection = {"schemaVersion": "ActorSelection/1", "type": "key", "key": normalized["key"]}
|
||
points = None
|
||
elif action_type == "wait":
|
||
selection = {"schemaVersion": "ActorSelection/1", "type": "wait", "ms": normalized["ms"]}
|
||
points = None
|
||
else:
|
||
raise AssertionError(f"v3 result fixture 暂不支持动作:{action_type}")
|
||
upgraded["request"] = {
|
||
"raw": json.dumps(selection, ensure_ascii=False, sort_keys=True),
|
||
"parsed": copy.deepcopy(selection),
|
||
}
|
||
upgraded["retryCount"] = 0
|
||
upgraded["protocolErrors"] = []
|
||
if points is None:
|
||
upgraded["targetResolution"] = {"status": "not_applicable"}
|
||
else:
|
||
upgraded["targetResolution"] = {
|
||
"status": "resolved", "targetSetRef": f"targets/action-{index}.json",
|
||
"targetSetHash": target_set_hash,
|
||
"sourceFrameRef": upgraded["preFrameRef"]["path"],
|
||
"sourceFrameHash": upgraded["preFrameRef"]["hash"],
|
||
"guideManifestRef": f"guides/action-{index}.json",
|
||
"guideManifestHash": _v3_fixture_hash(f"guide-{index}"),
|
||
"resolverVersion": "1.0.0", "selection": copy.deepcopy(selection),
|
||
"resolvedPoints": points, "error": None,
|
||
}
|
||
return upgraded
|
||
|
||
|
||
def _upgrade_history_result_to_v3(v2: dict) -> dict:
|
||
"""把冻结 v2 DECIDED 样本升级成 active/shadow 测试使用的 playtest/3 对象。"""
|
||
v3 = copy.deepcopy(v2)
|
||
v3.pop("_note", None)
|
||
v3["schemaVersion"] = "playtest/3"
|
||
v3["environment"]["virtualTime"]["waitMaxMs"] = 600
|
||
if isinstance(v3.get("actor"), dict):
|
||
v3["actor"]["contextProjectionVersion"] = "ActorView/3"
|
||
if v3.get("outcome") == "tester_error":
|
||
# 失败 selection 只属于 attempt audit,不得伪装为正式 actionRecord。
|
||
v3["actions"] = []
|
||
else:
|
||
v3["actions"] = [_upgrade_v3_action(action, index)
|
||
for index, action in enumerate(v3.get("actions") or [], start=1)]
|
||
|
||
legacy_judge = v2.get("judge") if isinstance(v2.get("judge"), dict) else None
|
||
source_roll_id = None
|
||
if legacy_judge is not None and v3.get("rolls"):
|
||
source_roll_id = str(v3["rolls"][-1]["rollId"])
|
||
package_hash = str(legacy_judge["packageHash"])
|
||
result_a = _v3_fixture_hash(f"{source_roll_id}-judge-a-result")
|
||
result_b = _v3_fixture_hash(f"{source_roll_id}-judge-b-result")
|
||
consensus_hash = _v3_fixture_hash(f"{source_roll_id}-consensus")
|
||
for roll in v3["rolls"]:
|
||
roll_id = str(roll["rollId"])
|
||
roll.update({
|
||
"proofPackageHash": _v3_fixture_hash(f"{roll_id}-proof"),
|
||
"judgePackageHash": package_hash,
|
||
"judgeARef": {
|
||
"rawRef": f"{roll_id}/judge-a/raw.json", "rawHash": _v3_fixture_hash(f"{roll_id}-a-raw"),
|
||
"parsedRef": f"{roll_id}/judge-a/parsed.json", "parsedHash": _v3_fixture_hash(f"{roll_id}-a-parsed"),
|
||
"normalizedRef": f"{roll_id}/judge-a/normalized.json", "resultHash": result_a,
|
||
},
|
||
"judgeAHash": _v3_fixture_hash(f"{roll_id}-a-manifest"),
|
||
"judgeBRef": {
|
||
"rawRef": f"{roll_id}/judge-b/raw.json", "rawHash": _v3_fixture_hash(f"{roll_id}-b-raw"),
|
||
"parsedRef": f"{roll_id}/judge-b/parsed.json", "parsedHash": _v3_fixture_hash(f"{roll_id}-b-parsed"),
|
||
"normalizedRef": f"{roll_id}/judge-b/normalized.json", "resultHash": result_b,
|
||
},
|
||
"judgeBHash": _v3_fixture_hash(f"{roll_id}-b-manifest"),
|
||
"judgeConsensusRef": f"{roll_id}/consensus.json",
|
||
"judgeConsensusHash": consensus_hash,
|
||
"costReservationRef": f"{roll_id}/cost-reservation.json",
|
||
"costReservationHash": _v3_fixture_hash(f"{roll_id}-reservation"),
|
||
})
|
||
v3["judge"] = {
|
||
"schemaVersion": "JudgeConsensus/1", "sourceRollId": source_roll_id,
|
||
"consensusRef": f"{source_roll_id}/consensus.json", "consensusHash": consensus_hash,
|
||
"packageHash": package_hash, "strategyVersion": "1.0.0",
|
||
"decision": legacy_judge["decision"], "degraded": bool(legacy_judge.get("degraded")),
|
||
"sameModel": True,
|
||
"independenceChecks": {
|
||
"differentLogicalClient": True, "differentSession": True, "differentPolicySeed": True,
|
||
"differentPromptArtifact": True, "differentEvidenceDir": True,
|
||
"differentRawFile": True, "sameJudgePackage": True,
|
||
"sameImageManifest": True, "sameModel": True,
|
||
},
|
||
"judgeAResultHash": result_a, "judgeBResultHash": result_b,
|
||
"obligationResults": copy.deepcopy(legacy_judge["obligationResults"]),
|
||
"conflicts": list(legacy_judge.get("contradictions") or []),
|
||
"failureSignature": None,
|
||
}
|
||
else:
|
||
v3["judge"] = None
|
||
# 旧 action-protocol tester_error 只有失败 attempt;正式 actionRecord 不应收录它。
|
||
if v3.get("outcome") == "tester_error":
|
||
v3["actions"], v3["rolls"], v3["events"] = [], [], []
|
||
v3["costRmb"] = 0
|
||
v3["decision"]["costRmb"] = 0
|
||
v3["sourceRollId"] = source_roll_id
|
||
accepted = v3["outcome"] == "accept"
|
||
v3["compatibility"] = {
|
||
"accepted": accepted and v3["acceptanceMode"] == "v3",
|
||
"ok": accepted and v3["acceptanceMode"] == "v3",
|
||
"acceptanceVersion": v3["acceptanceMode"],
|
||
"publishFrozen": bool(v3["decision"]["publishFrozen"]),
|
||
"schemaVersion": "playtest/3", "sourceRollId": source_roll_id,
|
||
}
|
||
return v3
|
||
|
||
|
||
def _set_v3_mode(v3: dict, mode: str) -> None:
|
||
"""在完整 canonical 样本上同步切换 active/shadow,保持所有镜像字段逐字一致。"""
|
||
accepted = v3["outcome"] == "accept"
|
||
frozen = mode == "v3_shadow" or not accepted
|
||
v3["acceptanceMode"] = mode
|
||
v3["decision"]["shadowAccepted"] = accepted if mode == "v3_shadow" else None
|
||
v3["decision"]["publishFrozen"] = frozen
|
||
compatibility = v3["compatibility"]
|
||
compatibility["accepted"] = accepted and mode == "v3"
|
||
compatibility["ok"] = accepted and mode == "v3"
|
||
compatibility["acceptanceVersion"] = mode
|
||
compatibility["publishFrozen"] = frozen
|
||
|
||
|
||
def _v3_result(outcome="accept", *, mode="v3", accepted=None, frozen=None, proof_complete=None):
|
||
"""把冻结样本升级为 v3;旧文件只作历史素材,active fixture 只返回 playtest/3。"""
|
||
path = _V2_HISTORY_SAMPLE_DIR / _V2_HISTORY_SAMPLE_BY_OUTCOME[outcome]
|
||
v3 = _upgrade_history_result_to_v3(json.loads(path.read_text(encoding="utf-8")))
|
||
_set_v3_mode(v3, mode)
|
||
if accepted is not None:
|
||
v3["decision"]["accepted"] = accepted
|
||
if frozen is not None:
|
||
v3["decision"]["publishFrozen"] = frozen
|
||
v3["compatibility"]["publishFrozen"] = frozen
|
||
if proof_complete is not None:
|
||
for obligation in v3["proofObligations"]:
|
||
if obligation.get("required") is True:
|
||
obligation["status"] = "satisfied" if proof_complete else "missing"
|
||
return v3
|
||
|
||
|
||
def _v3_summary(outcome="accept", **kwargs):
|
||
"""生产形态:完整结果在 acceptanceV3,顶层只镜像六字段 compatibility。"""
|
||
v3 = _v3_result(outcome, **kwargs)
|
||
compatibility = v3["compatibility"]
|
||
summary = {"verdict": {"pass": True}, "floor": copy.deepcopy(v3["floor"])}
|
||
for key in ("accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId"):
|
||
summary[key] = copy.deepcopy(compatibility[key])
|
||
summary["acceptanceV3"] = v3
|
||
summary["publishFrozen"] = v3["decision"]["publishFrozen"]
|
||
return summary
|
||
|
||
|
||
def _bind_v3_run(summary: dict, gd: Path, *, game_id: str = "t1", brief: str = "x") -> dict:
|
||
"""构造真实发布边界:当前 job、run-summary、staged artifact 与 release bundle 四者同源。"""
|
||
v3 = summary.get("acceptanceV3")
|
||
assert isinstance(v3, dict), "生产绑定夹具必须使用 acceptanceV3 包装态"
|
||
staged = gd.parent / "_wg1-gen" / game_id
|
||
staged.mkdir(parents=True, exist_ok=True)
|
||
release_bytes = (gd / "bundle.iife.js").read_bytes()
|
||
(staged / "bundle.iife.js").write_bytes(release_bytes)
|
||
staged_src = staged / "src"
|
||
staged_src.mkdir(parents=True, exist_ok=True)
|
||
for source_file in sorted((gd / "src").glob("*.js")):
|
||
(staged_src / source_file.name).write_bytes(source_file.read_bytes())
|
||
|
||
v3["gameId"] = game_id
|
||
v3["briefHash"] = hashlib.sha256(brief.encode("utf-8")).hexdigest()
|
||
trace_id = f"trace-{game_id}"
|
||
v3["taskBindingHash"] = hashlib.sha256(trace_id.encode("utf-8")).hexdigest()
|
||
v3["artifactHash"] = R.cheap_verify._artifact_hash_v3(game_id, staged)
|
||
v3["environment"]["buildRef"] = v3["artifactHash"]
|
||
summary["gameId"] = game_id
|
||
summary["acceptanceRunId"] = v3["runId"]
|
||
summary["acceptanceRequestHash"] = v3["acceptanceRequestHash"]
|
||
summary["acceptanceTaskBindingHash"] = v3["taskBindingHash"]
|
||
summary["acceptanceArtifactHash"] = v3["artifactHash"]
|
||
summary["acceptanceBriefHash"] = v3["briefHash"]
|
||
return {"traceId": trace_id, "gameId": game_id, "brief": brief}
|
||
|
||
|
||
def _v3_source_project(summary: dict, gd: Path, origin_brief: str) -> str:
|
||
"""从验收同源 staged/src 组 active v3 SourceProject 测试夹具。"""
|
||
v3 = summary["acceptanceV3"]
|
||
staged = gd.parent / "_wg1-gen" / v3["gameId"]
|
||
profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"}
|
||
source_project = R.build_source_project(
|
||
staged, profile, scaffold_template=v3["templateRoute"], origin_brief=origin_brief,
|
||
origin_brief_hash=hashlib.sha256(origin_brief.encode("utf-8")).hexdigest())
|
||
assert source_project is not None
|
||
return source_project
|
||
|
||
|
||
def test_v3_accept_projects_trusted_trace_and_succeeds():
|
||
"""active v3 accept + 完整硬证才成功;trace 只保留后端消费所需轻量字段。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
out = R.build_result_out(job, summary, gd,
|
||
source_project=_v3_source_project(summary, gd, "x"))
|
||
|
||
assert out["status"] == "succeeded"
|
||
playtest = out["trace"]["playtest"]
|
||
assert playtest["schemaVersion"] == "playtest/3"
|
||
assert playtest["acceptanceMode"] == "v3" and playtest["outcome"] == "accept"
|
||
assert playtest["accepted"] is True and playtest["publishFrozen"] is False
|
||
assert playtest["compatibilityAccepted"] is True and playtest["compatibilityOk"] is True
|
||
assert playtest["floorPass"] is True and playtest["finalPostguardPass"] is True
|
||
assert playtest["writeAuthority"] is True and all(playtest["finalChecks"].values())
|
||
assert playtest["blockingProblems"] == [] and playtest["contradictions"] == []
|
||
assert playtest["mergeConflicts"] == [] and playtest["proofComplete"] is True
|
||
assert all(playtest["evidenceRefs"][key] for key in ("actions", "frames", "events"))
|
||
assert playtest["firstPlay"]["loopClosed"] is True and playtest["firstPlay"]["proofRefs"]
|
||
assert out["trace"]["acceptanceState"] == "active"
|
||
identity = out["trace"]["playtest"]
|
||
assert {key: identity[key] for key in (
|
||
"runId", "gameId", "briefHash", "artifactHash", "taskBindingHash", "acceptanceRequestHash"
|
||
)} == {key: summary["acceptanceV3"][key] for key in (
|
||
"runId", "gameId", "briefHash", "artifactHash", "taskBindingHash", "acceptanceRequestHash"
|
||
)}
|
||
assert identity["engineBundleHash"] == hashlib.sha256(out["engineBundle"].encode("utf-8")).hexdigest()
|
||
assert set(summary["acceptanceV3"]["compatibility"]) == {
|
||
"accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId",
|
||
}
|
||
|
||
|
||
def test_v3_shadow_accept_is_frozen_even_when_other_fields_are_green():
|
||
"""shadow decision 可为 accept,但 publishFrozen 是硬闸,不能携包进入 succeeded。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept", mode="v3_shadow")
|
||
job = _bind_v3_run(summary, gd)
|
||
out = R.build_result_out(job, summary, gd)
|
||
|
||
assert out["status"] == "failed"
|
||
assert out["trace"]["playtest"]["outcome"] == "accept"
|
||
assert out["trace"]["playtest"]["accepted"] is True
|
||
assert out["trace"]["playtest"]["publishFrozen"] is True
|
||
assert out["trace"]["acceptanceState"] == "shadow"
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_v3_non_accept_outcomes_always_fail_despite_legacy_accept_true():
|
||
"""旧平铺 accepted 被污染为 true 也不能覆盖 v3 四态 decision。"""
|
||
gd = _make_game_dir(_tmp())
|
||
for outcome in ("reject", "inconclusive", "tester_error"):
|
||
summary = _v3_summary(outcome)
|
||
summary["accepted"] = True
|
||
summary["ok"] = True
|
||
job = _bind_v3_run(summary, gd)
|
||
job["traceId"] = f"v3-{outcome}"
|
||
out = R.build_result_out(job, summary, gd)
|
||
assert out["status"] == "failed", outcome
|
||
assert out["trace"]["playtest"]["outcome"] == outcome
|
||
assert out["trace"]["playtest"]["accepted"] is False
|
||
|
||
|
||
def _assert_v3_failed(summary: dict, trace_id: str) -> None:
|
||
"""统一断言 v3 单点变异不能携带产物进入 succeeded。"""
|
||
gd = _make_game_dir(_tmp())
|
||
job = _bind_v3_run(summary, gd)
|
||
job["traceId"] = trace_id
|
||
out = R.build_result_out(job, summary, gd)
|
||
assert out["status"] == "failed"
|
||
assert "engineBundle" not in out
|
||
assert out["trace"]["acceptanceState"] in ("downgrade_detected", "shadow", "frozen")
|
||
|
||
|
||
def test_v3_contradiction_1_top_outcome_fails_closed():
|
||
"""顶层 outcome 与 decision/compatibility 不一致,canonical 语义校验必须拦截。"""
|
||
summary = _v3_summary("accept")
|
||
summary["acceptanceV3"]["outcome"] = "reject"
|
||
_assert_v3_failed(summary, "v3-bad-outcome")
|
||
|
||
|
||
def test_v3_contradiction_2_floor_fails_closed():
|
||
"""accept 却让机械四门 floor 失败,不能靠旧 verdict.pass=true 放行。"""
|
||
summary = _v3_summary("accept")
|
||
summary["acceptanceV3"]["floor"]["pass"] = False
|
||
_assert_v3_failed(summary, "v3-bad-floor")
|
||
|
||
|
||
def test_v3_contradiction_3_final_postguard_fails_closed():
|
||
"""finalPostguard 任一检查、写权或矛盾项异常都必须 fail-closed。"""
|
||
variants = []
|
||
bad_check = _v3_summary("accept")
|
||
bad_check["acceptanceV3"]["finalPostguard"]["checks"]["floorPass"] = False
|
||
variants.append(bad_check)
|
||
bad_writer = _v3_summary("accept")
|
||
bad_writer["acceptanceV3"]["finalPostguard"]["writeAuthority"] = False
|
||
variants.append(bad_writer)
|
||
bad_problem = _v3_summary("accept")
|
||
bad_problem["acceptanceV3"]["finalPostguard"]["blockingProblems"] = ["仍有阻塞问题"]
|
||
variants.append(bad_problem)
|
||
for index, summary in enumerate(variants):
|
||
_assert_v3_failed(summary, f"v3-bad-final-{index}")
|
||
|
||
|
||
def test_v3_contradiction_4_merge_fails_closed():
|
||
"""mergeCandidate 被改写或仍有两掷冲突时,最终 accept 不成立。"""
|
||
bad_candidate = _v3_summary("accept")
|
||
bad_candidate["acceptanceV3"]["merge"]["mergeCandidate"] = "reject"
|
||
bad_conflict = _v3_summary("accept")
|
||
bad_conflict["acceptanceV3"]["merge"]["conflicts"] = ["两掷证据冲突"]
|
||
_assert_v3_failed(bad_candidate, "v3-bad-merge-candidate")
|
||
_assert_v3_failed(bad_conflict, "v3-bad-merge-conflict")
|
||
|
||
|
||
def test_v3_contradiction_5_judge_fails_closed():
|
||
"""独立 Judge 拒绝或降级时,兼容投影再绿也不能发布。"""
|
||
bad_decision = _v3_summary("accept")
|
||
bad_decision["acceptanceV3"]["judge"]["decision"] = "reject"
|
||
bad_degraded = _v3_summary("accept")
|
||
bad_degraded["acceptanceV3"]["judge"]["degraded"] = True
|
||
_assert_v3_failed(bad_decision, "v3-bad-judge-decision")
|
||
_assert_v3_failed(bad_degraded, "v3-bad-judge-degraded")
|
||
|
||
|
||
def test_v3_contradiction_6_compatibility_fails_closed():
|
||
"""compatibility 的 mode、冻结位或 accepted/ok 与权威 decision 矛盾时不得放行。"""
|
||
for key, value in (("ok", False), ("accepted", False), ("publishFrozen", True),
|
||
("acceptanceVersion", "v3_shadow")):
|
||
summary = _v3_summary("accept")
|
||
summary["acceptanceV3"]["compatibility"][key] = value
|
||
_assert_v3_failed(summary, f"v3-bad-compat-{key}")
|
||
|
||
|
||
def test_v3_contradiction_7_proof_refs_loop_fails_closed():
|
||
"""required proof、三类引用或首局闭环任一缺失,都不能被模型 accept 文本替代。"""
|
||
bad_proof = _v3_summary("accept")
|
||
bad_proof["acceptanceV3"]["proofObligations"][0]["status"] = "failed"
|
||
bad_refs = _v3_summary("accept")
|
||
bad_refs["acceptanceV3"]["proofObligations"][0]["actionRefs"] = []
|
||
bad_loop = _v3_summary("accept")
|
||
bad_loop["acceptanceV3"]["firstPlay"]["loopClosed"] = False
|
||
for index, summary in enumerate((bad_proof, bad_refs, bad_loop)):
|
||
_assert_v3_failed(summary, f"v3-bad-proof-{index}")
|
||
|
||
|
||
def test_v3_flat_projection_contradiction_fails_closed():
|
||
"""完整封存对象虽有效,run-summary 平铺 accepted 被污染也必须失败。"""
|
||
summary = _v3_summary("accept")
|
||
summary["accepted"] = False
|
||
_assert_v3_failed(summary, "v3-bad-flat")
|
||
|
||
|
||
def test_v3_declared_without_acceptance_payload_is_downgrade_detected():
|
||
"""任一 v3 声明已出现却缺 acceptanceV3 时,不得回落 legacy verdict/accepted。"""
|
||
claims = [
|
||
{"acceptanceVersion": "v3"},
|
||
{"acceptanceVersion": "historical_replay"},
|
||
{"acceptanceVersion": "v999"},
|
||
{"acceptanceVersion": {"corrupt": True}},
|
||
{"publishFrozen": False},
|
||
{"trace": {"playtest": {"schemaVersion": "playtest/2", "outcome": "accept"}}},
|
||
]
|
||
gd = _make_game_dir(_tmp())
|
||
for index, claim in enumerate(claims):
|
||
summary = {"verdict": {"pass": True}, "accepted": True, "ok": True, **claim}
|
||
out = R.build_result_out({"traceId": f"v3-downgrade-{index}", "brief": "x"}, summary, gd)
|
||
assert out["status"] == "failed"
|
||
assert out["trace"]["acceptanceState"] == "downgrade_detected"
|
||
assert out["trace"]["publishFrozen"] is True
|
||
assert "playtest" not in out["trace"]
|
||
|
||
|
||
def test_expected_v3_blocks_fully_stripped_summary_from_legacy_fallback():
|
||
"""生产边界已冻结 v3 时,即使 summary 所有 marker 都被剥掉,legacy pass=true 也不能放行。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = {"gameId": "t1", "verdict": {"pass": True}, "accepted": True, "ok": True}
|
||
out = R.build_result_out(
|
||
{"traceId": "v3-fully-stripped", "gameId": "t1", "brief": "x"}, summary, gd,
|
||
expected_acceptance_mode="v3")
|
||
assert out["status"] == "failed"
|
||
assert out["trace"]["acceptanceState"] == "downgrade_detected"
|
||
assert out["trace"]["publishFrozen"] is True
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_legacy_v1_stays_compatible_but_sealed_v2_is_observation_only():
|
||
"""无协议的 v1 存量仍兼容;冻结 playtest/2 可观察但不能作为 active 发布授权。"""
|
||
gd = _make_game_dir(_tmp())
|
||
legacy = {"verdict": {"pass": True}, "accepted": True, "playtest": {"accepted": True}}
|
||
legacy_out = R.build_result_out({"traceId": "legacy-v1", "brief": "x"}, legacy, gd)
|
||
assert legacy_out["status"] == "succeeded"
|
||
assert "acceptanceState" not in legacy_out["trace"]
|
||
|
||
frozen_v2 = json.loads(
|
||
(_V2_HISTORY_SAMPLE_DIR / _V2_HISTORY_SAMPLE_BY_OUTCOME["accept"]).read_text(encoding="utf-8")
|
||
)
|
||
observed = R.build_result_out({"traceId": "historical-v2", "brief": "x"}, frozen_v2, gd)
|
||
assert observed["status"] == "failed"
|
||
assert observed["trace"]["playtest"]["schemaVersion"] == "playtest/2"
|
||
assert "engineBundle" not in observed
|
||
|
||
|
||
def test_v3_unknown_schema_and_outcome_fail_closed():
|
||
"""未知 schema/outcome 不能回落顶层 accepted 或旧四门判定。"""
|
||
gd = _make_game_dir(_tmp())
|
||
bad_schema = _v3_summary("accept")
|
||
bad_schema["acceptanceV3"]["schemaVersion"] = "playtest/999"
|
||
bad_outcome = _v3_summary("accept")
|
||
bad_outcome["acceptanceV3"]["decision"]["outcome"] = "maybe"
|
||
|
||
schema_job = _bind_v3_run(bad_schema, gd)
|
||
outcome_job = _bind_v3_run(bad_outcome, gd)
|
||
assert R.build_result_out(schema_job, bad_schema, gd)["status"] == "failed"
|
||
assert R.build_result_out(outcome_job, bad_outcome, gd)["status"] == "failed"
|
||
|
||
|
||
def test_v3_failure_ignores_stale_legacy_failure_reason():
|
||
"""v3 outcome 已成为真相后,旧 worker 残留的合法枚举也不能把玩法失败误归为 unsafe_prompt。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("reject", mode="v3")
|
||
summary["failureReason"] = "unsafe_prompt"
|
||
job = _bind_v3_run(summary, gd)
|
||
|
||
out = R.build_result_out(job, summary, gd,
|
||
failure_reason="unsafe_prompt")
|
||
|
||
assert out["status"] == "failed"
|
||
assert out["failureReason"] == "llm_error"
|
||
assert out["trace"]["playtest"]["outcome"] == "reject"
|
||
|
||
|
||
def test_historical_replay_marker_never_falls_back_to_legacy_pass():
|
||
"""历史回放即使携带旧 pass/accepted=true 且 bundle 完整,也只能降级阻断,永不发布。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = {
|
||
"verdict": {"pass": True}, "accepted": True, "ok": True,
|
||
"acceptanceVersion": "historical_replay",
|
||
}
|
||
|
||
out = R.build_result_out({"traceId": "historical-no-publish", "brief": "x"}, summary, gd)
|
||
|
||
assert out["status"] == "failed"
|
||
assert out["trace"]["acceptanceState"] == "downgrade_detected"
|
||
assert out["trace"]["publishFrozen"] is True
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_v3_direct_sealed_result_is_observable_but_never_publishable():
|
||
"""裸 decision.json 可离线观察,但缺当前 run-summary 身份绑定,不能直接授权发布。"""
|
||
gd = _make_game_dir(_tmp())
|
||
out = R.build_result_out({"traceId": "v3-direct", "gameId": "t1", "brief": "x"},
|
||
_v3_result("accept"), gd)
|
||
assert out["status"] == "failed"
|
||
assert out["trace"]["playtest"]["schemaVersion"] == "playtest/3"
|
||
|
||
|
||
def test_v3_old_accept_cannot_publish_for_different_job():
|
||
"""同一封存 accept 被塞给另一 gameId 时必须 fail-closed。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
job["gameId"] = "t2"
|
||
assert R.build_result_out(job, summary, gd)["status"] == "failed"
|
||
|
||
|
||
def test_v3_accept_cannot_publish_for_different_create_brief():
|
||
"""create 当前 brief 与封存 briefHash 不同,旧 accept 不得跨题面重放。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
job["brief"] = "另一道题"
|
||
assert R.build_result_out(job, summary, gd)["status"] == "failed"
|
||
|
||
|
||
def test_v3_modify_keeps_original_brief_hash_binding():
|
||
"""modify job.brief 是修改指令;原题面 briefHash 由封存验收继承,不能按 create 规则误拒。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd, brief="原游戏题面")
|
||
job.update({"brief": "把速度调快", "modifyMode": "deterministic", "baseVersionId": "base-1"})
|
||
source_project = _v3_source_project(summary, gd, "原游戏题面")
|
||
assert R.build_result_out(job, summary, gd, source_project=source_project)["status"] == "succeeded"
|
||
|
||
|
||
def test_v3_modify_rejects_missing_empty_or_cross_brief_source_origin():
|
||
"""modify 不得用缺失/空/另一题面的 SourceProject 原题面授权 active v3 发布。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd, brief="原游戏题面")
|
||
job.update({"brief": "把速度调快", "modifyMode": "deterministic", "baseVersionId": "base-1"})
|
||
assert R.build_result_out(job, summary, gd, source_project=None)["status"] == "failed"
|
||
|
||
valid = json.loads(_v3_source_project(summary, gd, "原游戏题面"))
|
||
for brief in ("", "另一款游戏题面"):
|
||
forged = dict(valid)
|
||
forged["originBrief"] = brief
|
||
forged["originBriefHash"] = hashlib.sha256(brief.encode("utf-8")).hexdigest()
|
||
out = R.build_result_out(job, summary, gd, source_project=json.dumps(forged, ensure_ascii=False))
|
||
assert out["status"] == "failed"
|
||
|
||
|
||
def test_v3_regenerate_outer_assertion_failure_blocks_canonical_accept():
|
||
"""玩法验收虽 accept,模块重生成 no-op/越界的外层失败也必须阻断发布。"""
|
||
gd = _make_game_dir(_tmp())
|
||
accepted_summary = _v3_summary("accept")
|
||
job = _bind_v3_run(accepted_summary, gd, brief="原游戏题面")
|
||
job.update({"brief": "改玩法", "modifyMode": "regenerate-module", "baseVersionId": "base-1"})
|
||
failed_result = {
|
||
"status": "failed",
|
||
"stage": "play",
|
||
"summary": accepted_summary,
|
||
"verdict": accepted_summary.get("verdictFull"),
|
||
"error": "断言①失败:目标文件未变",
|
||
}
|
||
|
||
wrapped = W._regen_summary(failed_result, "t1")
|
||
out = R.build_result_out(job, wrapped, gd, expected_acceptance_mode="v3")
|
||
|
||
assert wrapped["ok"] is False, "外层三断言失败必须覆盖旧 run-summary.ok=true"
|
||
assert out["status"] == "failed"
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_v3_run_summary_identity_mismatch_fails_closed():
|
||
"""acceptanceRunId/requestHash 等平铺身份必须与本次 sealed decision 逐字段一致。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
summary["acceptanceRunId"] = "old-run"
|
||
assert R.build_result_out(job, summary, gd)["status"] == "failed"
|
||
|
||
|
||
def test_v3_accept_cannot_replay_same_content_across_tasks():
|
||
"""作品、题面与产物完全相同,只要当前 traceId 改变,旧封存裁决也不得授权新任务。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
replay = {**job, "traceId": "another-task-trace"}
|
||
source_project = _v3_source_project(summary, gd, "x")
|
||
assert R.build_result_out(replay, summary, gd, source_project=source_project)["status"] == "failed"
|
||
|
||
|
||
def test_v3_accept_cannot_publish_foreign_release_bundle():
|
||
"""发布目录 bundle 在验收后被替换,即使仍含 __GameBundle 也不得回调成功。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
(gd / "bundle.iife.js").write_text("var __GameBundle={foreign:true};", encoding="utf-8")
|
||
assert R.build_result_out(job, summary, gd)["status"] == "failed"
|
||
|
||
|
||
def test_v3_accept_cannot_publish_drifted_staged_artifact():
|
||
"""staged 目录任一普通文件漂移都会改变 artifactHash,不能只比 bundle 文件。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
staged = gd.parent / "_wg1-gen" / "t1"
|
||
(staged / "late-change.txt").write_text("drift", encoding="utf-8")
|
||
assert R.build_result_out(job, summary, gd)["status"] == "failed"
|
||
|
||
|
||
def test_v3_snapshot_rejects_bundle_rewrite_then_restore(monkeypatch):
|
||
"""快照若读到恶意中间态,即使文件随后恢复,内存快照 hash 仍不匹配旧验收,必须拒绝。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
source_project = _v3_source_project(summary, gd, "x")
|
||
staged_bundle = gd.parent / "_wg1-gen" / "t1" / "bundle.iife.js"
|
||
accepted_bytes = staged_bundle.read_bytes()
|
||
original_capture = R._capture_artifact_snapshot
|
||
|
||
def capture_evil_then_restore(root):
|
||
staged_bundle.write_text("var __GameBundle={evil:true};", encoding="utf-8")
|
||
try:
|
||
return original_capture(root)
|
||
finally:
|
||
staged_bundle.write_bytes(accepted_bytes)
|
||
|
||
monkeypatch.setattr(R, "_capture_artifact_snapshot", capture_evil_then_restore)
|
||
out = R.build_result_out(job, summary, gd, source_project=source_project)
|
||
assert out["status"] == "failed"
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_v3_snapshot_rejects_artifact_file_count_over_hard_cap(monkeypatch):
|
||
"""发布快照文件数超过与 Node 一致的 4096 硬帽时,必须 fail-closed。"""
|
||
assert R.artifact_snapshot.MAX_ARTIFACT_FILES == 4096
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
# 标准 staged 树已有 bundle + 两个 src 文件;缩小测试帽即可覆盖第 N+1 个文件前拒绝的路径。
|
||
monkeypatch.setattr(R.artifact_snapshot, "MAX_ARTIFACT_FILES", 2)
|
||
|
||
out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x"))
|
||
|
||
assert out["status"] == "failed"
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_v3_snapshot_rejects_artifact_bytes_over_hard_cap(monkeypatch):
|
||
"""发布快照总字节超过与 Node 一致的 128 MiB 硬帽时,必须在发布前拒绝。"""
|
||
assert R.artifact_snapshot.MAX_ARTIFACT_BYTES == 128 * 1024 * 1024
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
monkeypatch.setattr(R.artifact_snapshot, "MAX_ARTIFACT_BYTES", 1)
|
||
|
||
out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x"))
|
||
|
||
assert out["status"] == "failed"
|
||
assert "engineBundle" not in out
|
||
|
||
|
||
def test_v3_source_project_rejects_release_src_drift_and_accepts_staged_src():
|
||
"""v3 源血缘只认验收同源 staged/src;release/src 的后验改写不得落库。"""
|
||
gd = _make_game_dir(_tmp(), src_files={"game-logic.js": "// accepted logic\n",
|
||
"core.js": "export const SPEED=1;\n"})
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
staged = gd.parent / "_wg1-gen" / "t1"
|
||
profile = {"tickModel": "realtime", "inputModel": "discrete-choice", "progressModel": "metric"}
|
||
|
||
# 验收后仅漂移 release 源,bundle 与 staged artifact 仍保持不变。
|
||
(gd / "src" / "core.js").write_text("export const SPEED=999; // 未验收源\n", encoding="utf-8")
|
||
drifted = R.build_source_project(
|
||
gd, profile, scaffold_template="_template-story", origin_brief="x",
|
||
origin_brief_hash=hashlib.sha256(b"x").hexdigest())
|
||
rejected = R.build_result_out(job, summary, gd, source_project=drifted,
|
||
expected_acceptance_mode="v3")
|
||
assert rejected["status"] == "failed", "active v3 的源血缘属于跨请求可信边界,漂移时必须阻断"
|
||
assert "sourceProject" not in rejected, "release/src 后验漂移不得随成功回调落库"
|
||
|
||
accepted = R.build_source_project(
|
||
staged, profile, scaffold_template="_template-story", origin_brief="x",
|
||
origin_brief_hash=hashlib.sha256(b"x").hexdigest())
|
||
published = R.build_result_out(job, summary, gd, source_project=accepted,
|
||
expected_acceptance_mode="v3")
|
||
assert published["status"] == "succeeded"
|
||
assert json.loads(published["sourceProject"])["files"]["src/core.js"] == "export const SPEED=1;\n"
|
||
|
||
|
||
def test_v3_failed_result_never_carries_source_project():
|
||
"""v3 最终发布绑定失败时,即使 SourceProject 来自 staged,也不得落成可继承血缘。"""
|
||
gd = _make_game_dir(_tmp())
|
||
summary = _v3_summary("accept")
|
||
job = _bind_v3_run(summary, gd)
|
||
staged = gd.parent / "_wg1-gen" / "t1"
|
||
profile = {"tickModel": "realtime", "inputModel": "discrete-choice", "progressModel": "metric"}
|
||
source_project = R.build_source_project(
|
||
staged, profile, scaffold_template="_template-story", origin_brief="x",
|
||
origin_brief_hash=hashlib.sha256(b"x").hexdigest())
|
||
(gd / "bundle.iife.js").write_text("var __GameBundle={foreign:true};", encoding="utf-8")
|
||
|
||
out = R.build_result_out(job, summary, gd, source_project=source_project,
|
||
expected_acceptance_mode="v3")
|
||
|
||
assert out["status"] == "failed"
|
||
assert "sourceProject" not in out
|
||
|
||
|
||
if __name__ == "__main__":
|
||
_fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
|
||
_failed = 0
|
||
for _fn in _fns:
|
||
try:
|
||
_fn()
|
||
print(f" PASS {_fn.__name__}")
|
||
except Exception as e: # noqa: BLE001
|
||
_failed += 1
|
||
print(f" FAIL {_fn.__name__}: {type(e).__name__}: {e}")
|
||
print(f"\n{len(_fns) - _failed}/{len(_fns)} passed")
|
||
sys.exit(1 if _failed else 0)
|