"""test_result_out.py — §6.1 result-out 三字段组装单测(M3a U1)。 守的不变量(KTD2 BLOCKER 修正,镜像 SaaGraphDispatcher.buildCallbackReqVO + service.py build_callback_payload): · gameConfig = 最小占位 Map(templateId/title/theme/engineDriven),绝不塞源工程; · engineBundle = bundle.iife.js 全文(含 __GameBundle)= 运行时承重产物; · 成功门 = active v3 canonical accept,或真正 legacy 验收通过;两路都要求 engineBundle 含 __GameBundle; · sourceProject = best-effort 可选(profile 派生不出即省略,不伪造、不阻断成功); · traceId 取 job.traceId 或 job_id;纯函数无网络无 LLM、同输入恒同输出。 跑:cheap-worker/.venv/bin/python cheap-worker/tests/test_result_out.py """ import copy import hashlib import json import sys import tempfile from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # → cheap-worker/ import result_out as R # noqa: E402 import worker_service as W # noqa: E402 _GOOD_BUNDLE = "var __GameBundle=(function(){return{bootGameHost(){}}})();" def _tmp() -> Path: return Path(tempfile.mkdtemp(prefix="ro-test-")) def _make_game_dir(tmp: Path, *, bundle=_GOOD_BUNDLE, src_files=None) -> Path: """造 amgen-/ 形态产物目录(bundle.iife.js + src/)。""" gd = tmp / "amgen-t1" (gd / "src").mkdir(parents=True) if bundle is not None: (gd / "bundle.iife.js").write_text(bundle, encoding="utf-8") for name, content in (src_files or {"game-logic.js": "// logic\n", "core.js": "// core\n"}).items(): (gd / "src" / name).write_text(content, encoding="utf-8") return gd def test_succeeded_shape(): """成功路:verdict.pass ∧ bundle 含 __GameBundle → status=succeeded + gameConfig 占位 + engineBundle 承重。""" gd = _make_game_dir(_tmp()) job = {"job_id": "trace-abc", "traceId": "trace-abc", "templateId": "generic", "brief": "点点乐"} summary = {"ok": True, "finished": True, "verdict": {"pass": True, "failedGates": []}, "costRmb": 0.05} out = R.build_result_out(job, summary, gd) assert out["traceId"] == "trace-abc" assert out["status"] == "succeeded" assert out["templateId"] == "generic" # gameConfig 占位,绝不含源工程(BLOCKER 修正) assert out["gameConfig"]["templateId"] == "generic" assert out["gameConfig"]["theme"] == "generic" assert out["gameConfig"]["engineDriven"] is True assert "files" not in out["gameConfig"] and "entry" not in out["gameConfig"] # engineBundle 承重 assert "__GameBundle" in out["engineBundle"] assert out.get("failureReason") is None def test_failed_when_gates_not_pass(): """verdict.pass=False → status=failed + 七值 failureReason,不带 engineBundle。""" gd = _make_game_dir(_tmp()) job = {"traceId": "t2", "templateId": "generic", "brief": "x"} summary = {"finished": True, "verdict": {"pass": False, "failedGates": ["E_live"]}} out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" assert out.get("engineBundle") is None assert out["failureReason"] in R._FAILURE_REASONS # 对齐后端 FailureReasonEnum 真七值 def test_failed_when_engine_bundle_missing___gamebundle(): """即使 verdict.pass,bundle 无 __GameBundle → status=failed(承重门:engineBundle 含 __GameBundle)。""" gd = _make_game_dir(_tmp(), bundle="var notTheGlobal=1;") job = {"traceId": "t3", "templateId": "generic", "brief": "x"} summary = {"finished": True, "verdict": {"pass": True, "failedGates": []}} out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" assert out.get("engineBundle") is None def test_failed_breaker_maps_llm_error(): """预算熔断 / 内部真因不在七值内 → failureReason=llm_error(契约#6 catch-all;枚举无 budget 专项值)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "t4", "templateId": "generic", "brief": "x"} summary = {"finished": False, "breaker": True, "verdict": {"pass": None, "failedGates": None}} out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" assert out["failureReason"] == "llm_error" def test_explicit_failure_reason_wins(): """显式 failure_reason(在七值内)优先于推断。""" gd = _make_game_dir(_tmp()) job = {"traceId": "t5", "templateId": "generic", "brief": "x"} summary = {"finished": True, "verdict": {"pass": False, "failedGates": ["A_boot"]}} out = R.build_result_out(job, summary, gd, failure_reason="timeout") assert out["status"] == "failed" assert out["failureReason"] == "timeout" # ── WU2 §3.9:额度耗尽干净失败 + one-api 惯例分辨 ── def test_quota_exhausted_in_enum(): """quota_exhausted 已进 failureReason 枚举(跨契约共享枚举同批新增,与 aigc.yaml/dify-workflow-io.json 一致)。""" assert "quota_exhausted" in R._FAILURE_REASONS def test_summary_failure_reason_quota_maps_through(): """WU2 §3.9:summary.failureReason=quota_exhausted(Service 分辨→driver 透传)→ 落 failureReason=quota_exhausted。""" gd = _make_game_dir(_tmp()) job = {"traceId": "tq", "templateId": "generic", "brief": "x"} summary = {"finished": False, "verdict": {"pass": False, "failedGates": None}, "failureReason": "quota_exhausted"} out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" assert out["failureReason"] == "quota_exhausted" def test_summary_failure_reason_ignored_when_not_in_enum(): """summary.failureReason 非枚举值 → 不透传,回落 catch-all llm_error(防脏值污染契约)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "tq2", "templateId": "generic", "brief": "x"} summary = {"finished": False, "verdict": {"pass": False}, "failureReason": "some_garbage"} out = R.build_result_out(job, summary, gd) assert out["failureReason"] == "llm_error" def test_explicit_failure_reason_wins_over_summary(): """显式入参优先级高于 summary.failureReason。""" gd = _make_game_dir(_tmp()) job = {"traceId": "tq3", "templateId": "generic", "brief": "x"} summary = {"verdict": {"pass": False}, "failureReason": "quota_exhausted"} out = R.build_result_out(job, summary, gd, failure_reason="timeout") assert out["failureReason"] == "timeout" def test_classify_newapi_status_codes(): """状态码分辨:402/403(one-api 惯例额度码)→ quota_exhausted;裸 401 / 500 / None(无额度关键词)→ llm_error 可重试。""" assert R.classify_newapi_failure_reason(402) == "quota_exhausted" assert R.classify_newapi_failure_reason(403) == "quota_exhausted" assert R.classify_newapi_failure_reason(401) == "llm_error" # 裸 401 无 message = 真凭据失效 assert R.classify_newapi_failure_reason(500) == "llm_error" assert R.classify_newapi_failure_reason(None) == "llm_error" def test_classify_newapi_real_quota_exhausted_401(): """2026-07-08 真机耗尽探测坐实:new-api 对额度耗尽返 HTTP 401 + 「该令牌额度已用尽 …RemainQuota = 0」。 401 与凭据失效状态码重叠,故必须按 message 分辨——耗尽判 quota_exhausted(干净失败),真凭据失效仍 llm_error(可重试)。 这是关键回归:旧代码 401→llm_error 会把耗尽误判成可重试 → 反复重试到 watchdog 超时。""" real = "[sk-air***0AS] 该令牌额度已用尽 !token.UnlimitedQuota && token.RemainQuota = 0 (request id: 20260708...)" assert R.classify_newapi_failure_reason(401, real) == "quota_exhausted" assert R.classify_newapi_failure_reason(401, "无效的令牌,请检查") == "llm_error" # 真凭据失效仍可重试、不塌成耗尽 def test_classify_newapi_keyword_fallback(): """message 关键词分辨额度耗尽(状态码取不到 / 被框架包装时同样生效,优先于状态码)。""" assert R.classify_newapi_failure_reason(None, "error: insufficient user quota") == "quota_exhausted" assert R.classify_newapi_failure_reason(None, "当前分组余额不足,请充值") == "quota_exhausted" assert R.classify_newapi_failure_reason(None, "connection reset by peer") == "llm_error" def test_newapi_error_status_text_extraction(): """从带 status_code/response 的异常对象抽 (status, text);抽不到 status 返 None、文本兜底 str(exc)。""" class _Err(Exception): status_code = 402 message = "insufficient user quota" st, txt = R.newapi_error_status_text(_Err("boom")) assert st == 402 and "insufficient" in txt class _Resp: status_code = 403 class _Wrapped(Exception): response = _Resp() st2, _ = R.newapi_error_status_text(_Wrapped("x")) assert st2 == 403 st3, txt3 = R.newapi_error_status_text(ValueError("plain error")) assert st3 is None and "plain error" in txt3 def test_traceid_fallback_to_job_id(): """traceId 缺 → 用 job_id(贯穿键同值)。""" gd = _make_game_dir(_tmp()) job = {"job_id": "jid-9", "templateId": "generic", "brief": "x"} summary = {"verdict": {"pass": True}} out = R.build_result_out(job, summary, gd) assert out["traceId"] == "jid-9" def test_cost_goes_into_trace_not_toplevel(): """成本落 trace.cost.totalRmb——顶层不得有 costRmb(DifyCallbackReqVO 无此字段、后端严格拒未知)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "t6", "templateId": "generic", "brief": "x"} summary = {"verdict": {"pass": True}, "costRmb": 0.07} out = R.build_result_out(job, summary, gd) assert "costRmb" not in out, "顶层不得有 costRmb(会被后端 ObjectMapper 拒)" assert out["trace"]["cost"]["totalRmb"] == 0.07 def test_result_out_keys_within_vo_known_set(): """result-out 顶层键严格 ⊆ DifyCallbackReqVO 已知字段集(后端拒未知字段的硬约束)。""" vo_known = {"traceId", "status", "templateId", "gameConfig", "assets", "engineBundle", "failureReason", "qualityScore", "sourceProject", "trace"} gd = _make_game_dir(_tmp()) job = {"traceId": "t9", "templateId": "generic", "brief": "x"} out_ok = R.build_result_out(job, {"verdict": {"pass": True}, "costRmb": 0.07}, gd, source_project='{"schemaVersion":"2.0"}') out_fail = R.build_result_out(job, {"verdict": {"pass": False}}, gd) assert set(out_ok) <= vo_known, f"成功路含未知字段:{set(out_ok) - vo_known}" assert set(out_fail) <= vo_known, f"失败路含未知字段:{set(out_fail) - vo_known}" def test_build_source_project_valid_2_0(): """profile 可派生 + src/ 有文件 → sourceProject 是合法 2.0(必填齐 + sourceHash 64-hex)。""" import json import re gd = _make_game_dir(_tmp(), src_files={"game-logic.js": "// l\n", "core.js": "// c\n"}) profile = {"tickModel": "frame", "inputModel": "tap-targets", "progressModel": "score"} sp_str = R.build_source_project(gd, profile, entry="entry.js") assert sp_str is not None sp = json.loads(sp_str) assert sp["schemaVersion"] == "2.0" assert sp["globalName"] == "__GameBundle" assert sp["entry"] == "entry.js" assert sp["profile"] == profile assert "src/game-logic.js" in sp["files"] and "src/core.js" in sp["files"] assert re.fullmatch(r"[a-f0-9]{64}", sp["sourceHash"]) def test_build_source_project_carries_trusted_scaffold_without_changing_source_hash(): """scaffoldTemplate 是 additive 血缘;sourceHash 冻结为只覆盖 files,不能暗改既有寻址语义。""" gd = _make_game_dir(_tmp()) profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"} without_route = json.loads(R.build_source_project(gd, profile)) brief = "点击开始完成一局" brief_hash = hashlib.sha256(brief.encode("utf-8")).hexdigest() with_route = json.loads(R.build_source_project( gd, profile, scaffold_template="_template-story", origin_brief=brief, origin_brief_hash=brief_hash)) assert with_route["scaffoldTemplate"] == "_template-story" assert with_route["originBrief"] == brief and with_route["originBriefHash"] == brief_hash assert with_route["sourceHash"] == without_route["sourceHash"] assert with_route["files"] == without_route["files"] def test_build_source_project_rejects_unknown_scaffold_lineage(): """任意模板名不能进入受信 SourceProject 血缘。""" gd = _make_game_dir(_tmp()) profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"} assert R.build_source_project(gd, profile, scaffold_template="_template-generic") is None def test_build_source_project_rejects_empty_or_forged_origin_brief(): """active v3 原题面必须非空且正文与 sha256 自洽,不能只带一个格式正确的 hash。""" gd = _make_game_dir(_tmp()) profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"} assert R.build_source_project(gd, profile, scaffold_template="_template-story", origin_brief="", origin_brief_hash=hashlib.sha256(b"").hexdigest()) is None assert R.build_source_project(gd, profile, scaffold_template="_template-story", origin_brief="题面甲", origin_brief_hash="a" * 64) is None def test_source_project_contract_samples_cover_route_whitelist(): """新增正负样本必须由 Draft-07 schema 一正一拒,防 enum 只写文档未进机器门。""" from jsonschema import Draft7Validator root = Path(__file__).resolve().parents[2] schema = json.loads((root / "contracts/agent-loop/source-project.schema.json").read_text(encoding="utf-8")) valid = json.loads((root / "contracts/agent-loop/samples/source-project/valid" / "01-v2-with-scaffold-template.json").read_text(encoding="utf-8")) invalid = json.loads((root / "contracts/agent-loop/samples/source-project/invalid" / "01-unknown-scaffold-template.json").read_text(encoding="utf-8")) missing_origin_hash = json.loads((root / "contracts/agent-loop/samples/source-project/invalid" / "02-origin-brief-missing-hash.json").read_text(encoding="utf-8")) validator = Draft7Validator(schema) assert list(validator.iter_errors(valid)) == [] assert list(validator.iter_errors(invalid)), "未知 scaffoldTemplate 必须被 schema 拒绝" assert list(validator.iter_errors(missing_origin_hash)), "originBrief/hash 必须成对出现" def test_build_source_project_omitted_when_profile_incomplete(): """profile 派生不出(None / 三枚举不全)→ 返 None(省略、绝不伪造)。""" gd = _make_game_dir(_tmp()) assert R.build_source_project(gd, None, entry="entry.js") is None assert R.build_source_project(gd, {"tickModel": "frame"}, entry="entry.js") is None def test_build_source_project_omitted_when_no_src(): """src/ 无文件 → 返 None。""" empty = _tmp() / "amgen-e" (empty / "src").mkdir(parents=True) profile = {"tickModel": "frame", "inputModel": "tap-targets", "progressModel": "score"} assert R.build_source_project(empty, profile, entry="entry.js") is None def test_result_out_source_project_additive(): """build_result_out 收到 source_project(JSON str)→ payload 带;缺省不带(additive)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "t7", "templateId": "generic", "brief": "x"} summary = {"verdict": {"pass": True}} out_with = R.build_result_out(job, summary, gd, source_project='{"schemaVersion":"2.0"}') assert out_with["sourceProject"] == '{"schemaVersion":"2.0"}' out_without = R.build_result_out(job, summary, gd) assert "sourceProject" not in out_without def test_deterministic_zero_llm(): """纯函数:同输入恒同输出(无网络无 LLM)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "t8", "templateId": "generic", "brief": "稳定"} summary = {"verdict": {"pass": True}, "costRmb": 0.01} assert R.build_result_out(job, summary, gd) == R.build_result_out(job, summary, gd) # ────────────────────────────────────────────────────────────────────────────── # M3b U2:trace 9d 七项 + D11 首局维(sevenGateVerdict/gatespec)+ repairs 口径 + None 容错 # 守的不变量(镜像 wg1 _extract_trace + 后端 ReadinessScorer 读取口径): # · 必填七项 camelCase:pass/repairs/wallS/models/attempts/gameId/stage; # · sevenGateVerdict 只取 {pass,guards}(去 url/ts 噪声,guards.H_progress 保嵌套含 .pass); # · gatespec 键名必须是 `driver`(落 driverType 则后端 hasDriver=false、firstPlay 永中性); # · repairs = max(0, attempts-1)(首轮 attempts=1→repairs=0); # · 成功/失败路都抽 trace;verdictFull/driverType=None 时省略 sevenGateVerdict/gatespec(不塞 None、不崩)。 # ────────────────────────────────────────────────────────────────────────────── # U1 富化后的成功路 summary(合成;含完整 verdictFull 带 url/ts 噪声 + driverType + stage + models)。 _SUMMARY_SUCCESS = { "ok": True, "finished": True, "gameId": "amgen-t1", "attempts": 1, "verdict": {"pass": True, "failedGates": []}, # brief(既有键) "verdictFull": { # U1 新增:完整 verdict(带噪声 + guards.H_progress 嵌套) "pass": True, "url": "http://localhost:4320/?x", "ts": 1700000000, "guards": { "A_boot": {"pass": True}, "H_progress": {"pass": True, "checks": ["score+"], "latch": True, "state0": 0, "state1": 3}, }, }, "driverType": "tap-targets", "models": {"code": "MiniMax-M3"}, "stage": "play", "wallSec": 102.3, "costRmb": 0.0431, } def test_trace_seven_required_fields_and_d11_dims_success(): """成功路 trace:七项齐 + sevenGateVerdict.guards.H_progress.pass + gatespec.driver;噪声 url/ts 已去。""" gd = _make_game_dir(_tmp()) job = {"traceId": "u2-ok", "templateId": "generic", "brief": "点击得分小游戏"} out = R.build_result_out(job, _SUMMARY_SUCCESS, gd) tr = out["trace"] # 必填七项(camelCase 逐字镜像 _extract_trace) assert tr["pass"] is True assert tr["repairs"] == 0 # attempts=1 → max(0,0)=0(首轮 0) assert tr["wallS"] == 102.3 assert tr["models"] == {"code": "MiniMax-M3"} assert tr["attempts"] == 1 assert tr["gameId"] == "amgen-t1" assert tr["stage"] == "play" # D11 首局维:sevenGateVerdict 只 {pass,guards},guards.H_progress 嵌套含 .pass(后端 guardPass 读它) assert set(tr["sevenGateVerdict"].keys()) == {"pass", "guards"} assert "url" not in tr["sevenGateVerdict"] and "ts" not in tr["sevenGateVerdict"] assert tr["sevenGateVerdict"]["guards"]["H_progress"]["pass"] is True # gatespec 键名必须是 driver(落 driverType 会让后端 hasDriver=false) assert tr["gatespec"] == {"driver": "tap-targets"} assert "driverType" not in tr["gatespec"] # cost 维持现状 assert tr["cost"]["totalRmb"] == 0.0431 def test_trace_repairs_equals_max0_attempts_minus_1(): """repairs = max(0, attempts-1):attempts=1→0、attempts=3→2(对齐 wg1 len(attempts)-1)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "u2-rep", "templateId": "generic", "brief": "x"} s1 = dict(_SUMMARY_SUCCESS, attempts=1) assert R.build_result_out(job, s1, gd)["trace"]["repairs"] == 0 s3 = dict(_SUMMARY_SUCCESS, attempts=3) assert R.build_result_out(job, s3, gd)["trace"]["repairs"] == 2 def test_trace_failed_path_still_has_seven_required(): """失败路(verdictFull/driverType=None)trace 仍含七项(repairs/attempts/stage/wallS 排障价值)。""" gd = _make_game_dir(_tmp(), bundle="var notTheGlobal=1;") # 无 __GameBundle → status=failed job = {"traceId": "u2-fail", "templateId": "generic", "brief": "x"} summary = { "ok": False, "finished": False, "gameId": "amgen-t2", "attempts": 6, "verdict": None, "verdictFull": None, "driverType": None, "models": {"code": "MiniMax-M3"}, "stage": "code", "wallSec": 50.1, "costRmb": 0.02, } out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" tr = out["trace"] for k in ("pass", "repairs", "wallS", "models", "attempts", "gameId", "stage"): assert k in tr, f"失败路 trace 缺必填项 {k}" assert tr["repairs"] == 5 # attempts=6 → 5 assert tr["stage"] == "code" assert tr["pass"] is None # play 没跑 def test_trace_none_tolerance_omits_d11_dims(): """verdictFull/driverType=None(对照路)→ 省略 sevenGateVerdict/gatespec(不塞 None、不崩)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "u2-none", "templateId": "generic", "brief": "x"} summary = { "gameId": "amgen-t3", "attempts": 2, "verdict": None, "verdictFull": None, "driverType": None, "models": {"code": "MiniMax-M3"}, "stage": "smoke", "wallSec": 10.0, } out = R.build_result_out(job, summary, gd) tr = out["trace"] assert "sevenGateVerdict" not in tr # verdictFull=None → 省略 assert "gatespec" not in tr # driverType=None → 省略 assert tr["repairs"] == 1 # attempts=2 → 1 def test_trace_keys_camelcase_match_readiness_scorer(): """trace 键名逐字镜像 ReadinessScorer 读取口径(拼错会被后端静默接收 → 假绿)。""" gd = _make_game_dir(_tmp()) job = {"traceId": "u2-keys", "templateId": "generic", "brief": "x"} tr = R.build_result_out(job, _SUMMARY_SUCCESS, gd)["trace"] # ReadinessScorer 读取的确切键(camelCase):trace.pass / gatespec.driver / # sevenGateVerdict.guards.H_progress.pass / repairs / cost.totalRmb assert "pass" in tr and "gatespec" in tr and "sevenGateVerdict" in tr assert "wallS" in tr and "gameId" in tr # 非 wall_s / game_id(snake 会被后端读不到) assert "driver" in tr["gatespec"] assert "H_progress" in tr["sevenGateVerdict"]["guards"] # ────────────────────────────────────────────────────────────────────────────── # 生成验收 v3 消费层:decision/outcome 唯一写权、shadow 冻结、trace.playtest 轻量投影 # ────────────────────────────────────────────────────────────────────────────── _V2_HISTORY_SAMPLE_DIR = (Path(__file__).resolve().parents[2] / "contracts" / "play-loop" / "samples" / "playtest-evidence" / "valid") _V2_HISTORY_SAMPLE_BY_OUTCOME = { "accept": "01-accepted-narrative.json", "reject": "02-rejected-mechanical.json", "tester_error": "03-tester-error-action-protocol.json", "inconclusive": "04-inconclusive-insufficient-proof.json", } def _v3_fixture_hash(label: str) -> str: """为 A/B/consensus/目标解析 fixture 生成稳定 SHA-256。""" return hashlib.sha256(label.encode("utf-8")).hexdigest() def _upgrade_v3_action(action: dict, index: int) -> dict: """把历史坐标动作升级为 playtest/3 的结构化选择和确定性解析事实。""" upgraded = copy.deepcopy(action) normalized = upgraded["normalized"] action_type = normalized["type"] target_set_hash = _v3_fixture_hash(f"target-set-{index}") if action_type == "tap": target = {"id": f"r{index:02d}"} selection = { "schemaVersion": "ActorSelection/1", "type": "tap", "targetSetHash": target_set_hash, "target": target, } points = [{"role": "target", "target": target, "point": {"x": normalized["x"], "y": normalized["y"]}}] elif action_type == "key": selection = {"schemaVersion": "ActorSelection/1", "type": "key", "key": normalized["key"]} points = None elif action_type == "wait": selection = {"schemaVersion": "ActorSelection/1", "type": "wait", "ms": normalized["ms"]} points = None else: raise AssertionError(f"v3 result fixture 暂不支持动作:{action_type}") upgraded["request"] = { "raw": json.dumps(selection, ensure_ascii=False, sort_keys=True), "parsed": copy.deepcopy(selection), } upgraded["retryCount"] = 0 upgraded["protocolErrors"] = [] if points is None: upgraded["targetResolution"] = {"status": "not_applicable"} else: upgraded["targetResolution"] = { "status": "resolved", "targetSetRef": f"targets/action-{index}.json", "targetSetHash": target_set_hash, "sourceFrameRef": upgraded["preFrameRef"]["path"], "sourceFrameHash": upgraded["preFrameRef"]["hash"], "guideManifestRef": f"guides/action-{index}.json", "guideManifestHash": _v3_fixture_hash(f"guide-{index}"), "resolverVersion": "1.0.0", "selection": copy.deepcopy(selection), "resolvedPoints": points, "error": None, } return upgraded def _upgrade_history_result_to_v3(v2: dict) -> dict: """把冻结 v2 DECIDED 样本升级成 active/shadow 测试使用的 playtest/3 对象。""" v3 = copy.deepcopy(v2) v3.pop("_note", None) v3["schemaVersion"] = "playtest/3" v3["environment"]["virtualTime"]["waitMaxMs"] = 600 if isinstance(v3.get("actor"), dict): v3["actor"]["contextProjectionVersion"] = "ActorView/3" if v3.get("outcome") == "tester_error": # 失败 selection 只属于 attempt audit,不得伪装为正式 actionRecord。 v3["actions"] = [] else: v3["actions"] = [_upgrade_v3_action(action, index) for index, action in enumerate(v3.get("actions") or [], start=1)] legacy_judge = v2.get("judge") if isinstance(v2.get("judge"), dict) else None source_roll_id = None if legacy_judge is not None and v3.get("rolls"): source_roll_id = str(v3["rolls"][-1]["rollId"]) package_hash = str(legacy_judge["packageHash"]) result_a = _v3_fixture_hash(f"{source_roll_id}-judge-a-result") result_b = _v3_fixture_hash(f"{source_roll_id}-judge-b-result") consensus_hash = _v3_fixture_hash(f"{source_roll_id}-consensus") for roll in v3["rolls"]: roll_id = str(roll["rollId"]) roll.update({ "proofPackageHash": _v3_fixture_hash(f"{roll_id}-proof"), "judgePackageHash": package_hash, "judgeARef": { "rawRef": f"{roll_id}/judge-a/raw.json", "rawHash": _v3_fixture_hash(f"{roll_id}-a-raw"), "parsedRef": f"{roll_id}/judge-a/parsed.json", "parsedHash": _v3_fixture_hash(f"{roll_id}-a-parsed"), "normalizedRef": f"{roll_id}/judge-a/normalized.json", "resultHash": result_a, }, "judgeAHash": _v3_fixture_hash(f"{roll_id}-a-manifest"), "judgeBRef": { "rawRef": f"{roll_id}/judge-b/raw.json", "rawHash": _v3_fixture_hash(f"{roll_id}-b-raw"), "parsedRef": f"{roll_id}/judge-b/parsed.json", "parsedHash": _v3_fixture_hash(f"{roll_id}-b-parsed"), "normalizedRef": f"{roll_id}/judge-b/normalized.json", "resultHash": result_b, }, "judgeBHash": _v3_fixture_hash(f"{roll_id}-b-manifest"), "judgeConsensusRef": f"{roll_id}/consensus.json", "judgeConsensusHash": consensus_hash, "costReservationRef": f"{roll_id}/cost-reservation.json", "costReservationHash": _v3_fixture_hash(f"{roll_id}-reservation"), }) v3["judge"] = { "schemaVersion": "JudgeConsensus/1", "sourceRollId": source_roll_id, "consensusRef": f"{source_roll_id}/consensus.json", "consensusHash": consensus_hash, "packageHash": package_hash, "strategyVersion": "1.0.0", "decision": legacy_judge["decision"], "degraded": bool(legacy_judge.get("degraded")), "sameModel": True, "independenceChecks": { "differentLogicalClient": True, "differentSession": True, "differentPolicySeed": True, "differentPromptArtifact": True, "differentEvidenceDir": True, "differentRawFile": True, "sameJudgePackage": True, "sameImageManifest": True, "sameModel": True, }, "judgeAResultHash": result_a, "judgeBResultHash": result_b, "obligationResults": copy.deepcopy(legacy_judge["obligationResults"]), "conflicts": list(legacy_judge.get("contradictions") or []), "failureSignature": None, } else: v3["judge"] = None # 旧 action-protocol tester_error 只有失败 attempt;正式 actionRecord 不应收录它。 if v3.get("outcome") == "tester_error": v3["actions"], v3["rolls"], v3["events"] = [], [], [] v3["costRmb"] = 0 v3["decision"]["costRmb"] = 0 v3["sourceRollId"] = source_roll_id accepted = v3["outcome"] == "accept" v3["compatibility"] = { "accepted": accepted and v3["acceptanceMode"] == "v3", "ok": accepted and v3["acceptanceMode"] == "v3", "acceptanceVersion": v3["acceptanceMode"], "publishFrozen": bool(v3["decision"]["publishFrozen"]), "schemaVersion": "playtest/3", "sourceRollId": source_roll_id, } return v3 def _set_v3_mode(v3: dict, mode: str) -> None: """在完整 canonical 样本上同步切换 active/shadow,保持所有镜像字段逐字一致。""" accepted = v3["outcome"] == "accept" frozen = mode == "v3_shadow" or not accepted v3["acceptanceMode"] = mode v3["decision"]["shadowAccepted"] = accepted if mode == "v3_shadow" else None v3["decision"]["publishFrozen"] = frozen compatibility = v3["compatibility"] compatibility["accepted"] = accepted and mode == "v3" compatibility["ok"] = accepted and mode == "v3" compatibility["acceptanceVersion"] = mode compatibility["publishFrozen"] = frozen def _v3_result(outcome="accept", *, mode="v3", accepted=None, frozen=None, proof_complete=None): """把冻结样本升级为 v3;旧文件只作历史素材,active fixture 只返回 playtest/3。""" path = _V2_HISTORY_SAMPLE_DIR / _V2_HISTORY_SAMPLE_BY_OUTCOME[outcome] v3 = _upgrade_history_result_to_v3(json.loads(path.read_text(encoding="utf-8"))) _set_v3_mode(v3, mode) if accepted is not None: v3["decision"]["accepted"] = accepted if frozen is not None: v3["decision"]["publishFrozen"] = frozen v3["compatibility"]["publishFrozen"] = frozen if proof_complete is not None: for obligation in v3["proofObligations"]: if obligation.get("required") is True: obligation["status"] = "satisfied" if proof_complete else "missing" return v3 def _v3_summary(outcome="accept", **kwargs): """生产形态:完整结果在 acceptanceV3,顶层只镜像六字段 compatibility。""" v3 = _v3_result(outcome, **kwargs) compatibility = v3["compatibility"] summary = {"verdict": {"pass": True}, "floor": copy.deepcopy(v3["floor"])} for key in ("accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId"): summary[key] = copy.deepcopy(compatibility[key]) summary["acceptanceV3"] = v3 summary["publishFrozen"] = v3["decision"]["publishFrozen"] return summary def _bind_v3_run(summary: dict, gd: Path, *, game_id: str = "t1", brief: str = "x") -> dict: """构造真实发布边界:当前 job、run-summary、staged artifact 与 release bundle 四者同源。""" v3 = summary.get("acceptanceV3") assert isinstance(v3, dict), "生产绑定夹具必须使用 acceptanceV3 包装态" staged = gd.parent / "_wg1-gen" / game_id staged.mkdir(parents=True, exist_ok=True) release_bytes = (gd / "bundle.iife.js").read_bytes() (staged / "bundle.iife.js").write_bytes(release_bytes) staged_src = staged / "src" staged_src.mkdir(parents=True, exist_ok=True) for source_file in sorted((gd / "src").glob("*.js")): (staged_src / source_file.name).write_bytes(source_file.read_bytes()) v3["gameId"] = game_id v3["briefHash"] = hashlib.sha256(brief.encode("utf-8")).hexdigest() trace_id = f"trace-{game_id}" v3["taskBindingHash"] = hashlib.sha256(trace_id.encode("utf-8")).hexdigest() v3["artifactHash"] = R.cheap_verify._artifact_hash_v3(game_id, staged) v3["environment"]["buildRef"] = v3["artifactHash"] summary["gameId"] = game_id summary["acceptanceRunId"] = v3["runId"] summary["acceptanceRequestHash"] = v3["acceptanceRequestHash"] summary["acceptanceTaskBindingHash"] = v3["taskBindingHash"] summary["acceptanceArtifactHash"] = v3["artifactHash"] summary["acceptanceBriefHash"] = v3["briefHash"] return {"traceId": trace_id, "gameId": game_id, "brief": brief} def _v3_source_project(summary: dict, gd: Path, origin_brief: str) -> str: """从验收同源 staged/src 组 active v3 SourceProject 测试夹具。""" v3 = summary["acceptanceV3"] staged = gd.parent / "_wg1-gen" / v3["gameId"] profile = {"tickModel": "event", "inputModel": "discrete-choice", "progressModel": "narrative"} source_project = R.build_source_project( staged, profile, scaffold_template=v3["templateRoute"], origin_brief=origin_brief, origin_brief_hash=hashlib.sha256(origin_brief.encode("utf-8")).hexdigest()) assert source_project is not None return source_project def test_v3_accept_projects_trusted_trace_and_succeeds(): """active v3 accept + 完整硬证才成功;trace 只保留后端消费所需轻量字段。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x")) assert out["status"] == "succeeded" playtest = out["trace"]["playtest"] assert playtest["schemaVersion"] == "playtest/3" assert playtest["acceptanceMode"] == "v3" and playtest["outcome"] == "accept" assert playtest["accepted"] is True and playtest["publishFrozen"] is False assert playtest["compatibilityAccepted"] is True and playtest["compatibilityOk"] is True assert playtest["floorPass"] is True and playtest["finalPostguardPass"] is True assert playtest["writeAuthority"] is True and all(playtest["finalChecks"].values()) assert playtest["blockingProblems"] == [] and playtest["contradictions"] == [] assert playtest["mergeConflicts"] == [] and playtest["proofComplete"] is True assert all(playtest["evidenceRefs"][key] for key in ("actions", "frames", "events")) assert playtest["firstPlay"]["loopClosed"] is True and playtest["firstPlay"]["proofRefs"] assert out["trace"]["acceptanceState"] == "active" identity = out["trace"]["playtest"] assert {key: identity[key] for key in ( "runId", "gameId", "briefHash", "artifactHash", "taskBindingHash", "acceptanceRequestHash" )} == {key: summary["acceptanceV3"][key] for key in ( "runId", "gameId", "briefHash", "artifactHash", "taskBindingHash", "acceptanceRequestHash" )} assert identity["engineBundleHash"] == hashlib.sha256(out["engineBundle"].encode("utf-8")).hexdigest() assert set(summary["acceptanceV3"]["compatibility"]) == { "accepted", "ok", "acceptanceVersion", "publishFrozen", "schemaVersion", "sourceRollId", } def test_v3_shadow_accept_is_frozen_even_when_other_fields_are_green(): """shadow decision 可为 accept,但 publishFrozen 是硬闸,不能携包进入 succeeded。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept", mode="v3_shadow") job = _bind_v3_run(summary, gd) out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" assert out["trace"]["playtest"]["outcome"] == "accept" assert out["trace"]["playtest"]["accepted"] is True assert out["trace"]["playtest"]["publishFrozen"] is True assert out["trace"]["acceptanceState"] == "shadow" assert "engineBundle" not in out def test_v3_non_accept_outcomes_always_fail_despite_legacy_accept_true(): """旧平铺 accepted 被污染为 true 也不能覆盖 v3 四态 decision。""" gd = _make_game_dir(_tmp()) for outcome in ("reject", "inconclusive", "tester_error"): summary = _v3_summary(outcome) summary["accepted"] = True summary["ok"] = True job = _bind_v3_run(summary, gd) job["traceId"] = f"v3-{outcome}" out = R.build_result_out(job, summary, gd) assert out["status"] == "failed", outcome assert out["trace"]["playtest"]["outcome"] == outcome assert out["trace"]["playtest"]["accepted"] is False def _assert_v3_failed(summary: dict, trace_id: str) -> None: """统一断言 v3 单点变异不能携带产物进入 succeeded。""" gd = _make_game_dir(_tmp()) job = _bind_v3_run(summary, gd) job["traceId"] = trace_id out = R.build_result_out(job, summary, gd) assert out["status"] == "failed" assert "engineBundle" not in out assert out["trace"]["acceptanceState"] in ("downgrade_detected", "shadow", "frozen") def test_v3_contradiction_1_top_outcome_fails_closed(): """顶层 outcome 与 decision/compatibility 不一致,canonical 语义校验必须拦截。""" summary = _v3_summary("accept") summary["acceptanceV3"]["outcome"] = "reject" _assert_v3_failed(summary, "v3-bad-outcome") def test_v3_contradiction_2_floor_fails_closed(): """accept 却让机械四门 floor 失败,不能靠旧 verdict.pass=true 放行。""" summary = _v3_summary("accept") summary["acceptanceV3"]["floor"]["pass"] = False _assert_v3_failed(summary, "v3-bad-floor") def test_v3_contradiction_3_final_postguard_fails_closed(): """finalPostguard 任一检查、写权或矛盾项异常都必须 fail-closed。""" variants = [] bad_check = _v3_summary("accept") bad_check["acceptanceV3"]["finalPostguard"]["checks"]["floorPass"] = False variants.append(bad_check) bad_writer = _v3_summary("accept") bad_writer["acceptanceV3"]["finalPostguard"]["writeAuthority"] = False variants.append(bad_writer) bad_problem = _v3_summary("accept") bad_problem["acceptanceV3"]["finalPostguard"]["blockingProblems"] = ["仍有阻塞问题"] variants.append(bad_problem) for index, summary in enumerate(variants): _assert_v3_failed(summary, f"v3-bad-final-{index}") def test_v3_contradiction_4_merge_fails_closed(): """mergeCandidate 被改写或仍有两掷冲突时,最终 accept 不成立。""" bad_candidate = _v3_summary("accept") bad_candidate["acceptanceV3"]["merge"]["mergeCandidate"] = "reject" bad_conflict = _v3_summary("accept") bad_conflict["acceptanceV3"]["merge"]["conflicts"] = ["两掷证据冲突"] _assert_v3_failed(bad_candidate, "v3-bad-merge-candidate") _assert_v3_failed(bad_conflict, "v3-bad-merge-conflict") def test_v3_contradiction_5_judge_fails_closed(): """独立 Judge 拒绝或降级时,兼容投影再绿也不能发布。""" bad_decision = _v3_summary("accept") bad_decision["acceptanceV3"]["judge"]["decision"] = "reject" bad_degraded = _v3_summary("accept") bad_degraded["acceptanceV3"]["judge"]["degraded"] = True _assert_v3_failed(bad_decision, "v3-bad-judge-decision") _assert_v3_failed(bad_degraded, "v3-bad-judge-degraded") def test_v3_contradiction_6_compatibility_fails_closed(): """compatibility 的 mode、冻结位或 accepted/ok 与权威 decision 矛盾时不得放行。""" for key, value in (("ok", False), ("accepted", False), ("publishFrozen", True), ("acceptanceVersion", "v3_shadow")): summary = _v3_summary("accept") summary["acceptanceV3"]["compatibility"][key] = value _assert_v3_failed(summary, f"v3-bad-compat-{key}") def test_v3_contradiction_7_proof_refs_loop_fails_closed(): """required proof、三类引用或首局闭环任一缺失,都不能被模型 accept 文本替代。""" bad_proof = _v3_summary("accept") bad_proof["acceptanceV3"]["proofObligations"][0]["status"] = "failed" bad_refs = _v3_summary("accept") bad_refs["acceptanceV3"]["proofObligations"][0]["actionRefs"] = [] bad_loop = _v3_summary("accept") bad_loop["acceptanceV3"]["firstPlay"]["loopClosed"] = False for index, summary in enumerate((bad_proof, bad_refs, bad_loop)): _assert_v3_failed(summary, f"v3-bad-proof-{index}") def test_v3_flat_projection_contradiction_fails_closed(): """完整封存对象虽有效,run-summary 平铺 accepted 被污染也必须失败。""" summary = _v3_summary("accept") summary["accepted"] = False _assert_v3_failed(summary, "v3-bad-flat") def test_v3_declared_without_acceptance_payload_is_downgrade_detected(): """任一 v3 声明已出现却缺 acceptanceV3 时,不得回落 legacy verdict/accepted。""" claims = [ {"acceptanceVersion": "v3"}, {"acceptanceVersion": "historical_replay"}, {"acceptanceVersion": "v999"}, {"acceptanceVersion": {"corrupt": True}}, {"publishFrozen": False}, {"trace": {"playtest": {"schemaVersion": "playtest/2", "outcome": "accept"}}}, ] gd = _make_game_dir(_tmp()) for index, claim in enumerate(claims): summary = {"verdict": {"pass": True}, "accepted": True, "ok": True, **claim} out = R.build_result_out({"traceId": f"v3-downgrade-{index}", "brief": "x"}, summary, gd) assert out["status"] == "failed" assert out["trace"]["acceptanceState"] == "downgrade_detected" assert out["trace"]["publishFrozen"] is True assert "playtest" not in out["trace"] def test_expected_v3_blocks_fully_stripped_summary_from_legacy_fallback(): """生产边界已冻结 v3 时,即使 summary 所有 marker 都被剥掉,legacy pass=true 也不能放行。""" gd = _make_game_dir(_tmp()) summary = {"gameId": "t1", "verdict": {"pass": True}, "accepted": True, "ok": True} out = R.build_result_out( {"traceId": "v3-fully-stripped", "gameId": "t1", "brief": "x"}, summary, gd, expected_acceptance_mode="v3") assert out["status"] == "failed" assert out["trace"]["acceptanceState"] == "downgrade_detected" assert out["trace"]["publishFrozen"] is True assert "engineBundle" not in out def test_legacy_v1_stays_compatible_but_sealed_v2_is_observation_only(): """无协议的 v1 存量仍兼容;冻结 playtest/2 可观察但不能作为 active 发布授权。""" gd = _make_game_dir(_tmp()) legacy = {"verdict": {"pass": True}, "accepted": True, "playtest": {"accepted": True}} legacy_out = R.build_result_out({"traceId": "legacy-v1", "brief": "x"}, legacy, gd) assert legacy_out["status"] == "succeeded" assert "acceptanceState" not in legacy_out["trace"] frozen_v2 = json.loads( (_V2_HISTORY_SAMPLE_DIR / _V2_HISTORY_SAMPLE_BY_OUTCOME["accept"]).read_text(encoding="utf-8") ) observed = R.build_result_out({"traceId": "historical-v2", "brief": "x"}, frozen_v2, gd) assert observed["status"] == "failed" assert observed["trace"]["playtest"]["schemaVersion"] == "playtest/2" assert "engineBundle" not in observed def test_v3_unknown_schema_and_outcome_fail_closed(): """未知 schema/outcome 不能回落顶层 accepted 或旧四门判定。""" gd = _make_game_dir(_tmp()) bad_schema = _v3_summary("accept") bad_schema["acceptanceV3"]["schemaVersion"] = "playtest/999" bad_outcome = _v3_summary("accept") bad_outcome["acceptanceV3"]["decision"]["outcome"] = "maybe" schema_job = _bind_v3_run(bad_schema, gd) outcome_job = _bind_v3_run(bad_outcome, gd) assert R.build_result_out(schema_job, bad_schema, gd)["status"] == "failed" assert R.build_result_out(outcome_job, bad_outcome, gd)["status"] == "failed" def test_v3_failure_ignores_stale_legacy_failure_reason(): """v3 outcome 已成为真相后,旧 worker 残留的合法枚举也不能把玩法失败误归为 unsafe_prompt。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("reject", mode="v3") summary["failureReason"] = "unsafe_prompt" job = _bind_v3_run(summary, gd) out = R.build_result_out(job, summary, gd, failure_reason="unsafe_prompt") assert out["status"] == "failed" assert out["failureReason"] == "llm_error" assert out["trace"]["playtest"]["outcome"] == "reject" def test_historical_replay_marker_never_falls_back_to_legacy_pass(): """历史回放即使携带旧 pass/accepted=true 且 bundle 完整,也只能降级阻断,永不发布。""" gd = _make_game_dir(_tmp()) summary = { "verdict": {"pass": True}, "accepted": True, "ok": True, "acceptanceVersion": "historical_replay", } out = R.build_result_out({"traceId": "historical-no-publish", "brief": "x"}, summary, gd) assert out["status"] == "failed" assert out["trace"]["acceptanceState"] == "downgrade_detected" assert out["trace"]["publishFrozen"] is True assert "engineBundle" not in out def test_v3_direct_sealed_result_is_observable_but_never_publishable(): """裸 decision.json 可离线观察,但缺当前 run-summary 身份绑定,不能直接授权发布。""" gd = _make_game_dir(_tmp()) out = R.build_result_out({"traceId": "v3-direct", "gameId": "t1", "brief": "x"}, _v3_result("accept"), gd) assert out["status"] == "failed" assert out["trace"]["playtest"]["schemaVersion"] == "playtest/3" def test_v3_old_accept_cannot_publish_for_different_job(): """同一封存 accept 被塞给另一 gameId 时必须 fail-closed。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) job["gameId"] = "t2" assert R.build_result_out(job, summary, gd)["status"] == "failed" def test_v3_accept_cannot_publish_for_different_create_brief(): """create 当前 brief 与封存 briefHash 不同,旧 accept 不得跨题面重放。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) job["brief"] = "另一道题" assert R.build_result_out(job, summary, gd)["status"] == "failed" def test_v3_modify_keeps_original_brief_hash_binding(): """modify job.brief 是修改指令;原题面 briefHash 由封存验收继承,不能按 create 规则误拒。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd, brief="原游戏题面") job.update({"brief": "把速度调快", "modifyMode": "deterministic", "baseVersionId": "base-1"}) source_project = _v3_source_project(summary, gd, "原游戏题面") assert R.build_result_out(job, summary, gd, source_project=source_project)["status"] == "succeeded" def test_v3_modify_rejects_missing_empty_or_cross_brief_source_origin(): """modify 不得用缺失/空/另一题面的 SourceProject 原题面授权 active v3 发布。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd, brief="原游戏题面") job.update({"brief": "把速度调快", "modifyMode": "deterministic", "baseVersionId": "base-1"}) assert R.build_result_out(job, summary, gd, source_project=None)["status"] == "failed" valid = json.loads(_v3_source_project(summary, gd, "原游戏题面")) for brief in ("", "另一款游戏题面"): forged = dict(valid) forged["originBrief"] = brief forged["originBriefHash"] = hashlib.sha256(brief.encode("utf-8")).hexdigest() out = R.build_result_out(job, summary, gd, source_project=json.dumps(forged, ensure_ascii=False)) assert out["status"] == "failed" def test_v3_regenerate_outer_assertion_failure_blocks_canonical_accept(): """玩法验收虽 accept,模块重生成 no-op/越界的外层失败也必须阻断发布。""" gd = _make_game_dir(_tmp()) accepted_summary = _v3_summary("accept") job = _bind_v3_run(accepted_summary, gd, brief="原游戏题面") job.update({"brief": "改玩法", "modifyMode": "regenerate-module", "baseVersionId": "base-1"}) failed_result = { "status": "failed", "stage": "play", "summary": accepted_summary, "verdict": accepted_summary.get("verdictFull"), "error": "断言①失败:目标文件未变", } wrapped = W._regen_summary(failed_result, "t1") out = R.build_result_out(job, wrapped, gd, expected_acceptance_mode="v3") assert wrapped["ok"] is False, "外层三断言失败必须覆盖旧 run-summary.ok=true" assert out["status"] == "failed" assert "engineBundle" not in out def test_v3_run_summary_identity_mismatch_fails_closed(): """acceptanceRunId/requestHash 等平铺身份必须与本次 sealed decision 逐字段一致。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) summary["acceptanceRunId"] = "old-run" assert R.build_result_out(job, summary, gd)["status"] == "failed" def test_v3_accept_cannot_replay_same_content_across_tasks(): """作品、题面与产物完全相同,只要当前 traceId 改变,旧封存裁决也不得授权新任务。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) replay = {**job, "traceId": "another-task-trace"} source_project = _v3_source_project(summary, gd, "x") assert R.build_result_out(replay, summary, gd, source_project=source_project)["status"] == "failed" def test_v3_accept_cannot_publish_foreign_release_bundle(): """发布目录 bundle 在验收后被替换,即使仍含 __GameBundle 也不得回调成功。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) (gd / "bundle.iife.js").write_text("var __GameBundle={foreign:true};", encoding="utf-8") assert R.build_result_out(job, summary, gd)["status"] == "failed" def test_v3_accept_cannot_publish_drifted_staged_artifact(): """staged 目录任一普通文件漂移都会改变 artifactHash,不能只比 bundle 文件。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) staged = gd.parent / "_wg1-gen" / "t1" (staged / "late-change.txt").write_text("drift", encoding="utf-8") assert R.build_result_out(job, summary, gd)["status"] == "failed" def test_v3_snapshot_rejects_bundle_rewrite_then_restore(monkeypatch): """快照若读到恶意中间态,即使文件随后恢复,内存快照 hash 仍不匹配旧验收,必须拒绝。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) source_project = _v3_source_project(summary, gd, "x") staged_bundle = gd.parent / "_wg1-gen" / "t1" / "bundle.iife.js" accepted_bytes = staged_bundle.read_bytes() original_capture = R._capture_artifact_snapshot def capture_evil_then_restore(root): staged_bundle.write_text("var __GameBundle={evil:true};", encoding="utf-8") try: return original_capture(root) finally: staged_bundle.write_bytes(accepted_bytes) monkeypatch.setattr(R, "_capture_artifact_snapshot", capture_evil_then_restore) out = R.build_result_out(job, summary, gd, source_project=source_project) assert out["status"] == "failed" assert "engineBundle" not in out def test_v3_snapshot_rejects_artifact_file_count_over_hard_cap(monkeypatch): """发布快照文件数超过与 Node 一致的 4096 硬帽时,必须 fail-closed。""" assert R.artifact_snapshot.MAX_ARTIFACT_FILES == 4096 gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) # 标准 staged 树已有 bundle + 两个 src 文件;缩小测试帽即可覆盖第 N+1 个文件前拒绝的路径。 monkeypatch.setattr(R.artifact_snapshot, "MAX_ARTIFACT_FILES", 2) out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x")) assert out["status"] == "failed" assert "engineBundle" not in out def test_v3_snapshot_rejects_artifact_bytes_over_hard_cap(monkeypatch): """发布快照总字节超过与 Node 一致的 128 MiB 硬帽时,必须在发布前拒绝。""" assert R.artifact_snapshot.MAX_ARTIFACT_BYTES == 128 * 1024 * 1024 gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) monkeypatch.setattr(R.artifact_snapshot, "MAX_ARTIFACT_BYTES", 1) out = R.build_result_out(job, summary, gd, source_project=_v3_source_project(summary, gd, "x")) assert out["status"] == "failed" assert "engineBundle" not in out def test_v3_source_project_rejects_release_src_drift_and_accepts_staged_src(): """v3 源血缘只认验收同源 staged/src;release/src 的后验改写不得落库。""" gd = _make_game_dir(_tmp(), src_files={"game-logic.js": "// accepted logic\n", "core.js": "export const SPEED=1;\n"}) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) staged = gd.parent / "_wg1-gen" / "t1" profile = {"tickModel": "realtime", "inputModel": "discrete-choice", "progressModel": "metric"} # 验收后仅漂移 release 源,bundle 与 staged artifact 仍保持不变。 (gd / "src" / "core.js").write_text("export const SPEED=999; // 未验收源\n", encoding="utf-8") drifted = R.build_source_project( gd, profile, scaffold_template="_template-story", origin_brief="x", origin_brief_hash=hashlib.sha256(b"x").hexdigest()) rejected = R.build_result_out(job, summary, gd, source_project=drifted, expected_acceptance_mode="v3") assert rejected["status"] == "failed", "active v3 的源血缘属于跨请求可信边界,漂移时必须阻断" assert "sourceProject" not in rejected, "release/src 后验漂移不得随成功回调落库" accepted = R.build_source_project( staged, profile, scaffold_template="_template-story", origin_brief="x", origin_brief_hash=hashlib.sha256(b"x").hexdigest()) published = R.build_result_out(job, summary, gd, source_project=accepted, expected_acceptance_mode="v3") assert published["status"] == "succeeded" assert json.loads(published["sourceProject"])["files"]["src/core.js"] == "export const SPEED=1;\n" def test_v3_failed_result_never_carries_source_project(): """v3 最终发布绑定失败时,即使 SourceProject 来自 staged,也不得落成可继承血缘。""" gd = _make_game_dir(_tmp()) summary = _v3_summary("accept") job = _bind_v3_run(summary, gd) staged = gd.parent / "_wg1-gen" / "t1" profile = {"tickModel": "realtime", "inputModel": "discrete-choice", "progressModel": "metric"} source_project = R.build_source_project( staged, profile, scaffold_template="_template-story", origin_brief="x", origin_brief_hash=hashlib.sha256(b"x").hexdigest()) (gd / "bundle.iife.js").write_text("var __GameBundle={foreign:true};", encoding="utf-8") out = R.build_result_out(job, summary, gd, source_project=source_project, expected_acceptance_mode="v3") assert out["status"] == "failed" assert "sourceProject" not in out if __name__ == "__main__": _fns = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)] _failed = 0 for _fn in _fns: try: _fn() print(f" PASS {_fn.__name__}") except Exception as e: # noqa: BLE001 _failed += 1 print(f" FAIL {_fn.__name__}: {type(e).__name__}: {e}") print(f"\n{len(_fns) - _failed}/{len(_fns)} passed") sys.exit(1 if _failed else 0)