feat(tier2): B门-P1 验收质量层(advisory)——人可玩门 + L2 full + L3 升门判据
修「gate 通过≠人能玩」盲区(feie-005 实证:过九门+富三门却棋盘渲染空 + 8s 耐心人玩不了)。 补齐 D 族三层校验,全 advisory(写 verdict 不碰 decision;自带 Goodhart 守护:advisory 全挂也不翻 decision): - 人可玩门(play-phaser.cdp.cjs):render-reflects-state(状态有 item→该格 canvas 像素须非空) + render-sanity(无 [object Object]/undefined/NaN 漏底文本)+ human-timing(订单耐心 ≥ HUMAN_MIN_PATIENCE_MS=12000·★directional) - L2 设计符合 full(run.py compute_l2_signals):systemsPresent/mergeDag/ordersReachable/economyConsistent 四硬判据, 从落盘源工程确定性算(替桩) - L3 升门判据(run.py derive_l3_egregious + studio.py):消费 humanPlayability 的恶性渲染失败→suggestFatal 信号 - schema additive:humanPlayability 段 + L2.mismatches + L3.egregiousRenderFailure 离线验证:py_compile / node --check / JSON 合法 / 24 断言单测(含 Goodhart 守护:advisory 全挂 decision 字节不变)/ 金标 L2 离线 passed=True + 反例精确报错。 **fatal 提升 + 金标冒烟集成验证待 mini-desktop**:确认金标仍 ACCEPT(新检查不抛、全 pass)+ feie-005 被挂后, 逐条提 fatal(品类限定有 board/orders 档)。校准数据:金标 patience=16-20s / feie-005=8s / 阈值候选 12s。 Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
d6e977a5ac
commit
2c4e2e2248
@ -72,7 +72,7 @@
|
||||
}
|
||||
},
|
||||
"L2": {
|
||||
"description": "L2 设计符合层:玩法/关卡/UI vs 设计、品类约定。处置=尽量解非死循环。拒发权 observe→enforce:先只观测积累信号,判据证准才赋拒发权。九门里 H_progress 落本层。",
|
||||
"description": "L2 设计符合层:玩法/关卡/UI vs 设计、品类约定。处置=尽量解非死循环。拒发权 observe→enforce:先只观测积累信号,判据证准才赋拒发权。九门里 H_progress 落本层。【full 计算(零 LLM)】passed/signals/mismatches 由 judge 纯代码从已落盘源工程(data/datatable.gold.json + src/systems/)确定性算出:① 设计声明的三系统(资源/合成/订单)在数据表与源文件中都建出;② mergeChains 构成 DAG 无环;③ 每个 order.requires 可被合成产出∪开局库存满足(可达性);④ 经济胜负条件可达且自洽(赢线在 开局金币+订单 reward 之和 内可达、输线连续流失阈值≥1 且订单有有限耐心)。passed=①~④ 全 ok。spike 期 enforced=false(observe-only)。金标 fixture(mini-fei-e)先天满足 ①~④,故金标 L2 必 passed=true——这是将来提门时『金标必须过』的校准基线。",
|
||||
"type": "object",
|
||||
"required": ["enforced"],
|
||||
"additionalProperties": false,
|
||||
@ -82,11 +82,16 @@
|
||||
"type": "boolean"
|
||||
},
|
||||
"passed": {
|
||||
"description": "L2 设计符合度判定结果(确定性信号);enforced=false 时仅观测、不影响 decision。",
|
||||
"description": "L2 设计符合度判定结果(确定性信号,= 上述 ①~④ 全 ok);enforced=false 时仅观测、不影响 decision。",
|
||||
"type": ["boolean", "null"]
|
||||
},
|
||||
"signals": {
|
||||
"description": "L2 观测信号明细(设计符合度逐项,供 observe→enforce 积累)。",
|
||||
"description": "L2 观测信号明细(设计符合度逐项,供 observe→enforce 积累)。full 计算下含通过项的人读信号 + 以 '[不符]' 前缀列出的不符项摘要。",
|
||||
"type": "array",
|
||||
"items": { "type": "string" }
|
||||
},
|
||||
"mismatches": {
|
||||
"description": "L2 设计不符项明细(人读;additive·可选):声明的系统未建出 / 合成链空或成环 / 订单不可达 / 经济胜负不自洽等。仅观测,不影响 decision(enforced=false)。",
|
||||
"type": "array",
|
||||
"items": { "type": "string" }
|
||||
}
|
||||
@ -107,9 +112,30 @@
|
||||
"type": ["number", "null"]
|
||||
},
|
||||
"notes": {
|
||||
"description": "M3 软检/player panel 的劣化信号备注(可达性、空内容等确定性可逼近的;给人门减负,非判定)。",
|
||||
"description": "M3 软检/player panel 的劣化信号备注(可达性、空内容等确定性可逼近的;给人门减负,非判定)。也并入下方 egregiousRenderFailure 命中的人读条目。",
|
||||
"type": "array",
|
||||
"items": { "type": "string" }
|
||||
},
|
||||
"egregiousRenderFailure": {
|
||||
"description": "【L3 升门准备(advisory·additive·可选)】确定性『恶性渲染失败』判据,吃单元 A 的 humanPlayability.renderSanity / renderReflectsState 结果(漏底文本 [object Object]/undefined/NaN、或有 item 的格子画成空棋盘)。与 L3 主观打分分开——这两类是机器可确定性判定的恶性失败,非『好不好玩』。【纪律不让步】本字段绝不改变 scoreOnly(恒 true)、绝不参与 decision/runnableOk/L1.passed——egregious=true 时 suggestFatal=true 只是给主会话的升门信号,真提 fatal 须由主会话在 mini-desktop 实测金标(mini-fei-e)仍 ACCEPT、坏样本(feie-005)被正确挂后再据判据落门。",
|
||||
"type": ["object", "null"],
|
||||
"additionalProperties": false,
|
||||
"required": ["egregious", "suggestFatal"],
|
||||
"properties": {
|
||||
"egregious": {
|
||||
"description": "是否检出确定性恶性渲染失败(漏底文本 或 空棋盘)。observe-only,不改 decision。",
|
||||
"type": "boolean"
|
||||
},
|
||||
"suggestFatal": {
|
||||
"description": "升门信号:egregious=true 时为 true,建议主会话校准后把『egregious 即 fatal』提成门(当前仍 advisory)。",
|
||||
"type": "boolean"
|
||||
},
|
||||
"reasons": {
|
||||
"description": "恶性判定的人读理由明细(命中了哪类:render-sanity 漏底 / render-reflects-state 空棋盘)。",
|
||||
"type": "array",
|
||||
"items": { "type": "string" }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -169,6 +195,105 @@
|
||||
"evidenceDir": { "description": "四件套证据落盘目录(截图序列/驱动日志/帧 hash 等)。", "type": "string" },
|
||||
"promptVersions": { "description": "prompt id → 版本号映射(eval 回流路由用)。", "type": "object", "additionalProperties": { "type": "string" } }
|
||||
}
|
||||
},
|
||||
"humanPlayability": {
|
||||
"description": "【人可玩 advisory 段(additive·可选)】修『gate 通过≠人能玩』盲区:九门用 driver 按布局坐标点(不看屏幕)且点得飞快,抓不到『渲染不反映状态/漏底实现细节文本/反应窗口只为机器人调』这三类人玩不了但机器能过的缺陷。本段三项检查全部 advisory——只写进 verdict 供人审与后续校准,【绝不参与 decision 的 accept/fix/kill,也不计入 runnableOk / L1.passed】。提 fatal 须先由主会话在 mini-desktop 实测确认金标 fixture(mini-fei-e)仍过、坏样本(feie-005)被正确挂,再据金标实测值定阈值。状态=建(advisory;待主会话校准阈值后再议是否提 fatal)。",
|
||||
"type": ["object", "null"],
|
||||
"additionalProperties": false,
|
||||
"properties": {
|
||||
"advisory": {
|
||||
"description": "恒为 true 的不变量:本段全部 advisory,只报告不阻塞(防 Goodhart;与 L3 scoreOnly 同纪律,焊死『不当门』)。",
|
||||
"const": true
|
||||
},
|
||||
"renderReflectsState": {
|
||||
"description": "渲染反映状态检查:对 state.board 里每个『有 item 的格子』,按其 (x,y)+CELL 取 canvas 像素区域,与空格子基线区域比对(像素方差/与背景色差);若有 item 的格子区域与空格子无法区分→记 violation(feie-005 即此症:state 有原料但画空格子)。pass=无此类格子。",
|
||||
"type": ["object", "null"],
|
||||
"additionalProperties": false,
|
||||
"required": ["pass"],
|
||||
"properties": {
|
||||
"pass": { "description": "是否所有有 item 的格子都在渲染上区别于空格子。", "type": "boolean" },
|
||||
"checked": { "description": "实际比对的有 item 格子数。", "type": "integer", "minimum": 0 },
|
||||
"emptyButShouldHaveItem": {
|
||||
"description": "有 item 却渲染成空的格子明细。",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"required": ["i", "j", "item"],
|
||||
"additionalProperties": true,
|
||||
"properties": {
|
||||
"i": { "type": "integer" },
|
||||
"j": { "type": "integer" },
|
||||
"item": { "type": "string" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"note": { "description": "执行说明(如无法取像素/无 board 时的降级原因;非判定来源)。", "type": "string" }
|
||||
}
|
||||
},
|
||||
"renderSanity": {
|
||||
"description": "渲染漏底检查:扫描渲染出的文本(页面 DOM textContent + host 暴露的 canvas 文本若有)是否含 [object Object]/undefined/NaN 这类漏底实现细节。pass=无漏底文本。",
|
||||
"type": ["object", "null"],
|
||||
"additionalProperties": false,
|
||||
"required": ["pass"],
|
||||
"properties": {
|
||||
"pass": { "description": "是否无漏底实现细节文本。", "type": "boolean" },
|
||||
"violations": {
|
||||
"description": "命中的漏底标记明细。",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"required": ["marker"],
|
||||
"additionalProperties": true,
|
||||
"properties": {
|
||||
"marker": { "description": "命中的漏底标记(如 [object Object])。", "type": "string" },
|
||||
"sample": { "description": "命中处的文本片段(截断)。", "type": "string" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"note": { "description": "执行说明(非判定来源)。", "type": "string" }
|
||||
}
|
||||
},
|
||||
"humanTiming": {
|
||||
"description": "人类反应窗口检查:读数据表(经 state 或 datatable.gold.json)订单 patience 及任何反应窗口,记录 min/max/各单 patience;低于候选人类最小阈值 HUMAN_MIN_PATIENCE_MS(★ directional 待主会话用金标实测校准)的订单记 advisory violation(feie-005 即此症:8 秒耐心给机器人调的、人来不及)。",
|
||||
"type": ["object", "null"],
|
||||
"additionalProperties": false,
|
||||
"required": ["pass"],
|
||||
"properties": {
|
||||
"pass": { "description": "是否所有反应窗口都 ≥ 候选人类最小阈值。", "type": "boolean" },
|
||||
"minPatienceMs": { "description": "最短订单耐心(毫秒)。", "type": ["number", "null"] },
|
||||
"maxPatienceMs": { "description": "最长订单耐心(毫秒)。", "type": ["number", "null"] },
|
||||
"thresholdMs": { "description": "本次采用的候选人类最小阈值(HUMAN_MIN_PATIENCE_MS;★ directional)。", "type": "number" },
|
||||
"patiencesMs": {
|
||||
"description": "各订单耐心明细(毫秒)。",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"required": ["patienceMs"],
|
||||
"additionalProperties": true,
|
||||
"properties": {
|
||||
"orderId": { "type": "string" },
|
||||
"patienceMs": { "type": "number" },
|
||||
"belowThreshold": { "type": "boolean" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"violations": {
|
||||
"description": "低于阈值的订单明细(advisory)。",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"required": ["patienceMs"],
|
||||
"additionalProperties": true,
|
||||
"properties": {
|
||||
"orderId": { "type": "string" },
|
||||
"patienceMs": { "type": "number" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"note": { "description": "执行说明(如数据表来源/无 patience 字段时的降级;非判定来源)。", "type": "string" }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"$defs": {
|
||||
|
||||
@ -380,6 +380,41 @@ async def run_studio(
|
||||
breaker_tripped = {"kind": None, "reason": f"未捕获异常:{type(e).__name__}: {e}"}
|
||||
print(f"[tier2-studio] game={game_id} 异常:{type(e).__name__}: {e}", flush=True)
|
||||
|
||||
# ── L2 设计符合层 full(observe-only · 收口后算一次 · 零 LLM · 绝不参与 accept/reject)──────
|
||||
# 时机:ReAct 收敛、跑过门后(无论 finish/kill/熔断)算一次。它从已落盘源工程(数据表 + src/systems/)
|
||||
# 确定性算「设计声明的系统在不在 / 合成链 DAG / 订单可达 / 经济胜负自洽」四项硬结构信号 + 设计稿弱对账,
|
||||
# 写进 verdict.layerResults.L2.{passed(advisory),signals,mismatches}。本阶段纪律:enforced 恒 false、
|
||||
# **绝不改 decision**——只为 observe→enforce 积累信号。金标 fixture 先天满足四项判据,故金标 L2 必 passed=true,
|
||||
# 这是将来提门时「金标必须过」的校准基线。失败兜底:数据表不可读 → passed=False + mismatch,绝不中断主链。
|
||||
l2_passed = None
|
||||
l2_signals: list[str] = []
|
||||
l2_mismatches: list[str] = []
|
||||
try:
|
||||
l2 = run.compute_l2_signals(game_id, design_text)
|
||||
l2_passed = l2.get("passed")
|
||||
l2_signals = l2.get("signals") or []
|
||||
l2_mismatches = l2.get("mismatches") or []
|
||||
v = session.last_verdict
|
||||
if isinstance(v, dict):
|
||||
lr = v.setdefault("layerResults", {})
|
||||
l2seg = lr.setdefault("L2", {})
|
||||
l2seg["enforced"] = False # spike 期恒 observe-only(不计入 decision)
|
||||
l2seg["passed"] = l2_passed # boolean|null;advisory,不影响 decision
|
||||
# signals 累加(保留 harness 已写的 l2Signals,再并入 full 计算信号 + 不符项)。
|
||||
prev = l2seg.get("signals") or []
|
||||
merged = list(prev)
|
||||
for s in (l2_signals + [f"[不符] {m}" for m in l2_mismatches]):
|
||||
if s not in merged:
|
||||
merged.append(s)
|
||||
l2seg["signals"] = merged
|
||||
print(f"[tier2-studio][L2] game={game_id} passed(advisory)={l2_passed} "
|
||||
f"signals={len(l2_signals)} mismatches={len(l2_mismatches)}"
|
||||
"(observe-only,不参与 decision)", flush=True)
|
||||
except Exception as e: # noqa: BLE001 —— L2 整体兜底:任何异常都不中断主链、不翻 decision
|
||||
print(f"[tier2-studio][L2] 设计符合层计算异常(不影响 decision):"
|
||||
f"{type(e).__name__}: {e}", flush=True)
|
||||
l2_mismatches = [f"L2 计算异常:{type(e).__name__}: {e}"]
|
||||
|
||||
# ── L3 视觉软检(observe-only · 收口后调一次 · 绝不参与 accept/reject)───────────────
|
||||
# 时机:过完 L1 九门/富游戏门、ReAct 收敛后(无论 finish/kill/熔断)调一次。
|
||||
# 它用 player_system 提示词 + 真截图 + 真玩取证产 score/简评,只写进 verdict.L3.score/notes。
|
||||
@ -394,15 +429,33 @@ async def run_studio(
|
||||
l3_notes = l3.get("notes") or []
|
||||
# observe-only:把 L3 结果写回最近 verdict 的 layerResults.L3(scoreOnly 恒 true,只填 score/notes)。
|
||||
# 只在已有 verdict(真跑过门)时回写;verdict 缺失则 L3 结果只留在 result.l3(下方),不伪造 verdict。
|
||||
# L3 升门准备(确定性恶性渲染判据):吃单元 A 的 render-sanity / render-reflects-state
|
||||
# (harness 真玩落进 verdict.humanPlayability),提「漏底文本 / 空棋盘」这类**确定性**恶性失败,
|
||||
# 整理进 L3.notes 并标 egregious=true(信号:建议主会话校准后提 fatal)。**仍 advisory、不改 decision**。
|
||||
egr = run.derive_l3_egregious(session.last_verdict)
|
||||
l3_egregious_notes = egr.get("notes") or []
|
||||
if l3_egregious_notes:
|
||||
l3_notes = list(l3_notes) + l3_egregious_notes # 并入软检 notes(下方写回 verdict 与 result.l3)
|
||||
v = session.last_verdict
|
||||
if isinstance(v, dict):
|
||||
lr = v.setdefault("layerResults", {})
|
||||
l3seg = lr.setdefault("L3", {})
|
||||
l3seg["scoreOnly"] = True # schema const true 不变量(防 L3 退化成阻塞门)
|
||||
l3seg["score"] = l3_score # number|null;observe-only,不影响 decision
|
||||
# notes 累加(保留 harness 已写的 l3Notes,再并入软检备注)。
|
||||
# notes 累加(保留 harness 已写的 l3Notes,再并入软检备注 + 恶性渲染确定性条目)。
|
||||
prev = l3seg.get("notes") or []
|
||||
l3seg["notes"] = list(prev) + [n for n in l3_notes if n not in prev]
|
||||
# 恶性渲染确定性判据写成结构化 advisory 段(machine-readable 升门信号;仍不改 decision/scoreOnly)。
|
||||
l3seg["egregiousRenderFailure"] = {
|
||||
"egregious": bool(egr.get("egregious")),
|
||||
"suggestFatal": bool(egr.get("suggestFatal")),
|
||||
"reasons": egr.get("reasons") or [],
|
||||
}
|
||||
if egr.get("egregious"):
|
||||
# 仅日志告警(给主会话升门校准信号);不写任何字段进 decision/L1/runnableOk,严守 advisory。
|
||||
print(f"[tier2-studio][L3] game={game_id} 检出确定性恶性渲染失败 egregious=true "
|
||||
f"reasons={egr.get('reasons')} —— 建议主会话校准后提 fatal(当前 advisory,不改 decision)。",
|
||||
flush=True)
|
||||
except Exception as e: # noqa: BLE001 —— L3 整体兜底:任何异常都不中断主链、不翻 GREEN
|
||||
print(f"[tier2-studio][L3] 软检整体异常(score=null 兜底,不影响 decision):"
|
||||
f"{type(e).__name__}: {e}", flush=True)
|
||||
@ -460,6 +513,9 @@ async def run_studio(
|
||||
# 最近一次 run_gates 的 verdict(tier2-verdict 形状;judge 纯代码产出,零自评;若真跑过门,
|
||||
# 其 layerResults.L3.score/notes 已由上面 L3 软检 observe-only 回填)。
|
||||
"last_verdict": session.last_verdict,
|
||||
# L2 设计符合层结果(observe-only,绝不参与 decision;enforced 恒 false,passed 是 advisory 判定)。
|
||||
# verdict 缺失时仍在此可见(不伪造 verdict);金标 fixture 应 passed=true(将来提门校准基线)。
|
||||
"l2": {"passed": l2_passed, "signals": l2_signals, "mismatches": l2_mismatches},
|
||||
# L3 视觉软检结果(observe-only,绝不参与 decision;verdict 缺失时仍在此可见,不伪造 verdict)。
|
||||
"l3": {"score": l3_score, "notes": l3_notes},
|
||||
# 四熔断触发记录(对接 verdict.breakerKind:step_cap/budget/stuck/timeout)。
|
||||
|
||||
@ -480,6 +480,280 @@ def _derive_richgame_from_datatable(wd: Path) -> dict | None:
|
||||
return {"mergeChains": chains, "orders": orders}
|
||||
|
||||
|
||||
# ── 富游戏三系统语义锚(L2 设计符合层判据用)───────────────────────────────────
|
||||
# business-sim 品类的「设计意图」三系统:资源 / 合成 / 订单。L2 要对照「设计声明了哪些系统」
|
||||
# 与「实际建出的数据表 + 系统源文件」是否一致。这三者是 mini-肥鹅 fixture-spec 钉死的富游戏命门
|
||||
# (datatable.schema.json:currencies=资源 / mergeChains=合成 / orders=订单),不是质量打分。
|
||||
# 每条 = (人读系统名, 对应数据表段, 对应 src/systems/ 源文件名 stem)。
|
||||
_RICHGAME_SYSTEMS = (
|
||||
("资源系统", "currencies", "resource-system"),
|
||||
("合成系统", "mergeChains", "merge-system"),
|
||||
("订单系统", "orders", "order-system"),
|
||||
)
|
||||
|
||||
|
||||
def compute_l2_signals(game_id: str, design_text: str = "") -> dict:
|
||||
"""L2 设计符合层 full —— 从「设计意图」对照「实际建出的数据表 / 系统」算真信号(advisory)。
|
||||
|
||||
【为什么落这里、零 LLM】L2 判的是「玩法/系统实现得对不对、设计声明的系统在不在、经济胜负是否
|
||||
自洽可达」,全是可从**已落盘源工程**(data/datatable.gold.json + src/systems/*)确定性算出的结构信号,
|
||||
不需要看截图、不需要 LLM。这与现行 validate_datatable 同源(都读 agent 真填的表),但 validate_datatable
|
||||
是 build 前的**早断预检**(失败即拦 build),本函数是真玩后对 verdict 的**设计符合度观测**(advisory、不拦)。
|
||||
|
||||
【本阶段纪律 · advisory】本函数只产观测信号,**绝不改 decision 的 accept/reject**。返回的 passed 是
|
||||
advisory 判定(写进 verdict.layerResults.L2.passed 供 observe→enforce 积累),enforced 恒 false。
|
||||
金标 fixture(mini-fei-e)的数据表先天满足下列全部判据,故金标 L2 必 passed=true ——
|
||||
这正是将来把 L2 从 advisory 提成门时「金标必须过」的校准基线(见 followups / promotionPlan)。
|
||||
|
||||
判据五项(全确定性结构信号,不替 agent 设计数值,防 Goodhart):
|
||||
① systemsPresent —— 富游戏三系统(资源/合成/订单)既在数据表里声明(currencies/mergeChains/orders 非空),
|
||||
又有对应 src/systems/ 源文件(resource/merge/order-system),且可达性系统文件在。
|
||||
② mergeDag —— mergeChains 构成有向无环图(from→to 拓扑无环;与三联动门 dagAcyclic 同口径)。
|
||||
③ ordersReachable —— 每个 order.requires 的物品 ∈ mergeChains 可产出集合 ∪ 开局库存
|
||||
(与三联动门 requiresReachable 同口径:订单要的东西合成系统造得出来)。
|
||||
④ economyConsistent —— 经济胜负条件可达且自洽:赢线(coinsTarget)在「开局金币 + 全部订单 reward 之和」
|
||||
可达范围内(证盈利路数值上够得着,非死局);输线(consecutiveOrderFails≥1)且订单都有有限耐心(可流失)。
|
||||
⑤ designDeclares —— 设计稿(阶段 1 prose)是否点到了三系统(软交叉对账:命中即加分,**缺失绝不单独判负**——
|
||||
设计 prose 自由文本、措辞多样,只作弱信号,真信号在 ①~④ 的源工程结构)。
|
||||
|
||||
Args:
|
||||
game_id: 工程标识(定位 workdir 下 data/datatable.gold.json 与 src/systems/)。
|
||||
design_text: 阶段 1 设计稿原文(可选;只做 ⑤ 的弱交叉对账,缺失不影响 ①~④)。
|
||||
|
||||
Returns:
|
||||
{passed: bool(advisory), signals: [人读字符串], mismatches: [不符项人读字符串],
|
||||
checks: [{name, ok, detail}]}。
|
||||
- passed = 硬结构判据 ①~④ 全 ok(⑤ 不参与 passed,只进 signals)。
|
||||
- 任何 IO/解析异常都吞成「数据表不可读」一条 mismatch、passed=False,绝不抛(L2 advisory 不能中断主链)。
|
||||
"""
|
||||
wd = _workdir(game_id)
|
||||
checks: list[dict] = []
|
||||
signals: list[str] = []
|
||||
mismatches: list[str] = []
|
||||
|
||||
# ── 读 agent 真填的数据表(L2 的设计意图实物来源)──
|
||||
dt = wd / "data" / "datatable.gold.json"
|
||||
if not dt.exists():
|
||||
mismatches.append("缺 data/datatable.gold.json,无法对照设计符合度。")
|
||||
return {"passed": False, "signals": signals, "mismatches": mismatches,
|
||||
"checks": [{"name": "datatableReadable", "ok": False,
|
||||
"detail": f"缺数据表(workdir={wd})"}]}
|
||||
try:
|
||||
gold = json.loads(dt.read_text(encoding="utf-8"))
|
||||
except Exception as e: # noqa: BLE001 —— L2 advisory:表坏不抛,记 mismatch
|
||||
mismatches.append(f"datatable.gold.json 非合法 JSON:{type(e).__name__}: {e}")
|
||||
return {"passed": False, "signals": signals, "mismatches": mismatches,
|
||||
"checks": [{"name": "datatableReadable", "ok": False, "detail": str(e)[:160]}]}
|
||||
|
||||
cur = gold.get("currencies") if isinstance(gold.get("currencies"), dict) else {}
|
||||
chains_raw = gold.get("mergeChains") if isinstance(gold.get("mergeChains"), list) else []
|
||||
orders_raw = gold.get("orders") if isinstance(gold.get("orders"), list) else []
|
||||
|
||||
# ── 判据 ① systemsPresent:三系统既在数据表声明、又有源文件 ──
|
||||
systems_dir = wd / "src" / "systems"
|
||||
missing_sys: list[str] = []
|
||||
for human, table_key, file_stem in _RICHGAME_SYSTEMS:
|
||||
in_table = bool(gold.get(table_key)) # 数据表段非空(currencies/mergeChains/orders)
|
||||
in_src = (systems_dir / f"{file_stem}.js").exists() # 对应系统源文件在
|
||||
if not in_table:
|
||||
missing_sys.append(f"{human}(数据表缺 {table_key} 段或为空)")
|
||||
if not in_src:
|
||||
missing_sys.append(f"{human}(缺 src/systems/{file_stem}.js)")
|
||||
# 可达性系统文件(跨表可达性判据的承载;fixture 预建为平台锁定文件)。
|
||||
reach_ok = (systems_dir / "reachability.js").exists()
|
||||
if not reach_ok:
|
||||
missing_sys.append("可达性系统(缺 src/systems/reachability.js)")
|
||||
sys_ok = not missing_sys
|
||||
checks.append({"name": "systemsPresent", "ok": sys_ok,
|
||||
"detail": "三系统(资源/合成/订单)+ 可达性均在数据表与源文件中"
|
||||
if sys_ok else "缺:" + ";".join(missing_sys)})
|
||||
if sys_ok:
|
||||
signals.append("设计声明的富游戏三系统(资源/合成/订单)均在数据表与 src/systems/ 落实。")
|
||||
else:
|
||||
mismatches.append("声明的系统未全部建出:" + ";".join(missing_sys))
|
||||
|
||||
# ── 判据 ② mergeDag:合成链 DAG 无环(复用三联动门 dagAcyclic 口径)──
|
||||
edges = [(c.get("from"), c.get("to")) for c in chains_raw
|
||||
if isinstance(c, dict) and c.get("from") and c.get("to")]
|
||||
dag_ok = bool(edges) and _is_dag_acyclic(edges)
|
||||
if not edges:
|
||||
checks.append({"name": "mergeDag", "ok": False,
|
||||
"detail": "mergeChains 无合法 from→to 边(合成系统为空,三系统退化成孤岛)"})
|
||||
mismatches.append("合成链为空 —— 富游戏退化成无合成的孤岛,违背多系统耦合设计意图。")
|
||||
elif not dag_ok:
|
||||
checks.append({"name": "mergeDag", "ok": False,
|
||||
"detail": "mergeChains from→to 拓扑有环(合成依赖成圈,系统会死锁)"})
|
||||
mismatches.append("合成链存在环 —— 合成依赖成圈,不符合 DAG 设计意图。")
|
||||
else:
|
||||
checks.append({"name": "mergeDag", "ok": True,
|
||||
"detail": f"mergeChains {len(edges)} 条边构成 DAG 无环"})
|
||||
signals.append(f"合成链 {len(edges)} 条构成 DAG 无环(合成系统拓扑自洽)。")
|
||||
|
||||
# ── 判据 ③ ordersReachable:订单要的物品都能被合成产出 ∪ 开局库存(复用 requiresReachable 口径)──
|
||||
producible = {b for _, b in edges}
|
||||
start_inv = set(((cur.get("ingredients") or {}).get("initial") or {}).keys()) \
|
||||
if isinstance(cur.get("ingredients"), dict) else set()
|
||||
reachable = producible | start_inv
|
||||
unreachable: list[str] = []
|
||||
for o in orders_raw:
|
||||
req = (o or {}).get("requires") or {}
|
||||
if not isinstance(req, dict):
|
||||
unreachable.append(f"{(o or {}).get('id')}(requires 非 {{itemId:qty}} 映射)")
|
||||
continue
|
||||
for item_id in req:
|
||||
if item_id not in reachable:
|
||||
unreachable.append(f"{(o or {}).get('id')}→{item_id}")
|
||||
reach_orders_ok = bool(orders_raw) and not unreachable
|
||||
if not orders_raw:
|
||||
checks.append({"name": "ordersReachable", "ok": False,
|
||||
"detail": "orders 为空(订单系统退化成孤岛)"})
|
||||
mismatches.append("订单为空 —— 订单系统未真正接入,违背三系统耦合设计意图。")
|
||||
elif unreachable:
|
||||
checks.append({"name": "ordersReachable", "ok": False,
|
||||
"detail": "不可达订单物品:" + ",".join(unreachable[:8])})
|
||||
mismatches.append("以下订单物品合成系统造不出、也不在开局库存(订单↔合成未真正耦合):"
|
||||
+ ",".join(unreachable[:8]) + "。")
|
||||
else:
|
||||
checks.append({"name": "ordersReachable", "ok": True,
|
||||
"detail": f"{len(orders_raw)} 个订单的 requires 全部可达"})
|
||||
signals.append(f"{len(orders_raw)} 个订单要的物品全部可被合成产出或开局持有(订单↔合成真耦合)。")
|
||||
|
||||
# ── 判据 ④ economyConsistent:赢线在数值上够得着 + 输线自洽(经济门设计符合)──
|
||||
coins_init = (cur.get("coins") or {}).get("initial") if isinstance(cur.get("coins"), dict) else None
|
||||
coins_target = (gold.get("winCondition") or {}).get("coinsTarget") \
|
||||
if isinstance(gold.get("winCondition"), dict) else None
|
||||
lose_streak = (gold.get("loseCondition") or {}).get("consecutiveOrderFails") \
|
||||
if isinstance(gold.get("loseCondition"), dict) else None
|
||||
rewards = [o.get("reward") for o in orders_raw
|
||||
if isinstance((o or {}).get("reward"), (int, float))]
|
||||
patiences = [o.get("patience") for o in orders_raw
|
||||
if isinstance((o or {}).get("patience"), (int, float))]
|
||||
econ_problems: list[str] = []
|
||||
# 盈利路数值可达:开局金币 + 全部订单 reward 之和 ≥ 赢线(证攒得到 100,非死局)。
|
||||
if not isinstance(coins_init, (int, float)) or not isinstance(coins_target, (int, float)):
|
||||
econ_problems.append("缺 currencies.coins.initial 或 winCondition.coinsTarget")
|
||||
elif rewards and (coins_init + sum(rewards)) < coins_target:
|
||||
econ_problems.append(f"开局金币 {coins_init} + 全部订单 reward 之和 {sum(rewards)} "
|
||||
f"< 赢线 {coins_target}(盈利路数值上够不着,经济死局)")
|
||||
# 输线自洽:连续流失阈值 ≥1,且订单都有有限正耐心(能被流失)。
|
||||
if not isinstance(lose_streak, (int, float)) or lose_streak < 1:
|
||||
econ_problems.append("loseCondition.consecutiveOrderFails 缺失或 <1(无法判输)")
|
||||
if orders_raw and (not patiences or any(p is None or p <= 0 for p in patiences)):
|
||||
econ_problems.append("存在订单无有限正耐心(patience),破产路无法靠流失达成")
|
||||
econ_ok = not econ_problems
|
||||
checks.append({"name": "economyConsistent", "ok": econ_ok,
|
||||
"detail": ("赢线数值可达 + 输线自洽(可赢可输)" if econ_ok
|
||||
else ";".join(econ_problems))})
|
||||
if econ_ok:
|
||||
signals.append(f"经济胜负自洽:开局 {coins_init} 经订单正循环可达赢线 {coins_target},"
|
||||
f"连续 {lose_streak} 单流失可判输(可赢可输闭环)。")
|
||||
else:
|
||||
mismatches.append("经济胜负条件不可达或不自洽:" + ";".join(econ_problems) + "。")
|
||||
|
||||
# ── 判据 ⑤ designDeclares:设计稿弱交叉对账(命中加分,缺失绝不单独判负)──
|
||||
dt_lower = (design_text or "")
|
||||
declared_hits = [human for human, _key, _stem in _RICHGAME_SYSTEMS
|
||||
if human in dt_lower or human[:2] in dt_lower] # 命中「资源/合成/订单」任一表述
|
||||
if design_text:
|
||||
if len(declared_hits) >= 2:
|
||||
signals.append(f"设计稿点到了富游戏多系统({'、'.join(declared_hits)}),与建出的系统一致。")
|
||||
checks.append({"name": "designDeclares", "ok": True,
|
||||
"detail": f"设计稿声明系统:{declared_hits}"})
|
||||
else:
|
||||
# 弱信号:设计 prose 措辞多样,未命中关键词不代表设计错,只记观测、不进 mismatch、不影响 passed。
|
||||
signals.append("(弱信号)设计稿未明确点到三系统关键词,以源工程结构为准。")
|
||||
checks.append({"name": "designDeclares", "ok": True,
|
||||
"detail": "设计 prose 弱信号(不参与 passed);源工程结构为准"})
|
||||
|
||||
# ── 汇总:passed = 硬结构判据 ①~④ 全 ok(⑤ 弱信号不参与 passed)──
|
||||
hard_ok = sys_ok and dag_ok and reach_orders_ok and econ_ok
|
||||
return {"passed": hard_ok, "signals": signals, "mismatches": mismatches, "checks": checks}
|
||||
|
||||
|
||||
def _is_dag_acyclic(edges: list) -> bool:
|
||||
"""Kahn 拓扑判环:edges=[(from,to)] 无环返 True(L2/三联动门 dagAcyclic 共用)。"""
|
||||
from collections import defaultdict, deque
|
||||
adj: dict = defaultdict(list)
|
||||
indeg: dict = defaultdict(int)
|
||||
nodes = set()
|
||||
for a, b in edges:
|
||||
adj[a].append(b)
|
||||
indeg[b] += 1
|
||||
nodes.update((a, b))
|
||||
q = deque(n for n in nodes if indeg[n] == 0)
|
||||
seen = 0
|
||||
while q:
|
||||
n = q.popleft()
|
||||
seen += 1
|
||||
for m in adj[n]:
|
||||
indeg[m] -= 1
|
||||
if indeg[m] == 0:
|
||||
q.append(m)
|
||||
return seen == len(nodes)
|
||||
|
||||
|
||||
# ── L3 升门准备:确定性「恶性渲染失败」判据(吃单元 A 的 render-sanity 结果)──────────
|
||||
# 设计依据:tier2细节图说-D §图 D1(L3 只评分绝不当门)+ verdict.schema humanPlayability.renderSanity。
|
||||
# L3 主体是 M3 视觉软检(observe-only score,studio.py 收口接);本函数补一个**确定性**的恶性判据:
|
||||
# 单元 A 的 render-sanity 探针(harness evalRenderSanity)已在真玩时扫渲染文本里的漏底标记
|
||||
# ([object Object]/undefined/NaN)。这种漏底是「确定性可判的恶性渲染失败」——不是主观好不好玩,
|
||||
# 而是客观「画面糊了实现细节」。本函数把它从 humanPlayability 整理进 L3.notes,并标 egregious=true,
|
||||
# **建议主会话校准后提 fatal**(精确判据见 promotionPlan)。非恶性仍软,L3.score 仍 observe-only 不改 decision。
|
||||
def derive_l3_egregious(verdict: dict | None) -> dict:
|
||||
"""从 harness verdict 的 humanPlayability.renderSanity + playReport 提「恶性渲染失败」确定性判据。
|
||||
|
||||
【为什么吃 humanPlayability 而非自己重扫】render-sanity 的文本来源是真浏览器页面(DOM textContent +
|
||||
host 暴露的 __renderedText),只有 CDP harness 在 mini-desktop 真玩时取得到;6c6g 无 chrome 取不到。
|
||||
故本函数不重做扫描,只**消费** harness 已落进 verdict.humanPlayability.renderSanity 的结果
|
||||
(单元 A 产出),把恶性信号整理成 L3 升门准备的结构。
|
||||
|
||||
【恶性 egregious 定义(确定性,非主观)】满足任一:
|
||||
① renderSanity.pass===false —— 渲染文本含漏底实现细节([object Object]/undefined/NaN);
|
||||
② 空棋盘类:playReport.loaded 为真但 humanPlayability.renderReflectsState.pass===false
|
||||
(state 有 item 的格子在画面上与空格无法区分 —— feie-005 即此症)。
|
||||
这两类都是「机器能确定性判定的恶性渲染失败」,与 L3 主观打分(好不好玩)分开。
|
||||
|
||||
【本阶段纪律 · advisory】本函数**绝不改 decision**:只返回 {egregious, reasons, notes, suggestFatal}。
|
||||
egregious=true 时 suggestFatal=true(信号:建议主会话校准后把这条提成 fatal),但**真提 fatal 由主会话
|
||||
在 mini-desktop 实测金标仍过、坏样本被挂后做**——本函数永远只写进 L3.notes,decision 不动。
|
||||
|
||||
Returns:
|
||||
{egregious: bool, reasons: [人读], notes: [给 L3.notes 的人读条目], suggestFatal: bool}。
|
||||
"""
|
||||
if not isinstance(verdict, dict):
|
||||
return {"egregious": False, "reasons": [], "notes": [], "suggestFatal": False}
|
||||
hp = verdict.get("humanPlayability") or {}
|
||||
pr = verdict.get("playReport") or {}
|
||||
reasons: list[str] = []
|
||||
notes: list[str] = []
|
||||
|
||||
# ① 漏底文本(render-sanity):pass===false 即恶性。
|
||||
rs = hp.get("renderSanity")
|
||||
if isinstance(rs, dict) and rs.get("pass") is False:
|
||||
markers = [v.get("marker") for v in (rs.get("violations") or []) if isinstance(v, dict)]
|
||||
reasons.append("渲染文本含漏底实现细节(render-sanity):" + ",".join(m for m in markers if m))
|
||||
notes.append("L3 恶性渲染(确定性):画面出现漏底标记 "
|
||||
+ ",".join(m for m in markers if m)
|
||||
+ "([object Object]/undefined/NaN 类,非主观好不好玩)。")
|
||||
|
||||
# ② 空棋盘类(render-reflects-state):loaded 但有 item 的格子渲染成空。
|
||||
rrs = hp.get("renderReflectsState")
|
||||
if isinstance(rrs, dict) and rrs.get("pass") is False and pr.get("loaded"):
|
||||
empties = rrs.get("emptyButShouldHaveItem") or []
|
||||
sample = ",".join(f"({c.get('i')},{c.get('j')}):{c.get('item')}"
|
||||
for c in empties[:5] if isinstance(c, dict))
|
||||
reasons.append(f"有 item 的格子渲染成空棋盘(render-reflects-state):{sample}")
|
||||
notes.append("L3 恶性渲染(确定性):state 有原料但格子画成空 "
|
||||
+ f"({sample}),玩家看不到该看到的内容(feie-005 同症)。")
|
||||
|
||||
egregious = bool(reasons)
|
||||
if egregious:
|
||||
notes.append("【L3 升门准备】以上为确定性恶性渲染失败,egregious=true;"
|
||||
"建议主会话在 mini-desktop 实测金标仍 ACCEPT、坏样本被挂后,把『egregious=true 即 fatal』提成门"
|
||||
"(精确判据见 plan/promotionPlan)。当前仍 advisory,不改 decision。")
|
||||
return {"egregious": egregious, "reasons": reasons, "notes": notes, "suggestFatal": egregious}
|
||||
|
||||
|
||||
def run_gates(game_id: str, play_spec: dict | None = None, *,
|
||||
port: int = 4330, cdp_port: int = 9322, timeout_s: int = 300) -> dict:
|
||||
"""真浏览器真玩 L1 九门 + 富游戏三门,返回 tier2-verdict 形状(judge 纯代码,零 LLM)。
|
||||
|
||||
@ -65,6 +65,18 @@ function loadBrowserDeps() {
|
||||
/** 纯 Promise 延时(不走被拦的 sleep 命令)。 */
|
||||
const delay = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
/* ════════════════════════════════════════════════════════════════════════════
|
||||
* 【人可玩 advisory 阈值常量(单元 A)】
|
||||
* ★ DIRECTIONAL —— 候选值,待主会话用金标 fixture(mini-fei-e)实测校准后再定,并据此决定是否提 fatal。
|
||||
* 背景:九门用 driver 按布局坐标点(不看屏幕)且点得飞快,抓不到「反应窗口只为机器人调、人来不及」这类缺陷
|
||||
* (feie-005 即此症:订单只有 8 秒耐心)。本常量给「人类最小反应窗口」一个候选下限。
|
||||
* 金标实测值(datatable.gold.json):5 个订单 patience = 20/19/18/17/16 秒 → ms = 20000/19000/18000/17000/16000,
|
||||
* min=16000ms。注:金标这组 patience 已是「为有界 driver run 砍小后」的值(原 440~600s),仍 ≥ 16s。
|
||||
* 候选阈值取 12000ms(12s):低于金标 min(16s)留一档余量、让金标全过;feie-005 的 8s(=8000ms)< 12000ms 被挂。
|
||||
* 【纪律】本阈值仅产 advisory violation,绝不参与 decision。提 fatal/调阈值由主会话实测金标后做(见文件末 promotionPlan)。
|
||||
* ════════════════════════════════════════════════════════════════════════════ */
|
||||
const HUMAN_MIN_PATIENCE_MS = 12000;
|
||||
|
||||
/* ════════════════════════════════════════════════════════════════════════════
|
||||
* 【PHASER_PROBE —— per-engine 探针钩子抽象层(A6 EngineProbeHooks 的 Phaser 实现)】
|
||||
* 把四条钩子集中在这里,换引擎只重写这一段(不散落全文件);directional v1,spike 实证后收窄。
|
||||
@ -394,6 +406,182 @@ async function evalLatchGate(cdp, opts) {
|
||||
return { passed: checks.every((c) => c.ok), checks, terminalPhase: reachedTerminal ? phaseNow : null };
|
||||
}
|
||||
|
||||
/* ──────────────────────────────────────────────────────────────────────────
|
||||
* 【人可玩 advisory 检查(单元 A · 修「gate 通过≠人能玩」盲区)】
|
||||
* 三项全部 advisory:只写进 verdict.humanPlayability,绝不参与 decision/runnableOk/L1.passed。
|
||||
* 坐标系前提:host 用 Phaser Scale.NONE + width/height=逻辑视口(390×844)、resolution 默认 1,
|
||||
* 故 canvas backing-store 像素 = 逻辑像素(无 DPR 乘子;与 mapToClient 注释、boot-phaser-host 配置一致)。
|
||||
* 因此 state.board 的格中心逻辑坐标 (x,y) 直接就是 captureImageData 缓冲里的像素坐标 (x,y)。
|
||||
* ────────────────────────────────────────────────────────────────────────── */
|
||||
|
||||
/** 从 RGBA 缓冲取一个矩形 patch 的统计(均值 RGB + 各通道方差和 + 有效采样数)。
|
||||
* 越界/全透明像素跳过(透明=alpha 0)。patch 用「格内中心区」避开格边框 stroke,降低误差。
|
||||
* @param {Uint8Array} data RGBA 字节缓冲
|
||||
* @param {number} W 缓冲宽(像素)
|
||||
* @param {number} H 缓冲高(像素)
|
||||
* @param {number} cx patch 中心 x(像素=逻辑)
|
||||
* @param {number} cy patch 中心 y
|
||||
* @param {number} half patch 半边长(像素)
|
||||
* @returns {{ mean:[number,number,number], variance:number, samples:number }}
|
||||
*/
|
||||
function patchStats(data, W, H, cx, cy, half) {
|
||||
let sr = 0, sg = 0, sb = 0, srr = 0, sgg = 0, sbb = 0, n = 0;
|
||||
const x0 = Math.max(0, Math.round(cx - half)), x1 = Math.min(W - 1, Math.round(cx + half));
|
||||
const y0 = Math.max(0, Math.round(cy - half)), y1 = Math.min(H - 1, Math.round(cy + half));
|
||||
for (let yy = y0; yy <= y1; yy++) {
|
||||
for (let xx = x0; xx <= x1; xx++) {
|
||||
const i = (yy * W + xx) * 4;
|
||||
if (data[i + 3] === 0) continue; // 透明像素不计(Phaser 不透明黑底,正常无透明)
|
||||
const r = data[i], g = data[i + 1], b = data[i + 2];
|
||||
sr += r; sg += g; sb += b; srr += r * r; sgg += g * g; sbb += b * b; n++;
|
||||
}
|
||||
}
|
||||
if (n === 0) return { mean: [0, 0, 0], variance: 0, samples: 0 };
|
||||
const mr = sr / n, mg = sg / n, mb = sb / n;
|
||||
// 三通道方差之和(衡量 patch 内部「花不花」——精灵/文本会让方差升高,纯色占位方差近 0)。
|
||||
const variance = Math.max(0, (srr / n - mr * mr)) + Math.max(0, (sgg / n - mg * mg)) + Math.max(0, (sbb / n - mb * mb));
|
||||
return { mean: [mr, mg, mb], variance, samples: n };
|
||||
}
|
||||
|
||||
/** 两 RGB 均值的欧氏色差。 */
|
||||
function colorDist(a, b) {
|
||||
const dr = a[0] - b[0], dg = a[1] - b[1], db = a[2] - b[2];
|
||||
return Math.sqrt(dr * dr + dg * dg + db * db);
|
||||
}
|
||||
|
||||
/**
|
||||
* render-reflects-state(advisory):对 state.board 里每个有 item 的格子,取其 canvas 像素区域,
|
||||
* 与「空格子基线区域」比对;若有 item 的格子区域和空格子无法区分(基本等于背景/空格子)→记 violation。
|
||||
* 判据:item 格相对空格基线,要么色差够大(画了有色占位/精灵)、要么内部方差明显更高(画了精灵/文本/图标);
|
||||
* 两者都不满足 = 渲染没反映「这格有东西」(feie-005 即此症:state 有原料但画空格)。
|
||||
* @param {Uint8Array} data RGBA 缓冲
|
||||
* @param {number} W @param {number} H 缓冲尺寸(像素=逻辑)
|
||||
* @param {object} state 语义 state(含 board=[{i,j,x,y,item}])
|
||||
* @returns {{ pass:boolean, checked:number, emptyButShouldHaveItem:Array, note?:string }}
|
||||
*/
|
||||
function evalRenderReflectsState(data, W, H, state) {
|
||||
const board = (state && Array.isArray(state.board)) ? state.board : null;
|
||||
if (!board || board.length === 0) {
|
||||
return { pass: true, checked: 0, emptyButShouldHaveItem: [], note: 'state 无 board(非格子类游戏或未导出)→ advisory 跳过' };
|
||||
}
|
||||
// patch 半边长:CELL≈110,取格内中心 ~64×64 区(half=32),避开 4px stroke 与格间隙。
|
||||
const HALF = 32;
|
||||
// 收集每格 patch 统计。
|
||||
const itemCells = [], emptyCells = [];
|
||||
for (const c of board) {
|
||||
if (!c || typeof c.x !== 'number' || typeof c.y !== 'number') continue;
|
||||
const ps = patchStats(data, W, H, c.x, c.y, HALF);
|
||||
if (ps.samples === 0) continue; // 该格落在缓冲外/全透明,跳过
|
||||
const rec = { i: c.i, j: c.j, item: c.item, mean: ps.mean, variance: ps.variance };
|
||||
if (c.item) itemCells.push(rec); else emptyCells.push(rec);
|
||||
}
|
||||
if (itemCells.length === 0) {
|
||||
return { pass: true, checked: 0, emptyButShouldHaveItem: [], note: '当前帧无有 item 的格子(棋盘空/已合成清空)→ advisory 跳过' };
|
||||
}
|
||||
// 空格基线:优先用本帧空格 patch 的均值/方差;无空格时回落「整盘最暗 patch」近似空底(经营档空格=底色)。
|
||||
let baseMean, baseVar;
|
||||
if (emptyCells.length > 0) {
|
||||
const n = emptyCells.length;
|
||||
baseMean = [0, 0, 0]; baseVar = 0;
|
||||
for (const e of emptyCells) { baseMean[0] += e.mean[0]; baseMean[1] += e.mean[1]; baseMean[2] += e.mean[2]; baseVar += e.variance; }
|
||||
baseMean = [baseMean[0] / n, baseMean[1] / n, baseMean[2] / n]; baseVar = baseVar / n;
|
||||
} else {
|
||||
// 无空格基线:取所有 item 格里方差最低者当「最接近空」的近似底(保守,宁可漏判不可误挂)。
|
||||
const lowest = itemCells.reduce((a, b) => (b.variance < a.variance ? b : a), itemCells[0]);
|
||||
baseMean = lowest.mean; baseVar = lowest.variance;
|
||||
}
|
||||
// 判据阈值(advisory;directional):色差 > 18(8bit 通道感知可辨)或方差增量 > 120(画了精灵/文本)即「反映了状态」。
|
||||
const COLOR_DIST_MIN = 18, VAR_DELTA_MIN = 120;
|
||||
const emptyButShouldHaveItem = [];
|
||||
for (const c of itemCells) {
|
||||
const dist = colorDist(c.mean, baseMean);
|
||||
const varDelta = c.variance - baseVar;
|
||||
const reflects = dist > COLOR_DIST_MIN || varDelta > VAR_DELTA_MIN;
|
||||
if (!reflects) {
|
||||
emptyButShouldHaveItem.push({ i: c.i, j: c.j, item: c.item, colorDist: Math.round(dist), varDelta: Math.round(varDelta) });
|
||||
}
|
||||
}
|
||||
const pass = emptyButShouldHaveItem.length === 0;
|
||||
return {
|
||||
pass, checked: itemCells.length, emptyButShouldHaveItem,
|
||||
note: `比对 ${itemCells.length} 个有 item 格 vs ${emptyCells.length} 个空格基线(色差>${COLOR_DIST_MIN} 或 方差增量>${VAR_DELTA_MIN} 判为反映状态)`,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* render-sanity / no-[object Object](advisory):扫描渲染出的文本(页面 DOM textContent + host 暴露的 canvas 文本若有)
|
||||
* 是否含 [object Object]/undefined/NaN 这类漏底实现细节 → 记 violation。
|
||||
* 说明:Phaser canvas 文本不在 DOM 里,无法直接 textContent 抓;若 host/游戏把当前渲染文本暴露到
|
||||
* window.__renderedText(string[]) 则一并扫(可选,缺则只扫 DOM)。这是 advisory:抓到漏底=强信号、抓不到≠保证干净。
|
||||
* @param {object} cdp CDP 会话
|
||||
* @returns {Promise<{ pass:boolean, violations:Array, note?:string }>}
|
||||
*/
|
||||
async function evalRenderSanity(cdp) {
|
||||
// 在页面侧收集文本来源:① document.body.innerText(DOM 文本,如 #status、HTML overlay);
|
||||
// ② window.__renderedText(host/游戏可选暴露的「当前 canvas 渲染文本」数组,A6 可测性扩展,缺省 [])。
|
||||
const expr =
|
||||
'(function(){ try {' +
|
||||
' var out = [];' +
|
||||
' try { if (document && document.body) out.push(String(document.body.innerText || "")); } catch (e1) {}' +
|
||||
' try { var rt = window.__renderedText; if (Array.isArray(rt)) out = out.concat(rt.map(String)); else if (typeof rt === "string") out.push(rt); } catch (e2) {}' +
|
||||
' return out.join("\\n").slice(0, 20000);' +
|
||||
'} catch (e) { return ""; } })()';
|
||||
let text = '';
|
||||
try { text = String(await cdp.evaluate(expr) || ''); } catch (e) { return { pass: true, violations: [], note: '取页面文本异常(advisory 跳过):' + String((e && e.message) || e) }; }
|
||||
// 漏底标记词表(用边界匹配避免误伤正常词,如把数据里的 "undefined" 文案?——这里只抓裸标记,故配词边界)。
|
||||
const markers = [
|
||||
{ marker: '[object Object]', re: /\[object Object\]/g },
|
||||
{ marker: 'undefined', re: /(^|[^A-Za-z0-9_])undefined([^A-Za-z0-9_]|$)/g },
|
||||
{ marker: 'NaN', re: /(^|[^A-Za-z0-9_])NaN([^A-Za-z0-9_]|$)/g },
|
||||
];
|
||||
const violations = [];
|
||||
for (const m of markers) {
|
||||
const idx = text.search(m.re);
|
||||
if (idx >= 0) {
|
||||
// 截一段上下文(去掉换行,便于落证)。
|
||||
const sample = text.slice(Math.max(0, idx - 20), idx + 40).replace(/\s+/g, ' ').trim();
|
||||
violations.push({ marker: m.marker, sample });
|
||||
}
|
||||
}
|
||||
return { pass: violations.length === 0, violations, note: `扫描渲染文本 ${text.length} 字符(DOM innerText + window.__renderedText 若有)` };
|
||||
}
|
||||
|
||||
/**
|
||||
* human-timing(advisory):读数据表(经 state.orders 或 dataTable.orders)订单 patience(及反应窗口),
|
||||
* 记录 min/max/各单 patience;低于候选人类最小阈值 HUMAN_MIN_PATIENCE_MS 的订单记 advisory violation。
|
||||
* 单位归一:数据表 patience 单位=秒(game-core remainMs=patience*1000),故 *1000 转 ms 比对。
|
||||
* @param {object|null} state 语义 state(state.orders[].patience 单位秒)
|
||||
* @param {object|null} dataTable 数据表(datatable.gold.json;dataTable.orders[].patience 单位秒)
|
||||
* @returns {{ pass:boolean, minPatienceMs:number|null, maxPatienceMs:number|null, thresholdMs:number, patiencesMs:Array, violations:Array, note?:string }}
|
||||
*/
|
||||
function evalHumanTiming(state, dataTable) {
|
||||
// patience 来源优先级:数据表(权威·静态全量) > state.orders(运行时,可能已倒计时/只在场子集)。
|
||||
// 取「原始 patience」而非 remainMs(remainMs 会随时间衰减,不能当反应窗口设计值)。
|
||||
let orders = null, src = '';
|
||||
if (dataTable && Array.isArray(dataTable.orders) && dataTable.orders.length) { orders = dataTable.orders; src = 'datatable'; }
|
||||
else if (state && Array.isArray(state.orders) && state.orders.length) { orders = state.orders; src = 'state.orders'; }
|
||||
if (!orders) {
|
||||
return { pass: true, minPatienceMs: null, maxPatienceMs: null, thresholdMs: HUMAN_MIN_PATIENCE_MS, patiencesMs: [], violations: [], note: '无订单 patience 可读(非订单类游戏或未导出)→ advisory 跳过' };
|
||||
}
|
||||
const patiencesMs = [];
|
||||
for (const o of orders) {
|
||||
if (!o || typeof o.patience !== 'number') continue;
|
||||
const ms = o.patience * 1000; // 秒 → 毫秒
|
||||
patiencesMs.push({ orderId: (o.id != null ? String(o.id) : (o.instId != null ? String(o.instId) : '(无id)')), patienceMs: ms, belowThreshold: ms < HUMAN_MIN_PATIENCE_MS });
|
||||
}
|
||||
if (patiencesMs.length === 0) {
|
||||
return { pass: true, minPatienceMs: null, maxPatienceMs: null, thresholdMs: HUMAN_MIN_PATIENCE_MS, patiencesMs: [], violations: [], note: `订单无 patience 字段(来源=${src})→ advisory 跳过` };
|
||||
}
|
||||
const vals = patiencesMs.map((p) => p.patienceMs);
|
||||
const minMs = Math.min(...vals), maxMs = Math.max(...vals);
|
||||
const violations = patiencesMs.filter((p) => p.belowThreshold).map((p) => ({ orderId: p.orderId, patienceMs: p.patienceMs }));
|
||||
return {
|
||||
pass: violations.length === 0, minPatienceMs: minMs, maxPatienceMs: maxMs, thresholdMs: HUMAN_MIN_PATIENCE_MS,
|
||||
patiencesMs, violations,
|
||||
note: `patience 来源=${src};min=${minMs}ms max=${maxMs}ms 阈值=${HUMAN_MIN_PATIENCE_MS}ms(★ directional,待金标校准)`,
|
||||
};
|
||||
}
|
||||
|
||||
/** 组装 tier2-verdict.schema.json 形状(A5)。judge 纯代码产出,零 LLM、零 agent 自评。
|
||||
* @param ctx 收集到的全部判定材料 */
|
||||
function assembleVerdict(ctx) {
|
||||
@ -479,6 +667,18 @@ function assembleVerdict(ctx) {
|
||||
verdict.playReport = playReport;
|
||||
}
|
||||
if (ctx.breakerKind) verdict.breakerKind = ctx.breakerKind;
|
||||
|
||||
// ── 人可玩 advisory 段(单元 A):纯附加,绝不参与上面已算定的 decision/reasons/runnableOk/L1.passed ──
|
||||
// decision 在本函数上方早已算完;此处只把 advisory 观测结果挂上去供人审,一个字节不回改 decision。
|
||||
if (ctx.humanPlayability) {
|
||||
const hp = ctx.humanPlayability;
|
||||
verdict.humanPlayability = {
|
||||
advisory: true, // 不变量:本段全 advisory,只报告不阻塞(防 Goodhart)
|
||||
renderReflectsState: hp.renderReflectsState || null,
|
||||
renderSanity: hp.renderSanity || null,
|
||||
humanTiming: hp.humanTiming || null,
|
||||
};
|
||||
}
|
||||
return verdict;
|
||||
}
|
||||
|
||||
@ -502,6 +702,25 @@ async function main() {
|
||||
let spec = { driver: null, assertAfterPlay: [], expectLatch: false, richGame: {}, expectedEngineCallPrefixes: [] };
|
||||
try { spec = Object.assign(spec, JSON.parse(fs.readFileSync(specPath, 'utf8'))); } catch (_) { /* 用默认 */ }
|
||||
|
||||
// 人可玩 human-timing 用的数据表(patience 权威·静态全量)。来源优先级:
|
||||
// ① spec.dataTablePath 显式指定 → ② spec 所在目录下 data/datatable.gold.json(真跑约定:--spec 指向 <GAME_DIR>/play-spec.json,
|
||||
// 故数据表在 <GAME_DIR>/data/;本机 fixture 同构)→ ③ gameDir 兜底(本机直跑、无 --spec 时)→ ④ spec.richGame(含 orders.patience)。
|
||||
// 注:真跑 gameId='_tier2-gen/<id>',gameDir=path.resolve('games',gameId) 不指向真游戏目录(真目录在 game-runtime/games/);
|
||||
// 故以 --spec 所在目录(specDir)为准取数据表,才在真跑/本机两环境都对。读不到不报错(advisory):evalHumanTiming 回落 state.orders。
|
||||
const specDir = path.dirname(specPath);
|
||||
let dataTable = null;
|
||||
const dtCandidates = [
|
||||
spec.dataTablePath ? path.resolve(spec.dataTablePath) : null,
|
||||
path.join(specDir, 'data', 'datatable.gold.json'),
|
||||
path.join(specDir, 'data', 'datatable.json'),
|
||||
path.join(gameDir, 'data', 'datatable.gold.json'),
|
||||
].filter(Boolean);
|
||||
for (const p of dtCandidates) {
|
||||
try { dataTable = JSON.parse(fs.readFileSync(p, 'utf8')); break; } catch (_) { /* 试下一个 */ }
|
||||
}
|
||||
// 兜底:spec.richGame 自带 orders.patience(play-spec.example 即如此;worker 也会从数据表 derive 进 spec.richGame)。
|
||||
if (!dataTable && spec.richGame && Array.isArray(spec.richGame.orders)) dataTable = spec.richGame;
|
||||
|
||||
const url = `${base}/games/${gameId}/index.html`;
|
||||
const round = spec.round != null ? spec.round : 0;
|
||||
// driven 感知:有 driver 或非空 inputs → driven=true,E_live/H_progress 恢复致命。
|
||||
@ -514,6 +733,8 @@ async function main() {
|
||||
let totalDriven = 0;
|
||||
let forceKill = false, breakerKind = null;
|
||||
const l3Notes = [];
|
||||
// 人可玩 advisory 段累加器(单元 A;三项全 advisory,不参与 decision)。
|
||||
const humanPlayability = { renderReflectsState: null, renderSanity: null, humanTiming: null };
|
||||
|
||||
let cdp = null;
|
||||
try {
|
||||
@ -527,6 +748,16 @@ async function main() {
|
||||
const px0 = await brightPixels(cdp);
|
||||
const state0 = await readGameState(cdp);
|
||||
|
||||
// ── 人可玩 advisory ① render-reflects-state(首帧取证)──
|
||||
// 首帧棋盘已由 boot 的 refillFromInventory 铺满料(state0.board 多数格有 item),且 feie-005 的「画空格」
|
||||
// 缺陷首帧即可见 → 此刻抓 canvas 像素与 state0.board 比对最稳、最早(driver 还没扰动棋盘)。全 advisory。
|
||||
try {
|
||||
const img0 = await captureImageData(cdp, PHASER_PROBE.canvasSelector);
|
||||
humanPlayability.renderReflectsState = evalRenderReflectsState(img0.data, img0.width, img0.height, state0);
|
||||
} catch (eRR) {
|
||||
humanPlayability.renderReflectsState = { pass: true, checked: 0, emptyButShouldHaveItem: [], note: '取 canvas 像素异常(advisory 跳过):' + String((eRR && eRR.message) || eRR) };
|
||||
}
|
||||
|
||||
// 守卫C:Phaser 掌帧增量(@60fps 理论 ~30/500ms;区间 [20,45])。
|
||||
const fd = await frameDelta(cdp, 500);
|
||||
guards.C_frame = { pass: fd.delta != null && fd.delta >= 20 && fd.delta <= 45, detail: `Δframe=${fd.delta}`, ...fd };
|
||||
@ -671,6 +902,11 @@ async function main() {
|
||||
richGameGates.latch = { passed: lg.passed, checks: lg.checks };
|
||||
}
|
||||
|
||||
// ── 人可玩 advisory ② render-sanity / ③ human-timing(真玩末尾·session 仍活时取)──
|
||||
// ② 扫渲染文本漏底([object Object]/undefined/NaN);③ 读数据表 patience 与候选人类阈值比对。全 advisory。
|
||||
humanPlayability.renderSanity = await evalRenderSanity(cdp);
|
||||
humanPlayability.humanTiming = evalHumanTiming(state1, dataTable);
|
||||
|
||||
// 玩家取证报告(playReport,编排器对 CDP 捕获数据计算;非 agent 自评)。
|
||||
const sFinal = await readGameState(cdp);
|
||||
playReport = {
|
||||
@ -699,6 +935,7 @@ async function main() {
|
||||
structureOk, guards, richGameGates, playReport, drivenInputs: totalDriven,
|
||||
driven, findings: spec.findings, l2Signals: spec.l2Signals, l3Notes,
|
||||
forceKill, breakerKind,
|
||||
humanPlayability, // 人可玩 advisory 段(单元 A;assembleVerdict 纯附加,不碰 decision)
|
||||
});
|
||||
|
||||
fs.writeFileSync(path.join(evDir, 'verdict.json'), JSON.stringify(verdict, null, 2));
|
||||
@ -713,8 +950,33 @@ async function main() {
|
||||
console.log(` ${gate.passed ? '✅' : '❌'} 富[${name}] ${(gate.checks || []).map((c) => `${c.name}${c.ok ? '✓' : '✗'}`).join(' ')}`);
|
||||
};
|
||||
rgLine('三联动', rg.tripleLink); rgLine('经济', rg.economy); rgLine('latch', rg.latch);
|
||||
// 人可玩 advisory 摘要(不影响 decision/退出码;➖adv 标识=只报告)。
|
||||
const hp = verdict.humanPlayability;
|
||||
if (hp) {
|
||||
const hpLine = (name, r) => {
|
||||
if (!r) { console.log(` ◻ 人玩[${name}] 未执行`); return; }
|
||||
console.log(` ${r.pass ? '✅' : '➖adv✗'} 人玩[${name}] ${r.note || ''}`);
|
||||
};
|
||||
hpLine('渲染反映状态', hp.renderReflectsState);
|
||||
hpLine('渲染漏底', hp.renderSanity);
|
||||
hpLine('人类反应窗口', hp.humanTiming);
|
||||
// 把违规项摘要打出来(advisory,只为人审)。
|
||||
const rrs = hp.renderReflectsState;
|
||||
if (rrs && Array.isArray(rrs.emptyButShouldHaveItem) && rrs.emptyButShouldHaveItem.length) {
|
||||
console.log(` ↳ 有 item 却画空的格:${rrs.emptyButShouldHaveItem.map((e) => `(${e.i},${e.j})=${e.item}`).join(' ')}`);
|
||||
}
|
||||
const rsn = hp.renderSanity;
|
||||
if (rsn && Array.isArray(rsn.violations) && rsn.violations.length) {
|
||||
console.log(` ↳ 漏底文本:${rsn.violations.map((v) => v.marker).join(' ')}`);
|
||||
}
|
||||
const ht = hp.humanTiming;
|
||||
if (ht && Array.isArray(ht.violations) && ht.violations.length) {
|
||||
console.log(` ↳ 反应窗口过短(<${ht.thresholdMs}ms):${ht.violations.map((v) => `${v.orderId}=${v.patienceMs}ms`).join(' ')}`);
|
||||
}
|
||||
}
|
||||
console.log(` reasons=${JSON.stringify(verdict.reasons)} round=${verdict.round} driven=${driven}`);
|
||||
console.log(` evidence → ${path.relative(process.cwd(), evDir)}/`);
|
||||
// 退出码只看 decision(人可玩段 advisory,绝不改退出码)。
|
||||
process.exit(verdict.decision === 'accept' ? 0 : 1);
|
||||
}
|
||||
|
||||
@ -725,6 +987,9 @@ if (require.main === module) {
|
||||
|
||||
module.exports = {
|
||||
PHASER_PROBE,
|
||||
HUMAN_MIN_PATIENCE_MS,
|
||||
checkAssertion, getPath, stateProgressed,
|
||||
evalTripleLinkGate, evalEconomyGate, evalLatchGate, assembleVerdict,
|
||||
// 人可玩 advisory 检查(单元 A;纯函数部分可本机单测)。
|
||||
patchStats, colorDist, evalRenderReflectsState, evalRenderSanity, evalHumanTiming,
|
||||
};
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user