"""RoboGate VLA Leaderboard — HuggingFace Space. Industrial Pick & Place benchmark for Vision-Language-Action models. 68 adversarial scenarios · NVIDIA Isaac Sim 5.1 · Franka Panda 7-DOF. """ import json from pathlib import Path import gradio as gr import pandas as pd import plotly.graph_objects as go # ── Load data ── DATA_PATH = Path(__file__).parent / "data" / "leaderboard.json" with open(DATA_PATH) as f: DATA = json.load(f) MODELS = DATA["models"] # ── Dataframe ── df = pd.DataFrame( [ { "Rank": i + 1, "Model": m["name"], "Type": m["type"].upper(), "Params": m["params"], "Success Rate": f"{m['sr']:.1f}%", "Confidence": f"{m['confidence']}/100", "Collisions": m["collision"], "Grasp Miss": m["grasp_miss"], "Notes": m["notes"], } for i, m in enumerate( sorted(MODELS, key=lambda x: (-x["sr"], -x["confidence"])) ) ] ) # ── Charts ── def make_sr_chart() -> go.Figure: """Success rate bar chart.""" names = [m["name"] for m in MODELS] srs = [m["sr"] for m in MODELS] colors = ["#76B900" if s > 0 else "#ef4444" for s in srs] # Baseline green colors[0] = "#22c55e" fig = go.Figure( go.Bar( x=names, y=srs, marker_color=colors, text=[f"{s:.0f}%" for s in srs], textposition="outside", textfont=dict(size=14, color="white"), ) ) fig.update_layout( title=dict(text="Success Rate by Model", font=dict(size=16, color="white")), paper_bgcolor="#0B0B1A", plot_bgcolor="#13132B", font=dict(color="#7D8590"), yaxis=dict( title="Success Rate (%)", range=[0, 110], gridcolor="rgba(255,255,255,0.05)", ), xaxis=dict(title=""), margin=dict(t=50, b=80, l=50, r=20), height=400, ) return fig def make_confidence_chart() -> go.Figure: """Confidence score comparison.""" names = [m["name"] for m in MODELS] confs = [m["confidence"] for m in MODELS] colors = [ "#22c55e" if c >= 70 else "#eab308" if c >= 30 else "#ef4444" for c in confs ] fig = go.Figure( go.Bar( x=names, y=confs, marker_color=colors, text=[str(c) for c in confs], textposition="outside", textfont=dict(size=14, color="white"), ) ) fig.update_layout( title=dict( text="Confidence Score (0-100)", font=dict(size=16, color="white") ), paper_bgcolor="#0B0B1A", plot_bgcolor="#13132B", font=dict(color="#7D8590"), yaxis=dict( title="Confidence", range=[0, 110], gridcolor="rgba(255,255,255,0.05)", ), xaxis=dict(title=""), margin=dict(t=50, b=80, l=50, r=20), height=400, ) return fig def make_failure_chart() -> go.Figure: """Stacked failure mode chart (VLA models only).""" vla = [m for m in MODELS if m["type"] == "vla"] names = [m["name"] for m in vla] gm = [m["grasp_miss"] for m in vla] col = [m["collision"] for m in vla] fig = go.Figure( [ go.Bar( name="Grasp Miss", x=names, y=gm, marker_color="#FF8C00", text=gm, textposition="inside", textfont=dict(color="white"), ), go.Bar( name="Collision", x=names, y=col, marker_color="#ef4444", text=col, textposition="inside", textfont=dict(color="white"), ), ] ) fig.update_layout( barmode="stack", title=dict( text="Failure Modes (out of 68 scenarios)", font=dict(size=16, color="white"), ), paper_bgcolor="#0B0B1A", plot_bgcolor="#13132B", font=dict(color="#7D8590"), yaxis=dict( title="Episodes", range=[0, 75], gridcolor="rgba(255,255,255,0.05)", ), xaxis=dict(title=""), legend=dict( orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1 ), margin=dict(t=60, b=80, l=50, r=20), height=400, ) return fig # ── Gradio App ── HEADER_MD = """ # 🤖 RoboGate VLA Leaderboard **Industrial Pick & Place Benchmark** · 68 Adversarial Scenarios · NVIDIA Isaac Sim 5.1 · Franka Panda 7-DOF > **Key finding:** All four VLA models — including NVIDIA's GR00T N1.6 (3B) — achieve **0% success rate** vs. 100% scripted baseline. > 50,000+ experiments across 4 robots. The bottleneck is not model capacity — it's **training-deployment distribution mismatch**. """ ABOUT_MD = """ ## About This Benchmark RoboGate evaluates robot manipulation policies on **68 adversarial industrial scenarios** in NVIDIA Isaac Sim: | Category | Scenarios | Description | |----------|-----------|-------------| | Nominal | 20 | Standard pick-and-place conditions | | Edge Cases | 15 | Low friction, small objects, heavy mass | | Adversarial | 10 | Cluttered workspace, extreme parameters | | Domain Rand | 23 | Randomized physics (friction, mass, obstacles) | **Pipeline:** Isaac Sim (Python 3.11) ↔ ZMQ socket ↔ VLA inference (Python 3.10) **Camera:** mss screen capture 256×256 RGB at 20Hz **Action:** 7-DOF delta end-effector pose → IK solver → joint targets """ SUBMIT_MD = """ ## 📬 Submit Your Model Want to add your VLA model to this leaderboard? 1. Fork [github.com/liveplex-cpu/robogate](https://github.com/liveplex-cpu/robogate) 2. Implement your VLA client following `scripts/vla_octo_client.py` 3. Run against our 68-scenario suite 4. Submit results via [robogate.io/vla](https://robogate.io/vla) or open a GitHub issue **Requirements:** - Must run on the same 68 scenarios (no cherry-picking) - Must use the provided Isaac Sim environment (no modifications) - Must report all metrics: SR, Confidence, failure breakdown """ CITATION_MD = """ ## 📄 Citation ```bibtex @misc{agentai2026robogate, title = {ROBOGATE: Adaptive Failure Discovery for Safe Robot Policy Deployment via Two-Stage Boundary-Focused Sampling}, author = {{AgentAI Co., Ltd.}}, year = {2026}, doi = {10.5281/zenodo.19166967}, url = {https://robogate.io/paper} } ``` **Links:** [Paper](https://robogate.io/paper) · [DOI: 10.5281/zenodo.19166967](https://doi.org/10.5281/zenodo.19166967) · [Dataset (50K+)](https://huggingface.co/datasets/liveplex/robogate-failure-dictionary) · [GitHub](https://github.com/liveplex-cpu/robogate) · [Interactive Explorer](https://robogate.io/failures) """ CSS = """ .gradio-container { background-color: #0B0B1A !important; } .gr-box { background-color: #13132B !important; border-color: #1e1e3a !important; } footer { display: none !important; } """ with gr.Blocks(css=CSS, title="RoboGate VLA Leaderboard") as demo: gr.Markdown(HEADER_MD) with gr.Tabs(): with gr.TabItem("🏆 Leaderboard"): gr.Dataframe( value=df, headers=list(df.columns), interactive=False, wrap=True, ) with gr.TabItem("📊 Charts"): with gr.Row(): gr.Plot(make_sr_chart()) gr.Plot(make_confidence_chart()) gr.Plot(make_failure_chart()) with gr.TabItem("ℹ️ About"): gr.Markdown(ABOUT_MD) with gr.TabItem("📬 Submit"): gr.Markdown(SUBMIT_MD) gr.Markdown(CITATION_MD) if __name__ == "__main__": demo.launch(server_name="0.0.0.0")