Spaces:
Sleeping
Sleeping
| """RoboGate VLA Leaderboard — HuggingFace Space. | |
| Industrial Pick & Place benchmark for Vision-Language-Action models. | |
| 68 adversarial scenarios · NVIDIA Isaac Sim 5.1 · Franka Panda 7-DOF. | |
| """ | |
| import json | |
| from pathlib import Path | |
| import gradio as gr | |
| import pandas as pd | |
| import plotly.graph_objects as go | |
| # ── Load data ── | |
| DATA_PATH = Path(__file__).parent / "data" / "leaderboard.json" | |
| with open(DATA_PATH) as f: | |
| DATA = json.load(f) | |
| MODELS = DATA["models"] | |
| # ── Dataframe ── | |
| df = pd.DataFrame( | |
| [ | |
| { | |
| "Rank": i + 1, | |
| "Model": m["name"], | |
| "Type": m["type"].upper(), | |
| "Params": m["params"], | |
| "Success Rate": f"{m['sr']:.1f}%", | |
| "Confidence": f"{m['confidence']}/100", | |
| "Collisions": m["collision"], | |
| "Grasp Miss": m["grasp_miss"], | |
| "Notes": m["notes"], | |
| } | |
| for i, m in enumerate( | |
| sorted(MODELS, key=lambda x: (-x["sr"], -x["confidence"])) | |
| ) | |
| ] | |
| ) | |
| # ── Charts ── | |
| def make_sr_chart() -> go.Figure: | |
| """Success rate bar chart.""" | |
| names = [m["name"] for m in MODELS] | |
| srs = [m["sr"] for m in MODELS] | |
| colors = ["#76B900" if s > 0 else "#ef4444" for s in srs] | |
| # Baseline green | |
| colors[0] = "#22c55e" | |
| fig = go.Figure( | |
| go.Bar( | |
| x=names, | |
| y=srs, | |
| marker_color=colors, | |
| text=[f"{s:.0f}%" for s in srs], | |
| textposition="outside", | |
| textfont=dict(size=14, color="white"), | |
| ) | |
| ) | |
| fig.update_layout( | |
| title=dict(text="Success Rate by Model", font=dict(size=16, color="white")), | |
| paper_bgcolor="#0B0B1A", | |
| plot_bgcolor="#13132B", | |
| font=dict(color="#7D8590"), | |
| yaxis=dict( | |
| title="Success Rate (%)", | |
| range=[0, 110], | |
| gridcolor="rgba(255,255,255,0.05)", | |
| ), | |
| xaxis=dict(title=""), | |
| margin=dict(t=50, b=80, l=50, r=20), | |
| height=400, | |
| ) | |
| return fig | |
| def make_confidence_chart() -> go.Figure: | |
| """Confidence score comparison.""" | |
| names = [m["name"] for m in MODELS] | |
| confs = [m["confidence"] for m in MODELS] | |
| colors = [ | |
| "#22c55e" if c >= 70 else "#eab308" if c >= 30 else "#ef4444" for c in confs | |
| ] | |
| fig = go.Figure( | |
| go.Bar( | |
| x=names, | |
| y=confs, | |
| marker_color=colors, | |
| text=[str(c) for c in confs], | |
| textposition="outside", | |
| textfont=dict(size=14, color="white"), | |
| ) | |
| ) | |
| fig.update_layout( | |
| title=dict( | |
| text="Confidence Score (0-100)", font=dict(size=16, color="white") | |
| ), | |
| paper_bgcolor="#0B0B1A", | |
| plot_bgcolor="#13132B", | |
| font=dict(color="#7D8590"), | |
| yaxis=dict( | |
| title="Confidence", | |
| range=[0, 110], | |
| gridcolor="rgba(255,255,255,0.05)", | |
| ), | |
| xaxis=dict(title=""), | |
| margin=dict(t=50, b=80, l=50, r=20), | |
| height=400, | |
| ) | |
| return fig | |
| def make_failure_chart() -> go.Figure: | |
| """Stacked failure mode chart (VLA models only).""" | |
| vla = [m for m in MODELS if m["type"] == "vla"] | |
| names = [m["name"] for m in vla] | |
| gm = [m["grasp_miss"] for m in vla] | |
| col = [m["collision"] for m in vla] | |
| fig = go.Figure( | |
| [ | |
| go.Bar( | |
| name="Grasp Miss", | |
| x=names, | |
| y=gm, | |
| marker_color="#FF8C00", | |
| text=gm, | |
| textposition="inside", | |
| textfont=dict(color="white"), | |
| ), | |
| go.Bar( | |
| name="Collision", | |
| x=names, | |
| y=col, | |
| marker_color="#ef4444", | |
| text=col, | |
| textposition="inside", | |
| textfont=dict(color="white"), | |
| ), | |
| ] | |
| ) | |
| fig.update_layout( | |
| barmode="stack", | |
| title=dict( | |
| text="Failure Modes (out of 68 scenarios)", | |
| font=dict(size=16, color="white"), | |
| ), | |
| paper_bgcolor="#0B0B1A", | |
| plot_bgcolor="#13132B", | |
| font=dict(color="#7D8590"), | |
| yaxis=dict( | |
| title="Episodes", | |
| range=[0, 75], | |
| gridcolor="rgba(255,255,255,0.05)", | |
| ), | |
| xaxis=dict(title=""), | |
| legend=dict( | |
| orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1 | |
| ), | |
| margin=dict(t=60, b=80, l=50, r=20), | |
| height=400, | |
| ) | |
| return fig | |
| # ── Gradio App ── | |
| HEADER_MD = """ | |
| # 🤖 RoboGate VLA Leaderboard | |
| **Industrial Pick & Place Benchmark** · 68 Adversarial Scenarios · NVIDIA Isaac Sim 5.1 · Franka Panda 7-DOF | |
| > **Key finding:** All four VLA models — including NVIDIA's GR00T N1.6 (3B) — achieve **0% success rate** vs. 100% scripted baseline. | |
| > 50,000+ experiments across 4 robots. The bottleneck is not model capacity — it's **training-deployment distribution mismatch**. | |
| """ | |
| ABOUT_MD = """ | |
| ## About This Benchmark | |
| RoboGate evaluates robot manipulation policies on **68 adversarial industrial scenarios** in NVIDIA Isaac Sim: | |
| | Category | Scenarios | Description | | |
| |----------|-----------|-------------| | |
| | Nominal | 20 | Standard pick-and-place conditions | | |
| | Edge Cases | 15 | Low friction, small objects, heavy mass | | |
| | Adversarial | 10 | Cluttered workspace, extreme parameters | | |
| | Domain Rand | 23 | Randomized physics (friction, mass, obstacles) | | |
| **Pipeline:** Isaac Sim (Python 3.11) ↔ ZMQ socket ↔ VLA inference (Python 3.10) | |
| **Camera:** mss screen capture 256×256 RGB at 20Hz | |
| **Action:** 7-DOF delta end-effector pose → IK solver → joint targets | |
| """ | |
| SUBMIT_MD = """ | |
| ## 📬 Submit Your Model | |
| Want to add your VLA model to this leaderboard? | |
| 1. Fork [github.com/liveplex-cpu/robogate](https://github.com/liveplex-cpu/robogate) | |
| 2. Implement your VLA client following `scripts/vla_octo_client.py` | |
| 3. Run against our 68-scenario suite | |
| 4. Submit results via [robogate.io/vla](https://robogate.io/vla) or open a GitHub issue | |
| **Requirements:** | |
| - Must run on the same 68 scenarios (no cherry-picking) | |
| - Must use the provided Isaac Sim environment (no modifications) | |
| - Must report all metrics: SR, Confidence, failure breakdown | |
| """ | |
| CITATION_MD = """ | |
| ## 📄 Citation | |
| ```bibtex | |
| @misc{agentai2026robogate, | |
| title = {ROBOGATE: Adaptive Failure Discovery for Safe Robot | |
| Policy Deployment via Two-Stage Boundary-Focused Sampling}, | |
| author = {{AgentAI Co., Ltd.}}, | |
| year = {2026}, | |
| doi = {10.5281/zenodo.19166967}, | |
| url = {https://robogate.io/paper} | |
| } | |
| ``` | |
| **Links:** | |
| [Paper](https://robogate.io/paper) · | |
| [DOI: 10.5281/zenodo.19166967](https://doi.org/10.5281/zenodo.19166967) · | |
| [Dataset (50K+)](https://huggingface.co/datasets/liveplex/robogate-failure-dictionary) · | |
| [GitHub](https://github.com/liveplex-cpu/robogate) · | |
| [Interactive Explorer](https://robogate.io/failures) | |
| """ | |
| CSS = """ | |
| .gradio-container { background-color: #0B0B1A !important; } | |
| .gr-box { background-color: #13132B !important; border-color: #1e1e3a !important; } | |
| footer { display: none !important; } | |
| """ | |
| with gr.Blocks(css=CSS, title="RoboGate VLA Leaderboard") as demo: | |
| gr.Markdown(HEADER_MD) | |
| with gr.Tabs(): | |
| with gr.TabItem("🏆 Leaderboard"): | |
| gr.Dataframe( | |
| value=df, | |
| headers=list(df.columns), | |
| interactive=False, | |
| wrap=True, | |
| ) | |
| with gr.TabItem("📊 Charts"): | |
| with gr.Row(): | |
| gr.Plot(make_sr_chart()) | |
| gr.Plot(make_confidence_chart()) | |
| gr.Plot(make_failure_chart()) | |
| with gr.TabItem("ℹ️ About"): | |
| gr.Markdown(ABOUT_MD) | |
| with gr.TabItem("📬 Submit"): | |
| gr.Markdown(SUBMIT_MD) | |
| gr.Markdown(CITATION_MD) | |
| if __name__ == "__main__": | |
| demo.launch(server_name="0.0.0.0") | |