Download data.js from RLE-Bench/blog: direct link, hf CLI and curl.
- Browser
- Download file 10.9 kB
-
https://huggingface.co/spaces/RLE-Bench/blog/resolve/main/data.js
- Command line
-
hf download hf://spaces/RLE-Bench/blog/data.js
-
curl -L -o data.js https://huggingface.co/spaces/RLE-Bench/blog/resolve/main/data.js
10.9 kB
| /* --------------------------------------------------------------------------- | |
| * RLE-Bench leaderboard data | |
| * | |
| * Public task descriptions, numbering and evaluation summaries follow blog/. | |
| * Task IDs equal the public task numbers (task01 = T01) and key scores and deep links. | |
| * | |
| * Models and results live in assets/data/leaderboard.json, generated by | |
| * assets/data/export.py from the per-task run dumps next to it. leaderboard.js merges | |
| * that file into BENCH at load time: it supplies `models`, `scores`, `costs` | |
| * and `results`, overrides each task's `splits` / `splitLabel` / `aggregate` | |
| * with the exported split catalog, narrows presentation.snapshotTaskIds to | |
| * the tasks that actually have results, and sets meta.dataStatus from the | |
| * export's `status`. The `splits` below are the fallback used when the JSON | |
| * carries no catalog for a task. | |
| * ------------------------------------------------------------------------- */ | |
| const BENCH = { | |
| // Task ids equal the public task numbers (task01 = T01 Agentic Control, ...). | |
| presentation: { | |
| taskOrder: ["task01", "task02", "task03", "task04", "task05", "task06", "task07", "task08", "task09"], | |
| snapshotTaskIds: ["task01", "task02", "task03", "task04", "task06", "task07", "task08", "task09"], | |
| taskNumbers: {task01: "01", task02: "02", task03: "03", task04: "04", task05: "05", task06: "06", task07: "07", task08: "08", task09: "09"}, | |
| // Two-level breakdown: each workflow groups the public tasks listed under it, in display order. | |
| workflows: [ | |
| { id: "control", name: "Interactive Control", tasks: ["task01", "task02", "task03"] }, | |
| { id: "policy", name: "Policy Development", tasks: ["task04", "task05"] }, | |
| { id: "perception", name: "Perception and Estimation", tasks: ["task06", "task07"] }, | |
| { id: "design", name: "Mechanical Design", tasks: ["task08", "task09"] }, | |
| ], | |
| }, | |
| meta: { | |
| name: "RLE-Bench", | |
| subtitle: "A full-stack robotics engineering benchmark for coding agents", | |
| version: "v1.0-dev", | |
| updated: "2026-09-02", | |
| dataStatus: "placeholder", // overridden by leaderboard.json `status` ("measured") once it loads | |
| // Header links. Fill `arxiv` in when the report is up — until then the | |
| // nav shows the link greyed out rather than pointing nowhere. | |
| github: "https://github.com/RLE-Bench/RLE-Bench", | |
| arxiv: "https://arxiv.org/pdf/2609.34210v2", | |
| }, | |
| /* --- agents under evaluation and their scores --------------------------- | |
| * Loaded at runtime from assets/data/leaderboard.json (see leaderboard.js). | |
| * That file holds `models` (id, name, org, harness, open, baseline), | |
| * `scores[taskId][modelId]` (arrays aligned to each task's `splits`) and | |
| * `costs[taskId][modelId]` ({cost, hours, context_tokens, ...}), each a mean | |
| * over the task's subtask runs. Only evaluated model/harness combinations are | |
| * listed there. Overall cost figures average those per-task means within each | |
| * workflow, then across workflows; Context Length is the mean of input_tokens | |
| * minus cached_tokens per run. | |
| */ | |
| /* --- the task families -------------------------------------------------- */ | |
| tasks: [ | |
| { | |
| id: "task08", | |
| num: "08", | |
| name: "Mobile Base Design", | |
| short: "Mobile Base Design", | |
| tagline: "One mobile base and controller for three robot arms", | |
| description: "The agent designs a common mobile base and controller for three types of arms, Panda, UR5e, and xArm7, subject to physical and resource constraints.", | |
| product: "MJCF model of the mobile base and controller code.", | |
| // The verifier reports weighted stage credit (each checkpoint already the minimum over the three arms). | |
| aggregate: "weighted", | |
| splitLabel: "Scoring Stage", | |
| splits: ["Validity", "Design", "Static Stability", "Dynamic Stability", "Integration"], | |
| weights: [0.15, 0.35, 0.15, 0.20, 0.15], | |
| development: "2 h", | |
| compute: "4 CPUs", | |
| evaluation: "Worst-case performance across the three arms under different shelf targets, payloads, and static and dynamic checks.", | |
| }, | |
| { | |
| id: "task09", | |
| num: "09", | |
| name: "Gravity Compensation for Gello", | |
| short: "Gravity Compensation for Gello", | |
| tagline: "Passive gravity compensation and adaptive control for leader arms", | |
| description: "The agent co-designs passive gravity compensation and adaptive feedforward control for GELLO leader arms, covering three different arm types used in teleoperation.", | |
| product: "Mechanical designs and calibration programs.", | |
| aggregate: "mean", | |
| splitLabel: "Follower Arm", | |
| splits: ["Franka (7 DoF)", "UR5e (6 DoF)", "xArm7 (7 DoF)"], | |
| development: "3 h", | |
| compute: "4 CPUs", | |
| evaluation: "Unseen physical instances, poses, and payloads test holding, backdrivability, torque headroom, and recovery.", | |
| }, | |
| { | |
| id: "task01", | |
| num: "01", | |
| name: "Agentic Control", | |
| short: "Agentic Control", | |
| tagline: "Agent-in-the-loop control under three harness levels", | |
| description: "Agent-in-the-loop control for five kitchen tasks. The agent works under one of three harness levels and must solve each task in as few interaction steps as possible: L1 provides bare action APIs; L2 adds camera calibration and learned models such as SAM 3 and Contact-GraspNet; L3 adds privileged object information.", | |
| product: "Agent context and control experience.", | |
| aggregate: "mean", | |
| splitLabel: "Harness Level", | |
| splits: ["L1 — action API only", "L2 — + harness library", "L3 — + privileged state"], | |
| development: "8 h", | |
| compute: "50,000 steps · 8 CPUs · 1 GPU", | |
| evaluation: "Success rate over five trials of the same task in held-out kitchen scenes.", | |
| }, | |
| { | |
| id: "task02", | |
| num: "02", | |
| name: "Harness Engineering", | |
| short: "Harness Engineering", | |
| tagline: "Reusable perception and control tools for fresh agents", | |
| description: "The agent builds its own harness, including perception tools, controllers, and a manual, so that an independent agent can reuse it to solve a new held-out task zero-shot.", | |
| product: "Perception tools, controllers, and a manual.", | |
| aggregate: "mean", | |
| splitLabel: "Difficulty Band", | |
| splits: ["EASY (5 groups)", "MEDIUM (5 groups)", "HARD (5 groups)"], | |
| development: "8 h", | |
| compute: "75,000 steps · 8 CPUs · 1 GPU", | |
| evaluation: "Success rate of a fresh agent solving a new held-out task with the produced harness, over five trials.", | |
| }, | |
| { | |
| id: "task06", | |
| num: "06", | |
| name: "Pose Estimation", | |
| short: "Pose Estimation", | |
| tagline: "Planar object pose under motion and occlusion", | |
| description: "The agent develops an estimator that identifies asymmetric objects and recovers their planar position and orientation during motion and partial occlusion, across four variants that differ in sensing modality and allowed methods.", | |
| product: "An estimator or trained TorchScript model.", | |
| aggregate: "mean", | |
| splitLabel: "Subtask", | |
| splits: ["rgb-only", "rgb-depth", "model-training", "method-agnostic"], | |
| development: "2 h", | |
| compute: "Subtask-specific", | |
| evaluation: "A hidden battery of 100 static frames and ten push episodes in which the arm crosses the line of sight, scored by translation and rotation error. A wrong shape identification zeroes the group, and inference slower than 10 Hz on CPU is penalized.", | |
| }, | |
| { | |
| id: "task07", | |
| num: "07", | |
| name: "Bin Clearing", | |
| short: "Bin Clearing", | |
| tagline: "Closed-loop manipulation with visual and force feedback", | |
| description: "Agents integrate visual and force feedback into a policy that transfers steel brackets from a cluttered bin to a conveyor.", | |
| product: "A standalone closed-loop policy package.", | |
| // The verifier reward is the task score; the two 0-1 episode-mean components in the dump are shown as splits. | |
| aggregate: "weighted", | |
| splitLabel: "Reward Component", | |
| splits: ["Clearance", "Throughput"], | |
| weights: [1, 1], | |
| development: "4 h", | |
| compute: "4 CPUs", | |
| evaluation: "Eight hidden piles. Clearance, throughput, and penalties for drops, damage, and impacts determine the score.", | |
| }, | |
| { | |
| id: "task03", | |
| num: "03", | |
| name: "Embodied Reasoning", | |
| short: "Embodied Reasoning", | |
| tagline: "Interaction-based reasoning about physical properties", | |
| description: "The agent must interact with the simulator to gather further observations, then reason about the scene to decide how to solve the task or what the answer is. Unlike static visual question answering, these problems cannot be solved from a single observation; the agent has to act in the world.", | |
| product: "Task-specific decision-making or answer.", | |
| aggregate: "mean", | |
| splitLabel: "Scenario", | |
| splits: ["Tower Max Height", "Cantilever Overhang", "Balance Coins", "Rubik Cube", "Hidden Center of Mass"], | |
| development: "Up to 9 h", | |
| compute: "50,000 steps · 8 CPUs · 1 GPU", | |
| evaluation: "Average success rate or performance over five subtasks.", | |
| }, | |
| { | |
| id: "task04", | |
| num: "04", | |
| name: "Whole-Body Motion Tracking", | |
| short: "Whole-Body Motion Tracking", | |
| tagline: "Humanoid policy training and sim-to-sim deployment", | |
| description: "Agents train a humanoid motion-tracking policy to follow reference motions while maintaining stability under various deployment conditions.", | |
| product: "An exported ONNX-format tracking policy.", | |
| aggregate: "mean", | |
| splitLabel: "Motion (20 s Clip)", | |
| splits: ["Dance", "Fight", "Fall & Get Up", "Run", "Sprint"], | |
| development: "4 h", | |
| compute: "8 CPUs · 1 RTX 5090 GPU", | |
| evaluation: "Sim-to-sim transfer from MuJoCo-Warp to MuJoCo-C under hidden perturbations.", | |
| }, | |
| { | |
| id: "task05", | |
| num: "05", | |
| name: "NanoVLA Recipe", | |
| short: "NanoVLA Recipe", | |
| tagline: "Training recipes for vision-language-action models", | |
| description: "Agents develop vision-language-action (VLA) models in LIBERO-10 and RoboTwin 2.0 across four tracks covering open design and robustness.", | |
| product: "One training recipe per subtask.", | |
| development: "4 h", | |
| compute: "1 H100 GPU", | |
| evaluation: "The delivered recipe is used to train a VLA model, which is then evaluated under different circumstances. Each subtask is scored by the success rate of that VLA model.", | |
| aggregate: "mean", | |
| splits: ["LIBERO Open Design", "LIBERO Robustness", "RoboTwin Open Design", "RoboTwin Robustness"], | |
| splitLabel: "Track", | |
| }, | |
| ], | |
| }; | |
| if (typeof module !== "undefined") module.exports = BENCH; | |