Buckets:
| <meta charset="utf-8" /><meta name="hf:doc:metadata" content="{"title":"MolmoAct2 Policy","local":"molmoact2-policy","sections":[{"title":"Installation Requirements","local":"installation-requirements","sections":[],"depth":2},{"title":"Usage","local":"usage","sections":[],"depth":2},{"title":"Training","local":"training","sections":[{"title":"Training With Original MolmoAct2 Weight","local":"training-with-original-molmoact2-weight","sections":[],"depth":3},{"title":"Training With LeRobot MolmoAct2 Weight","local":"training-with-lerobot-molmoact2-weight","sections":[],"depth":3},{"title":"Common Practices","local":"common-practices","sections":[],"depth":3},{"title":"Common Policy Options","local":"common-policy-options","sections":[],"depth":3},{"title":"Learning Rates","local":"learning-rates","sections":[],"depth":3},{"title":"Dataset Quantile Statistics","local":"dataset-quantile-statistics","sections":[],"depth":3}],"depth":2},{"title":"Evaluation","local":"evaluation","sections":[{"title":"Evaluation With LeRobot MolmoAct2 Weight","local":"evaluation-with-lerobot-molmoact2-weight","sections":[],"depth":3},{"title":"Evaluation With Original MolmoAct2 Weight","local":"evaluation-with-original-molmoact2-weight","sections":[],"depth":3},{"title":"Common Evaluation Options","local":"common-evaluation-options","sections":[],"depth":3}],"depth":2},{"title":"Performance Results","local":"performance-results","sections":[{"title":"LIBERO Benchmark Results","local":"libero-benchmark-results","sections":[],"depth":3}],"depth":2},{"title":"Hardware Deployment (lerobot-rollout)","local":"hardware-deployment-lerobot-rollout","sections":[{"title":"Camera naming convention","local":"camera-naming-convention","sections":[],"depth":3},{"title":"Joint frame transform (SO-100/101 zero-shot)","local":"joint-frame-transform-so-100101-zero-shot","sections":[],"depth":3}],"depth":2},{"title":"Differences From the Original Implementation","local":"differences-from-the-original-implementation","sections":[],"depth":2},{"title":"Citation","local":"citation","sections":[],"depth":2},{"title":"License","local":"license","sections":[],"depth":2}],"depth":1}"/> | |
| <link href="/docs/lerobot/main/en/_app/immutable/entry/start.Dk__j1Q3.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/BYEZFv3_.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/ByoQ5dZT.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/entry/app.-pyB7NrA.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/jBs4H5jA.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/BwNHWMUY.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/DoKxYzxk.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/nodes/0.CKhCqrIb.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/BEuEZEXY.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/nodes/2.CvOmdGcd.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/BSz-biwr.js" rel="modulepreload"> | |
| <link href="/docs/lerobot/main/en/_app/immutable/chunks/CT4ZCfxV.js" rel="modulepreload"> | |
| <!--ozuu19--><meta name="hf:doc:metadata" content="{"title":"MolmoAct2 Policy","local":"molmoact2-policy","sections":[{"title":"Installation Requirements","local":"installation-requirements","sections":[],"depth":2},{"title":"Usage","local":"usage","sections":[],"depth":2},{"title":"Training","local":"training","sections":[{"title":"Training With Original MolmoAct2 Weight","local":"training-with-original-molmoact2-weight","sections":[],"depth":3},{"title":"Training With LeRobot MolmoAct2 Weight","local":"training-with-lerobot-molmoact2-weight","sections":[],"depth":3},{"title":"Common Practices","local":"common-practices","sections":[],"depth":3},{"title":"Common Policy Options","local":"common-policy-options","sections":[],"depth":3},{"title":"Learning Rates","local":"learning-rates","sections":[],"depth":3},{"title":"Dataset Quantile Statistics","local":"dataset-quantile-statistics","sections":[],"depth":3}],"depth":2},{"title":"Evaluation","local":"evaluation","sections":[{"title":"Evaluation With LeRobot MolmoAct2 Weight","local":"evaluation-with-lerobot-molmoact2-weight","sections":[],"depth":3},{"title":"Evaluation With Original MolmoAct2 Weight","local":"evaluation-with-original-molmoact2-weight","sections":[],"depth":3},{"title":"Common Evaluation Options","local":"common-evaluation-options","sections":[],"depth":3}],"depth":2},{"title":"Performance Results","local":"performance-results","sections":[{"title":"LIBERO Benchmark Results","local":"libero-benchmark-results","sections":[],"depth":3}],"depth":2},{"title":"Hardware Deployment (lerobot-rollout)","local":"hardware-deployment-lerobot-rollout","sections":[{"title":"Camera naming convention","local":"camera-naming-convention","sections":[],"depth":3},{"title":"Joint frame transform (SO-100/101 zero-shot)","local":"joint-frame-transform-so-100101-zero-shot","sections":[],"depth":3}],"depth":2},{"title":"Differences From the Original Implementation","local":"differences-from-the-original-implementation","sections":[],"depth":2},{"title":"Citation","local":"citation","sections":[],"depth":2},{"title":"License","local":"license","sections":[],"depth":2}],"depth":1}"/><!----> | |
| <link href="/docs/lerobot/main/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="molmoact2-policy" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#molmoact2-policy"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MolmoAct2 Policy</span></h1><!--]--><!----> <p>MolmoAct2 is the LeRobot policy implementation of <a href="https://allenai.org/blog/molmoact2" rel="nofollow">MolmoAct2</a>, ported into the LeRobot | |
| training, evaluation, checkpointing, and dataset interfaces for easier use with | |
| LeRobot datasets.</p> <p>This implementation currently supports training and evaluation for the regular | |
| MolmoAct2 model. MolmoAct2-Think, which supports adaptive depth reasoning, is | |
| not included in this LeRobot policy yet and is coming soon.</p> <p>For the original MolmoAct2 training code used for the experiments reported in | |
| the paper, see <a href="https://github.com/allenai/molmoact2" rel="nofollow">allenai/molmoact2</a>.</p> <!--[1--><h2 class="relative group"><a id="installation-requirements" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#installation-requirements"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Installation Requirements</span></h2><!--]--><!----> <p>Install LeRobot with the MolmoAct2 optional dependencies:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->uv <span class="hljs-built_in">sync</span> --locked --extra molmoact2<!----></pre></div><!----> <p>To run the models in this repository, you need an NVIDIA GPU. The measurements | |
| below were taken on a single NVIDIA H100 80GB with bf16 model loading, LIBERO with two RGB cameras. MolmoAct2 rows use <code>chunk_size=10</code>, action dim 7 | |
| padded to <code>expected_max_action_dim=32</code>, and <code>num_flow_timesteps=8</code>. Training measurements use <code>gradient_checkpointing=true</code> and include the forward pass, backward pass, | |
| gradient clipping, optimizer step, and optimizer state allocation. Values are | |
| peak GPU memory sampled with <code>nvidia-smi</code>. Leave a few GiB of headroom for | |
| dataloader workers, CUDA context, and fragmentation.</p> <p>Multi-GPU training through <code>accelerate</code> increases throughput and global batch | |
| size, but this LeRobot port does not currently expose the original MolmoAct2 <code>fsdp_devices</code> model-parallel training path. The current training script has | |
| not been tested for multi-node training.</p> <table><thead><tr><th>Mode</th><th align="right">Peak Memory, bs=8</th><th align="right">Peak Memory, bs=16</th><th align="right">Peak Memory, bs=32</th></tr></thead><tbody><tr><td>Inference, continuous, CUDA graph enabled (bs=1)</td><td align="right">12.1 GiB</td><td align="right">-</td><td align="right">-</td></tr><tr><td>Fine-tuning, action expert only, continuous</td><td align="right">16.5 GiB</td><td align="right">18.3 GiB</td><td align="right">21.4 GiB</td></tr><tr><td>Fine-tuning, LoRA VLM, both action modes</td><td align="right">20.2 GiB</td><td align="right">26.8 GiB</td><td align="right">41.3 GiB</td></tr><tr><td>Fine-tuning, full model, both action modes</td><td align="right">48.3 GiB</td><td align="right">49.8 GiB</td><td align="right">60.1 GiB</td></tr></tbody></table> <p>The repo has been tested with Ubuntu 22.04.</p> <!--[1--><h2 class="relative group"><a id="usage" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#usage"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Usage</span></h2><!--]--><!----> <p>To use MolmoAct2 in a LeRobot training config, set:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->--policy.type=molmoact2<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="training" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training</span></h2><!--]--><!----> <p>MolmoAct2 can be fine-tuned from either the released MolmoAct2 Hugging Face | |
| checkpoint format or from a checkpoint already saved by LeRobot. Both routes use | |
| the same LeRobot training loop, dataset transforms, checkpoint saving, and | |
| logging. The difference is only how the initial policy weights and processor | |
| state are loaded.</p> <!--[2--><h3 class="relative group"><a id="training-with-original-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training-with-original-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training With Original MolmoAct2 Weight</span></h3><!--]--><!----> <p>Use <code>policy.checkpoint_path</code> when starting from a released MolmoAct2 checkpoint, | |
| for example <code>allenai/MolmoAct2</code> or <code>allenai/MolmoAct2-LIBERO</code>. LeRobot will load | |
| the original HF model files, then build its own policy processor from the | |
| dataset metadata and the policy options below.</p> <p>The command below shows full fine-tuning on the merged LIBERO dataset. It uses | |
| bf16 model loading, 8 flow timesteps, LeRobot dataset statistics, image | |
| augmentation, and LeRobot’s checkpointing/logging path.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->accelerate launch \ | |
| --num_processes=8 \ | |
| --mixed_precision=bf16 \ | |
| -m lerobot.scripts.lerobot_train \ | |
| --dataset.repo_id=allenai/MolmoAct2-LIBERO-Dataset \ | |
| --dataset.root=/path/to/lerobot/data/allenai/MolmoAct2-LIBERO-Dataset \ | |
| --dataset.video_backend=pyav \ | |
| --dataset.image_transforms.enable=<span class="hljs-literal">true</span> \ | |
| --policy.type=molmoact2 \ | |
| --policy.checkpoint_path=allenai/MolmoAct2-LIBERO \ | |
| --policy.device=cuda \ | |
| --policy.action_mode=both \ | |
| --policy.chunk_size=10 \ | |
| --policy.n_action_steps=10 \ | |
| --policy.setup_type=<span class="hljs-string">"single franka robotic arm in libero"</span> \ | |
| --policy.control_mode=<span class="hljs-string">"delta end-effector pose"</span> \ | |
| --policy.image_keys=<span class="hljs-string">'["observation.images.image","observation.images.wrist_image"]'</span> \ | |
| --policy.model_dtype=bfloat16 \ | |
| --policy.num_flow_timesteps=8 \ | |
| --policy.gradient_checkpointing=<span class="hljs-literal">true</span> \ | |
| --policy.freeze_embedding=<span class="hljs-literal">true</span> \ | |
| --policy.normalize_gripper=<span class="hljs-literal">false</span> \ | |
| --policy.enable_knowledge_insulation=<span class="hljs-literal">false</span> \ | |
| --policy.push_to_hub=<span class="hljs-literal">false</span> \ | |
| --wandb.enable=<span class="hljs-literal">true</span> \ | |
| --wandb.entity=<wandb_entity> \ | |
| --wandb.project=<wandb_project> \ | |
| --job_name=<job_name> \ | |
| --output_dir=outputs/<job_name> \ | |
| --steps=10000 \ | |
| --batch_size=32 \ | |
| --num_workers=4 \ | |
| --log_freq=20 \ | |
| --env_eval_freq=-1 \ | |
| --save_checkpoint=<span class="hljs-literal">true</span> \ | |
| --save_freq=2000<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="training-with-lerobot-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training-with-lerobot-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training With LeRobot MolmoAct2 Weight</span></h3><!--]--><!----> <p>Use <code>policy.path</code> when starting from a MolmoAct2 checkpoint that was saved by | |
| LeRobot, either from a local <code>pretrained_model</code> directory or from the Hub. This | |
| restores the saved LeRobot policy config, model weights, processor, and | |
| normalization statistics. You can still override training-time options such as <code>batch_size</code>, <code>steps</code>, LoRA flags, or <code>policy.action_mode</code>.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->accelerate launch \ | |
| --num_processes=8 \ | |
| --mixed_precision=bf16 \ | |
| -m lerobot.scripts.lerobot_train \ | |
| --dataset.repo_id=allenai/MolmoAct2-LIBERO-Dataset \ | |
| --dataset.root=/path/to/lerobot/data/allenai/MolmoAct2-LIBERO-Dataset \ | |
| --dataset.video_backend=pyav \ | |
| --dataset.image_transforms.enable=<span class="hljs-literal">true</span> \ | |
| --policy.path=/path/to/pretrained_model \ | |
| --policy.device=cuda \ | |
| --policy.action_mode=both \ | |
| --policy.chunk_size=10 \ | |
| --policy.n_action_steps=10 \ | |
| --policy.model_dtype=bfloat16 \ | |
| --policy.num_flow_timesteps=8 \ | |
| --policy.gradient_checkpointing=<span class="hljs-literal">true</span> \ | |
| --wandb.enable=<span class="hljs-literal">true</span> \ | |
| --wandb.entity=<wandb_entity> \ | |
| --wandb.project=<wandb_project> \ | |
| --job_name=<job_name> \ | |
| --output_dir=outputs/<job_name> \ | |
| --steps=10000 \ | |
| --batch_size=32 \ | |
| --num_workers=4 \ | |
| --log_freq=20 \ | |
| --env_eval_freq=-1 \ | |
| --save_checkpoint=<span class="hljs-literal">true</span> \ | |
| --save_freq=2000<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="common-practices" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-practices"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Practices</span></h3><!--]--><!----> <p>For fine-tuning on a comparatively small dataset, such as a single LIBERO suite | |
| or a real-world dataset with less than 200 demonstrations, a global batch size of | |
| 16 to 32 is a good starting point. In these settings, <code>policy.enable_lora_vlm=true</code> or <code>policy.train_action_expert_only=true</code> is also a practical choice. In both | |
| cases, we intentionally keep the action expert fully trainable, which we found | |
| to be crucial for model performance. For larger fine-tuning datasets, larger | |
| global batch sizes and full fine-tuning are usually preferred.</p> <!--[2--><h3 class="relative group"><a id="common-policy-options" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-policy-options"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Policy Options</span></h3><!--]--><!----> <ul><li><code>policy.checkpoint_path</code>: original MolmoAct2 HF checkpoint to initialize from. | |
| Use this for released MolmoAct2 weights.</li> <li><code>policy.path</code>: LeRobot checkpoint to initialize from. Use this for checkpoints | |
| created by LeRobot training.</li> <li><code>policy.action_mode</code>: training target, one of <code>continuous</code>, <code>discrete</code>, or <code>both</code>. <code>both</code> trains the flow-matching action expert and the discrete | |
| action-token loss.</li> <li><code>policy.train_action_expert_only</code>: trains only parameters whose names contain <code>action_expert</code>. It requires <code>policy.action_mode=continuous</code>.</li> <li><code>policy.enable_lora_vlm</code>: enables LoRA on VLM linear layers. Use <code>policy.enable_lora_action_expert=true</code> only if LoRA should also cover action | |
| expert linear layers. When <code>policy.enable_lora_action_expert=false</code>, the | |
| action expert base weights remain fully trainable while the VLM is trained | |
| through LoRA adapters. When <code>policy.enable_lora_action_expert=true</code>, the | |
| action expert is also adapter-tuned instead of fully fine-tuned.</li> <li><code>policy.enable_knowledge_insulation</code>: when <code>true</code>, detaches action-expert | |
| context K/V states before the action loss. The default is <code>false</code>.</li> <li><code>policy.chunk_size</code>: action horizon used by the policy. For LIBERO we use <code>10</code>. This LeRobot port overrides the loaded checkpoint’s <code>max_action_horizon</code> with this value.</li> <li><code>policy.n_action_steps</code>: number of actions consumed from each predicted | |
| chunk before querying the policy again. For LIBERO, set it to <code>chunk_size</code>.</li> <li><code>policy.setup_type</code>: text inserted into the prompt to describe the robot and | |
| scene, e.g. <code>single franka robotic arm in libero</code>. More examples are listed | |
| in the <code>metadata_by_tag</code> entries of <a href="https://huggingface.co/allenai/MolmoAct2/blob/main/norm_stats.json" rel="nofollow"><code>norm_stats.json</code></a>.</li> <li><code>policy.control_mode</code>: text inserted into the prompt to describe the action | |
| space, e.g. <code>delta end-effector pose</code> or <code>absolute joint pose</code>.</li> <li><code>policy.image_keys</code>: ordered LeRobot image observation keys passed to the | |
| processor.</li> <li><code>policy.model_dtype</code>: checkpoint/forward dtype, one of <code>float32</code>, <code>bfloat16</code>, or <code>float16</code>. Use <code>bfloat16</code> for normal training.</li> <li><code>policy.num_flow_timesteps</code>: number of flow-matching timesteps sampled per | |
| example during training. We use <code>8</code> for fine-tuning.</li> <li><code>policy.num_inference_steps</code>: optional override for continuous action | |
| generation steps at inference time.</li> <li><code>policy.gradient_checkpointing</code>: enables checkpointing in the VLM/action path | |
| to reduce activation memory.</li> <li><code>policy.freeze_embedding</code>: freezes input embeddings. The default is <code>true</code>.</li> <li><code>policy.normalize_gripper</code>: controls whether gripper dimensions are included | |
| in state/action quantile normalization. The default is <code>false</code>.</li> <li><code>policy.normalize_language</code>: normalizes task strings before prompt | |
| construction. The default is <code>true</code>.</li> <li><code>policy.mask_action_dim_padding</code>: masks padded dimensions in the flow loss. | |
| Released checkpoints use <code>policy.expected_max_action_dim=32</code>.</li> <li><code>policy.max_sequence_length</code>: optional manual sequence cap. Leave unset to | |
| infer it from images, state dimension, action dimension, action horizon, and | |
| discrete-action mode.</li></ul> <!--[2--><h3 class="relative group"><a id="learning-rates" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#learning-rates"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Learning Rates</span></h3><!--]--><!----> <p>MolmoAct2 uses parameter-group learning rates to match the original MolmoAct2 | |
| fine-tuning experiments.</p> <ul><li>Full fine-tuning uses <code>policy.optimizer_lr=1e-5</code> for the VLM, <code>policy.optimizer_vit_lr=5e-6</code> for the vision tower, <code>policy.optimizer_connector_lr=5e-6</code> for image connector layers, and <code>policy.optimizer_action_expert_lr=5e-5</code> for the action expert.</li> <li>LoRA VLM fine-tuning sets the VLM, vision, and connector LoRA parameter | |
| groups to <code>5e-5</code> when <code>policy.enable_lora_vlm=true</code>. By default, <code>policy.enable_lora_action_expert=false</code>, so the action expert is still fully | |
| fine-tuned with <code>policy.optimizer_action_expert_lr</code>. If <code>policy.enable_lora_action_expert=true</code>, the action expert is trained through | |
| LoRA adapters instead.</li> <li>Action-expert-only fine-tuning trains only the action expert and uses <code>policy.optimizer_action_expert_lr=5e-5</code>.</li></ul> <p>You can override the full fine-tuning and action-expert learning rates with <code>policy.optimizer_lr</code>, <code>policy.optimizer_vit_lr</code>, <code>policy.optimizer_connector_lr</code>, and <code>policy.optimizer_action_expert_lr</code>. | |
| Scheduler settings can be changed with <code>policy.scheduler_warmup_steps</code>, <code>policy.scheduler_decay_steps</code>, and <code>policy.scheduler_decay_lr</code>.</p> <!--[2--><h3 class="relative group"><a id="dataset-quantile-statistics" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#dataset-quantile-statistics"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Dataset Quantile Statistics</span></h3><!--]--><!----> <p>MolmoAct2 defaults to quantile normalization for state and action features. If | |
| your dataset has not been converted with quantile statistics, you can add them | |
| with:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->python src/lerobot/scripts/augment_dataset_quantile_stats.py \ | |
| --repo-id=your_dataset<!----></pre></div><!----> <p>Recording, resuming, and merging aggregate quantiles from per-episode summaries, so <code>meta/stats.json</code> ends up holding a conservative envelope (<code>min</code> for <code>q <= 50</code>, <code>max</code> for <code>q > 50</code>) rather than whole-dataset quantiles. To estimate the latter, scan every episode with a running histogram:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->python src/lerobot/scripts/augment_dataset_quantile_stats.py \ | |
| --repo-id=your_dataset \ | |
| --overwrite \ | |
| --skip-images<!----></pre></div><!----> <p><code>--skip-images</code> keeps the existing image statistics and avoids video decoding when only <code>STATE</code>/<code>ACTION</code> need recomputing, and <code>--root</code> reads a local dataset instead of the Hub. These values are histogram estimates, subject to discretization and rebinning error, so they can differ from the conservative ones — which changes MolmoAct2’s normalized targets and therefore its loss scale. Statistics already saved inside an existing checkpoint are not affected.</p> <p>Alternatively, train MolmoAct2 with mean/std normalization:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->--policy.normalization_mapping=<span class="hljs-string">'{"ACTION": "MEAN_STD", "STATE": "MEAN_STD", "VISUAL": "IDENTITY"}'</span><!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="evaluation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#evaluation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Evaluation</span></h2><!--]--><!----> <p>Evaluation also supports both LeRobot-saved checkpoints and original MolmoAct2 | |
| HF checkpoints. For LIBERO replication, keep the EGL rendering environment | |
| fixed and use <code>policy.per_episode_seed=true</code>.</p> <p><strong>Important:</strong> We found that <code>num_steps_wait=10</code> does not reliably let the | |
| LIBERO scene stabilize and can degrade measured success. All LIBERO evaluation | |
| results reported here use <code>num_steps_wait=50</code>.</p> <!--[2--><h3 class="relative group"><a id="evaluation-with-lerobot-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#evaluation-with-lerobot-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Evaluation With LeRobot MolmoAct2 Weight</span></h3><!--]--><!----> <p>Use <code>policy.path</code> for a checkpoint saved by LeRobot. The saved processor and | |
| normalization statistics are restored together with the model.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!----><span class="hljs-built_in">export</span> MUJOCO_GL=egl | |
| <span class="hljs-built_in">export</span> PYOPENGL_PLATFORM=egl | |
| <span class="hljs-built_in">export</span> OMP_NUM_THREADS=1 | |
| <span class="hljs-built_in">export</span> MKL_NUM_THREADS=1 | |
| lerobot-eval \ | |
| --policy.path=allenai/MolmoAct2-LIBERO-LeRobot \ | |
| --policy.inference_action_mode=continuous \ | |
| --policy.model_dtype=bfloat16 \ | |
| --policy.use_amp=<span class="hljs-literal">true</span> \ | |
| --policy.enable_inference_cuda_graph=<span class="hljs-literal">true</span> \ | |
| --policy.device=cuda \ | |
| --policy.per_episode_seed=<span class="hljs-literal">true</span> \ | |
| --policy.eval_seed=1000 \ | |
| --env.type=libero \ | |
| --env.task=libero_10,libero_goal,libero_object,libero_spatial \ | |
| --env.camera_name_mapping=<span class="hljs-string">'{"agentview_image":"image","robot0_eye_in_hand_image":"wrist_image"}'</span> \ | |
| --eval.batch_size=1 \ | |
| --eval.n_episodes=50 \ | |
| --seed=1000<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="evaluation-with-original-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#evaluation-with-original-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Evaluation With Original MolmoAct2 Weight</span></h3><!--]--><!----> <p>You can evaluate a released Hugging Face checkpoint directly without first | |
| converting it to a LeRobot checkpoint. In this case, set <code>policy.checkpoint_path</code> to the HF model repo and provide <code>policy.norm_tag</code>. | |
| For LIBERO, <code>policy.norm_tag=libero</code> loads the LIBERO action/state | |
| normalization statistics, action horizon, prompt metadata, and image-key order | |
| from the checkpoint’s <code>norm_stats.json</code>.</p> <p>To fully replicate the MolmoAct2 paper results with released Hugging Face | |
| checkpoints, we recommend using the v0.5.1-pinned <a href="https://github.com/allenai/lerobot/tree/molmoact2-hf-inference" rel="nofollow"><code>allenai/lerobot</code> <code>molmoact2-hf-inference</code></a> branch. That branch matches the original evaluation settings used for the | |
| reported numbers.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!----><span class="hljs-built_in">export</span> MUJOCO_GL=egl | |
| <span class="hljs-built_in">export</span> PYOPENGL_PLATFORM=egl | |
| <span class="hljs-built_in">export</span> OMP_NUM_THREADS=1 | |
| <span class="hljs-built_in">export</span> MKL_NUM_THREADS=1 | |
| lerobot-eval \ | |
| --policy.type=molmoact2 \ | |
| --policy.checkpoint_path=allenai/MolmoAct2-LIBERO \ | |
| --policy.norm_tag=libero \ | |
| --policy.inference_action_mode=continuous \ | |
| --policy.model_dtype=float32 \ | |
| --policy.use_amp=<span class="hljs-literal">false</span> \ | |
| --policy.enable_inference_cuda_graph=<span class="hljs-literal">true</span> \ | |
| --policy.device=cuda \ | |
| --policy.per_episode_seed=<span class="hljs-literal">true</span> \ | |
| --policy.eval_seed=1000 \ | |
| --env.type=libero \ | |
| --env.task=libero_goal \ | |
| --env.camera_name_mapping=<span class="hljs-string">'{"agentview_image":"image","robot0_eye_in_hand_image":"wrist_image"}'</span> \ | |
| --eval.batch_size=1 \ | |
| --eval.n_episodes=50 \ | |
| --seed=1000<!----></pre></div><!----> <p>Use <code>--env.task=libero_10,libero_goal,libero_object,libero_spatial</code> to run the | |
| full LIBERO suite. The same command works for other released MolmoAct2 | |
| checkpoints as long as the requested <code>policy.norm_tag</code> exists in that | |
| checkpoint’s <code>norm_stats.json</code>.</p> <!--[2--><h3 class="relative group"><a id="common-evaluation-options" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-evaluation-options"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Evaluation Options</span></h3><!--]--><!----> <ul><li><code>policy.inference_action_mode</code>: required for rollout. Use <code>continuous</code> for | |
| flow-matching inference or <code>discrete</code> for action-token inference. It must be | |
| compatible with the training-time <code>policy.action_mode</code> saved in the | |
| checkpoint.</li> <li><code>policy.path</code>: LeRobot checkpoint path or Hub repo. Use this for checkpoints | |
| saved by LeRobot.</li> <li><code>policy.checkpoint_path</code>: original MolmoAct2 HF checkpoint path or Hub repo. | |
| Use this with <code>policy.type=molmoact2</code> and <code>policy.norm_tag</code>.</li> <li><code>policy.norm_tag</code>: selects normalization statistics, prompt metadata, | |
| image-key order, and action horizon from the original checkpoint’s <code>norm_stats.json</code>. It is required for direct original-HF checkpoint | |
| evaluation.</li> <li><code>policy.model_dtype</code>: model load/forward dtype. Use <code>bfloat16</code> for normal | |
| GPU evaluation. Use <code>float32</code> only when you explicitly want fp32 inference.</li> <li><code>policy.use_amp</code>: runs the policy forward under autocast during eval. For <code>model_dtype=bfloat16</code>, keep this enabled.</li> <li><code>policy.enable_inference_cuda_graph</code>: enables the MolmoAct2 inference CUDA | |
| graph path for faster repeated continuous-action rollout.</li> <li><code>policy.per_episode_seed</code> and <code>policy.eval_seed</code>: make stochastic continuous | |
| action generation deterministic per episode for replication.</li> <li><code>env.task</code>: comma-separated LIBERO suites or a single suite. Use <code>libero_10,libero_goal,libero_object,libero_spatial</code> for the full benchmark.</li> <li><code>env.camera_name_mapping</code>: maps LIBERO camera names to the image keys expected | |
| by the policy processor.</li></ul> <!--[1--><h2 class="relative group"><a id="performance-results" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#performance-results"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Performance Results</span></h2><!--]--><!----> <!--[2--><h3 class="relative group"><a id="libero-benchmark-results" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#libero-benchmark-results"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>LIBERO Benchmark Results</span></h3><!--]--><!----> <p>MolmoAct2 has demonstrated strong performance on the LIBERO benchmark suite. To | |
| compare and test its LeRobot implementation, we fine-tuned <a href="https://huggingface.co/allenai/MolmoAct2-LIBERO" rel="nofollow"><code>allenai/MolmoAct2-LIBERO</code></a> for an additional 10k steps on the LIBERO dataset with per-GPU batch size 32 on | |
| 8 H100 GPUs, then compared the results to the original MolmoAct2 reference | |
| results.</p> <p>The LeRobot fine-tuned checkpoint reported here is available at <a href="https://huggingface.co/allenai/MolmoAct2-LIBERO-LeRobot" rel="nofollow"><code>allenai/MolmoAct2-LIBERO-LeRobot</code></a> and was trained on <a href="https://huggingface.co/datasets/allenai/MolmoAct2-LIBERO-Dataset" rel="nofollow"><code>allenai/MolmoAct2-LIBERO-Dataset</code></a>.</p> <table><thead><tr><th>Benchmark</th><th align="right">LeRobot Implementation</th><th align="right">MolmoAct2 Original</th></tr></thead><tbody><tr><td>LIBERO Spatial</td><td align="right">98.4%</td><td align="right">97.8%</td></tr><tr><td>LIBERO Object</td><td align="right">100.0%</td><td align="right">100.0%</td></tr><tr><td>LIBERO Goal</td><td align="right">98.0%</td><td align="right">97.8%</td></tr><tr><td>LIBERO 10</td><td align="right">96.6%</td><td align="right">93.2%</td></tr><tr><td>Average</td><td align="right">98.25%</td><td align="right">97.20%</td></tr></tbody></table> <p>These results demonstrate MolmoAct2’s strong performance across diverse robotic | |
| manipulation tasks. To reproduce them, follow the instructions in the LIBERO | |
| evaluation section.</p> <!--[1--><h2 class="relative group"><a id="hardware-deployment-lerobot-rollout" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#hardware-deployment-lerobot-rollout"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Hardware Deployment (lerobot-rollout)</span></h2><!--]--><!----> <p>LeRobot-format checkpoints are available on the Hub for direct use with <code>lerobot-rollout</code>. Each checkpoint uses specific camera names that must | |
| match your robot’s camera configuration.</p> <!--[2--><h3 class="relative group"><a id="camera-naming-convention" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#camera-naming-convention"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Camera naming convention</span></h3><!--]--><!----> <p>Each checkpoint expects specific <code>observation.images.*</code> keys. | |
| If your robot cameras have different names, use <code>--rename_map</code> to map them:</p> <table><thead><tr><th>Checkpoint</th><th>Camera keys</th><th>Description</th></tr></thead><tbody><tr><td>MolmoAct2-LIBERO-LeRobot</td><td><code>image</code>, <code>wrist_image</code></td><td>LIBERO sim cameras</td></tr><tr><td>MolmoAct2-BimanualYAM-LeRobot</td><td><code>top</code>, <code>left</code>, <code>right</code></td><td>YAM 3-camera setup</td></tr><tr><td>MolmoAct2-DROID-LeRobot</td><td><code>cam0</code>, <code>cam1</code></td><td>External + wrist</td></tr><tr><td>MolmoAct2-SO100_101-LeRobot</td><td><code>cam0</code>, <code>cam1</code></td><td>Primary + secondary view</td></tr></tbody></table> <p>Example with an SO-100 robot using top and side cameras:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->lerobot-rollout \ | |
| --policy.path=lerobot/MolmoAct2-SO100_101-LeRobot \ | |
| --rename_map=<span class="hljs-string">'{"observation.images.top": "observation.images.cam0", "observation.images.side": "observation.images.cam1"}'</span> \ | |
| --robot.type=so100_follower \ | |
| --robot.port=/dev/ttyACM0 \ | |
| --robot.cameras=<span class="hljs-string">'{ | |
| top: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}, | |
| side: {type: opencv, index_or_path: 2, width: 640, height: 480, fps: 30} | |
| }'</span> \ | |
| --task=<span class="hljs-string">"pick up the red cube"</span> --duration=30<!----></pre></div><!----> <p>To use a wrist camera instead, just change the rename mapping:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->--rename_map=<span class="hljs-string">'{"observation.images.top": "observation.images.cam0", "observation.images.wrist": "observation.images.cam1"}'</span><!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="joint-frame-transform-so-100101-zero-shot" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#joint-frame-transform-so-100101-zero-shot"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Joint frame transform (SO-100/101 zero-shot)</span></h3><!--]--><!----> <blockquote class="warning"><!---->The MolmoAct2-SO100_101 checkpoint was trained on data that uses a different | |
| joint calibration convention than LeRobot >= 0.5.0. Without a frame | |
| correction, the arm may move in the wrong direction. <p>This affects both <strong>zero-shot deployment</strong> and <strong>fine-tuning</strong> from the | |
| original checkpoint. The pretrained weights expect the old convention, so | |
| all joint data (observations and actions) must be transformed to match.</p> <p>The converted LeRobot checkpoint (<code>lerobot/MolmoAct2-SO100_101-LeRobot</code>) | |
| already includes this correction in its processor pipeline. If you convert | |
| or fine-tune the checkpoint yourself, set the following in the policy config (<code>configuration_molmoact2.py</code>):</p> <ul><li><code>joint_signs</code>: <code>[1, -1, 1, 1, 1, 1]</code> (flips shoulder_lift direction)</li> <li><code>joint_offsets</code>: <code>[0, 90, 90, 0, 0, 0]</code> (shifts shoulder_lift and elbow_flex by 90°)</li></ul> <p>See the <a href="./backwardcomp">backward compatibility guide</a> for details on the | |
| calibration change.</p><!----></blockquote><!----> <!--[1--><h2 class="relative group"><a id="differences-from-the-original-implementation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#differences-from-the-original-implementation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Differences From the Original Implementation</span></h2><!--]--><!----> <p>This LeRobot port is intended to match MolmoAct2 behavior while using LeRobot’s | |
| dataset, training, evaluation, checkpoint, and logging infrastructure. The main | |
| differences from the original training repository are:</p> <ul><li>The original paper training stack loads the model in fp32 and trains under | |
| mixed precision. This LeRobot port usually loads the checkpoint directly in <code>policy.model_dtype=bfloat16</code> for lower memory use.</li> <li>The original repository uses its own FSDP/model-parallel training path. The | |
| LeRobot port uses the standard LeRobot/Accelerate training path and has not | |
| been tested for multi-node training.</li> <li>The original repository supports sequence packing. The LeRobot port trains on | |
| one LeRobot sample per item and pads to an inferred fixed sequence budget.</li> <li>The LeRobot port follows LeRobot’s optimizer, scheduler, checkpoint saving, | |
| dataset transforms, image augmentation, and Weights & Biases logging | |
| conventions.</li> <li>The original training path supports mixed action horizons by padding to <code>max_action_horizon</code> and masking padded horizon slots in the action expert | |
| self-attention. This is useful when training across datasets with different | |
| control frequencies. The LeRobot port currently targets single-dataset | |
| fine-tuning, so <code>policy.chunk_size</code> overrides the checkpoint <code>max_action_horizon</code> and horizon masking is not implemented yet. Support for | |
| this mixed-horizon path is planned.</li></ul> <!--[1--><h2 class="relative group"><a id="citation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#citation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Citation</span></h2><!--]--><!----> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bibtex "><!----><span class="hljs-comment">@misc{fang2026molmoact2actionreasoningmodels,</span> | |
| title={MolmoAct2: Action Reasoning Models for Real-world Deployment}, | |
| author={Haoquan Fang <span class="hljs-keyword">and</span> Jiafei Duan <span class="hljs-keyword">and</span> Donovan Clay <span class="hljs-keyword">and</span> Sam Wang <span class="hljs-keyword">and</span> Shuo Liu <span class="hljs-keyword">and</span> Weikai Huang <span class="hljs-keyword">and</span> Xiang Fan <span class="hljs-keyword">and</span> Wei-Chuan Tsai <span class="hljs-keyword">and</span> Shirui Chen <span class="hljs-keyword">and</span> Yi Ru Wang <span class="hljs-keyword">and</span> Shanli Xing <span class="hljs-keyword">and</span> Jaemin Cho <span class="hljs-keyword">and</span> Jae Sung Park <span class="hljs-keyword">and</span> Ainaz Eftekhar <span class="hljs-keyword">and</span> Peter Sushko <span class="hljs-keyword">and</span> Karen Farley <span class="hljs-keyword">and</span> Angad Wadhwa <span class="hljs-keyword">and</span> Cole Harrison <span class="hljs-keyword">and</span> Winson Han <span class="hljs-keyword">and</span> Ying-Chun Lee <span class="hljs-keyword">and</span> Eli VanderBilt <span class="hljs-keyword">and</span> Rose Hendrix <span class="hljs-keyword">and</span> Suveen Ellawela <span class="hljs-keyword">and</span> Lucas Ngoo <span class="hljs-keyword">and</span> Joyce Chai <span class="hljs-keyword">and</span> Zhongzheng Ren <span class="hljs-keyword">and</span> Ali Farhadi <span class="hljs-keyword">and</span> Dieter Fox <span class="hljs-keyword">and</span> Ranjay Krishna}, | |
| year={<span class="hljs-number">2026</span>}, | |
| eprint={<span class="hljs-number">2605</span>.<span class="hljs-number">02881</span>}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.RO}, | |
| url={https:<span class="hljs-comment">//arxiv.org/abs/2605.02881},</span> | |
| }<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="license" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#license"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>License</span></h2><!--]--><!----> <p>This model is licensed under Apache 2.0. It is intended for research and | |
| educational use in accordance with <a href="https://allenai.org/responsible-use" rel="nofollow">Ai2’s Responsible Use Guidelines</a>, | |
| consistent with <a href="https://github.com/allenai/molmoact2" rel="nofollow">allenai/molmoact2</a>.</p> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/lerobot/blob/main/docs/source/molmoact2.mdx" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]--> | |
| <script> | |
| { | |
| __sveltekit_tk9oje = { | |
| base: "/docs/lerobot/main/en", | |
| assets: "/docs/lerobot/main/en" | |
| }; | |
| const element = document.currentScript.parentElement; | |
| Promise.all([ | |
| import("/docs/lerobot/main/en/_app/immutable/entry/start.Dk__j1Q3.js"), | |
| import("/docs/lerobot/main/en/_app/immutable/entry/app.-pyB7NrA.js") | |
| ]).then(([kit, app]) => { | |
| kit.start(app, element, { | |
| node_ids: [0, 2], | |
| data: [null,null], | |
| form: null, | |
| error: null | |
| }); | |
| }); | |
| } | |
| </script> | |
Xet Storage Details
- Size:
- 78.6 kB
- Xet hash:
- 5dd9b6449110121752c3d44b443aedc1047f222b89a1e008b795e90343226775
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.