Buckets:

hf-doc-build/doc-dev / lerobot /pr_3613 /en /molmoact2.html
download
raw
78.6 kB
<meta charset="utf-8" /><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;MolmoAct2 Policy&quot;,&quot;local&quot;:&quot;molmoact2-policy&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Installation Requirements&quot;,&quot;local&quot;:&quot;installation-requirements&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Usage&quot;,&quot;local&quot;:&quot;usage&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Training&quot;,&quot;local&quot;:&quot;training&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Training With Original MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;training-with-original-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Training With LeRobot MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;training-with-lerobot-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Common Practices&quot;,&quot;local&quot;:&quot;common-practices&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Common Policy Options&quot;,&quot;local&quot;:&quot;common-policy-options&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Learning Rates&quot;,&quot;local&quot;:&quot;learning-rates&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Dataset Quantile Statistics&quot;,&quot;local&quot;:&quot;dataset-quantile-statistics&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Evaluation&quot;,&quot;local&quot;:&quot;evaluation&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Evaluation With LeRobot MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;evaluation-with-lerobot-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Evaluation With Original MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;evaluation-with-original-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Common Evaluation Options&quot;,&quot;local&quot;:&quot;common-evaluation-options&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Performance Results&quot;,&quot;local&quot;:&quot;performance-results&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;LIBERO Benchmark Results&quot;,&quot;local&quot;:&quot;libero-benchmark-results&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Hardware Deployment (lerobot-rollout)&quot;,&quot;local&quot;:&quot;hardware-deployment-lerobot-rollout&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Camera naming convention&quot;,&quot;local&quot;:&quot;camera-naming-convention&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Joint frame transform (SO-100/101 zero-shot)&quot;,&quot;local&quot;:&quot;joint-frame-transform-so-100101-zero-shot&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Differences From the Original Implementation&quot;,&quot;local&quot;:&quot;differences-from-the-original-implementation&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Citation&quot;,&quot;local&quot;:&quot;citation&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;License&quot;,&quot;local&quot;:&quot;license&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/>
<link href="/docs/lerobot/main/en/_app/immutable/entry/start.Dk__j1Q3.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/BYEZFv3_.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/ByoQ5dZT.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/entry/app.-pyB7NrA.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/jBs4H5jA.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/BwNHWMUY.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/DoKxYzxk.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/nodes/0.CKhCqrIb.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/BEuEZEXY.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/nodes/2.CvOmdGcd.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/BSz-biwr.js" rel="modulepreload">
<link href="/docs/lerobot/main/en/_app/immutable/chunks/CT4ZCfxV.js" rel="modulepreload">
<!--ozuu19--><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;MolmoAct2 Policy&quot;,&quot;local&quot;:&quot;molmoact2-policy&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Installation Requirements&quot;,&quot;local&quot;:&quot;installation-requirements&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Usage&quot;,&quot;local&quot;:&quot;usage&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Training&quot;,&quot;local&quot;:&quot;training&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Training With Original MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;training-with-original-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Training With LeRobot MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;training-with-lerobot-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Common Practices&quot;,&quot;local&quot;:&quot;common-practices&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Common Policy Options&quot;,&quot;local&quot;:&quot;common-policy-options&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Learning Rates&quot;,&quot;local&quot;:&quot;learning-rates&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Dataset Quantile Statistics&quot;,&quot;local&quot;:&quot;dataset-quantile-statistics&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Evaluation&quot;,&quot;local&quot;:&quot;evaluation&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Evaluation With LeRobot MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;evaluation-with-lerobot-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Evaluation With Original MolmoAct2 Weight&quot;,&quot;local&quot;:&quot;evaluation-with-original-molmoact2-weight&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Common Evaluation Options&quot;,&quot;local&quot;:&quot;common-evaluation-options&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Performance Results&quot;,&quot;local&quot;:&quot;performance-results&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;LIBERO Benchmark Results&quot;,&quot;local&quot;:&quot;libero-benchmark-results&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Hardware Deployment (lerobot-rollout)&quot;,&quot;local&quot;:&quot;hardware-deployment-lerobot-rollout&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Camera naming convention&quot;,&quot;local&quot;:&quot;camera-naming-convention&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Joint frame transform (SO-100/101 zero-shot)&quot;,&quot;local&quot;:&quot;joint-frame-transform-so-100101-zero-shot&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Differences From the Original Implementation&quot;,&quot;local&quot;:&quot;differences-from-the-original-implementation&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Citation&quot;,&quot;local&quot;:&quot;citation&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;License&quot;,&quot;local&quot;:&quot;license&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/><!---->
<link href="/docs/lerobot/main/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="molmoact2-policy" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#molmoact2-policy"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MolmoAct2 Policy</span></h1><!--]--><!----> <p>MolmoAct2 is the LeRobot policy implementation of <a href="https://allenai.org/blog/molmoact2" rel="nofollow">MolmoAct2</a>, ported into the LeRobot
training, evaluation, checkpointing, and dataset interfaces for easier use with
LeRobot datasets.</p> <p>This implementation currently supports training and evaluation for the regular
MolmoAct2 model. MolmoAct2-Think, which supports adaptive depth reasoning, is
not included in this LeRobot policy yet and is coming soon.</p> <p>For the original MolmoAct2 training code used for the experiments reported in
the paper, see <a href="https://github.com/allenai/molmoact2" rel="nofollow">allenai/molmoact2</a>.</p> <!--[1--><h2 class="relative group"><a id="installation-requirements" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#installation-requirements"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Installation Requirements</span></h2><!--]--><!----> <p>Install LeRobot with the MolmoAct2 optional dependencies:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->uv <span class="hljs-built_in">sync</span> --locked --extra molmoact2<!----></pre></div><!----> <p>To run the models in this repository, you need an NVIDIA GPU. The measurements
below were taken on a single NVIDIA H100 80GB with bf16 model loading, LIBERO with two RGB cameras. MolmoAct2 rows use <code>chunk_size=10</code>, action dim 7
padded to <code>expected_max_action_dim=32</code>, and <code>num_flow_timesteps=8</code>. Training measurements use <code>gradient_checkpointing=true</code> and include the forward pass, backward pass,
gradient clipping, optimizer step, and optimizer state allocation. Values are
peak GPU memory sampled with <code>nvidia-smi</code>. Leave a few GiB of headroom for
dataloader workers, CUDA context, and fragmentation.</p> <p>Multi-GPU training through <code>accelerate</code> increases throughput and global batch
size, but this LeRobot port does not currently expose the original MolmoAct2 <code>fsdp_devices</code> model-parallel training path. The current training script has
not been tested for multi-node training.</p> <table><thead><tr><th>Mode</th><th align="right">Peak Memory, bs=8</th><th align="right">Peak Memory, bs=16</th><th align="right">Peak Memory, bs=32</th></tr></thead><tbody><tr><td>Inference, continuous, CUDA graph enabled (bs=1)</td><td align="right">12.1 GiB</td><td align="right">-</td><td align="right">-</td></tr><tr><td>Fine-tuning, action expert only, continuous</td><td align="right">16.5 GiB</td><td align="right">18.3 GiB</td><td align="right">21.4 GiB</td></tr><tr><td>Fine-tuning, LoRA VLM, both action modes</td><td align="right">20.2 GiB</td><td align="right">26.8 GiB</td><td align="right">41.3 GiB</td></tr><tr><td>Fine-tuning, full model, both action modes</td><td align="right">48.3 GiB</td><td align="right">49.8 GiB</td><td align="right">60.1 GiB</td></tr></tbody></table> <p>The repo has been tested with Ubuntu 22.04.</p> <!--[1--><h2 class="relative group"><a id="usage" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#usage"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Usage</span></h2><!--]--><!----> <p>To use MolmoAct2 in a LeRobot training config, set:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->--policy.type=molmoact2<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="training" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training</span></h2><!--]--><!----> <p>MolmoAct2 can be fine-tuned from either the released MolmoAct2 Hugging Face
checkpoint format or from a checkpoint already saved by LeRobot. Both routes use
the same LeRobot training loop, dataset transforms, checkpoint saving, and
logging. The difference is only how the initial policy weights and processor
state are loaded.</p> <!--[2--><h3 class="relative group"><a id="training-with-original-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training-with-original-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training With Original MolmoAct2 Weight</span></h3><!--]--><!----> <p>Use <code>policy.checkpoint_path</code> when starting from a released MolmoAct2 checkpoint,
for example <code>allenai/MolmoAct2</code> or <code>allenai/MolmoAct2-LIBERO</code>. LeRobot will load
the original HF model files, then build its own policy processor from the
dataset metadata and the policy options below.</p> <p>The command below shows full fine-tuning on the merged LIBERO dataset. It uses
bf16 model loading, 8 flow timesteps, LeRobot dataset statistics, image
augmentation, and LeRobot’s checkpointing/logging path.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->accelerate launch \
--num_processes=8 \
--mixed_precision=bf16 \
-m lerobot.scripts.lerobot_train \
--dataset.repo_id=allenai/MolmoAct2-LIBERO-Dataset \
--dataset.root=/path/to/lerobot/data/allenai/MolmoAct2-LIBERO-Dataset \
--dataset.video_backend=pyav \
--dataset.image_transforms.enable=<span class="hljs-literal">true</span> \
--policy.type=molmoact2 \
--policy.checkpoint_path=allenai/MolmoAct2-LIBERO \
--policy.device=cuda \
--policy.action_mode=both \
--policy.chunk_size=10 \
--policy.n_action_steps=10 \
--policy.setup_type=<span class="hljs-string">&quot;single franka robotic arm in libero&quot;</span> \
--policy.control_mode=<span class="hljs-string">&quot;delta end-effector pose&quot;</span> \
--policy.image_keys=<span class="hljs-string">&#x27;[&quot;observation.images.image&quot;,&quot;observation.images.wrist_image&quot;]&#x27;</span> \
--policy.model_dtype=bfloat16 \
--policy.num_flow_timesteps=8 \
--policy.gradient_checkpointing=<span class="hljs-literal">true</span> \
--policy.freeze_embedding=<span class="hljs-literal">true</span> \
--policy.normalize_gripper=<span class="hljs-literal">false</span> \
--policy.enable_knowledge_insulation=<span class="hljs-literal">false</span> \
--policy.push_to_hub=<span class="hljs-literal">false</span> \
--wandb.enable=<span class="hljs-literal">true</span> \
--wandb.entity=&lt;wandb_entity&gt; \
--wandb.project=&lt;wandb_project&gt; \
--job_name=&lt;job_name&gt; \
--output_dir=outputs/&lt;job_name&gt; \
--steps=10000 \
--batch_size=32 \
--num_workers=4 \
--log_freq=20 \
--env_eval_freq=-1 \
--save_checkpoint=<span class="hljs-literal">true</span> \
--save_freq=2000<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="training-with-lerobot-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training-with-lerobot-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training With LeRobot MolmoAct2 Weight</span></h3><!--]--><!----> <p>Use <code>policy.path</code> when starting from a MolmoAct2 checkpoint that was saved by
LeRobot, either from a local <code>pretrained_model</code> directory or from the Hub. This
restores the saved LeRobot policy config, model weights, processor, and
normalization statistics. You can still override training-time options such as <code>batch_size</code>, <code>steps</code>, LoRA flags, or <code>policy.action_mode</code>.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->accelerate launch \
--num_processes=8 \
--mixed_precision=bf16 \
-m lerobot.scripts.lerobot_train \
--dataset.repo_id=allenai/MolmoAct2-LIBERO-Dataset \
--dataset.root=/path/to/lerobot/data/allenai/MolmoAct2-LIBERO-Dataset \
--dataset.video_backend=pyav \
--dataset.image_transforms.enable=<span class="hljs-literal">true</span> \
--policy.path=/path/to/pretrained_model \
--policy.device=cuda \
--policy.action_mode=both \
--policy.chunk_size=10 \
--policy.n_action_steps=10 \
--policy.model_dtype=bfloat16 \
--policy.num_flow_timesteps=8 \
--policy.gradient_checkpointing=<span class="hljs-literal">true</span> \
--wandb.enable=<span class="hljs-literal">true</span> \
--wandb.entity=&lt;wandb_entity&gt; \
--wandb.project=&lt;wandb_project&gt; \
--job_name=&lt;job_name&gt; \
--output_dir=outputs/&lt;job_name&gt; \
--steps=10000 \
--batch_size=32 \
--num_workers=4 \
--log_freq=20 \
--env_eval_freq=-1 \
--save_checkpoint=<span class="hljs-literal">true</span> \
--save_freq=2000<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="common-practices" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-practices"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Practices</span></h3><!--]--><!----> <p>For fine-tuning on a comparatively small dataset, such as a single LIBERO suite
or a real-world dataset with less than 200 demonstrations, a global batch size of
16 to 32 is a good starting point. In these settings, <code>policy.enable_lora_vlm=true</code> or <code>policy.train_action_expert_only=true</code> is also a practical choice. In both
cases, we intentionally keep the action expert fully trainable, which we found
to be crucial for model performance. For larger fine-tuning datasets, larger
global batch sizes and full fine-tuning are usually preferred.</p> <!--[2--><h3 class="relative group"><a id="common-policy-options" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-policy-options"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Policy Options</span></h3><!--]--><!----> <ul><li><code>policy.checkpoint_path</code>: original MolmoAct2 HF checkpoint to initialize from.
Use this for released MolmoAct2 weights.</li> <li><code>policy.path</code>: LeRobot checkpoint to initialize from. Use this for checkpoints
created by LeRobot training.</li> <li><code>policy.action_mode</code>: training target, one of <code>continuous</code>, <code>discrete</code>, or <code>both</code>. <code>both</code> trains the flow-matching action expert and the discrete
action-token loss.</li> <li><code>policy.train_action_expert_only</code>: trains only parameters whose names contain <code>action_expert</code>. It requires <code>policy.action_mode=continuous</code>.</li> <li><code>policy.enable_lora_vlm</code>: enables LoRA on VLM linear layers. Use <code>policy.enable_lora_action_expert=true</code> only if LoRA should also cover action
expert linear layers. When <code>policy.enable_lora_action_expert=false</code>, the
action expert base weights remain fully trainable while the VLM is trained
through LoRA adapters. When <code>policy.enable_lora_action_expert=true</code>, the
action expert is also adapter-tuned instead of fully fine-tuned.</li> <li><code>policy.enable_knowledge_insulation</code>: when <code>true</code>, detaches action-expert
context K/V states before the action loss. The default is <code>false</code>.</li> <li><code>policy.chunk_size</code>: action horizon used by the policy. For LIBERO we use <code>10</code>. This LeRobot port overrides the loaded checkpoint’s <code>max_action_horizon</code> with this value.</li> <li><code>policy.n_action_steps</code>: number of actions consumed from each predicted
chunk before querying the policy again. For LIBERO, set it to <code>chunk_size</code>.</li> <li><code>policy.setup_type</code>: text inserted into the prompt to describe the robot and
scene, e.g. <code>single franka robotic arm in libero</code>. More examples are listed
in the <code>metadata_by_tag</code> entries of <a href="https://huggingface.co/allenai/MolmoAct2/blob/main/norm_stats.json" rel="nofollow"><code>norm_stats.json</code></a>.</li> <li><code>policy.control_mode</code>: text inserted into the prompt to describe the action
space, e.g. <code>delta end-effector pose</code> or <code>absolute joint pose</code>.</li> <li><code>policy.image_keys</code>: ordered LeRobot image observation keys passed to the
processor.</li> <li><code>policy.model_dtype</code>: checkpoint/forward dtype, one of <code>float32</code>, <code>bfloat16</code>, or <code>float16</code>. Use <code>bfloat16</code> for normal training.</li> <li><code>policy.num_flow_timesteps</code>: number of flow-matching timesteps sampled per
example during training. We use <code>8</code> for fine-tuning.</li> <li><code>policy.num_inference_steps</code>: optional override for continuous action
generation steps at inference time.</li> <li><code>policy.gradient_checkpointing</code>: enables checkpointing in the VLM/action path
to reduce activation memory.</li> <li><code>policy.freeze_embedding</code>: freezes input embeddings. The default is <code>true</code>.</li> <li><code>policy.normalize_gripper</code>: controls whether gripper dimensions are included
in state/action quantile normalization. The default is <code>false</code>.</li> <li><code>policy.normalize_language</code>: normalizes task strings before prompt
construction. The default is <code>true</code>.</li> <li><code>policy.mask_action_dim_padding</code>: masks padded dimensions in the flow loss.
Released checkpoints use <code>policy.expected_max_action_dim=32</code>.</li> <li><code>policy.max_sequence_length</code>: optional manual sequence cap. Leave unset to
infer it from images, state dimension, action dimension, action horizon, and
discrete-action mode.</li></ul> <!--[2--><h3 class="relative group"><a id="learning-rates" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#learning-rates"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Learning Rates</span></h3><!--]--><!----> <p>MolmoAct2 uses parameter-group learning rates to match the original MolmoAct2
fine-tuning experiments.</p> <ul><li>Full fine-tuning uses <code>policy.optimizer_lr=1e-5</code> for the VLM, <code>policy.optimizer_vit_lr=5e-6</code> for the vision tower, <code>policy.optimizer_connector_lr=5e-6</code> for image connector layers, and <code>policy.optimizer_action_expert_lr=5e-5</code> for the action expert.</li> <li>LoRA VLM fine-tuning sets the VLM, vision, and connector LoRA parameter
groups to <code>5e-5</code> when <code>policy.enable_lora_vlm=true</code>. By default, <code>policy.enable_lora_action_expert=false</code>, so the action expert is still fully
fine-tuned with <code>policy.optimizer_action_expert_lr</code>. If <code>policy.enable_lora_action_expert=true</code>, the action expert is trained through
LoRA adapters instead.</li> <li>Action-expert-only fine-tuning trains only the action expert and uses <code>policy.optimizer_action_expert_lr=5e-5</code>.</li></ul> <p>You can override the full fine-tuning and action-expert learning rates with <code>policy.optimizer_lr</code>, <code>policy.optimizer_vit_lr</code>, <code>policy.optimizer_connector_lr</code>, and <code>policy.optimizer_action_expert_lr</code>.
Scheduler settings can be changed with <code>policy.scheduler_warmup_steps</code>, <code>policy.scheduler_decay_steps</code>, and <code>policy.scheduler_decay_lr</code>.</p> <!--[2--><h3 class="relative group"><a id="dataset-quantile-statistics" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#dataset-quantile-statistics"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Dataset Quantile Statistics</span></h3><!--]--><!----> <p>MolmoAct2 defaults to quantile normalization for state and action features. If
your dataset has not been converted with quantile statistics, you can add them
with:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->python src/lerobot/scripts/augment_dataset_quantile_stats.py \
--repo-id=your_dataset<!----></pre></div><!----> <p>Recording, resuming, and merging aggregate quantiles from per-episode summaries, so <code>meta/stats.json</code> ends up holding a conservative envelope (<code>min</code> for <code>q &lt;= 50</code>, <code>max</code> for <code>q > 50</code>) rather than whole-dataset quantiles. To estimate the latter, scan every episode with a running histogram:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->python src/lerobot/scripts/augment_dataset_quantile_stats.py \
--repo-id=your_dataset \
--overwrite \
--skip-images<!----></pre></div><!----> <p><code>--skip-images</code> keeps the existing image statistics and avoids video decoding when only <code>STATE</code>/<code>ACTION</code> need recomputing, and <code>--root</code> reads a local dataset instead of the Hub. These values are histogram estimates, subject to discretization and rebinning error, so they can differ from the conservative ones — which changes MolmoAct2’s normalized targets and therefore its loss scale. Statistics already saved inside an existing checkpoint are not affected.</p> <p>Alternatively, train MolmoAct2 with mean/std normalization:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->--policy.normalization_mapping=<span class="hljs-string">&#x27;{&quot;ACTION&quot;: &quot;MEAN_STD&quot;, &quot;STATE&quot;: &quot;MEAN_STD&quot;, &quot;VISUAL&quot;: &quot;IDENTITY&quot;}&#x27;</span><!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="evaluation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#evaluation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Evaluation</span></h2><!--]--><!----> <p>Evaluation also supports both LeRobot-saved checkpoints and original MolmoAct2
HF checkpoints. For LIBERO replication, keep the EGL rendering environment
fixed and use <code>policy.per_episode_seed=true</code>.</p> <p><strong>Important:</strong> We found that <code>num_steps_wait=10</code> does not reliably let the
LIBERO scene stabilize and can degrade measured success. All LIBERO evaluation
results reported here use <code>num_steps_wait=50</code>.</p> <!--[2--><h3 class="relative group"><a id="evaluation-with-lerobot-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#evaluation-with-lerobot-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Evaluation With LeRobot MolmoAct2 Weight</span></h3><!--]--><!----> <p>Use <code>policy.path</code> for a checkpoint saved by LeRobot. The saved processor and
normalization statistics are restored together with the model.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!----><span class="hljs-built_in">export</span> MUJOCO_GL=egl
<span class="hljs-built_in">export</span> PYOPENGL_PLATFORM=egl
<span class="hljs-built_in">export</span> OMP_NUM_THREADS=1
<span class="hljs-built_in">export</span> MKL_NUM_THREADS=1
lerobot-eval \
--policy.path=allenai/MolmoAct2-LIBERO-LeRobot \
--policy.inference_action_mode=continuous \
--policy.model_dtype=bfloat16 \
--policy.use_amp=<span class="hljs-literal">true</span> \
--policy.enable_inference_cuda_graph=<span class="hljs-literal">true</span> \
--policy.device=cuda \
--policy.per_episode_seed=<span class="hljs-literal">true</span> \
--policy.eval_seed=1000 \
--env.type=libero \
--env.task=libero_10,libero_goal,libero_object,libero_spatial \
--env.camera_name_mapping=<span class="hljs-string">&#x27;{&quot;agentview_image&quot;:&quot;image&quot;,&quot;robot0_eye_in_hand_image&quot;:&quot;wrist_image&quot;}&#x27;</span> \
--eval.batch_size=1 \
--eval.n_episodes=50 \
--seed=1000<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="evaluation-with-original-molmoact2-weight" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#evaluation-with-original-molmoact2-weight"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Evaluation With Original MolmoAct2 Weight</span></h3><!--]--><!----> <p>You can evaluate a released Hugging Face checkpoint directly without first
converting it to a LeRobot checkpoint. In this case, set <code>policy.checkpoint_path</code> to the HF model repo and provide <code>policy.norm_tag</code>.
For LIBERO, <code>policy.norm_tag=libero</code> loads the LIBERO action/state
normalization statistics, action horizon, prompt metadata, and image-key order
from the checkpoint’s <code>norm_stats.json</code>.</p> <p>To fully replicate the MolmoAct2 paper results with released Hugging Face
checkpoints, we recommend using the v0.5.1-pinned <a href="https://github.com/allenai/lerobot/tree/molmoact2-hf-inference" rel="nofollow"><code>allenai/lerobot</code> <code>molmoact2-hf-inference</code></a> branch. That branch matches the original evaluation settings used for the
reported numbers.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!----><span class="hljs-built_in">export</span> MUJOCO_GL=egl
<span class="hljs-built_in">export</span> PYOPENGL_PLATFORM=egl
<span class="hljs-built_in">export</span> OMP_NUM_THREADS=1
<span class="hljs-built_in">export</span> MKL_NUM_THREADS=1
lerobot-eval \
--policy.type=molmoact2 \
--policy.checkpoint_path=allenai/MolmoAct2-LIBERO \
--policy.norm_tag=libero \
--policy.inference_action_mode=continuous \
--policy.model_dtype=float32 \
--policy.use_amp=<span class="hljs-literal">false</span> \
--policy.enable_inference_cuda_graph=<span class="hljs-literal">true</span> \
--policy.device=cuda \
--policy.per_episode_seed=<span class="hljs-literal">true</span> \
--policy.eval_seed=1000 \
--env.type=libero \
--env.task=libero_goal \
--env.camera_name_mapping=<span class="hljs-string">&#x27;{&quot;agentview_image&quot;:&quot;image&quot;,&quot;robot0_eye_in_hand_image&quot;:&quot;wrist_image&quot;}&#x27;</span> \
--eval.batch_size=1 \
--eval.n_episodes=50 \
--seed=1000<!----></pre></div><!----> <p>Use <code>--env.task=libero_10,libero_goal,libero_object,libero_spatial</code> to run the
full LIBERO suite. The same command works for other released MolmoAct2
checkpoints as long as the requested <code>policy.norm_tag</code> exists in that
checkpoint’s <code>norm_stats.json</code>.</p> <!--[2--><h3 class="relative group"><a id="common-evaluation-options" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-evaluation-options"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Evaluation Options</span></h3><!--]--><!----> <ul><li><code>policy.inference_action_mode</code>: required for rollout. Use <code>continuous</code> for
flow-matching inference or <code>discrete</code> for action-token inference. It must be
compatible with the training-time <code>policy.action_mode</code> saved in the
checkpoint.</li> <li><code>policy.path</code>: LeRobot checkpoint path or Hub repo. Use this for checkpoints
saved by LeRobot.</li> <li><code>policy.checkpoint_path</code>: original MolmoAct2 HF checkpoint path or Hub repo.
Use this with <code>policy.type=molmoact2</code> and <code>policy.norm_tag</code>.</li> <li><code>policy.norm_tag</code>: selects normalization statistics, prompt metadata,
image-key order, and action horizon from the original checkpoint’s <code>norm_stats.json</code>. It is required for direct original-HF checkpoint
evaluation.</li> <li><code>policy.model_dtype</code>: model load/forward dtype. Use <code>bfloat16</code> for normal
GPU evaluation. Use <code>float32</code> only when you explicitly want fp32 inference.</li> <li><code>policy.use_amp</code>: runs the policy forward under autocast during eval. For <code>model_dtype=bfloat16</code>, keep this enabled.</li> <li><code>policy.enable_inference_cuda_graph</code>: enables the MolmoAct2 inference CUDA
graph path for faster repeated continuous-action rollout.</li> <li><code>policy.per_episode_seed</code> and <code>policy.eval_seed</code>: make stochastic continuous
action generation deterministic per episode for replication.</li> <li><code>env.task</code>: comma-separated LIBERO suites or a single suite. Use <code>libero_10,libero_goal,libero_object,libero_spatial</code> for the full benchmark.</li> <li><code>env.camera_name_mapping</code>: maps LIBERO camera names to the image keys expected
by the policy processor.</li></ul> <!--[1--><h2 class="relative group"><a id="performance-results" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#performance-results"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Performance Results</span></h2><!--]--><!----> <!--[2--><h3 class="relative group"><a id="libero-benchmark-results" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#libero-benchmark-results"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>LIBERO Benchmark Results</span></h3><!--]--><!----> <p>MolmoAct2 has demonstrated strong performance on the LIBERO benchmark suite. To
compare and test its LeRobot implementation, we fine-tuned <a href="https://huggingface.co/allenai/MolmoAct2-LIBERO" rel="nofollow"><code>allenai/MolmoAct2-LIBERO</code></a> for an additional 10k steps on the LIBERO dataset with per-GPU batch size 32 on
8 H100 GPUs, then compared the results to the original MolmoAct2 reference
results.</p> <p>The LeRobot fine-tuned checkpoint reported here is available at <a href="https://huggingface.co/allenai/MolmoAct2-LIBERO-LeRobot" rel="nofollow"><code>allenai/MolmoAct2-LIBERO-LeRobot</code></a> and was trained on <a href="https://huggingface.co/datasets/allenai/MolmoAct2-LIBERO-Dataset" rel="nofollow"><code>allenai/MolmoAct2-LIBERO-Dataset</code></a>.</p> <table><thead><tr><th>Benchmark</th><th align="right">LeRobot Implementation</th><th align="right">MolmoAct2 Original</th></tr></thead><tbody><tr><td>LIBERO Spatial</td><td align="right">98.4%</td><td align="right">97.8%</td></tr><tr><td>LIBERO Object</td><td align="right">100.0%</td><td align="right">100.0%</td></tr><tr><td>LIBERO Goal</td><td align="right">98.0%</td><td align="right">97.8%</td></tr><tr><td>LIBERO 10</td><td align="right">96.6%</td><td align="right">93.2%</td></tr><tr><td>Average</td><td align="right">98.25%</td><td align="right">97.20%</td></tr></tbody></table> <p>These results demonstrate MolmoAct2’s strong performance across diverse robotic
manipulation tasks. To reproduce them, follow the instructions in the LIBERO
evaluation section.</p> <!--[1--><h2 class="relative group"><a id="hardware-deployment-lerobot-rollout" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#hardware-deployment-lerobot-rollout"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Hardware Deployment (lerobot-rollout)</span></h2><!--]--><!----> <p>LeRobot-format checkpoints are available on the Hub for direct use with <code>lerobot-rollout</code>. Each checkpoint uses specific camera names that must
match your robot’s camera configuration.</p> <!--[2--><h3 class="relative group"><a id="camera-naming-convention" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#camera-naming-convention"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Camera naming convention</span></h3><!--]--><!----> <p>Each checkpoint expects specific <code>observation.images.*</code> keys.
If your robot cameras have different names, use <code>--rename_map</code> to map them:</p> <table><thead><tr><th>Checkpoint</th><th>Camera keys</th><th>Description</th></tr></thead><tbody><tr><td>MolmoAct2-LIBERO-LeRobot</td><td><code>image</code>, <code>wrist_image</code></td><td>LIBERO sim cameras</td></tr><tr><td>MolmoAct2-BimanualYAM-LeRobot</td><td><code>top</code>, <code>left</code>, <code>right</code></td><td>YAM 3-camera setup</td></tr><tr><td>MolmoAct2-DROID-LeRobot</td><td><code>cam0</code>, <code>cam1</code></td><td>External + wrist</td></tr><tr><td>MolmoAct2-SO100_101-LeRobot</td><td><code>cam0</code>, <code>cam1</code></td><td>Primary + secondary view</td></tr></tbody></table> <p>Example with an SO-100 robot using top and side cameras:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->lerobot-rollout \
--policy.path=lerobot/MolmoAct2-SO100_101-LeRobot \
--rename_map=<span class="hljs-string">&#x27;{&quot;observation.images.top&quot;: &quot;observation.images.cam0&quot;, &quot;observation.images.side&quot;: &quot;observation.images.cam1&quot;}&#x27;</span> \
--robot.type=so100_follower \
--robot.port=/dev/ttyACM0 \
--robot.cameras=<span class="hljs-string">&#x27;{
top: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30},
side: {type: opencv, index_or_path: 2, width: 640, height: 480, fps: 30}
}&#x27;</span> \
--task=<span class="hljs-string">&quot;pick up the red cube&quot;</span> --duration=30<!----></pre></div><!----> <p>To use a wrist camera instead, just change the rename mapping:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->--rename_map=<span class="hljs-string">&#x27;{&quot;observation.images.top&quot;: &quot;observation.images.cam0&quot;, &quot;observation.images.wrist&quot;: &quot;observation.images.cam1&quot;}&#x27;</span><!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="joint-frame-transform-so-100101-zero-shot" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#joint-frame-transform-so-100101-zero-shot"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Joint frame transform (SO-100/101 zero-shot)</span></h3><!--]--><!----> <blockquote class="warning"><!---->The MolmoAct2-SO100_101 checkpoint was trained on data that uses a different
joint calibration convention than LeRobot >= 0.5.0. Without a frame
correction, the arm may move in the wrong direction. <p>This affects both <strong>zero-shot deployment</strong> and <strong>fine-tuning</strong> from the
original checkpoint. The pretrained weights expect the old convention, so
all joint data (observations and actions) must be transformed to match.</p> <p>The converted LeRobot checkpoint (<code>lerobot/MolmoAct2-SO100_101-LeRobot</code>)
already includes this correction in its processor pipeline. If you convert
or fine-tune the checkpoint yourself, set the following in the policy config (<code>configuration_molmoact2.py</code>):</p> <ul><li><code>joint_signs</code>: <code>[1, -1, 1, 1, 1, 1]</code> (flips shoulder_lift direction)</li> <li><code>joint_offsets</code>: <code>[0, 90, 90, 0, 0, 0]</code> (shifts shoulder_lift and elbow_flex by 90°)</li></ul> <p>See the <a href="./backwardcomp">backward compatibility guide</a> for details on the
calibration change.</p><!----></blockquote><!----> <!--[1--><h2 class="relative group"><a id="differences-from-the-original-implementation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#differences-from-the-original-implementation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Differences From the Original Implementation</span></h2><!--]--><!----> <p>This LeRobot port is intended to match MolmoAct2 behavior while using LeRobot’s
dataset, training, evaluation, checkpoint, and logging infrastructure. The main
differences from the original training repository are:</p> <ul><li>The original paper training stack loads the model in fp32 and trains under
mixed precision. This LeRobot port usually loads the checkpoint directly in <code>policy.model_dtype=bfloat16</code> for lower memory use.</li> <li>The original repository uses its own FSDP/model-parallel training path. The
LeRobot port uses the standard LeRobot/Accelerate training path and has not
been tested for multi-node training.</li> <li>The original repository supports sequence packing. The LeRobot port trains on
one LeRobot sample per item and pads to an inferred fixed sequence budget.</li> <li>The LeRobot port follows LeRobot’s optimizer, scheduler, checkpoint saving,
dataset transforms, image augmentation, and Weights &amp; Biases logging
conventions.</li> <li>The original training path supports mixed action horizons by padding to <code>max_action_horizon</code> and masking padded horizon slots in the action expert
self-attention. This is useful when training across datasets with different
control frequencies. The LeRobot port currently targets single-dataset
fine-tuning, so <code>policy.chunk_size</code> overrides the checkpoint <code>max_action_horizon</code> and horizon masking is not implemented yet. Support for
this mixed-horizon path is planned.</li></ul> <!--[1--><h2 class="relative group"><a id="citation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#citation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Citation</span></h2><!--]--><!----> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bibtex "><!----><span class="hljs-comment">@misc{fang2026molmoact2actionreasoningmodels,</span>
title={MolmoAct2: Action Reasoning Models for Real-world Deployment},
author={Haoquan Fang <span class="hljs-keyword">and</span> Jiafei Duan <span class="hljs-keyword">and</span> Donovan Clay <span class="hljs-keyword">and</span> Sam Wang <span class="hljs-keyword">and</span> Shuo Liu <span class="hljs-keyword">and</span> Weikai Huang <span class="hljs-keyword">and</span> Xiang Fan <span class="hljs-keyword">and</span> Wei-Chuan Tsai <span class="hljs-keyword">and</span> Shirui Chen <span class="hljs-keyword">and</span> Yi Ru Wang <span class="hljs-keyword">and</span> Shanli Xing <span class="hljs-keyword">and</span> Jaemin Cho <span class="hljs-keyword">and</span> Jae Sung Park <span class="hljs-keyword">and</span> Ainaz Eftekhar <span class="hljs-keyword">and</span> Peter Sushko <span class="hljs-keyword">and</span> Karen Farley <span class="hljs-keyword">and</span> Angad Wadhwa <span class="hljs-keyword">and</span> Cole Harrison <span class="hljs-keyword">and</span> Winson Han <span class="hljs-keyword">and</span> Ying-Chun Lee <span class="hljs-keyword">and</span> Eli VanderBilt <span class="hljs-keyword">and</span> Rose Hendrix <span class="hljs-keyword">and</span> Suveen Ellawela <span class="hljs-keyword">and</span> Lucas Ngoo <span class="hljs-keyword">and</span> Joyce Chai <span class="hljs-keyword">and</span> Zhongzheng Ren <span class="hljs-keyword">and</span> Ali Farhadi <span class="hljs-keyword">and</span> Dieter Fox <span class="hljs-keyword">and</span> Ranjay Krishna},
year={<span class="hljs-number">2026</span>},
eprint={<span class="hljs-number">2605</span>.<span class="hljs-number">02881</span>},
archivePrefix={arXiv},
primaryClass={cs.RO},
url={https:<span class="hljs-comment">//arxiv.org/abs/2605.02881},</span>
}<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="license" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#license"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>License</span></h2><!--]--><!----> <p>This model is licensed under Apache 2.0. It is intended for research and
educational use in accordance with <a href="https://allenai.org/responsible-use" rel="nofollow">Ai2’s Responsible Use Guidelines</a>,
consistent with <a href="https://github.com/allenai/molmoact2" rel="nofollow">allenai/molmoact2</a>.</p> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/lerobot/blob/main/docs/source/molmoact2.mdx" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]-->
<script>
{
__sveltekit_tk9oje = {
base: "/docs/lerobot/main/en",
assets: "/docs/lerobot/main/en"
};
const element = document.currentScript.parentElement;
Promise.all([
import("/docs/lerobot/main/en/_app/immutable/entry/start.Dk__j1Q3.js"),
import("/docs/lerobot/main/en/_app/immutable/entry/app.-pyB7NrA.js")
]).then(([kit, app]) => {
kit.start(app, element, {
node_ids: [0, 2],
data: [null,null],
form: null,
error: null
});
});
}
</script>

Xet Storage Details

Size:
78.6 kB
·
Xet hash:
5dd9b6449110121752c3d44b443aedc1047f222b89a1e008b795e90343226775

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.