Buckets:
| <meta charset="utf-8" /><meta name="hf:doc:metadata" content="{"title":"Cosmos 3","local":"cosmos-3","sections":[{"title":"What’s new in Cosmos 3","local":"whats-new-in-cosmos-3","sections":[],"depth":2},{"title":"Available checkpoints","local":"available-checkpoints","sections":[],"depth":2},{"title":"Prompt upsampling","local":"prompt-upsampling","sections":[],"depth":2},{"title":"Text-to-video","local":"text-to-video","sections":[],"depth":2},{"title":"Text-to-image","local":"text-to-image","sections":[],"depth":2},{"title":"Image-to-video","local":"image-to-video","sections":[],"depth":2},{"title":"Video-to-video","local":"video-to-video","sections":[],"depth":2},{"title":"Video-to-video with sound","local":"video-to-video-with-sound","sections":[],"depth":2},{"title":"Text-to-video with sound","local":"text-to-video-with-sound","sections":[],"depth":2},{"title":"Action-conditioned generation","local":"action-conditioned-generation","sections":[{"title":"Action policy","local":"action-policy","sections":[],"depth":3}],"depth":2},{"title":"Context parallelism","local":"context-parallelism","sections":[{"title":"Run it","local":"run-it","sections":[],"depth":3},{"title":"Fitting large models with tensor parallelism","local":"fitting-large-models-with-tensor-parallelism","sections":[],"depth":3},{"title":"Use it in your own modular pipeline","local":"use-it-in-your-own-modular-pipeline","sections":[],"depth":3}],"depth":2},{"title":"Metadata templates","local":"metadata-templates","sections":[],"depth":2},{"title":"Safety checker","local":"safety-checker","sections":[],"depth":2},{"title":"Cosmos3OmniPipeline","local":"diffusers.Cosmos3OmniPipeline","sections":[],"depth":2},{"title":"Cosmos3OmniModularPipeline","local":"cosmos3omnimodularpipeline","sections":[{"title":"Modular examples for all existing workflows","local":"modular-examples-for-all-existing-workflows","sections":[],"depth":3},{"title":"Modular transfer (structural control)","local":"modular-transfer-structural-control","sections":[],"depth":3},{"title":"Distilled (few-step) text-to-image and image-to-video","local":"diffusers.Cosmos3OmniModularPipeline","sections":[],"depth":3}],"depth":2},{"title":"Cosmos3DistilledModularPipeline","local":"diffusers.Cosmos3DistilledModularPipeline","sections":[],"depth":2},{"title":"CosmosActionCondition","local":"diffusers.CosmosActionCondition","sections":[],"depth":2},{"title":"Cosmos3OmniPipelineOutput","local":"diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput","sections":[],"depth":2}],"depth":1}"/> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/entry/start.Cpc5Vo8y.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/DPK2mSIa.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/jDjavuwI.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/entry/app.DMayJ2U5.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/CMF6C6Mt.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/LxvcAq1p.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/i0LL_sTA.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/nodes/0.BYHb3kF5.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/9Uo8d8lw.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/nodes/147.BpYtScoZ.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/BtE7mKSK.js" rel="modulepreload"> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/chunks/BoRAdXUU.js" rel="modulepreload"> | |
| <!--if5dit--><meta name="hf:doc:metadata" content="{"title":"Cosmos 3","local":"cosmos-3","sections":[{"title":"What’s new in Cosmos 3","local":"whats-new-in-cosmos-3","sections":[],"depth":2},{"title":"Available checkpoints","local":"available-checkpoints","sections":[],"depth":2},{"title":"Prompt upsampling","local":"prompt-upsampling","sections":[],"depth":2},{"title":"Text-to-video","local":"text-to-video","sections":[],"depth":2},{"title":"Text-to-image","local":"text-to-image","sections":[],"depth":2},{"title":"Image-to-video","local":"image-to-video","sections":[],"depth":2},{"title":"Video-to-video","local":"video-to-video","sections":[],"depth":2},{"title":"Video-to-video with sound","local":"video-to-video-with-sound","sections":[],"depth":2},{"title":"Text-to-video with sound","local":"text-to-video-with-sound","sections":[],"depth":2},{"title":"Action-conditioned generation","local":"action-conditioned-generation","sections":[{"title":"Action policy","local":"action-policy","sections":[],"depth":3}],"depth":2},{"title":"Context parallelism","local":"context-parallelism","sections":[{"title":"Run it","local":"run-it","sections":[],"depth":3},{"title":"Fitting large models with tensor parallelism","local":"fitting-large-models-with-tensor-parallelism","sections":[],"depth":3},{"title":"Use it in your own modular pipeline","local":"use-it-in-your-own-modular-pipeline","sections":[],"depth":3}],"depth":2},{"title":"Metadata templates","local":"metadata-templates","sections":[],"depth":2},{"title":"Safety checker","local":"safety-checker","sections":[],"depth":2},{"title":"Cosmos3OmniPipeline","local":"diffusers.Cosmos3OmniPipeline","sections":[],"depth":2},{"title":"Cosmos3OmniModularPipeline","local":"cosmos3omnimodularpipeline","sections":[{"title":"Modular examples for all existing workflows","local":"modular-examples-for-all-existing-workflows","sections":[],"depth":3},{"title":"Modular transfer (structural control)","local":"modular-transfer-structural-control","sections":[],"depth":3},{"title":"Distilled (few-step) text-to-image and image-to-video","local":"diffusers.Cosmos3OmniModularPipeline","sections":[],"depth":3}],"depth":2},{"title":"Cosmos3DistilledModularPipeline","local":"diffusers.Cosmos3DistilledModularPipeline","sections":[],"depth":2},{"title":"CosmosActionCondition","local":"diffusers.CosmosActionCondition","sections":[],"depth":2},{"title":"Cosmos3OmniPipelineOutput","local":"diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput","sections":[],"depth":2}],"depth":1}"/><!----> | |
| <link href="/docs/diffusers/pr_14421/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <!--[0--><h1 class="relative group"><a id="cosmos-3" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#cosmos-3"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Cosmos 3</span></h1><!--]--><!----> <p>NVIDIA Cosmos 3 is a unified world foundation model (WFM) for Physical AI — a single omni-model that combines world generation, physical reasoning, and action generation. It replaces the separate Predict, Reason, and Transfer models from earlier Cosmos releases: whether you’re building for robotics, autonomous vehicles, or smart spaces, Cosmos 3 gives you one foundation to simulate and understand the physical world.</p> <p>What’s shipping with this release:</p> <ul><li>Models on the Hugging Face Hub with model cards and licensing</li> <li>Cosmos 3 Diffusers integration for generation pipelines (this page)</li> <li>Post-training scripts for fine-tuning Cosmos 3 on your own data</li> <li>Open synthetic data generation (SDG) datasets for Physical AI</li></ul> <!--[1--><h2 class="relative group"><a id="whats-new-in-cosmos-3" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#whats-new-in-cosmos-3"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>What’s new in Cosmos 3</span></h2><!--]--><!----> <p>The biggest change from previous Cosmos releases is that Cosmos 3 is an <em>omni-model</em>, built on a Mixture-of-Transformers (MoT) architecture. Previously, developers worked with separate models for world generation (Predict), controlled generation (Transfer), scene understanding (Reason), and action-policy generation. Cosmos 3 unifies all of these in one model that reasons and generates across modalities in a single forward pass.</p> <p>From one model you can:</p> <ul><li>Generate physically plausible video worlds from text, images, or action inputs (image-to-video, text-to-video, action-conditioned video generation).</li> <li>Reason about physical properties like motion, causality, and spatial relationships.</li> <li>Predict future video and action sequences from the current state.</li> <li>Transfer scenes across viewpoints and conditions with structural control <em>(coming soon)</em>.</li></ul> <p>Under the hood, a single <code>Cosmos3OmniTransformer</code> runs a Qwen-style language model in parallel with a diffusion generation pathway: text tokens flow through a causal “understanding” stream while video and sound latents flow through a bi-directionally-attended “generation” stream, joined by a 3D multimodal RoPE. See the <a href="https://huggingface.co/papers/2501.03575" rel="nofollow">Cosmos World Foundation Model Platform paper</a> for the architectural background.</p> <!--[1--><h2 class="relative group"><a id="available-checkpoints" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#available-checkpoints"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Available checkpoints</span></h2><!--]--><!----> <p>Two checkpoints are released on the Hub — <a href="https://huggingface.co/nvidia/Cosmos3-Nano" rel="nofollow"><code>nvidia/Cosmos3-Nano</code></a> (smaller, faster) and <a href="https://huggingface.co/nvidia/Cosmos3-Super" rel="nofollow"><code>nvidia/Cosmos3-Super</code></a> (larger, higher quality). The same pipeline class supports text-to-image, text-to-video, image-to-video, and (with a sound-capable checkpoint) text+image-to-video-with-sound — pick a repo and use the per-model tab in each workflow below.</p> <blockquote class="tip"><p>Make sure to check out the Schedulers <a href="../../using-diffusers/schedulers">guide</a> to learn how to explore the tradeoff between scheduler speed and quality, and see the <a href="../../using-diffusers/loading#reuse-a-pipeline">reuse components across pipelines</a> section to learn how to efficiently load the same components into multiple pipelines.</p></blockquote> <!--[1--><h2 class="relative group"><a id="prompt-upsampling" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#prompt-upsampling"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Prompt upsampling</span></h2><!--]--><!----> <p>Cosmos 3 was trained on long, highly descriptive captions. For optimal quality, short text prompts should be <strong>upsampled into a specific JSON structure</strong> before they are passed to the pipeline. The upsampler lives in the <a href="https://github.com/NVIDIA/cosmos-framework" rel="nofollow">cosmos-framework</a> package.</p> <p>Start from a short, plain-text prompt and save it to <code>assets/prompt.txt</code>. For the text-to-video example below, the original prompt is <em>“A robotic arm is cleaning a plate in a kitchen”</em>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!----><span class="hljs-built_in">mkdir</span> -p assets | |
| <span class="hljs-built_in">echo</span> <span class="hljs-string">"A robotic arm is cleaning a plate in a kitchen"</span> > assets/prompt.txt<!----></pre></div><!----> <p>Then install the framework and run the upsampler. The example below upsamples for text-to-video using Opus-4.6:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->git <span class="hljs-built_in">clone</span> https://github.com/NVIDIA/cosmos-framework.git packages/cosmos-framework | |
| pip install -e packages/cosmos-framework | |
| <span class="hljs-built_in">export</span> PROMPT_UPSAMPLER_ENDPOINT_URL=<span class="hljs-string">"https://api.anthropic.com/v1/"</span> | |
| <span class="hljs-built_in">export</span> PROMPT_UPSAMPLER_MODEL_NAME=<span class="hljs-string">"claude-opus-4-6"</span> | |
| <span class="hljs-built_in">export</span> PROMPT_UPSAMPLER_API_TOKEN=<span class="hljs-string">"<your_token>"</span> | |
| python -m cosmos_framework.inference.prompt_upsampling \ | |
| --input assets/prompt.txt \ | |
| --output assets/example_t2v_prompt.json \ | |
| --mode text2video \ | |
| --endpoint-url <span class="hljs-string">"<span class="hljs-variable">${PROMPT_UPSAMPLER_ENDPOINT_URL}</span>"</span> \ | |
| --model <span class="hljs-string">"<span class="hljs-variable">${PROMPT_UPSAMPLER_MODEL_NAME}</span>"</span> \ | |
| --api-token <span class="hljs-string">"<span class="hljs-variable">${PROMPT_UPSAMPLER_API_TOKEN}</span>"</span> \ | |
| --resolution 720 \ | |
| --aspect-ratio <span class="hljs-string">"16,9"</span><!----></pre></div><!----> <p>Switch <code>--mode</code> to match the workflow you are targeting (<code>text2image</code>, <code>text2video</code>, <code>image2video</code>). The command writes the upsampled prompt(s) to the <code>--output</code> file as a JSON array (one object per non-empty line in <code>--input</code>); pass a <code>.jsonl</code> path instead to get one JSON object per line. For <code>image2video</code>, you must also supply the conditioning image via <code>--image-url</code> (a URL or local path) or <code>--image-list</code> (one image per prompt).</p> <p>A pre-upsampled positive prompt (<code>assets/example_t2v_prompt.json</code>) and negative prompt (<code>assets/negative_prompt.json</code>) are provided for convenience, and are used by the generation examples below. The examples load these JSON files and pass them to the pipeline as JSON strings via <code>json.dumps(...)</code>.</p> <!--[1--><h2 class="relative group"><a id="text-to-video" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#text-to-video"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Text-to-video</span></h2><!--]--><!----> <p>Multi-frame generation conditioned on text alone. Pick <code>num_frames</code> based on the target duration — the default <code>num_frames=189</code> produces ≈ 7.9 s at 24 FPS. The prompt and negative prompt are read from the JSON-upsampled files described in <a href="#prompt-upsampling">Prompt upsampling</a>.</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| <span class="hljs-comment"># JSON-upsampled positive and negative prompts (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt.json"</span>)) | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| pipe.scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| result = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| num_inference_steps=<span class="hljs-number">35</span>, | |
| guidance_scale=<span class="hljs-number">6.0</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| ) | |
| <span class="hljs-comment"># macro_block_size=1 allows arbitrary frame sizes (Cosmos3 outputs are not always divisible by 16).</span> | |
| export_to_video(result.video, <span class="hljs-string">"cosmos3_t2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>)<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="text-to-image" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#text-to-image"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Text-to-image</span></h2><!--]--><!----> <p>Single-frame generation. The model is conditioned only on the text prompt; pass <code>num_frames=1</code>. Upsample with <code>--mode text2image</code> to produce the JSON prompt.</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| <span class="hljs-comment"># JSON-upsampled prompt (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2i_prompt.json"</span>)) | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| result = pipe(prompt=json.dumps(json_prompt), num_frames=<span class="hljs-number">1</span>, height=<span class="hljs-number">720</span>, width=<span class="hljs-number">1280</span>) | |
| result.video[<span class="hljs-number">0</span>].save(<span class="hljs-string">"cosmos3_t2i.jpg"</span>, <span class="hljs-built_in">format</span>=<span class="hljs-string">"JPEG"</span>, quality=<span class="hljs-number">85</span>)<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="image-to-video" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#image-to-video"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Image-to-video</span></h2><!--]--><!----> <p>Pass a conditioning image via <code>image=</code>. The pipeline anchors frame 0 to the supplied image and denoises the rest. Upsample with <code>--mode image2video</code> to produce the JSON prompt.</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_image | |
| <span class="hljs-comment"># JSON-upsampled positive and negative prompts (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_i2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt_i2v.json"</span>)) | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| image = load_image( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/releases/download/assets/robot_153.jpg"</span> | |
| ) | |
| result = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| image=image, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| ) | |
| <span class="hljs-comment"># macro_block_size=1 allows arbitrary frame sizes (Cosmos3 outputs are not always divisible by 16).</span> | |
| export_to_video(result.video, <span class="hljs-string">"cosmos3_i2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>)<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="video-to-video" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#video-to-video"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Video-to-video</span></h2><!--]--><!----> <p>Pass a conditioning clip via <code>video=</code> (e.g. from <code>load_video</code>). The pipeline anchors the leading latent frames given by <code>condition_frame_indexes_vision</code> (default <code>[0, 1]</code>) to the clip and denoises the rest. Use <code>condition_video_keep</code> (<code>"first"</code> or <code>"last"</code>) to choose which end of a longer source clip the conditioning frames are taken from. As with the other modes, the prompt should follow the descriptive JSON structure described in <a href="#prompt-upsampling">Prompt upsampling</a>.</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_video | |
| <span class="hljs-comment"># JSON-upsampled positive and negative prompts (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_v2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt_i2v.json"</span>)) | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| pipe.scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| video = load_video( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/raw/refs/heads/assets/cosmos3/inputs/vision/robot_pouring.mp4"</span> | |
| ) | |
| result = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| video=video, | |
| condition_frame_indexes_vision=[<span class="hljs-number">0</span>, <span class="hljs-number">1</span>], | |
| condition_video_keep=<span class="hljs-string">"first"</span>, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| num_inference_steps=<span class="hljs-number">35</span>, | |
| guidance_scale=<span class="hljs-number">6.0</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| ) | |
| <span class="hljs-comment"># macro_block_size=1 allows arbitrary frame sizes (Cosmos3 outputs are not always divisible by 16).</span> | |
| export_to_video(result.video, <span class="hljs-string">"cosmos3_v2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>)<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="video-to-video-with-sound" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#video-to-video-with-sound"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Video-to-video with sound</span></h2><!--]--><!----> <p>When the checkpoint carries a <code>sound_tokenizer</code>, add <code>enable_sound=True</code> to the video-to-video call to jointly generate a synchronized audio track. The waveform is returned alongside the video and can be muxed into the MP4 with <a href="/docs/diffusers/pr_14421/en/api/utilities#diffusers.utils.encode_video">encode_video()</a>.</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> encode_video, load_video | |
| <span class="hljs-comment"># JSON-upsampled positive and negative prompts (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_v2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt_i2v.json"</span>)) | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| pipe.scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| video = load_video( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/raw/refs/heads/assets/cosmos3/inputs/vision/robot_pouring.mp4"</span> | |
| ) | |
| result = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| video=video, | |
| condition_frame_indexes_vision=[<span class="hljs-number">0</span>, <span class="hljs-number">1</span>], | |
| condition_video_keep=<span class="hljs-string">"first"</span>, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| enable_sound=<span class="hljs-literal">True</span>, | |
| ) | |
| encode_video( | |
| result.video, | |
| fps=<span class="hljs-number">24</span>, | |
| audio=result.sound, | |
| audio_sample_rate=pipe.sound_tokenizer.config.sampling_rate, | |
| output_path=<span class="hljs-string">"cosmos3_v2v_with_sound.mp4"</span>, | |
| )<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="text-to-video-with-sound" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#text-to-video-with-sound"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Text-to-video with sound</span></h2><!--]--><!----> <p>When the checkpoint carries a <code>sound_tokenizer</code>, pass <code>enable_sound=True</code> to jointly generate a synchronized audio track. The waveform is returned alongside the video and can be muxed into the MP4 with <a href="/docs/diffusers/pr_14421/en/api/utilities#diffusers.utils.encode_video">encode_video()</a>.</p> <p>This is the same call as the text-to-video example above with <code>enable_sound=True</code> added:</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> encode_video | |
| <span class="hljs-comment"># JSON-upsampled positive and negative prompts (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2v_sound_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt.json"</span>)) | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| result = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| enable_sound=<span class="hljs-literal">True</span>, | |
| ) | |
| encode_video( | |
| result.video, | |
| fps=<span class="hljs-number">24</span>, | |
| audio=result.sound, | |
| audio_sample_rate=pipe.sound_tokenizer.config.sampling_rate, | |
| output_path=<span class="hljs-string">"cosmos3_with_sound.mp4"</span>, | |
| )<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="action-conditioned-generation" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#action-conditioned-generation"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Action-conditioned generation</span></h2><!--]--><!----> <p>Action runs group every action-specific input into a <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.CosmosActionCondition">CosmosActionCondition</a> passed via the <code>action</code> argument instead of the top-level <code>image</code> / <code>video</code> / <code>height</code> / <code>width</code> arguments. Set <code>resolution_tier</code> (<code>256</code>/<code>480</code>/<code>704</code>/<code>720</code>) close to the input video’s native resolution; it selects the conditioning canvas. Cosmos 3 supports three action modes — <code>policy</code>, <code>forward_dynamics</code>, and <code>inverse_dynamics</code>. <code>policy</code> and <code>forward_dynamics</code> condition only on the first frame (so an <code>image</code> or a <code>video</code> both work), while <code>inverse_dynamics</code> requires a <code>video</code>. The conditioning video for an action run is set on <code>action.video</code> (or <code>action.image</code>), not on the pipeline’s top-level <code>video</code> argument.</p> <p>Pass a plain task description as <code>prompt</code> and pick the camera with <code>action.view_point</code> (default <code>"ego_view"</code>; also <code>"third_person_view"</code>, <code>"wrist_view"</code>, <code>"concat_view"</code>). The pipeline turns these into the structured JSON caption the model was trained on, so action prompts should not be LLM-upsampled.</p> <!--[2--><h3 class="relative group"><a id="action-policy" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#action-policy"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Action policy</span></h3><!--]--><!----> <p>Action policy generation predicts future video and action tokens from the first observation frame, text prompt, and action domain metadata. The example below uses the Bridge robot domain and writes the predicted action chunk to JSON in model-normalized action space.</p> <div class="flex space-x-2 items-center my-1.5 mr-8 h-7 !pl-0 -mx-3 md:mx-0"><!--[--><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd border-gray-800 bg-black dark:bg-gray-700 text-white">Nano</div><div class="flex items-center border rounded-lg px-1.5 py-1 leading-none select-none text-smd text-gray-500 cursor-pointer opacity-90 hover:text-gray-700 dark:hover:text-gray-200 hover:shadow-sm">Super</div><!--]--></div> <div class="language-select"><!--[0--><div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline, CosmosActionCondition | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_video | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16, device_map=<span class="hljs-string">"cuda"</span> | |
| ) | |
| pipe.scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| prompt = <span class="hljs-string">"Put the pot to the left of the purple item."</span> | |
| video = load_video( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/raw/refs/heads/assets/cosmos3/inputs/action/bridge_20260501_0.mp4"</span> | |
| ) | |
| result = pipe( | |
| prompt=prompt, | |
| action=CosmosActionCondition( | |
| mode=<span class="hljs-string">"policy"</span>, | |
| chunk_size=<span class="hljs-number">16</span>, | |
| domain_name=<span class="hljs-string">"bridge_orig_lerobot"</span>, | |
| resolution_tier=<span class="hljs-number">480</span>, | |
| video=video, | |
| view_point=<span class="hljs-string">"ego_view"</span>, | |
| ), | |
| fps=<span class="hljs-number">5</span>, | |
| num_inference_steps=<span class="hljs-number">30</span>, | |
| guidance_scale=<span class="hljs-number">1.0</span>, | |
| use_system_prompt=<span class="hljs-literal">False</span>, | |
| ) | |
| <span class="hljs-comment"># macro_block_size=1 allows arbitrary frame sizes (Cosmos3 outputs are not always divisible by 16).</span> | |
| export_to_video(result.video, <span class="hljs-string">"sample.mp4"</span>, fps=<span class="hljs-number">5</span>, macro_block_size=<span class="hljs-number">1</span>) | |
| <span class="hljs-keyword">if</span> result.action <span class="hljs-keyword">is</span> <span class="hljs-keyword">not</span> <span class="hljs-literal">None</span>: | |
| <span class="hljs-keyword">with</span> <span class="hljs-built_in">open</span>(<span class="hljs-string">"sample_action.json"</span>, <span class="hljs-string">"w"</span>) <span class="hljs-keyword">as</span> f: | |
| json.dump(result.action[<span class="hljs-number">0</span>].tolist(), f)<!----></pre></div><!----><!--]--><!----> <!--[-1--><!--]--><!----><!----></div><!----> <!--[1--><h2 class="relative group"><a id="context-parallelism" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#context-parallelism"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Context parallelism</span></h2><!--]--><!----> <p>For long videos or high resolutions, a single forward pass can exceed the memory and latency budget of one GPU. Cosmos 3 supports <strong>context parallelism (CP)</strong> to shard the sequence dimension across multiple GPUs, splitting the attention computation so each device holds only a slice of the tokens.</p> <p>Cosmos 3 supports <strong>Ulysses</strong> context parallelism (all-to-all sequence/head exchange). Ring attention is not supported.</p> <p>Unlike most diffusers models, Cosmos 3 does <strong>not</strong> wire CP into the transformer or the declarative <code>enable_parallelism()</code> path: its grouped-query attention, separate understanding/generation streams (the generation stream attends to both), and ragged per-stream lengths can’t be expressed as a <code>_cp_plan</code>. Instead, the model exposes small no-op shard/gather seams, and the implementation lives in <a href="https://github.com/huggingface/diffusers/blob/main/examples/cosmos3/cosmos_parallel.py" rel="nofollow"><code>examples/cosmos3/cosmos_parallel.py</code></a> — a self-contained module you can read end to end and adapt. It offers two orthogonal, composable sharding axes:</p> <table><thead><tr><th>Helper</th><th>Shards</th><th>Use for</th></tr></thead><tbody><tr><td><code>enable_cosmos3_context_parallel(transformer, cp_mesh)</code></td><td>sequence (CP / Ulysses)</td><td>latency on a model that fits one GPU (<code>Nano</code>)</td></tr><tr><td><code>enable_cosmos3_tensor_parallel(transformer, tp_mesh)</code></td><td>weights (TP)</td><td>fitting a model that doesn’t fit one GPU (<code>Super</code>)</td></tr></tbody></table> <p>Use either alone or both together on a 2-D <code>(tp, cp)</code> mesh (see <a href="#fitting-large-models-with-tensor-parallelism">Fitting large models with tensor parallelism</a>).</p> <p>Two requirements are specific to Cosmos 3:</p> <ul><li>Use the <code>native</code> attention backend. Cosmos 3 uses grouped-query attention (GQA), and the native SDPA backend is the only one that accepts <code>enable_gqa</code> (cuDNN and flash reject it). The helpers expand the KV heads to the query-head count and call SDPA with <code>enable_gqa=False</code> so it still dispatches to the flash kernel (the math fallback would materialize the full <code>[S, S]</code> scores and OOM on long sequences).</li> <li>The CP (Ulysses) degree must divide the query-head count (32 for <code>Nano</code>, 64 for <code>Super</code>); for TP, the degree must divide the KV heads (8). The understanding (text) and generation (video/sound) streams are sharded independently along the sequence, and ragged lengths are zero-padded internally to a multiple of the world size.</li></ul> <!--[2--><h3 class="relative group"><a id="run-it" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#run-it"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Run it</span></h3><!--]--><!----> <p>The full CLI <a href="https://github.com/huggingface/diffusers/blob/main/examples/cosmos3/inference_cosmos3.py" rel="nofollow"><code>examples/cosmos3/inference_cosmos3.py</code></a> uses <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniModularPipeline">Cosmos3OmniModularPipeline</a> and reuses these helpers, so <strong>any modality</strong> (text-to-image/video, image-to-video, sound, action modes) runs multi-GPU via <code>--tp-degree</code> / <code>--cp-degree</code>. Launch with <a href="https://docs.pytorch.org/docs/stable/elastic/run.html" rel="nofollow">torchrun</a>; <code>--tp-degree * --cp-degree</code> must equal <code>--nproc_per_node</code>. Every rank produces the same output; rank 0 writes it.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!----><span class="hljs-comment"># CP only — Nano (fits one GPU); CP degree must divide 32 query heads.</span> | |
| torchrun --nproc_per_node=4 examples/cosmos3/inference_cosmos3.py --model nano --cp-degree 4 --prompt <span class="hljs-string">"..."</span> | |
| <span class="hljs-comment"># TP only — Super; TP degree must divide 64 query heads and 8 KV heads.</span> | |
| torchrun --nproc_per_node=4 examples/cosmos3/inference_cosmos3.py --model super --tp-degree 4 --prompt <span class="hljs-string">"..."</span> | |
| <span class="hljs-comment"># TP + CP — Super, with sound (TP=2 x CP=2 across 4 GPUs).</span> | |
| torchrun --nproc_per_node=4 examples/cosmos3/inference_cosmos3.py \ | |
| --model super --tp-degree 2 --cp-degree 2 --enable-sound --prompt <span class="hljs-string">"..."</span><!----></pre></div><!----> <p><code>Super</code>’s ~120 GB of weights do not fit on one 96 GB GPU, so it needs TP; <code>Nano</code> fits on a single GPU, so CP for it is a pure latency optimization. (Omit both flags to run single-GPU.)</p> <!--[2--><h3 class="relative group"><a id="fitting-large-models-with-tensor-parallelism" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#fitting-large-models-with-tensor-parallelism"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Fitting large models with tensor parallelism</span></h3><!--]--><!----> <p>CP shards <em>activations</em> but replicates every weight on every rank, so it does not reduce a model’s weight footprint — a model that doesn’t fit on one GPU still won’t fit under CP alone. To shard the <strong>weights</strong>, <code>enable_cosmos3_tensor_parallel(transformer, tp_mesh)</code> applies Megatron-style tensor parallelism on a second, orthogonal mesh axis:</p> <ul><li>The attention and MLP projections are column/row sharded across the TP group (<code>to_q/to_k/to_v</code> + <code>add_q/k/v</code> and the MLPs’ <code>gate/up</code> are column-parallel; <code>to_out/to_add_out</code> and the MLPs’ <code>down</code> are row-parallel with an all-reduce). Each rank ends up owning <code>query_heads / tp</code> query heads and <code>kv_heads / tp</code> KV heads.</li> <li>TP composes with CP on a 2-D <code>(tp, cp)</code> device mesh: TP splits heads/weights persistently, CP shards the sequence on top. The constraints are <code>tp</code> divides the KV heads (8), and <code>tp * cp</code> divides the query heads (32 for <code>Nano</code>, 64 for <code>Super</code>).</li> <li>Weights are loaded to CPU and sharded onto the GPUs layer by layer, so the full model is never materialized on a single device.</li></ul> <blockquote class="tip"><p>TP issues an all-reduce on every attention and MLP block, so it is bandwidth-heavy. On hosts without NVLink it is the dominant cost; prefer the smallest TP degree that makes the weights fit and put the remaining GPUs into CP.</p></blockquote> <!--[2--><h3 class="relative group"><a id="use-it-in-your-own-modular-pipeline" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#use-it-in-your-own-modular-pipeline"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Use it in your own modular pipeline</span></h3><!--]--><!----> <p>The CLI flags are convenient, but you can call the helpers directly with <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniModularPipeline">Cosmos3OmniModularPipeline</a>. Load the pipeline configuration and components on CPU, apply TP <em>before</em> moving the pipeline to the rank-local GPU, switch to the <code>native</code> backend, and then enable CP. Do not use <code>device_map</code> for this flow:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> os | |
| <span class="hljs-keyword">import</span> sys | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">import</span> torch.distributed <span class="hljs-keyword">as</span> dist | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniModularPipeline | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> torch.distributed.device_mesh <span class="hljs-keyword">import</span> init_device_mesh | |
| <span class="hljs-comment"># Make the helper module importable.</span> | |
| sys.path.insert(<span class="hljs-number">0</span>, <span class="hljs-string">"examples/cosmos3"</span>) | |
| <span class="hljs-keyword">from</span> cosmos_parallel <span class="hljs-keyword">import</span> ( | |
| enable_cosmos3_context_parallel, | |
| enable_cosmos3_flash_attention, | |
| enable_cosmos3_tensor_parallel, | |
| ) | |
| <span class="hljs-comment"># torchrun sets RANK / WORLD_SIZE / LOCAL_RANK. Pick tp_degree * cp_degree == world size.</span> | |
| local_rank = <span class="hljs-built_in">int</span>(os.environ[<span class="hljs-string">"LOCAL_RANK"</span>]) | |
| torch.cuda.set_device(local_rank) | |
| dist.init_process_group(<span class="hljs-string">"nccl"</span>) | |
| mesh = init_device_mesh(<span class="hljs-string">"cuda"</span>, (tp_degree, cp_degree), mesh_dim_names=(<span class="hljs-string">"tp"</span>, <span class="hljs-string">"cp"</span>)) | |
| <span class="hljs-comment"># Load components on CPU first; a TP-sharded model may not fit one GPU.</span> | |
| pipe = Cosmos3OmniModularPipeline.from_pretrained(model_id) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.enable_safety_checker() | |
| <span class="hljs-keyword">if</span> tp_degree > <span class="hljs-number">1</span>: | |
| enable_cosmos3_tensor_parallel(pipe.transformer, mesh[<span class="hljs-string">"tp"</span>]) <span class="hljs-comment"># shard weights -> GPUs</span> | |
| pipe.to(<span class="hljs-string">f"cuda:<span class="hljs-subst">{local_rank}</span>"</span>) <span class="hljs-comment"># move the replicated remainder</span> | |
| pipe.transformer.set_attention_backend(<span class="hljs-string">"native"</span>) | |
| <span class="hljs-keyword">if</span> cp_degree > <span class="hljs-number">1</span>: | |
| enable_cosmos3_context_parallel(pipe.transformer, mesh[<span class="hljs-string">"cp"</span>]) <span class="hljs-comment"># shard the sequence</span> | |
| <span class="hljs-keyword">elif</span> tp_degree > <span class="hljs-number">1</span>: | |
| enable_cosmos3_flash_attention(pipe.transformer) <span class="hljs-comment"># GQA-safe dense attention</span> | |
| <span class="hljs-comment"># Modular pipelines replace components through update_components().</span> | |
| scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| pipe.update_components(scheduler=scheduler) | |
| <span class="hljs-comment"># A single output name returns that value; a list returns a dictionary.</span> | |
| outputs = pipe( | |
| prompt=<span class="hljs-string">'{"scene":"A robot arm in a kitchen"}'</span>, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| output=[<span class="hljs-string">"videos"</span>, <span class="hljs-string">"sound"</span>, <span class="hljs-string">"sampling_rate"</span>, <span class="hljs-string">"action"</span>], | |
| ) | |
| videos = outputs[<span class="hljs-string">"videos"</span>] | |
| sound = outputs[<span class="hljs-string">"sound"</span>] <span class="hljs-comment"># None unless sound generation was requested.</span> | |
| action = outputs[<span class="hljs-string">"action"</span>] <span class="hljs-comment"># None unless an action workflow produced actions.</span><!----></pre></div><!----> <p>For CP only (no weight sharding), use a 1-D mesh: <code>init_device_mesh("cuda", (world_size,), mesh_dim_names=("cp",))</code> and just <code>enable_cosmos3_context_parallel</code>.</p> <p><code>enable_safety_checker()</code> loads and enables the default checker; <code>disable_safety_checker()</code> explicitly disables it. Use those pipeline methods instead of the task-pipeline <code>enable_safety_checker=</code> construction argument or <code>enable_safety_check=</code> call argument. Modular pipelines also do not return <code>Cosmos3OmniPipelineOutput</code>: use <code>output="videos"</code> for frames alone, or an output list and its returned dictionary as shown above instead of <code>result.video</code>, <code>result.sound</code>, or <code>result.action</code>.</p> <blockquote class="tip"><p>On some multi-GPU topologies the first NCCL all-to-all can hang. If a CP run stalls at the start of the first denoising step, set <code>NCCL_P2P_DISABLE=1</code> in the environment before launching <code>torchrun</code>.</p></blockquote> <p>CP and TP compose with all the workflows above (text-to-video, image-to-video, text-to-video with sound, and action-conditioned generation) and with both the <code>Nano</code> and <code>Super</code> checkpoints — only the pipeline construction and the parallelism setup lines change.</p> <!--[1--><h2 class="relative group"><a id="metadata-templates" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#metadata-templates"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Metadata templates</span></h2><!--]--><!----> <p><code>tokenize_prompt</code> appends short metadata sentences inside the user message so the LLM sees the conditioning the model was trained with. The positive prompt gets sentences like <em>“The video is 7.9 seconds long and is of 24 FPS.”</em> and <em>“This video is of 720x1280 resolution.”</em>; the negative prompt gets the inverse (<em>”… is not …”</em>).</p> <p>Both are on by default. Disable either pair through <code>__call__</code>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->result = pipe( | |
| prompt=prompt, | |
| negative_prompt=negative_prompt, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| add_duration_template=<span class="hljs-literal">False</span>, <span class="hljs-comment"># skip the duration sentence on both prompts</span> | |
| add_resolution_template=<span class="hljs-literal">False</span>, <span class="hljs-comment"># skip the resolution sentence on both prompts</span> | |
| )<!----></pre></div><!----> <p><code>add_duration_template</code> has no effect when <code>num_frames == 1</code> (image mode); only the resolution sentence is appended in that case.</p> <!--[1--><h2 class="relative group"><a id="safety-checker" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#safety-checker"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Safety checker</span></h2><!--]--><!----> <p>Cosmos3 wires up the <a href="https://pypi.org/project/cosmos-guardrail/" rel="nofollow"><code>cosmos_guardrail</code></a> <code>CosmosSafetyChecker</code> and runs it <strong>by default</strong>. The text guardrail rejects unsafe prompts before generation (<code>ValueError</code>); the video guardrail runs on the decoded frames and either pixelates detected faces or rejects the output. Audio output is not guardrailed.</p> <p>Install the optional dependency to enable the default checker:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class=" "><!---->pip <span class="hljs-keyword">install</span> cosmos_guardrail<!----></pre></div><!----> <p>The checker is mandatory under the NVIDIA Open Model License Agreement. The two flags below exist for tests and development workflows where the guardrail would be redundant (e.g., the input has already been cleared, or you are intentionally exercising the pipeline on edge inputs).</p> <p><strong>Disable at construction</strong> (no checker is instantiated, so no guardrail models are downloaded or loaded into memory):</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline | |
| pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, | |
| dtype=torch.bfloat16, | |
| device_map=<span class="hljs-string">"cuda"</span>, | |
| enable_safety_checker=<span class="hljs-literal">False</span>, | |
| )<!----></pre></div><!----> <p><strong>Disable for a single call</strong> (checker stays loaded — useful for one-off bypass while keeping the default on for subsequent calls):</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->result = pipe( | |
| prompt=prompt, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| enable_safety_check=<span class="hljs-literal">False</span>, | |
| )<!----></pre></div><!----> <p>To supply a custom checker (e.g., a no-op subclass for fast tests), pass it as <code>safety_checker=</code>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->pipe = Cosmos3OmniPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, | |
| dtype=torch.bfloat16, | |
| device_map=<span class="hljs-string">"cuda"</span>, | |
| safety_checker=MyCustomSafetyChecker(), | |
| )<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="diffusers.Cosmos3OmniPipeline" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.Cosmos3OmniPipeline"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Cosmos3OmniPipeline</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.Cosmos3OmniPipeline"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.</span><span class="font-semibold">Cosmos3OmniPipeline</span></span></h3><!----> <a id="diffusers.Cosmos3OmniPipeline" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.Cosmos3OmniPipeline"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/pipelines/cosmos/pipeline_cosmos3_omni.py#L365" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">transformer<span class="opacity-60">: Cosmos3OmniTransformer</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">text_tokenizer<span class="opacity-60">: AutoTokenizer</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">vae<span class="opacity-60">: AutoencoderKLWan</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">scheduler<span class="opacity-60">: UniPCMultistepScheduler</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">sound_tokenizer<span class="opacity-60">: diffusers.models.autoencoders.autoencoder_cosmos3_audio.Cosmos3AVAEAudioTokenizer | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">safety_checker<span class="opacity-60">: diffusers.pipelines.cosmos.pipeline_cosmos3_omni.CosmosSafetyChecker | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">enable_safety_checker<span class="opacity-60">: bool = True</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">default_use_system_prompt<span class="opacity-60">: bool = True</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">use_native_flow_schedule<span class="opacity-60">: bool = False</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.Cosmos3OmniPipeline.decode_sound"><!----><h4 class="!m-0"><span class="flex-1 rounded-xl py-0.5 break-all bg-gradient-to-r from-blue-50/60 to-white dark:from-gray-900 dark:to-gray-950 text-blue-700 dark:text-blue-300 font-medium px-2"><svg width="1em" height="1em" viewBox="0 0 32 33" class="mr-1 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg"><path d="M5.80566 18.3545C4.90766 17.4565 4.90766 16.0005 5.80566 15.1025L14.3768 6.53142C15.2748 5.63342 16.7307 5.63342 17.6287 6.53142L26.1999 15.1025C27.0979 16.0005 27.0979 17.4565 26.1999 18.3545L17.6287 26.9256C16.7307 27.8236 15.2748 27.8236 14.3768 26.9256L5.80566 18.3545Z" fill="currentColor" fill-opacity="0.25"/><path fill-rule="evenodd" clip-rule="evenodd" d="M16.4801 13.9619C16.4801 12.9761 16.7467 12.5436 16.9443 12.3296C17.1764 12.078 17.5731 11.8517 18.2275 11.707C18.8821 11.5623 19.638 11.5342 20.4038 11.5582C20.7804 11.57 21.1341 11.5932 21.4719 11.6156L21.5263 11.6193C21.8195 11.6389 22.1626 11.6618 22.4429 11.6618V7.40825C22.3209 7.40825 22.1219 7.39596 21.7544 7.37149C21.4202 7.34925 20.9976 7.32115 20.5371 7.30672C19.6286 7.27824 18.4672 7.29779 17.3093 7.55377C16.1512 7.8098 14.8404 8.33724 13.8181 9.4452C12.7612 10.5907 12.2266 12.1236 12.2266 13.9619V15.0127H10.6836V19.2662H12.2266V26.6332H16.4801V19.2662H20.3394V15.0127H16.4801V13.9619Z" fill="currentColor"/></svg>decode_sound</span></h4><!----> <a id="diffusers.Cosmos3OmniPipeline.decode_sound" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.Cosmos3OmniPipeline.decode_sound"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/pipelines/cosmos/pipeline_cosmos3_omni.py#L469" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">latent<span class="opacity-60">: Tensor</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Decode a sound latent <code>[C, T]</code> to a waveform <code>[audio_ch, N]</code>.</p> <p>Adds/removes the batch dimension expected by the sound tokenizer decoder.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.Cosmos3OmniPipeline.prepare_latents"><!----><h4 class="!m-0"><span class="flex-1 rounded-xl py-0.5 break-all bg-gradient-to-r from-blue-50/60 to-white dark:from-gray-900 dark:to-gray-950 text-blue-700 dark:text-blue-300 font-medium px-2"><svg width="1em" height="1em" viewBox="0 0 32 33" class="mr-1 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg"><path d="M5.80566 18.3545C4.90766 17.4565 4.90766 16.0005 5.80566 15.1025L14.3768 6.53142C15.2748 5.63342 16.7307 5.63342 17.6287 6.53142L26.1999 15.1025C27.0979 16.0005 27.0979 17.4565 26.1999 18.3545L17.6287 26.9256C16.7307 27.8236 15.2748 27.8236 14.3768 26.9256L5.80566 18.3545Z" fill="currentColor" fill-opacity="0.25"/><path fill-rule="evenodd" clip-rule="evenodd" d="M16.4801 13.9619C16.4801 12.9761 16.7467 12.5436 16.9443 12.3296C17.1764 12.078 17.5731 11.8517 18.2275 11.707C18.8821 11.5623 19.638 11.5342 20.4038 11.5582C20.7804 11.57 21.1341 11.5932 21.4719 11.6156L21.5263 11.6193C21.8195 11.6389 22.1626 11.6618 22.4429 11.6618V7.40825C22.3209 7.40825 22.1219 7.39596 21.7544 7.37149C21.4202 7.34925 20.9976 7.32115 20.5371 7.30672C19.6286 7.27824 18.4672 7.29779 17.3093 7.55377C16.1512 7.8098 14.8404 8.33724 13.8181 9.4452C12.7612 10.5907 12.2266 12.1236 12.2266 13.9619V15.0127H10.6836V19.2662H12.2266V26.6332H16.4801V19.2662H20.3394V15.0127H16.4801V13.9619Z" fill="currentColor"/></svg>prepare_latents</span></h4><!----> <a id="diffusers.Cosmos3OmniPipeline.prepare_latents" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.Cosmos3OmniPipeline.prepare_latents"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/pipelines/cosmos/pipeline_cosmos3_omni.py#L715" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">image<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">video<span class="opacity-60">: typing.Union[list[PIL.Image.Image], torch.Tensor, numpy.ndarray, NoneType] = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">condition_frame_indexes_vision<span class="opacity-60">: Iterable = (0, 1)</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">condition_video_keep<span class="opacity-60">: typing.Literal['first', 'last'] = 'first'</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">num_frames<span class="opacity-60">: int | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">height<span class="opacity-60">: int | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">width<span class="opacity-60">: int | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">fps<span class="opacity-60">: float = 24.0</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">latents<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">sound_latents<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">action_latents<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">generator<span class="opacity-60">: typing.Optional[torch.Generator] = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">device<span class="opacity-60">: str = 'cuda'</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">dtype<span class="opacity-60">: dtype = torch.bfloat16</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">enable_sound<span class="opacity-60">: bool = False</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">action<span class="opacity-60">: CosmosActionCondition | None = None</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Build conditioning + initial noise for a single sample.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.Cosmos3OmniPipeline.tokenize_prompt"><!----><h4 class="!m-0"><span class="flex-1 rounded-xl py-0.5 break-all bg-gradient-to-r from-blue-50/60 to-white dark:from-gray-900 dark:to-gray-950 text-blue-700 dark:text-blue-300 font-medium px-2"><svg width="1em" height="1em" viewBox="0 0 32 33" class="mr-1 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg"><path d="M5.80566 18.3545C4.90766 17.4565 4.90766 16.0005 5.80566 15.1025L14.3768 6.53142C15.2748 5.63342 16.7307 5.63342 17.6287 6.53142L26.1999 15.1025C27.0979 16.0005 27.0979 17.4565 26.1999 18.3545L17.6287 26.9256C16.7307 27.8236 15.2748 27.8236 14.3768 26.9256L5.80566 18.3545Z" fill="currentColor" fill-opacity="0.25"/><path fill-rule="evenodd" clip-rule="evenodd" d="M16.4801 13.9619C16.4801 12.9761 16.7467 12.5436 16.9443 12.3296C17.1764 12.078 17.5731 11.8517 18.2275 11.707C18.8821 11.5623 19.638 11.5342 20.4038 11.5582C20.7804 11.57 21.1341 11.5932 21.4719 11.6156L21.5263 11.6193C21.8195 11.6389 22.1626 11.6618 22.4429 11.6618V7.40825C22.3209 7.40825 22.1219 7.39596 21.7544 7.37149C21.4202 7.34925 20.9976 7.32115 20.5371 7.30672C19.6286 7.27824 18.4672 7.29779 17.3093 7.55377C16.1512 7.8098 14.8404 8.33724 13.8181 9.4452C12.7612 10.5907 12.2266 12.1236 12.2266 13.9619V15.0127H10.6836V19.2662H12.2266V26.6332H16.4801V19.2662H20.3394V15.0127H16.4801V13.9619Z" fill="currentColor"/></svg>tokenize_prompt</span></h4><!----> <a id="diffusers.Cosmos3OmniPipeline.tokenize_prompt" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.Cosmos3OmniPipeline.tokenize_prompt"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/pipelines/cosmos/pipeline_cosmos3_omni.py#L1085" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">prompt<span class="opacity-60">: str</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">negative_prompt<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">num_frames<span class="opacity-60">: int = 189</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">height<span class="opacity-60">: int = 720</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">width<span class="opacity-60">: int = 1280</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">fps<span class="opacity-60">: float = 24.0</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">use_system_prompt<span class="opacity-60">: bool | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">add_resolution_template<span class="opacity-60">: bool = True</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">add_duration_template<span class="opacity-60">: bool = True</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">action_mode<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">action_view_point<span class="opacity-60">: str | None = None</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Apply prompt-augmentation templates and tokenize cond/uncond prompts via the configured chat template.</p> <p>This pipeline does not run a separate text encoder: the joint Cosmos3 transformer consumes raw token IDs | |
| alongside vision (and optionally sound) tokens.</p> <p>When <code>negative_prompt</code> is <code>None</code>, an empty string is used; the Cosmos3 docs page documents recommended | |
| quality-control negative prompts to pass explicitly for text2video / image2video. The duration and resolution | |
| templates are appended to the prompt, and inverse templates are appended to the negative prompt, when enabled.</p> <p>When <code>action_mode</code> is set, the prompt is instead converted to the structured action JSON caption the model | |
| was trained on (see <code>_build_action_json_prompt</code>), using <code>action_view_point</code> for the framing field; the | |
| flat metadata templates are skipped because the JSON already carries duration/fps/resolution/aspect_ratio.</p></div></div> <ul><li>all</li> <li><strong>call</strong></li></ul> <!--[1--><h2 class="relative group"><a id="cosmos3omnimodularpipeline" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#cosmos3omnimodularpipeline"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Cosmos3OmniModularPipeline</span></h2><!--]--><!----> <p>Cosmos 3 is also available as a Modular Diffusers pipeline. The task-based <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniPipeline">Cosmos3OmniPipeline</a> remains available; the modular pipeline coexists with it and covers the same modes (<code>text2image</code>, <code>text2video</code>, <code>image2video</code>, <code>video2video</code>, action-conditioned generation, and <code>transfer</code> (structural control), with optional sound when supported by the checkpoint).</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniModularPipeline | |
| pipe = Cosmos3OmniModularPipeline.from_pretrained( | |
| <span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16 | |
| ) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.enable_safety_checker() | |
| videos = pipe( | |
| prompt=<span class="hljs-string">'{"scene":"A robot arm in a kitchen"}'</span>, | |
| num_frames=<span class="hljs-number">1</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| <span class="hljs-comment"># Modular pipelines expose declared outputs directly instead of using the task pipeline's</span> | |
| <span class="hljs-comment"># `return_dict`/`Cosmos3OmniPipelineOutput` API.</span> | |
| image = videos[<span class="hljs-number">0</span>]<!----></pre></div><!----> <p>You can also load through <a href="/docs/diffusers/pr_14421/en/api/modular_diffusers/pipeline#diffusers.ModularPipeline">ModularPipeline</a> and let the repository config select the blocks class:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ModularPipeline | |
| pipe = ModularPipeline.from_pretrained(<span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.enable_safety_checker() | |
| videos = pipe( | |
| prompt=<span class="hljs-string">'{"scene":"A robot arm in a kitchen"}'</span>, num_frames=<span class="hljs-number">1</span>, height=<span class="hljs-number">720</span>, width=<span class="hljs-number">1280</span>, output=<span class="hljs-string">"videos"</span> | |
| )<!----></pre></div><!----> <p>To inspect or customize a specific Cosmos modular workflow, use <code>available_workflows</code> + <code>get_workflow()</code>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->available = pipe.blocks.available_workflows | |
| image2video_blocks = pipe.blocks.get_workflow(<span class="hljs-string">"image2video"</span>)<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="modular-examples-for-all-existing-workflows" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#modular-examples-for-all-existing-workflows"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Modular examples for all existing workflows</span></h3><!--]--><!----> <p>The modular pipeline supports the same call signatures as the task pipeline. The snippets below mirror every generation example shown above (<code>text2video</code>, <code>text2image</code>, <code>image2video</code>, <code>video2video</code>, <code>video2video_sound</code>, <code>text2video_sound</code>, and <code>action_policy</code>). Transfer (structural control) has its own inputs and is shown separately in <a href="#modular-transfer-structural-control">Modular transfer</a> below.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniModularPipeline, CosmosActionCondition | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> encode_video, export_to_video, load_image, load_video | |
| pipe = Cosmos3OmniModularPipeline.from_pretrained(<span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.enable_safety_checker() | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| pipe.scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| <span class="hljs-comment"># text2video</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt.json"</span>)) | |
| videos = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| num_inference_steps=<span class="hljs-number">35</span>, | |
| guidance_scale=<span class="hljs-number">6.0</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| export_to_video(videos, <span class="hljs-string">"cosmos3_modular_t2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>) | |
| <span class="hljs-comment"># text2image</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2i_prompt.json"</span>)) | |
| videos = pipe(prompt=json.dumps(json_prompt), num_frames=<span class="hljs-number">1</span>, height=<span class="hljs-number">720</span>, width=<span class="hljs-number">1280</span>, output=<span class="hljs-string">"videos"</span>) | |
| videos[<span class="hljs-number">0</span>].save(<span class="hljs-string">"cosmos3_modular_t2i.jpg"</span>, <span class="hljs-built_in">format</span>=<span class="hljs-string">"JPEG"</span>, quality=<span class="hljs-number">85</span>) | |
| <span class="hljs-comment"># image2video</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_i2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt_i2v.json"</span>)) | |
| image = load_image(<span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/releases/download/assets/robot_153.jpg"</span>) | |
| videos = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| image=image, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| export_to_video(videos, <span class="hljs-string">"cosmos3_modular_i2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>) | |
| <span class="hljs-comment"># video2video</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_v2v_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt_i2v.json"</span>)) | |
| video = load_video( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/raw/refs/heads/assets/cosmos3/inputs/vision/robot_pouring.mp4"</span> | |
| ) | |
| videos = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| video=video, | |
| condition_frame_indexes_vision=[<span class="hljs-number">0</span>, <span class="hljs-number">1</span>], | |
| condition_video_keep=<span class="hljs-string">"first"</span>, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| num_inference_steps=<span class="hljs-number">35</span>, | |
| guidance_scale=<span class="hljs-number">6.0</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| export_to_video(videos, <span class="hljs-string">"cosmos3_modular_v2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>) | |
| <span class="hljs-comment"># video2video_sound</span> | |
| outputs = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| video=video, | |
| condition_frame_indexes_vision=[<span class="hljs-number">0</span>, <span class="hljs-number">1</span>], | |
| condition_video_keep=<span class="hljs-string">"first"</span>, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| enable_sound=<span class="hljs-literal">True</span>, | |
| output=[<span class="hljs-string">"videos"</span>, <span class="hljs-string">"sound"</span>, <span class="hljs-string">"sampling_rate"</span>], | |
| ) | |
| encode_video( | |
| outputs[<span class="hljs-string">"videos"</span>], | |
| fps=<span class="hljs-number">24</span>, | |
| audio=outputs[<span class="hljs-string">"sound"</span>], | |
| audio_sample_rate=outputs[<span class="hljs-string">"sampling_rate"</span>], | |
| output_path=<span class="hljs-string">"cosmos3_modular_v2v_with_sound.mp4"</span>, | |
| ) | |
| <span class="hljs-comment"># text2video_sound</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2v_sound_prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt.json"</span>)) | |
| outputs = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">24.0</span>, | |
| enable_sound=<span class="hljs-literal">True</span>, | |
| output=[<span class="hljs-string">"videos"</span>, <span class="hljs-string">"sound"</span>, <span class="hljs-string">"sampling_rate"</span>], | |
| ) | |
| encode_video( | |
| outputs[<span class="hljs-string">"videos"</span>], | |
| fps=<span class="hljs-number">24</span>, | |
| audio=outputs[<span class="hljs-string">"sound"</span>], | |
| audio_sample_rate=outputs[<span class="hljs-string">"sampling_rate"</span>], | |
| output_path=<span class="hljs-string">"cosmos3_modular_t2v_with_sound.mp4"</span>, | |
| ) | |
| <span class="hljs-comment"># action_policy</span> | |
| prompt = <span class="hljs-string">"Put the pot to the left of the purple item."</span> | |
| action_video = load_video( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/raw/refs/heads/assets/cosmos3/inputs/action/bridge_20260501_0.mp4"</span> | |
| ) | |
| outputs = pipe( | |
| prompt=prompt, | |
| action=CosmosActionCondition( | |
| mode=<span class="hljs-string">"policy"</span>, | |
| chunk_size=<span class="hljs-number">16</span>, | |
| domain_name=<span class="hljs-string">"bridge_orig_lerobot"</span>, | |
| resolution_tier=<span class="hljs-number">480</span>, | |
| video=action_video, | |
| view_point=<span class="hljs-string">"ego_view"</span>, | |
| ), | |
| fps=<span class="hljs-number">5</span>, | |
| num_inference_steps=<span class="hljs-number">30</span>, | |
| guidance_scale=<span class="hljs-number">1.0</span>, | |
| use_system_prompt=<span class="hljs-literal">False</span>, | |
| output=[<span class="hljs-string">"videos"</span>, <span class="hljs-string">"action"</span>], | |
| ) | |
| export_to_video(outputs[<span class="hljs-string">"videos"</span>], <span class="hljs-string">"cosmos3_modular_action_policy.mp4"</span>, fps=<span class="hljs-number">5</span>, macro_block_size=<span class="hljs-number">1</span>) | |
| <span class="hljs-keyword">if</span> outputs[<span class="hljs-string">"action"</span>] <span class="hljs-keyword">is</span> <span class="hljs-keyword">not</span> <span class="hljs-literal">None</span>: | |
| <span class="hljs-keyword">with</span> <span class="hljs-built_in">open</span>(<span class="hljs-string">"cosmos3_modular_action_policy.json"</span>, <span class="hljs-string">"w"</span>) <span class="hljs-keyword">as</span> f: | |
| json.dump(outputs[<span class="hljs-string">"action"</span>][<span class="hljs-number">0</span>].tolist(), f)<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="modular-transfer-structural-control" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#modular-transfer-structural-control"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Modular transfer (structural control)</span></h3><!--]--><!----> <p>Transfer follows a <strong>precomputed control video</strong> (edge, blur, depth, segmentation, or a world-scenario map) passed through <code>control_videos=</code> as a <code>{hint: video}</code> mapping. It is video-only (no <code>image</code> / <code>video</code> / <code>action</code> / <code>enable_sound</code>), the prompt is a pre-upsampled JSON caption (see <a href="#prompt-upsampling">Prompt upsampling</a>), and long clips are generated autoregressively in chunks of <code>num_video_frames_per_chunk</code> and stitched automatically. <code>guidance_scale</code> is the usual text CFG; <code>control_guidance</code> (<code>!= 1.0</code>) additionally amplifies the control signal. Recommended starting values per hint:</p> <table><thead><tr><th>Hint</th><th><code>guidance_scale</code></th><th><code>control_guidance</code></th><th><code>flow_shift</code></th><th>Geometry</th></tr></thead><tbody><tr><td>Edge / Blur / Depth</td><td>3.0</td><td>1.5</td><td>10.0</td><td>121 frames @ 30 FPS</td></tr><tr><td>Segmentation</td><td>3.0</td><td>2.0</td><td>10.0</td><td>121 frames @ 30 FPS</td></tr><tr><td>World scenario (WSM)</td><td>1.0</td><td>3.0</td><td>10.0</td><td>101 frames @ 10 FPS</td></tr></tbody></table> <p>Diffusers does not ship the control assets. Ready-made ones (a control video + matching <code>prompt.json</code> per hint, plus a shared <code>negative_prompt.json</code>) live in the <a href="https://github.com/NVIDIA/cosmos/tree/main/cookbooks/cosmos3/generator/transfer/assets" rel="nofollow">Cosmos cookbook</a>. For the edge example below, download them into a local <code>assets/</code> folder:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->base=https://github.com/NVIDIA/cosmos/raw/refs/heads/main/cookbooks/cosmos3/generator/transfer/assets | |
| <span class="hljs-built_in">mkdir</span> -p assets/edge | |
| curl -sL <span class="hljs-string">"<span class="hljs-variable">$base</span>/edge/control_edge.mp4"</span> -o assets/edge/control_edge.mp4 | |
| curl -sL <span class="hljs-string">"<span class="hljs-variable">$base</span>/edge/prompt.json"</span> -o assets/edge/prompt.json | |
| curl -sL <span class="hljs-string">"<span class="hljs-variable">$base</span>/negative_prompt.json"</span> -o assets/negative_prompt.json<!----></pre></div><!----> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniModularPipeline | |
| <span class="hljs-keyword">from</span> diffusers.schedulers.scheduling_unipc_multistep <span class="hljs-keyword">import</span> UniPCMultistepScheduler | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_video | |
| pipe = Cosmos3OmniModularPipeline.from_pretrained(<span class="hljs-string">"nvidia/Cosmos3-Nano"</span>, dtype=torch.bfloat16) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| pipe.scheduler = UniPCMultistepScheduler.from_config( | |
| pipe.scheduler.config, flow_shift=<span class="hljs-number">10.0</span>, use_karras_sigmas=<span class="hljs-literal">False</span> | |
| ) | |
| <span class="hljs-comment"># Downloaded into assets/ from the Cosmos cookbook (see the curl snippet above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/edge/prompt.json"</span>)) | |
| negative_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/negative_prompt.json"</span>)) | |
| control_edge = load_video(<span class="hljs-string">"assets/edge/control_edge.mp4"</span>) | |
| videos = pipe( | |
| prompt=json.dumps(json_prompt), | |
| negative_prompt=json.dumps(negative_prompt), | |
| control_videos={<span class="hljs-string">"edge"</span>: control_edge}, | |
| num_frames=<span class="hljs-number">121</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| fps=<span class="hljs-number">30.0</span>, | |
| num_inference_steps=<span class="hljs-number">35</span>, | |
| guidance_scale=<span class="hljs-number">3.0</span>, | |
| control_guidance=<span class="hljs-number">1.5</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| export_to_video(videos, <span class="hljs-string">"cosmos3_modular_transfer_edge.mp4"</span>, fps=<span class="hljs-number">30</span>, macro_block_size=<span class="hljs-number">1</span>)<!----></pre></div><!----> <!--[2--><h3 class="relative group"><a id="diffusers.Cosmos3OmniModularPipeline" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.Cosmos3OmniModularPipeline"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Distilled (few-step) text-to-image and image-to-video</span></h3><!--]--><!----> <p>Few-step distilled checkpoints are served by <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3DistilledModularPipeline">Cosmos3DistilledModularPipeline</a> (blocks: <code>Cosmos3DistilledBlocks</code>); the base <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniModularPipeline">Cosmos3OmniModularPipeline</a> and <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniPipeline">Cosmos3OmniPipeline</a> do | |
| not support them. <code>num_inference_steps</code> is fixed to the length of the <code>distilled_sigmas</code> pipeline | |
| config (from the checkpoint’s <code>modular_model_index.json</code>) and <code>guidance_scale</code> is forced to | |
| 1.0 since guidance is baked into the weights — passing any other value for either raises an error, | |
| and <code>negative_prompt</code> is warned about and ignored.</p> <p>Prompts follow the same descriptive JSON structure as the non-distilled models, so short text | |
| must be upsampled first — use <code>--mode text2image</code> (T2I) or <code>--mode image2video</code> (I2V) as | |
| described in <a href="#prompt-upsampling">Prompt upsampling</a>, then pass the JSON via <code>json.dumps(...)</code>.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> json | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3DistilledModularPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_image | |
| <span class="hljs-comment"># JSON-upsampled prompt (see "Prompt upsampling" above).</span> | |
| json_prompt = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_t2i_prompt.json"</span>)) | |
| repo = <span class="hljs-string">"nvidia/Cosmos3-Super-Text2Image-4Step"</span> | |
| pipe = Cosmos3DistilledModularPipeline.from_pretrained(repo, dtype=torch.bfloat16) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-comment"># text-to-image (distilled)</span> | |
| videos = pipe( | |
| prompt=json.dumps(json_prompt), | |
| num_frames=<span class="hljs-number">1</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| videos[<span class="hljs-number">0</span>].save(<span class="hljs-string">"cosmos3_distilled_t2i.jpg"</span>, <span class="hljs-built_in">format</span>=<span class="hljs-string">"JPEG"</span>, quality=<span class="hljs-number">85</span>) | |
| <span class="hljs-comment"># image-to-video (distilled) — load the I2V repo instead</span> | |
| <span class="hljs-comment"># JSON-upsampled prompt (see "Prompt upsampling" above); upsampled from the source prompt</span> | |
| <span class="hljs-comment"># "The right robotic hand picks up the red sphere on the shelf."</span> | |
| json_prompt_i2v = json.load(<span class="hljs-built_in">open</span>(<span class="hljs-string">"assets/example_i2v_prompt.json"</span>)) | |
| repo_i2v = <span class="hljs-string">"nvidia/Cosmos3-Super-Image2Video-4Step"</span> | |
| pipe = Cosmos3DistilledModularPipeline.from_pretrained(repo_i2v, dtype=torch.bfloat16) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| image = load_image( | |
| <span class="hljs-string">"https://github.com/nvidia-cosmos/cosmos-dependencies/raw/refs/heads/assets/cosmos3/inputs/vision/robot_153.jpg"</span> | |
| ) | |
| videos = pipe( | |
| prompt=json.dumps(json_prompt_i2v), | |
| image=image, | |
| num_frames=<span class="hljs-number">189</span>, | |
| height=<span class="hljs-number">720</span>, | |
| width=<span class="hljs-number">1280</span>, | |
| output=<span class="hljs-string">"videos"</span>, | |
| ) | |
| export_to_video(videos, <span class="hljs-string">"cosmos3_distilled_i2v.mp4"</span>, fps=<span class="hljs-number">24</span>, macro_block_size=<span class="hljs-number">1</span>)<!----></pre></div><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.Cosmos3OmniModularPipeline"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.</span><span class="font-semibold">Cosmos3OmniModularPipeline</span></span></h3><!----> <a id="diffusers.Cosmos3OmniModularPipeline" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.Cosmos3OmniModularPipeline"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/modular_pipelines/cosmos/modular_pipeline.py#L7" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">blocks<span class="opacity-60">: diffusers.modular_pipelines.modular_pipeline.ModularPipelineBlocks | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">pretrained_model_name_or_path<span class="opacity-60">: str | os.PathLike | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">components_manager<span class="opacity-60">: diffusers.modular_pipelines.components_manager.ComponentsManager | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">collection<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">workflow<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">modular_config_dict<span class="opacity-60">: dict[str, typing.Any] | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">config_dict<span class="opacity-60">: dict[str, typing.Any] | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">**kwargs<span class="opacity-60"></span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>A ModularPipeline for Cosmos 3 omni generation.</p></div> <ul><li>all</li> <li><strong>call</strong></li></ul> <!--[1--><h2 class="relative group"><a id="diffusers.Cosmos3DistilledModularPipeline" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.Cosmos3DistilledModularPipeline"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Cosmos3DistilledModularPipeline</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.Cosmos3DistilledModularPipeline"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.</span><span class="font-semibold">Cosmos3DistilledModularPipeline</span></span></h3><!----> <a id="diffusers.Cosmos3DistilledModularPipeline" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.Cosmos3DistilledModularPipeline"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/modular_pipelines/cosmos/modular_pipeline.py#L121" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">blocks<span class="opacity-60">: diffusers.modular_pipelines.modular_pipeline.ModularPipelineBlocks | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">pretrained_model_name_or_path<span class="opacity-60">: str | os.PathLike | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">components_manager<span class="opacity-60">: diffusers.modular_pipelines.components_manager.ComponentsManager | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">collection<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">workflow<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">modular_config_dict<span class="opacity-60">: dict[str, typing.Any] | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">config_dict<span class="opacity-60">: dict[str, typing.Any] | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">**kwargs<span class="opacity-60"></span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>A ModularPipeline for distilled (few-step) Cosmos 3 omni generation.</p> <p>Distilled checkpoints bake classifier-free guidance into the weights and sample on a fixed schedule read from the | |
| pipeline’s <code>distilled_sigmas</code> config (populated from <code>modular_model_index.json</code>), so <code>guidance_scale</code> and <code>num_inference_steps</code> are fixed and <code>negative_prompt</code> is not supported.</p></div> <ul><li>all</li> <li><strong>call</strong></li></ul> <!--[1--><h2 class="relative group"><a id="diffusers.CosmosActionCondition" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>CosmosActionCondition</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.CosmosActionCondition"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.</span><span class="font-semibold">CosmosActionCondition</span></span></h3><!----> <a id="diffusers.CosmosActionCondition" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.CosmosActionCondition"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/pipelines/cosmos/pipeline_cosmos3_omni.py#L254" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">mode<span class="opacity-60">: typing.Literal['policy', 'forward_dynamics', 'inverse_dynamics']</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">chunk_size<span class="opacity-60">: int</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">domain_name<span class="opacity-60">: str</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">resolution_tier<span class="opacity-60">: int = 480</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">raw_actions<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">image<span class="opacity-60">: typing.Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, NoneType] = None</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">video<span class="opacity-60">: typing.Union[list, numpy.ndarray, torch.Tensor, NoneType] = None</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">view_point<span class="opacity-60">: str = 'ego_view'</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.mode" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.mode"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>mode</strong> (<code>str</code>) — | |
| The action task. One of <code>"forward_dynamics"</code> (roll out future video from a first frame and a given | |
| <code>raw_actions</code> sequence), <code>"inverse_dynamics"</code> (infer the actions connecting the conditioning frames), or | |
| <code>"policy"</code> (jointly roll out future video and actions from the first frame).<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.chunk_size" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.chunk_size"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>chunk_size</strong> (<code>int</code>) — | |
| Number of action transition steps in the chunk. The paired conditioning video spans <code>chunk_size + 1</code> | |
| frames.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.domain_name" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.domain_name"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>domain_name</strong> (<code>str</code>) — | |
| Embodiment domain selecting the domain-aware action projection weights. Must be one of the registered | |
| Cosmos 3 embodiment domains. It also fixes the unpadded action width used to slice predicted actions, | |
| resolved internally from this name (see <code>_EMBODIMENT_TO_RAW_ACTION_DIM</code>).<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.resolution_tier" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.resolution_tier"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>resolution_tier</strong> (<code>int</code>, defaults to <code>480</code>) — | |
| Action conditioning resolution <em>tier</em> (one of <code>256</code>, <code>480</code>, <code>704</code>, <code>720</code>). The tier picks a predefined | |
| canvas whose aspect ratio is closest to the input; the input is downscaled (never upscaled) and padded into | |
| it for conditioning. This is not the output frame size, which tracks the input content. Match the tier to | |
| the input’s native resolution: a lower tier discards detail, while a higher tier adds no resolution (no | |
| upscaling), wastes compute on padding, and is a train/inference mismatch that can hurt quality.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.raw_actions" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.raw_actions"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>raw_actions</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Raw domain action vectors of shape <code>[T, raw_action_dim]</code> driving <code>"forward_dynamics"</code>. Sequences shorter | |
| than <code>chunk_size</code> repeat the last action; longer ones are truncated. Channels beyond the model’s | |
| <code>action_dim</code> are rejected, and narrower inputs are zero-padded up to <code>action_dim</code>.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.image" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.image"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>image</strong> (<code>PIL.Image.Image</code>, <code>np.ndarray</code>, or <code>torch.Tensor</code>, <em>optional</em>) — | |
| Conditioning frame for <code>"policy"</code> / <code>"forward_dynamics"</code>. Mutually exclusive with <code>video</code>.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.video" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.video"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>video</strong> (<code>list</code>, <code>np.ndarray</code>, or <code>torch.Tensor</code>, <em>optional</em>) — | |
| Conditioning video, required for <code>"inverse_dynamics"</code>. For <code>"policy"</code> / <code>"forward_dynamics"</code> only its first | |
| frame is used. Mutually exclusive with <code>image</code>.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.CosmosActionCondition.view_point" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.CosmosActionCondition.view_point"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>view_point</strong> (<code>str</code>, defaults to <code>"ego_view"</code>) — | |
| Camera perspective label used to populate the action caption’s <code>cinematography.framing</code> field. One of | |
| <code>"ego_view"</code>, <code>"third_person_view"</code>, <code>"wrist_view"</code>, or <code>"concat_view"</code>. The action model was trained on | |
| structured JSON captions that carry this viewpoint sentence; an unrecognized label drops the framing field | |
| (with a warning).<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Groups every input required for a Cosmos 3 action-conditioned generation task.</p> <p>Pass this to <code>Cosmos3OmniPipeline.__call__()</code> via the <code>action</code> argument instead of the top-level <code>image</code> / <code>height</code> / <code>width</code> arguments, which are reserved for t2v, i2v runs.</p></div> <!--[1--><h2 class="relative group"><a id="diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Cosmos3OmniPipelineOutput</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.pipelines.cosmos.pipeline_cosmos3_omni.</span><span class="font-semibold">Cosmos3OmniPipelineOutput</span></span></h3><!----> <a id="diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14421/src/diffusers/pipelines/cosmos/pipeline_cosmos3_omni.py#L235" target="_blank"><span><</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">video<span class="opacity-60">: typing.Any</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">sound<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">action<span class="opacity-60">: list[torch.Tensor] | None = None</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput.video" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput.video"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>video</strong> — The generated video. The exact type depends on <code>output_type</code> | |
| passed to the pipeline: a list of PIL frames for <code>"pil"</code> (default), an <code>np.ndarray</code> of shape <code>[T, H, W, C]</code> for <code>"np"</code>, a <code>torch.Tensor</code> of shape <code>[T, C, H, W]</code> for <code>"pt"</code>, or a raw latent tensor | |
| when <code>output_type="latent"</code>.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput.sound" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput.sound"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>sound</strong> — Decoded audio waveform of shape <code>[C, N]</code>. <code>None</code> when | |
| <code>enable_sound=False</code>.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput.action" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.pipelines.cosmos.pipeline_cosmos3_omni.Cosmos3OmniPipelineOutput.action"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>action</strong> — Predicted action tokens. <code>None</code> unless an action mode predicts actions.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Output dataclass for <a href="/docs/diffusers/pr_14421/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniPipeline">Cosmos3OmniPipeline</a>.</p></div> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/cosmos3.md" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]--> | |
| <script> | |
| { | |
| __sveltekit_8gkdeh = { | |
| base: "/docs/diffusers/pr_14421/en", | |
| assets: "/docs/diffusers/pr_14421/en" | |
| }; | |
| const element = document.currentScript.parentElement; | |
| Promise.all([ | |
| import("/docs/diffusers/pr_14421/en/_app/immutable/entry/start.Cpc5Vo8y.js"), | |
| import("/docs/diffusers/pr_14421/en/_app/immutable/entry/app.DMayJ2U5.js") | |
| ]).then(([kit, app]) => { | |
| kit.start(app, element, { | |
| node_ids: [0, 147], | |
| data: [null,null], | |
| form: null, | |
| error: null | |
| }); | |
| }); | |
| } | |
| </script> | |
Xet Storage Details
- Size:
- 191 kB
- Xet hash:
- bdf5177ce652289873a211baa2de58fc64e0f2b82344017c42fa89f37139d453
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.