Buckets:
| import"../chunks/DsnmJJEf.js";import{i as I,h as D,C as A,H as i,a as l,D as m,E as L,s as R}from"../chunks/BtE7mKSK.js";import{p as P,o as V,s as e,f as r,a as t,b as X,c as u,d,r as f,n as Y}from"../chunks/jDjavuwI.js";import{E as h}from"../chunks/SrSJA0zO.js";const S='{"title":"LongCat-AudioDiT","local":"longcat-audiodit","sections":[{"title":"Usage","local":"usage","sections":[],"depth":2},{"title":"Tips","local":"tips","sections":[],"depth":2},{"title":"LongCatAudioDiTPipeline","local":"diffusers.LongCatAudioDiTPipeline","sections":[],"depth":2}],"depth":1}';var N=d('<meta name="hf:doc:metadata"/>'),Z=d("<p>Examples:</p> <!>",1),Q=d("<p>If you get the error message below, you need to finetune the weights for your downstream task:</p> <!>",1),F=d(`<p></p> <!> <!> <p>LongCat-AudioDiT is a text-to-audio diffusion model from Meituan LongCat. The diffusers integration exposes a standard <a href="/docs/diffusers/pr_14229/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a> interface for text-conditioned audio generation.</p> <p>This pipeline was adapted from the LongCat-AudioDiT reference implementation: <a href="https://github.com/meituan-longcat/LongCat-AudioDiT" rel="nofollow">https://github.com/meituan-longcat/LongCat-AudioDiT</a></p> <p>This pipeline supports loading from a local directory or Hugging Face Hub repository in diffusers format (containing <code>text_encoder/</code>, <code>transformer/</code>, <code>vae/</code>, <code>tokenizer/</code>, and <code>scheduler/</code> subfolders).</p> <!> <!> <!> <ul><li><code>audio_duration_s</code> is the most direct way to control output duration.</li> <li>Use <code>generator=torch.Generator("cuda").manual_seed(42)</code> to make generation reproducible.</li> <li>Output shape is <code>(batch, channels, samples)</code> - use <code>.audios[0, 0]</code> to get a single audio sample.</li> <li>The pipeline outputs mono audio (1 channel). If you need stereo, you can duplicate the channel: <code>audio.unsqueeze(0).repeat(1, 2, 1)</code>.</li></ul> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Function invoked when calling the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Instantiate a PyTorch diffusion pipeline from pretrained pipeline weights.</p> <p>The pipeline is set in evaluation mode (<code>model.eval()</code>) by default.</p> <!> <blockquote class="tip"><p>> To use private or <a href="https://huggingface.co/docs/hub/models-gated#gated-models" rel="nofollow">gated</a> models, log-in | |
| with <code>hf > auth login</code>.</p></blockquote> <!></div></div> <!> <p></p>`,1);function O(k,x){P(x,!1),V(()=>{new URLSearchParams(window.location.search).get("fw")}),I();var g=F();D("14fusn5",o=>{var a=N();R(a,"content",S),t(o,a)});var _=e(r(g),2);A(_,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var y=e(_,2);i(y,{title:"LongCat-AudioDiT",local:"longcat-audiodit",headingTag:"h1"});var b=e(y,8);i(b,{title:"Usage",local:"usage",headingTag:"h2"});var w=e(b,2);l(w,{code:"aW1wb3J0JTIwc291bmRmaWxlJTIwYXMlMjBzZiUwQWltcG9ydCUyMHRvcmNoJTBBZnJvbSUyMGRpZmZ1c2VycyUyMGltcG9ydCUyMExvbmdDYXRBdWRpb0RpVFBpcGVsaW5lJTBBJTBBcGlwZWxpbmUlMjAlM0QlMjBMb25nQ2F0QXVkaW9EaVRQaXBlbGluZS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIycnVpeGlhbmdtYSUyRkxvbmdDYXQtQXVkaW9EaVQtMUItRGlmZnVzZXJzJTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5mbG9hdDE2JTJDJTBBKSUwQXBpcGVsaW5lJTIwJTNEJTIwcGlwZWxpbmUudG8oJTIyY3VkYSUyMiklMEElMEFwcm9tcHQlMjAlM0QlMjAlMjJBJTIwY2FsbSUyMG9jZWFuJTIwd2F2ZSUyMGFtYmllbmNlJTIwd2l0aCUyMHNvZnQlMjB3aW5kJTIwaW4lMjB0aGUlMjBiYWNrZ3JvdW5kLiUyMiUwQWF1ZGlvJTIwJTNEJTIwcGlwZWxpbmUoJTBBJTIwJTIwJTIwJTIwcHJvbXB0JTJDJTBBJTIwJTIwJTIwJTIwYXVkaW9fZHVyYXRpb25fcyUzRDUuMCUyQyUwQSUyMCUyMCUyMCUyMG51bV9pbmZlcmVuY2Vfc3RlcHMlM0QxNiUyQyUwQSUyMCUyMCUyMCUyMGd1aWRhbmNlX3NjYWxlJTNENC4wJTJDJTBBJTIwJTIwJTIwJTIwZ2VuZXJhdG9yJTNEdG9yY2guR2VuZXJhdG9yKCUyMmN1ZGElMjIpLm1hbnVhbF9zZWVkKDQyKSUyQyUwQSkuYXVkaW9zJTVCMCUyQyUyMDAlNUQlMEElMEFzZi53cml0ZSglMjJsb25nY2F0LndhdiUyMiUyQyUyMGF1ZGlvJTJDJTIwcGlwZWxpbmUuc2FtcGxlX3JhdGUp",highlighted:`<span class="hljs-keyword">import</span> soundfile <span class="hljs-keyword">as</span> sf | |
| <span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> LongCatAudioDiTPipeline | |
| pipeline = LongCatAudioDiTPipeline.from_pretrained( | |
| <span class="hljs-string">"ruixiangma/LongCat-AudioDiT-1B-Diffusers"</span>, | |
| torch_dtype=torch.float16, | |
| ) | |
| pipeline = pipeline.to(<span class="hljs-string">"cuda"</span>) | |
| prompt = <span class="hljs-string">"A calm ocean wave ambience with soft wind in the background."</span> | |
| audio = pipeline( | |
| prompt, | |
| audio_duration_s=<span class="hljs-number">5.0</span>, | |
| num_inference_steps=<span class="hljs-number">16</span>, | |
| guidance_scale=<span class="hljs-number">4.0</span>, | |
| generator=torch.Generator(<span class="hljs-string">"cuda"</span>).manual_seed(<span class="hljs-number">42</span>), | |
| ).audios[<span class="hljs-number">0</span>, <span class="hljs-number">0</span>] | |
| sf.write(<span class="hljs-string">"longcat.wav"</span>, audio, pipeline.sample_rate)`,lang:"py",wrap:!1});var T=e(w,2);i(T,{title:"Tips",local:"tips",headingTag:"h2"});var M=e(T,4);i(M,{title:"LongCatAudioDiTPipeline",local:"diffusers.LongCatAudioDiTPipeline",headingTag:"h2"});var c=e(M,2),v=u(c);m(v,{name:"class diffusers.LongCatAudioDiTPipeline",anchor:"diffusers.LongCatAudioDiTPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/longcat_audio_dit/pipeline_longcat_audio_dit.py#L99",parameters:[{name:"vae",val:": LongCatAudioDiTVae"},{name:"text_encoder",val:": UMT5EncoderModel"},{name:"tokenizer",val:": PreTrainedTokenizerBase"},{name:"transformer",val:": LongCatAudioDiTTransformer"},{name:"scheduler",val:": diffusers.schedulers.scheduling_flow_match_euler_discrete.FlowMatchEulerDiscreteScheduler | None = None"}]});var p=e(v,2),j=u(p);m(j,{name:"__call__",anchor:"diffusers.LongCatAudioDiTPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/longcat_audio_dit/pipeline_longcat_audio_dit.py#L219",parameters:[{name:"prompt",val:": str | list[str]"},{name:"negative_prompt",val:": str | list[str] | None = None"},{name:"audio_duration_s",val:": float | None = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"num_inference_steps",val:": int = 16"},{name:"guidance_scale",val:": float = 4.0"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"output_type",val:": str = 'np'"},{name:"return_dict",val:": bool = True"},{name:"callback_on_step_end",val:": typing.Optional[typing.Callable[[int, int], NoneType]] = None"},{name:"callback_on_step_end_tensor_inputs",val:": list = ['latents']"}],parametersDescription:[{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.prompt",description:"<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>) — Prompt or prompts that guide audio generation.",name:"prompt"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.negative_prompt",description:"<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — Negative prompt(s) for classifier-free guidance.",name:"negative_prompt"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.audio_duration_s",description:`<strong>audio_duration_s</strong> (<code>float</code>, <em>optional</em>) — | |
| Target audio duration in seconds. Ignored when <code>latents</code> is provided.`,name:"audio_duration_s"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated noisy latents of shape <code>(batch_size, duration, latent_dim)</code>.`,name:"latents"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.num_inference_steps",description:"<strong>num_inference_steps</strong> (<code>int</code>, defaults to 16) — Number of denoising steps.",name:"num_inference_steps"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.guidance_scale",description:"<strong>guidance_scale</strong> (<code>float</code>, defaults to 4.0) — Guidance scale for classifier-free guidance.",name:"guidance_scale"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.generator",description:"<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) — Random generator(s).",name:"generator"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.output_type",description:"<strong>output_type</strong> (<code>str</code>, defaults to <code>"np"</code>) — Output format: <code>"np"</code>, <code>"pt"</code>, or <code>"latent"</code>.",name:"output_type"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.return_dict",description:"<strong>return_dict</strong> (<code>bool</code>, defaults to <code>True</code>) — Whether to return <code>AudioPipelineOutput</code>.",name:"return_dict"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) — | |
| A function called at the end of each denoising step with the pipeline, step index, timestep, and tensor | |
| inputs specified by <code>callback_on_step_end_tensor_inputs</code>.`,name:"callback_on_step_end"},{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>list</code>, defaults to <code>["latents"]</code>) — | |
| Tensor inputs passed to <code>callback_on_step_end</code>.`,name:"callback_on_step_end_tensor_inputs"}]});var G=e(j,4);h(G,{anchor:"diffusers.LongCatAudioDiTPipeline.__call__.example",children:(o,a)=>{var n=Z(),s=e(r(n),2);l(s,{code:"aW1wb3J0JTIwc291bmRmaWxlJTIwYXMlMjBzZiUwQWltcG9ydCUyMHRvcmNoJTBBZnJvbSUyMGRpZmZ1c2VycyUyMGltcG9ydCUyMExvbmdDYXRBdWRpb0RpVFBpcGVsaW5lJTBBJTBBcGlwZSUyMCUzRCUyMExvbmdDYXRBdWRpb0RpVFBpcGVsaW5lLmZyb21fcHJldHJhaW5lZCglMjJydWl4aWFuZ21hJTJGTG9uZ0NhdC1BdWRpb0RpVC0xQi1EaWZmdXNlcnMlMjIpJTBBcGlwZS50byglMjJjdWRhJTIyKSUwQSUwQXByb21wdCUyMCUzRCUyMCUyMkElMjBjYWxtJTIwb2NlYW4lMjB3YXZlJTIwYW1iaWVuY2UlMjB3aXRoJTIwc29mdCUyMHdpbmQlMjBpbiUyMHRoZSUyMGJhY2tncm91bmQuJTIyJTBBYXVkaW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMGF1ZGlvX2R1cmF0aW9uX3MlM0Q1LjAlMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNEMjAlMkMlMEElMjAlMjAlMjAlMjBndWlkYW5jZV9zY2FsZSUzRDQuMCUyQyUwQSUyMCUyMCUyMCUyMGdlbmVyYXRvciUzRHRvcmNoLkdlbmVyYXRvciglMjJjdWRhJTIyKS5tYW51YWxfc2VlZCg0MiklMkMlMEEpLmF1ZGlvcyU1QjAlMkMlMjAwJTVEJTBBc2Yud3JpdGUoJTIyb3V0cHV0LndhdiUyMiUyQyUyMGF1ZGlvJTJDJTIwcGlwZS5zYW1wbGVfcmF0ZSk=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> soundfile <span class="hljs-keyword">as</span> sf | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> LongCatAudioDiTPipeline | |
| <span class="hljs-meta">>>> </span>pipe = LongCatAudioDiTPipeline.from_pretrained(<span class="hljs-string">"ruixiangma/LongCat-AudioDiT-1B-Diffusers"</span>) | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"A calm ocean wave ambience with soft wind in the background."</span> | |
| <span class="hljs-meta">>>> </span>audio = pipe( | |
| <span class="hljs-meta">... </span> prompt, | |
| <span class="hljs-meta">... </span> audio_duration_s=<span class="hljs-number">5.0</span>, | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">20</span>, | |
| <span class="hljs-meta">... </span> guidance_scale=<span class="hljs-number">4.0</span>, | |
| <span class="hljs-meta">... </span> generator=torch.Generator(<span class="hljs-string">"cuda"</span>).manual_seed(<span class="hljs-number">42</span>), | |
| <span class="hljs-meta">... </span>).audios[<span class="hljs-number">0</span>, <span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>sf.write(<span class="hljs-string">"output.wav"</span>, audio, pipe.sample_rate)`,lang:"py",wrap:!1}),t(o,n)},$$slots:{default:!0}}),f(p);var U=e(p,2),J=u(U);m(J,{name:"from_pretrained",anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/pipeline_utils.py#L629",parameters:[{name:"pretrained_model_name_or_path",val:": str | os.PathLike"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.pretrained_model_name_or_path",description:`<strong>pretrained_model_name_or_path</strong> (<code>str</code> or <code>os.PathLike</code>, <em>optional</em>) — | |
| Can be either:</p> | |
| <ul> | |
| <li>A string, the <em>repo id</em> (for example <code>CompVis/ldm-text2im-large-256</code>) of a pretrained pipeline | |
| hosted on the Hub.</li> | |
| <li>A path to a <em>directory</em> (for example <code>./my_pipeline_directory/</code>) containing pipeline weights | |
| saved using | |
| <a href="/docs/diffusers/pr_14229/en/api/pipelines/overview#diffusers.DiffusionPipeline.save_pretrained">save_pretrained()</a>.</li> | |
| <li>A path to a <em>directory</em> (for example <code>./my_pipeline_directory/</code>) containing a dduf file</li> | |
| </ul>`,name:"pretrained_model_name_or_path"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.dtype",description:`<strong>dtype</strong> (<code>torch.dtype</code> or <code>dict[str, Union[str, torch.dtype]]</code>, <em>optional</em>) — | |
| Override the default <code>torch.dtype</code> and load the model with another dtype. To load submodels with | |
| different dtype pass a <code>dict</code> (for example <code>{'transformer': torch.bfloat16, 'vae': torch.float16}</code>). | |
| Set the default dtype for unspecified components with <code>default</code> (for example <code>{'transformer': torch.bfloat16, 'default': torch.float16}</code>). If a component is not specified and no default is set, | |
| <code>torch.float32</code> is used.`,name:"dtype"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.custom_pipeline",description:`<strong>custom_pipeline</strong> (<code>str</code>, <em>optional</em>) —</p> | |
| <blockquote class="warning"> | |
| <p>> 🧪 This is an experimental feature and may change in the future.</p> | |
| </blockquote> | |
| <p>Can be either:</p> | |
| <ul> | |
| <li>A string, the <em>repo id</em> (for example <code>hf-internal-testing/diffusers-dummy-pipeline</code>) of a custom | |
| pipeline hosted on the Hub. The repository must contain a file called pipeline.py that defines | |
| the custom pipeline.</li> | |
| <li>A string, the <em>file name</em> of a community pipeline hosted on GitHub under | |
| <a href="https://github.com/huggingface/diffusers/tree/main/examples/community" rel="nofollow">Community</a>. Valid file | |
| names must match the file name and not the pipeline script (<code>clip_guided_stable_diffusion</code> | |
| instead of <code>clip_guided_stable_diffusion.py</code>). Community pipelines are always loaded from the | |
| current main branch of GitHub.</li> | |
| <li>A path to a directory (<code>./my_pipeline_directory/</code>) containing a custom pipeline. The directory | |
| must contain a file called <code>pipeline.py</code> that defines the custom pipeline.</li> | |
| </ul> | |
| <p>For more information on how to load and create custom pipelines, please have a look at <a href="https://huggingface.co/docs/diffusers/using-diffusers/custom_pipeline_overview" rel="nofollow">Loading and | |
| Adding Custom | |
| Pipelines</a>`,name:"custom_pipeline"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.force_download",description:`<strong>force_download</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to force the (re-)download of the model weights and configuration files, overriding the | |
| cached versions if they exist.`,name:"force_download"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.cache_dir",description:`<strong>cache_dir</strong> (<code>Union[str, os.PathLike]</code>, <em>optional</em>) — | |
| Path to a directory where a downloaded pretrained model configuration is cached if the standard cache | |
| is not used.`,name:"cache_dir"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.proxies",description:`<strong>proxies</strong> (<code>Dict[str, str]</code>, <em>optional</em>) — | |
| A dictionary of proxy servers to use by protocol or endpoint, for example, <code>{'http': 'foo.bar:3128', 'http://hostname': 'foo.bar:4012'}</code>. The proxies are used on each request.`,name:"proxies"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.output_loading_info(bool,",description:`<strong>output_loading_info(<code>bool</code>,</strong> <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to also return a dictionary containing missing keys, unexpected keys and error messages.`,name:"output_loading_info(bool,"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.local_files_only",description:`<strong>local_files_only</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to only load local model weights and configuration files or not. If set to <code>True</code>, the model | |
| won’t be downloaded from the Hub.`,name:"local_files_only"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.token",description:`<strong>token</strong> (<code>str</code> or <em>bool</em>, <em>optional</em>) — | |
| The token to use as HTTP bearer authorization for remote files. If <code>True</code>, the token generated from | |
| <code>diffusers-cli login</code> (stored in <code>~/.huggingface</code>) is used.`,name:"token"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.revision",description:`<strong>revision</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"main"</code>) — | |
| The specific model version to use. It can be a branch name, a tag name, a commit id, or any identifier | |
| allowed by Git.`,name:"revision"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.custom_revision",description:`<strong>custom_revision</strong> (<code>str</code>, <em>optional</em>) — | |
| The specific model version to use. It can be a branch name, a tag name, or a commit id similar to | |
| <code>revision</code> when loading a custom pipeline from the Hub. Defaults to the latest stable 🤗 Diffusers | |
| version.`,name:"custom_revision"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.mirror",description:`<strong>mirror</strong> (<code>str</code>, <em>optional</em>) — | |
| Mirror source to resolve accessibility issues if you’re downloading a model in China. We do not | |
| guarantee the timeliness or safety of the source, and you should refer to the mirror site for more | |
| information.`,name:"mirror"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.device_map",description:`<strong>device_map</strong> (<code>str</code>, <em>optional</em>) — | |
| Strategy that dictates how the different components of a pipeline should be placed on available | |
| devices. Currently, only “balanced” <code>device_map</code> is supported. Check out | |
| <a href="https://huggingface.co/docs/diffusers/main/en/tutorials/inference_with_big_models#device-placement" rel="nofollow">this</a> | |
| to know more.`,name:"device_map"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.max_memory",description:`<strong>max_memory</strong> (<code>Dict</code>, <em>optional</em>) — | |
| A dictionary device identifier for the maximum memory. Will default to the maximum memory available for | |
| each GPU and the available CPU RAM if unset.`,name:"max_memory"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.offload_folder",description:`<strong>offload_folder</strong> (<code>str</code> or <code>os.PathLike</code>, <em>optional</em>) — | |
| The path to offload weights if device_map contains the value <code>"disk"</code>.`,name:"offload_folder"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.offload_state_dict",description:`<strong>offload_state_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| If <code>True</code>, temporarily offloads the CPU state dict to the hard drive to avoid running out of CPU RAM if | |
| the weight of the CPU state dict + the biggest shard of the checkpoint does not fit. Defaults to <code>True</code> | |
| when there is some disk offload.`,name:"offload_state_dict"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.low_cpu_mem_usage",description:`<strong>low_cpu_mem_usage</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code> if torch version >= 1.9.0 else <code>False</code>) — | |
| Speed up model loading only loading the pretrained weights and not initializing the weights. This also | |
| tries to not use more than 1x model size in CPU memory (including peak memory) while loading the model. | |
| Only supported for PyTorch >= 1.9.0. If you are using an older version of PyTorch, setting this | |
| argument to <code>True</code> will raise an error.`,name:"low_cpu_mem_usage"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.use_safetensors",description:`<strong>use_safetensors</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>None</code>) — | |
| If set to <code>None</code>, the safetensors weights are downloaded if they’re available <strong>and</strong> if the | |
| safetensors library is installed. If set to <code>True</code>, the model is forcibly loaded from safetensors | |
| weights. If set to <code>False</code>, safetensors weights are not loaded.`,name:"use_safetensors"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.use_onnx",description:`<strong>use_onnx</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>None</code>) — | |
| If set to <code>True</code>, ONNX weights will always be downloaded if present. If set to <code>False</code>, ONNX weights | |
| will never be downloaded. By default <code>use_onnx</code> defaults to the <code>_is_onnx</code> class attribute which is | |
| <code>False</code> for non-ONNX pipelines and <code>True</code> for ONNX pipelines. ONNX weights include both files ending | |
| with <code>.onnx</code> and <code>.pb</code>.`,name:"use_onnx"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.kwargs",description:`<strong>kwargs</strong> (remaining dictionary of keyword arguments, <em>optional</em>) — | |
| Can be used to overwrite load and saveable variables (the pipeline components of the specific pipeline | |
| class). The overwritten components are passed directly to the pipelines <code>__init__</code> method. See example | |
| below for more information.`,name:"kwargs"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.variant",description:`<strong>variant</strong> (<code>str</code>, <em>optional</em>) — | |
| Load weights from a specified variant filename such as <code>"fp16"</code> or <code>"ema"</code>. This is ignored when | |
| loading <code>from_flax</code>.`,name:"variant"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.dduf_file(str,",description:`<strong>dduf_file(<code>str</code>,</strong> <em>optional</em>) — | |
| Load weights from the specified dduf file. <deprecated> This argument is deprecated and will be removed | |
| in version 0.41.0. </deprecated>`,name:"dduf_file(str,"},{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.disable_mmap",description:`<strong>disable_mmap</strong> (‘bool’, <em>optional</em>, defaults to ‘False’) — | |
| Whether to disable mmap when loading a Safetensors model. This option can perform better when the model | |
| is on a network mount or hard drive, which may not handle the seeky-ness of mmap very well.`,name:"disable_mmap"}]});var C=e(J,6);h(C,{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.example",children:(o,a)=>{var n=Q(),s=e(r(n),2);l(s,{code:"U29tZSUyMHdlaWdodHMlMjBvZiUyMFVOZXQyRENvbmRpdGlvbk1vZGVsJTIwd2VyZSUyMG5vdCUyMGluaXRpYWxpemVkJTIwZnJvbSUyMHRoZSUyMG1vZGVsJTIwY2hlY2twb2ludCUyMGF0JTIwc3RhYmxlLWRpZmZ1c2lvbi12MS01JTJGc3RhYmxlLWRpZmZ1c2lvbi12MS01JTIwYW5kJTIwYXJlJTIwbmV3bHklMjBpbml0aWFsaXplZCUyMGJlY2F1c2UlMjB0aGUlMjBzaGFwZXMlMjBkaWQlMjBub3QlMjBtYXRjaCUzQSUwQS0lMjBjb252X2luLndlaWdodCUzQSUyMGZvdW5kJTIwc2hhcGUlMjB0b3JjaC5TaXplKCU1QjMyMCUyQyUyMDQlMkMlMjAzJTJDJTIwMyU1RCklMjBpbiUyMHRoZSUyMGNoZWNrcG9pbnQlMjBhbmQlMjB0b3JjaC5TaXplKCU1QjMyMCUyQyUyMDklMkMlMjAzJTJDJTIwMyU1RCklMjBpbiUyMHRoZSUyMG1vZGVsJTIwaW5zdGFudGlhdGVkJTBBWW91JTIwc2hvdWxkJTIwcHJvYmFibHklMjBUUkFJTiUyMHRoaXMlMjBtb2RlbCUyMG9uJTIwYSUyMGRvd24tc3RyZWFtJTIwdGFzayUyMHRvJTIwYmUlMjBhYmxlJTIwdG8lMjB1c2UlMjBpdCUyMGZvciUyMHByZWRpY3Rpb25zJTIwYW5kJTIwaW5mZXJlbmNlLg==",highlighted:`Some weights of UNet2DConditionModel were not initialized from the model checkpoint <span class="hljs-built_in">at</span> stable-<span class="hljs-keyword">diffusion-v1-5/stable-diffusion-v1-5 </span><span class="hljs-keyword">and </span>are newly initialized <span class="hljs-keyword">because </span>the <span class="hljs-keyword">shapes </span><span class="hljs-keyword">did </span>not match: | |
| - conv_in.weight: found <span class="hljs-keyword">shape </span>torch.Size([<span class="hljs-number">320</span>, <span class="hljs-number">4</span>, <span class="hljs-number">3</span>, <span class="hljs-number">3</span>]) in the checkpoint <span class="hljs-keyword">and </span>torch.Size([<span class="hljs-number">320</span>, <span class="hljs-number">9</span>, <span class="hljs-number">3</span>, <span class="hljs-number">3</span>]) in the model <span class="hljs-keyword">instantiated | |
| </span>You <span class="hljs-keyword">should </span>probably TRAIN this model on a down-stream task to <span class="hljs-keyword">be </span>able to use it for predictions <span class="hljs-keyword">and </span>inference.`,lang:"",wrap:!1}),t(o,n)},$$slots:{default:!0}});var B=e(C,4);h(B,{anchor:"diffusers.LongCatAudioDiTPipeline.from_pretrained.example-2",children:(o,a)=>{var n=Z(),s=e(r(n),2);l(s,{code:"ZnJvbSUyMGRpZmZ1c2VycyUyMGltcG9ydCUyMERpZmZ1c2lvblBpcGVsaW5lJTBBJTBBJTIzJTIwRG93bmxvYWQlMjBwaXBlbGluZSUyMGZyb20lMjBodWdnaW5nZmFjZS5jbyUyMGFuZCUyMGNhY2hlLiUwQXBpcGVsaW5lJTIwJTNEJTIwRGlmZnVzaW9uUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUyMkNvbXBWaXMlMkZsZG0tdGV4dDJpbS1sYXJnZS0yNTYlMjIpJTBBJTBBJTIzJTIwRG93bmxvYWQlMjBwaXBlbGluZSUyMHRoYXQlMjByZXF1aXJlcyUyMGFuJTIwYXV0aG9yaXphdGlvbiUyMHRva2VuJTBBJTIzJTIwRm9yJTIwbW9yZSUyMGluZm9ybWF0aW9uJTIwb24lMjBhY2Nlc3MlMjB0b2tlbnMlMkMlMjBwbGVhc2UlMjByZWZlciUyMHRvJTIwdGhpcyUyMHNlY3Rpb24lMEElMjMlMjBvZiUyMHRoZSUyMGRvY3VtZW50YXRpb24lNUQoaHR0cHMlM0ElMkYlMkZodWdnaW5nZmFjZS5jbyUyRmRvY3MlMkZodWIlMkZzZWN1cml0eS10b2tlbnMpJTBBcGlwZWxpbmUlMjAlM0QlMjBEaWZmdXNpb25QaXBlbGluZS5mcm9tX3ByZXRyYWluZWQoJTIyc3RhYmxlLWRpZmZ1c2lvbi12MS01JTJGc3RhYmxlLWRpZmZ1c2lvbi12MS01JTIyKSUwQSUwQSUyMyUyMFVzZSUyMGElMjBkaWZmZXJlbnQlMjBzY2hlZHVsZXIlMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTE1TRGlzY3JldGVTY2hlZHVsZXIlMEElMEFzY2hlZHVsZXIlMjAlM0QlMjBMTVNEaXNjcmV0ZVNjaGVkdWxlci5mcm9tX2NvbmZpZyhwaXBlbGluZS5zY2hlZHVsZXIuY29uZmlnKSUwQXBpcGVsaW5lLnNjaGVkdWxlciUyMCUzRCUyMHNjaGVkdWxlcg==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> DiffusionPipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Download pipeline from huggingface.co and cache.</span> | |
| <span class="hljs-meta">>>> </span>pipeline = DiffusionPipeline.from_pretrained(<span class="hljs-string">"CompVis/ldm-text2im-large-256"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Download pipeline that requires an authorization token</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># For more information on access tokens, please refer to this section</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># of the documentation](https://huggingface.co/docs/hub/security-tokens)</span> | |
| <span class="hljs-meta">>>> </span>pipeline = DiffusionPipeline.from_pretrained(<span class="hljs-string">"stable-diffusion-v1-5/stable-diffusion-v1-5"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Use a different scheduler</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> LMSDiscreteScheduler | |
| <span class="hljs-meta">>>> </span>scheduler = LMSDiscreteScheduler.from_config(pipeline.scheduler.config) | |
| <span class="hljs-meta">>>> </span>pipeline.scheduler = scheduler`,lang:"py",wrap:!1}),t(o,n)},$$slots:{default:!0}}),f(U),f(c);var W=e(c,2);L(W,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/longcat_audio_dit.md"}),Y(2),t(k,g),X()}export{O as component}; | |
Xet Storage Details
- Size:
- 29.6 kB
- Xet hash:
- 3b036f081cefc299f23ace7b0f4e07f417083b1a09dde6b8554e4db3214255b7
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.