Buckets:

download
raw
23.8 kB
import"../chunks/DsnmJJEf.js";import{i as W,h as C,C as N,H as c,a as j,D as t,E as G,s as V}from"../chunks/BtE7mKSK.js";import{p as q,o as A,s as e,f as v,a as m,b as R,c as n,d as g,r as s,n as o}from"../chunks/jDjavuwI.js";import{E as X}from"../chunks/SrSJA0zO.js";const F='{"title":"JoyAI-Image-Edit-Plus","local":"joyai-image-edit-plus","sections":[{"title":"JoyImageEditPlusPipeline","local":"diffusers.JoyImageEditPlusPipeline","sections":[],"depth":2},{"title":"JoyImageEditPlusPipelineOutput","local":"diffusers.JoyImageEditPlusPipelineOutput","sections":[],"depth":2}],"depth":1}';var Q=g('<meta name="hf:doc:metadata"/>'),S=g("<p>Examples:</p> <!>",1),L=g(`<p></p> <!> <!> <p><a href="https://github.com/jd-opensource/JoyAI-Image" rel="nofollow">JoyAI-Image</a> is a unified multimodal foundation model for image understanding, text-to-image generation, and instruction-guided image editing. It combines an 8B Multimodal Large Language Model (MLLM) with a 16B Multimodal Diffusion Transformer (MMDiT).</p> <p>JoyAI-Image-Edit-Plus is a multi-image instruction-guided editing model that accepts <strong>multiple reference images</strong> and a text instruction to generate a new image that combines elements from the references according to the instruction. It supports 1–5 reference images per sample.</p> <table><thead><tr><th align="center">Model</th><th align="center">Description</th><th align="center">Download</th></tr></thead><tbody><tr><td align="center">JoyAI-Image-Edit-Plus</td><td align="center">Multi-image instruction-guided editing with element composition from multiple references</td><td align="center"><a href="https://huggingface.co/jdopensource/JoyAI-Image-Edit-Plus-Diffusers" rel="nofollow">Hugging Face</a></td></tr></tbody></table> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Diffusion pipeline for multi-image instruction-guided editing using JoyImage Edit Plus.</p> <p>Supports multiple reference images with different resolutions. Each reference image is independently VAE-encoded
and patchified, then concatenated with the target noise patches for joint denoising.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Function invoked when calling the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encode prompts with inline &#60;image> tokens via the Qwen3-VL processor.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Prepare 6D padded latent tensor with target noise + reference image latents.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for JoyImage Edit Plus multi-image editing pipelines.</p></div> <!> <p></p>`,1);function O(P,T){q(T,!1),A(()=>{new URLSearchParams(window.location.search).get("fw")}),W();var u=L();C("ka77ad",a=>{var p=Q();V(p,"content",F),m(a,p)});var h=e(v(u),2);N(h,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var _=e(h,2);c(_,{title:"JoyAI-Image-Edit-Plus",local:"joyai-image-edit-plus",headingTag:"h1"});var f=e(_,8);j(f,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwUElMJTIwaW1wb3J0JTIwSW1hZ2UlMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSm95SW1hZ2VFZGl0UGx1c1BpcGVsaW5lJTBBJTBBcGlwZWxpbmUlMjAlM0QlMjBKb3lJbWFnZUVkaXRQbHVzUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMmpkb3BlbnNvdXJjZSUyRkpveUFJLUltYWdlLUVkaXQtUGx1cy1EaWZmdXNlcnMlMjIlMkMlMjB0b3JjaF9kdHlwZSUzRHRvcmNoLmJmbG9hdDE2JTBBKSUwQXBpcGVsaW5lLnRvKCUyMmN1ZGElMjIpJTBBJTBBaW1hZ2VzJTIwJTNEJTIwJTVCJTBBJTIwJTIwJTIwJTIwSW1hZ2Uub3BlbiglMjJyZWZlcmVuY2VfMC5wbmclMjIpLmNvbnZlcnQoJTIyUkdCJTIyKSUyQyUwQSUyMCUyMCUyMCUyMEltYWdlLm9wZW4oJTIycmVmZXJlbmNlXzEucG5nJTIyKS5jb252ZXJ0KCUyMlJHQiUyMiklMkMlMEElNUQlMEElMEF0YXJnZXRfaCUyQyUyMHRhcmdldF93JTIwJTNEJTIwcGlwZWxpbmUuaW1hZ2VfcHJvY2Vzc29yLmdldF9kZWZhdWx0X2hlaWdodF93aWR0aChpbWFnZXMlNUItMSU1RCklMEElMEFvdXRwdXQlMjAlM0QlMjBwaXBlbGluZSglMEElMjAlMjAlMjAlMjBpbWFnZXMlM0RpbWFnZXMlMkMlMEElMjAlMjAlMjAlMjBwcm9tcHQlM0QlMjJDb21iaW5lJTIwdGhlJTIwcGVyc29uJTIwZnJvbSUyMHRoZSUyMHNlY29uZCUyMGltYWdlJTIwd2l0aCUyMHRoZSUyMHNjZW5lJTIwZnJvbSUyMHRoZSUyMGZpcnN0JTIwaW1hZ2UuJTIyJTJDJTBBJTIwJTIwJTIwJTIwbmVnYXRpdmVfcHJvbXB0JTNEJTIybG93JTIwcXVhbGl0eSUyQyUyMGJsdXJyeSUyQyUyMGRlZm9ybWVkJTIyJTJDJTBBJTIwJTIwJTIwJTIwaGVpZ2h0JTNEdGFyZ2V0X2glMkMlMEElMjAlMjAlMjAlMjB3aWR0aCUzRHRhcmdldF93JTJDJTBBJTIwJTIwJTIwJTIwbnVtX2luZmVyZW5jZV9zdGVwcyUzRDMwJTJDJTBBJTIwJTIwJTIwJTIwZ3VpZGFuY2Vfc2NhbGUlM0Q0LjAlMkMlMEElMjAlMjAlMjAlMjBnZW5lcmF0b3IlM0R0b3JjaC5HZW5lcmF0b3IoJTIyY3VkYSUyMikubWFudWFsX3NlZWQoNDIpJTJDJTBBKS5pbWFnZXMlNUIwJTVEJTBBb3V0cHV0LnNhdmUoJTIyam95aW1hZ2VfZWRpdF9wbHVzX291dHB1dC5wbmclMjIp",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> JoyImageEditPlusPipeline
pipeline = JoyImageEditPlusPipeline.from_pretrained(
<span class="hljs-string">&quot;jdopensource/JoyAI-Image-Edit-Plus-Diffusers&quot;</span>, torch_dtype=torch.bfloat16
)
pipeline.to(<span class="hljs-string">&quot;cuda&quot;</span>)
images = [
Image.<span class="hljs-built_in">open</span>(<span class="hljs-string">&quot;reference_0.png&quot;</span>).convert(<span class="hljs-string">&quot;RGB&quot;</span>),
Image.<span class="hljs-built_in">open</span>(<span class="hljs-string">&quot;reference_1.png&quot;</span>).convert(<span class="hljs-string">&quot;RGB&quot;</span>),
]
target_h, target_w = pipeline.image_processor.get_default_height_width(images[-<span class="hljs-number">1</span>])
output = pipeline(
images=images,
prompt=<span class="hljs-string">&quot;Combine the person from the second image with the scene from the first image.&quot;</span>,
negative_prompt=<span class="hljs-string">&quot;low quality, blurry, deformed&quot;</span>,
height=target_h,
width=target_w,
num_inference_steps=<span class="hljs-number">30</span>,
guidance_scale=<span class="hljs-number">4.0</span>,
generator=torch.Generator(<span class="hljs-string">&quot;cuda&quot;</span>).manual_seed(<span class="hljs-number">42</span>),
).images[<span class="hljs-number">0</span>]
output.save(<span class="hljs-string">&quot;joyimage_edit_plus_output.png&quot;</span>)`,lang:"python",wrap:!1});var y=e(f,2);c(y,{title:"JoyImageEditPlusPipeline",local:"diffusers.JoyImageEditPlusPipeline",headingTag:"h2"});var i=e(y,2),M=n(i);t(M,{name:"class diffusers.JoyImageEditPlusPipeline",anchor:"diffusers.JoyImageEditPlusPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit_plus.py#L129",parameters:[{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"},{name:"vae",val:": AutoencoderKLWan"},{name:"text_encoder",val:": Qwen3VLForConditionalGeneration"},{name:"tokenizer",val:": Qwen2Tokenizer"},{name:"transformer",val:": JoyImageEditPlusTransformer3DModel"},{name:"processor",val:": Qwen3VLProcessor"},{name:"text_token_max_length",val:": int = 2048"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPlusPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14230/en/api/schedulers/flow_match_euler_discrete#diffusers.FlowMatchEulerDiscreteScheduler">FlowMatchEulerDiscreteScheduler</a>) &#x2014;
A scheduler to be used in combination with <code>transformer</code> to denoise the encoded image latents.`,name:"scheduler"},{anchor:"diffusers.JoyImageEditPlusPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/autoencoder_kl_wan#diffusers.AutoencoderKLWan">AutoencoderKLWan</a>) &#x2014;
Variational Auto-Encoder (VAE) model to encode and decode images to and from latent representations.`,name:"vae"},{anchor:"diffusers.JoyImageEditPlusPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>Qwen3VLForConditionalGeneration</code>) &#x2014;
Multimodal text encoder for prompt encoding with inline image understanding.`,name:"text_encoder"},{anchor:"diffusers.JoyImageEditPlusPipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>Qwen2Tokenizer</code>) &#x2014;
Tokenizer for text processing.`,name:"tokenizer"},{anchor:"diffusers.JoyImageEditPlusPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/transformer_joyimage_edit_plus#diffusers.JoyImageEditPlusTransformer3DModel">JoyImageEditPlusTransformer3DModel</a>) &#x2014;
Conditional Transformer (MMDiT) architecture to denoise the encoded image latents.`,name:"transformer"},{anchor:"diffusers.JoyImageEditPlusPipeline.processor",description:`<strong>processor</strong> (<code>Qwen3VLProcessor</code>) &#x2014;
Processor for multimodal inputs (text + images).`,name:"processor"},{anchor:"diffusers.JoyImageEditPlusPipeline.text_token_max_length",description:`<strong>text_token_max_length</strong> (<code>int</code>, defaults to <code>2048</code>) &#x2014;
Maximum token length for text encoding.`,name:"text_token_max_length"}]});var l=e(M,6),J=n(l);t(J,{name:"__call__",anchor:"diffusers.JoyImageEditPlusPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit_plus.py#L441",parameters:[{name:"images",val:": list[PIL.Image.Image] | list[list[PIL.Image.Image]] | None = None"},{name:"prompt",val:": str | list[str] = None"},{name:"height",val:": int | None = None"},{name:"width",val:": int | None = None"},{name:"num_inference_steps",val:": int = 30"},{name:"timesteps",val:": list = None"},{name:"sigmas",val:": list = None"},{name:"guidance_scale",val:": float = 4.0"},{name:"negative_prompt",val:": str | list[str] | None = None"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": str | None = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": list = ['latents']"},{name:"max_sequence_length",val:": int = 4096"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.images",description:`<strong>images</strong> (<code>list[Image.Image]</code> or <code>list[list[Image.Image]]</code>, <em>optional</em>) &#x2014;
Reference images for editing. Each image can have a different resolution. If a flat list is provided,
it is treated as one sample with multiple references.`,name:"images"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts to guide the image generation. If not defined, one has to pass <code>prompt_embeds</code>
instead.`,name:"prompt"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, <em>optional</em>) &#x2014;
The height in pixels of the generated image. If <code>None</code>, determined from the last reference image.`,name:"height"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, <em>optional</em>) &#x2014;
The width in pixels of the generated image. If <code>None</code>, determined from the last reference image.`,name:"width"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to <code>30</code>) &#x2014;
The number of denoising steps. More denoising steps usually lead to a higher quality image at the
expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>list[int]</code>, <em>optional</em>) &#x2014;
Custom timesteps to use for the denoising process. If not defined, equal spacing is used.`,name:"timesteps"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<code>list[float]</code>, <em>optional</em>) &#x2014;
Custom sigmas to use for the denoising process.`,name:"sigmas"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.guidance_scale",description:`<strong>guidance_scale</strong> (<code>float</code>, <em>optional</em>, defaults to <code>4.0</code>) &#x2014;
Classifier-free guidance scale. Higher values encourage the model to generate images more aligned with
the <code>prompt</code> at the expense of lower image quality.`,name:"guidance_scale"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the image generation. If not defined, a blank prompt is used for
classifier-free guidance.`,name:"negative_prompt"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) &#x2014;
One or a list of <a href="https://pytorch.org/docs/stable/generated/torch.Generator.html" rel="nofollow">torch generator(s)</a>
to make generation deterministic.`,name:"generator"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated noisy latents to be used as inputs for image generation.`,name:"latents"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings. Can be used to easily tweak text inputs.`,name:"prompt_embeds"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Attention mask for pre-generated text embeddings.`,name:"prompt_embeds_mask"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings.`,name:"negative_prompt_embeds"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.negative_prompt_embeds_mask",description:`<strong>negative_prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Attention mask for pre-generated negative text embeddings.`,name:"negative_prompt_embeds_mask"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>&quot;pil&quot;</code>) &#x2014;
The output format of the generated image. Choose between <code>&quot;pil&quot;</code> (<code>PIL.Image.Image</code>), <code>&quot;np&quot;</code>
(<code>np.ndarray</code>), <code>&quot;pt&quot;</code> (<code>torch.Tensor</code>), or <code>&quot;latent&quot;</code> for raw latent output.`,name:"output_type"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether or not to return a <a href="/docs/diffusers/pr_14230/en/api/pipelines/joyimage_edit_plus#diffusers.JoyImageEditPlusPipelineOutput">JoyImageEditPlusPipelineOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) &#x2014;
A function called at the end of each denoising step with arguments: the pipeline, step index, timestep,
and a dict of callback tensor inputs.`,name:"callback_on_step_end"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>list[str]</code>, <em>optional</em>, defaults to <code>[&quot;latents&quot;]</code>) &#x2014;
The list of tensor inputs for the <code>callback_on_step_end</code> function.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, <em>optional</em>, defaults to <code>4096</code>) &#x2014;
Maximum sequence length for the text encoder.`,name:"max_sequence_length"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>If <code>return_dict</code> is <code>True</code>, <a
href="/docs/diffusers/pr_14230/en/api/pipelines/joyimage_edit_plus#diffusers.JoyImageEditPlusPipelineOutput"
>JoyImageEditPlusPipelineOutput</a> is returned, otherwise a <code>tuple</code> is
returned where the first element is a list of generated images.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><a
href="/docs/diffusers/pr_14230/en/api/pipelines/joyimage_edit_plus#diffusers.JoyImageEditPlusPipelineOutput"
>JoyImageEditPlusPipelineOutput</a> or <code>tuple</code></p>
`});var E=e(J,4);X(E,{anchor:"diffusers.JoyImageEditPlusPipeline.__call__.example",children:(a,p)=>{var w=S(),B=e(v(w),2);j(B,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSm95SW1hZ2VFZGl0UGx1c1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGxvYWRfaW1hZ2UlMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMmpkb3BlbnNvdXJjZSUyRkpveUFJLUltYWdlLUVkaXQtUGx1cy1EaWZmdXNlcnMlMjIlMEFwaXBlJTIwJTNEJTIwSm95SW1hZ2VFZGl0UGx1c1BpcGVsaW5lLmZyb21fcHJldHJhaW5lZChtb2RlbF9pZCUyQyUyMHRvcmNoX2R0eXBlJTNEdG9yY2guYmZsb2F0MTYpJTBBcGlwZS50byglMjJjdWRhJTIyKSUwQSUwQWltYWdlcyUyMCUzRCUyMCU1QiUwQSUyMCUyMCUyMCUyMGxvYWRfaW1hZ2UoJTIyZG9nLnBuZyUyMiklMkMlMEElMjAlMjAlMjAlMjBsb2FkX2ltYWdlKCUyMnBlcnNvbi5wbmclMjIpJTJDJTBBJTVEJTBBb3V0cHV0JTIwJTNEJTIwcGlwZSglMEElMjAlMjAlMjAlMjBpbWFnZXMlM0RpbWFnZXMlMkMlMEElMjAlMjAlMjAlMjBwcm9tcHQlM0QlMjJMZXQlMjB0aGUlMjBwZXJzb24lMjBsb3ZpbmdseSUyMHBsYXklMjB3aXRoJTIwdGhlJTIwZG9nLiUyMiUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDEwMjQlMkMlMEElMjAlMjAlMjAlMjB3aWR0aCUzRDEwMjQlMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNEMzAlMkMlMEElMjAlMjAlMjAlMjBndWlkYW5jZV9zY2FsZSUzRDQuMCUyQyUwQSUyMCUyMCUyMCUyMGdlbmVyYXRvciUzRHRvcmNoLm1hbnVhbF9zZWVkKDQyKSUyQyUwQSklMEFvdXRwdXQuaW1hZ2VzJTVCMCU1RC5zYXZlKCUyMm91dHB1dC5wbmclMjIp",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> torch
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> JoyImageEditPlusPipeline
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> load_image
<span class="hljs-meta">&gt;&gt;&gt; </span>model_id = <span class="hljs-string">&quot;jdopensource/JoyAI-Image-Edit-Plus-Diffusers&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe = JoyImageEditPlusPipeline.from_pretrained(model_id, torch_dtype=torch.bfloat16)
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>images = [
<span class="hljs-meta">... </span> load_image(<span class="hljs-string">&quot;dog.png&quot;</span>),
<span class="hljs-meta">... </span> load_image(<span class="hljs-string">&quot;person.png&quot;</span>),
<span class="hljs-meta">... </span>]
<span class="hljs-meta">&gt;&gt;&gt; </span>output = pipe(
<span class="hljs-meta">... </span> images=images,
<span class="hljs-meta">... </span> prompt=<span class="hljs-string">&quot;Let the person lovingly play with the dog.&quot;</span>,
<span class="hljs-meta">... </span> height=<span class="hljs-number">1024</span>,
<span class="hljs-meta">... </span> width=<span class="hljs-number">1024</span>,
<span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">30</span>,
<span class="hljs-meta">... </span> guidance_scale=<span class="hljs-number">4.0</span>,
<span class="hljs-meta">... </span> generator=torch.manual_seed(<span class="hljs-number">42</span>),
<span class="hljs-meta">... </span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>output.images[<span class="hljs-number">0</span>].save(<span class="hljs-string">&quot;output.png&quot;</span>)`,lang:"python",wrap:!1}),m(a,w)},$$slots:{default:!0}}),s(l);var r=e(l,2),U=n(r);t(U,{name:"encode_prompt_multiple_images",anchor:"diffusers.JoyImageEditPlusPipeline.encode_prompt_multiple_images",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit_plus.py#L229",parameters:[{name:"prompt",val:": str | list[str]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"images",val:": list[PIL.Image.Image] | None = None"},{name:"max_sequence_length",val:": int | None = None"}]}),o(2),s(r);var b=e(r,2),x=n(b);t(x,{name:"prepare_latents",anchor:"diffusers.JoyImageEditPlusPipeline.prepare_latents",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit_plus.py#L274",parameters:[{name:"batch_size",val:": int"},{name:"num_channels_latents",val:": int"},{name:"height",val:": int"},{name:"width",val:": int"},{name:"dtype",val:": dtype"},{name:"device",val:": device"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType]"},{name:"reference_images",val:": list[list[PIL.Image.Image]] | None = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPlusPipeline.prepare_latents.latents",description:`<strong>latents</strong> &#x2014; Optional pre-computed noise for the target slot. Shape <code>(B, C, 1, H&apos;, W&apos;)</code> where
<code>H&apos;</code> and <code>W&apos;</code> are the latent-space dimensions. When <code>None</code>, random noise is sampled.`,name:"latents"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>[B, max_patches, C, pt, ph, pw] target_mask: [B, max_patches] (True for target patches)
shape_list: per-sample list of (t, h, w) tuples for each component</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p>padded_latents</p>
`}),o(2),s(b),s(i);var I=e(i,2);c(I,{title:"JoyImageEditPlusPipelineOutput",local:"diffusers.JoyImageEditPlusPipelineOutput",headingTag:"h2"});var d=e(I,2),Z=n(d);t(Z,{name:"class diffusers.JoyImageEditPlusPipelineOutput",anchor:"diffusers.JoyImageEditPlusPipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/joyimage/pipeline_output.py#L20",parameters:[{name:"images",val:": typing.Union[typing.List[PIL.Image.Image], numpy.ndarray]"}]}),o(2),s(d);var k=e(d,2);G(k,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/joyimage_edit_plus.md"}),o(2),m(P,u),R()}export{O as component};

Xet Storage Details

Size:
23.8 kB
·
Xet hash:
89753da6c9967a63df2fc84fa5fcc40a9cb8ec3fb36c2f9099dfe9748258d815

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.