Buckets:
| import"../chunks/DsnmJJEf.js";import{i as Y,h as A,C as H,H as i,a as s,D as t,E as K,s as $}from"../chunks/BtE7mKSK.js";import{p as ee,o as te,s as e,f as C,a as f,b as ne,c as n,d as y,r as a,n as o}from"../chunks/jDjavuwI.js";import{E as ae}from"../chunks/SrSJA0zO.js";const oe='{"title":"JoyAI-Image-Edit","local":"joyai-image-edit","sections":[{"title":"Spatial editing","local":"spatial-editing","sections":[{"title":"Object Move","local":"object-move","sections":[],"depth":3},{"title":"Object Rotation","local":"object-rotation","sections":[],"depth":3},{"title":"Camera Control","local":"camera-control","sections":[],"depth":3}],"depth":2},{"title":"JoyImageEditPipeline","local":"diffusers.JoyImageEditPipeline","sections":[],"depth":2},{"title":"JoyImageEditPipelineOutput","local":"diffusers.JoyImageEditPipelineOutput","sections":[],"depth":2}],"depth":1}';var ie=y('<meta name="hf:doc:metadata"/>'),se=y("<p>Examples:</p> <!>",1),re=y(`<p></p> <!> <!> <p><a href="https://github.com/jd-opensource/JoyAI-Image" rel="nofollow">JoyAI-Image</a> is a unified multimodal foundation model for image understanding, text-to-image generation, and instruction-guided image editing. It combines an 8B Multimodal Large Language Model (MLLM) with a 16B Multimodal Diffusion Transformer (MMDiT). A central principle of JoyAI-Image is the closed-loop collaboration between understanding, generation, and editing.</p> <p>JoyAI-Image-Edit supports general image editing as well as spatial editing capabilities including object move, object rotation, and camera control.</p> <table><thead><tr><th align="center">Model</th><th align="center">Description</th><th align="center">Download</th></tr></thead><tbody><tr><td align="center">JoyAI-Image-Edit</td><td align="center">Instruction-guided image editing with precise and controllable spatial manipulation</td><td align="center"><a href="https://huggingface.co/jdopensource/JoyAI-Image-Edit-Diffusers" rel="nofollow">Hugging Face</a></td></tr></tbody></table> <!> <!> <p>JoyAI-Image supports three spatial editing prompt patterns: <strong>Object Move</strong>, <strong>Object Rotation</strong>, and <strong>Camera Control</strong>. For best results, follow the prompt templates below as closely as possible. For more information, refer to <a href="https://github.com/EasonXiao-888/SpatialEdit" rel="nofollow">SpatialEdit</a>.</p> <!> <p>Move a target object into a specified region marked by a red box in the input image.</p> <!> <!> <p>Rotate an object to a specific canonical view. Supported <code><view></code> values: <code>front</code>, <code>right</code>, <code>left</code>, <code>rear</code>, <code>front right</code>, <code>front left</code>, <code>rear right</code>, <code>rear left</code>.</p> <!> <!> <p>Change the camera viewpoint while keeping the 3D scene unchanged.</p> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Diffusion pipeline for image editing using the JoyImage architecture.</p> <p>The pipeline encodes text and image conditioning via a Qwen3-VL text encoder, denoises latents with a 3-D | |
| transformer, and decodes the result with a WAN VAE.</p> <p>Model offloading order: text_encoder -> transformer -> vae.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Generate an edited image conditioned on a reference image and a text prompt.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Validate pipeline inputs before the forward pass.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Invert <code>normalize_latents</code> to recover the original latent scale.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encode a text prompt into embeddings (text-only path).</p> <p>Pre-computed <code>prompt_embeds</code> bypass encoding entirely.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encode prompts that contain inline image tokens via the Qwen processor.</p> <p><code><image>\\n</code> placeholders in each prompt string are replaced by the Qwen vision special tokens before being | |
| fed to the multimodal encoder.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Normalise latents using per-channel statistics from the VAE config.</p> <p>Uses (latent - mean) / std when the VAE exposes <code>latents_mean</code> and <code>latents_std</code>; otherwise falls back to | |
| scaling by <code>scaling_factor</code>.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Prepare the initial noisy latent tensor for the denoising loop.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for JoyImageEdit generation pipelines.</p></div> <!> <p></p>`,1);function ce(V,B){ee(B,!1),te(()=>{new URLSearchParams(window.location.search).get("fw")}),Y();var b=re();A("16rvsos",r=>{var h=ie();$(h,"content",oe),f(r,h)});var J=e(C(b),2);H(J,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var v=e(J,2);i(v,{title:"JoyAI-Image-Edit",local:"joyai-image-edit",headingTag:"h1"});var I=e(v,8);s(I,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSm95SW1hZ2VFZGl0UGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwbG9hZF9pbWFnZSUwQSUwQXBpcGVsaW5lJTIwJTNEJTIwSm95SW1hZ2VFZGl0UGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMmpkb3BlbnNvdXJjZSUyRkpveUFJLUltYWdlLUVkaXQtRGlmZnVzZXJzJTIyJTJDJTIwZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUwQSklMEFwaXBlbGluZS50byglMjJjdWRhJTIyKSUwQSUwQWltYWdlJTIwJTNEJTIwbG9hZF9pbWFnZSglMjJodHRwcyUzQSUyRiUyRmh1Z2dpbmdmYWNlLmNvJTJGZGF0YXNldHMlMkZodWdnaW5nZmFjZSUyRmRvY3VtZW50YXRpb24taW1hZ2VzJTJGcmVzb2x2ZSUyRm1haW4lMkZkaWZmdXNlcnMlMkZhc3Ryb25hdXQuanBnJTIyKSUwQXByb21wdCUyMCUzRCUyMCUyMkFkZCUyMHdpbmdzJTIwdG8lMjB0aGUlMjBhc3Ryb25hdXQuJTIyJTBBJTBBb3V0cHV0JTIwJTNEJTIwcGlwZWxpbmUoJTBBJTIwJTIwJTIwJTIwaW1hZ2UlM0RpbWFnZSUyQyUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG51bV9pbmZlcmVuY2Vfc3RlcHMlM0Q0MCUyQyUwQSUyMCUyMCUyMCUyMGd1aWRhbmNlX3NjYWxlJTNENC4wJTJDJTBBJTIwJTIwJTIwJTIwZ2VuZXJhdG9yJTNEdG9yY2guR2VuZXJhdG9yKCUyMmN1ZGElMjIpLm1hbnVhbF9zZWVkKDApJTJDJTBBKS5pbWFnZXMlNUIwJTVEJTBBb3V0cHV0LnNhdmUoJTIyam95aW1hZ2VfZWRpdF9vdXRwdXQucG5nJTIyKQ==",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> JoyImageEditPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> load_image | |
| pipeline = JoyImageEditPipeline.from_pretrained( | |
| <span class="hljs-string">"jdopensource/JoyAI-Image-Edit-Diffusers"</span>, dtype=torch.bfloat16 | |
| ) | |
| pipeline.to(<span class="hljs-string">"cuda"</span>) | |
| image = load_image(<span class="hljs-string">"https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"</span>) | |
| prompt = <span class="hljs-string">"Add wings to the astronaut."</span> | |
| output = pipeline( | |
| image=image, | |
| prompt=prompt, | |
| num_inference_steps=<span class="hljs-number">40</span>, | |
| guidance_scale=<span class="hljs-number">4.0</span>, | |
| generator=torch.Generator(<span class="hljs-string">"cuda"</span>).manual_seed(<span class="hljs-number">0</span>), | |
| ).images[<span class="hljs-number">0</span>] | |
| output.save(<span class="hljs-string">"joyimage_edit_output.png"</span>)`,lang:"python",wrap:!1});var w=e(I,2);i(w,{title:"Spatial editing",local:"spatial-editing",headingTag:"h2"});var T=e(w,4);i(T,{title:"Object Move",local:"object-move",headingTag:"h3"});var M=e(T,4);s(M,{code:"TW92ZSUyMHRoZSUyMCUzQ29iamVjdCUzRSUyMGludG8lMjB0aGUlMjByZWQlMjBib3glMjBhbmQlMjBmaW5hbGx5JTIwcmVtb3ZlJTIwdGhlJTIwcmVkJTIwYm94Lg==",highlighted:"Move the <object> into the red box and finally remove the red box.",lang:"text",wrap:!1});var x=e(M,2);i(x,{title:"Object Rotation",local:"object-rotation",headingTag:"h3"});var j=e(x,4);s(j,{code:"Um90YXRlJTIwdGhlJTIwJTNDb2JqZWN0JTNFJTIwdG8lMjBzaG93JTIwdGhlJTIwJTNDdmlldyUzRSUyMHNpZGUlMjB2aWV3Lg==",highlighted:"Rotate the <object> to show the <view> side view.",lang:"text",wrap:!1});var U=e(j,2);i(U,{title:"Camera Control",local:"camera-control",headingTag:"h3"});var E=e(U,4);s(E,{code:"TW92ZSUyMHRoZSUyMGNhbWVyYS4lMEEtJTIwQ2FtZXJhJTIwcm90YXRpb24lM0ElMjBZYXclMjAlN0J5X3JvdGF0aW9uJTdEJUMyJUIwJTJDJTIwUGl0Y2glMjAlN0JwX3JvdGF0aW9uJTdEJUMyJUIwLiUwQS0lMjBDYW1lcmElMjB6b29tJTNBJTIwaW4lMkZvdXQlMkZ1bmNoYW5nZWQuJTBBLSUyMEtlZXAlMjB0aGUlMjAzRCUyMHNjZW5lJTIwc3RhdGljJTNCJTIwb25seSUyMGNoYW5nZSUyMHRoZSUyMHZpZXdwb2ludC4=",highlighted:`Move the camera. | |
| - Camera rotation: Yaw {y_rotation}°, Pitch {p_rotation}°. | |
| - Camera zoom: in/out/unchanged. | |
| - Keep the 3D scene static; only change the viewpoint.`,lang:"text",wrap:!1});var Z=e(E,2);i(Z,{title:"JoyImageEditPipeline",local:"diffusers.JoyImageEditPipeline",headingTag:"h2"});var p=e(Z,2),P=n(p);t(P,{name:"class diffusers.JoyImageEditPipeline",anchor:"diffusers.JoyImageEditPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L100",parameters:[{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"},{name:"vae",val:": AutoencoderKLWan"},{name:"text_encoder",val:": Qwen3VLForConditionalGeneration"},{name:"tokenizer",val:": Qwen2Tokenizer"},{name:"transformer",val:": JoyImageEditTransformer3DModel"},{name:"processor",val:": Qwen3VLProcessor"},{name:"text_token_max_length",val:": int = 2048"}]});var l=e(P,8),k=n(l);t(k,{name:"__call__",anchor:"diffusers.JoyImageEditPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L600",parameters:[{name:"image",val:": typing.Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, list[PIL.Image.Image], list[numpy.ndarray], list[torch.Tensor], NoneType] = None"},{name:"prompt",val:": str | list[str] = None"},{name:"height",val:": int | None = None"},{name:"width",val:": int | None = None"},{name:"num_inference_steps",val:": int = 40"},{name:"timesteps",val:": typing.List[int] = None"},{name:"sigmas",val:": typing.List[float] = None"},{name:"guidance_scale",val:": float = 4.0"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"num_images_per_prompt",val:": typing.Optional[int] = 1"},{name:"generator",val:": typing.Union[torch.Generator, typing.List[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": typing.Optional[str] = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, typing.Dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": typing.List[str] = ['latents']"},{name:"max_sequence_length",val:": int = 4096"},{name:"enable_denormalization",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPipeline.__call__.prompt",description:`<strong>prompt</strong> (<em>str</em> or <em>List[str]</em>) — | |
| The prompt or prompts to guide generation.`,name:"prompt"},{anchor:"diffusers.JoyImageEditPipeline.__call__.height",description:`<strong>height</strong> (<em>int</em>) — | |
| Height of the generated output in pixels.`,name:"height"},{anchor:"diffusers.JoyImageEditPipeline.__call__.width",description:`<strong>width</strong> (<em>int</em>) — | |
| Width of the generated output in pixels.`,name:"width"},{anchor:"diffusers.JoyImageEditPipeline.__call__.image",description:`<strong>image</strong> (<em>PipelineImageInput</em>, <em>optional</em>) — | |
| Reference image used for conditioning. When provided the pipeline operates in image-editing mode with | |
| <code>num_items=2</code>.`,name:"image"},{anchor:"diffusers.JoyImageEditPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<em>int</em>, <em>optional</em>, defaults to 40) — | |
| Number of denoising steps. More steps generally improve quality at the cost of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.JoyImageEditPipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<em>List[int]</em>, <em>optional</em>) — | |
| Custom timesteps for the denoising process. When provided, <code>num_inference_steps</code> is inferred from the | |
| list length.`,name:"timesteps"},{anchor:"diffusers.JoyImageEditPipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<em>List[float]</em>, <em>optional</em>) — | |
| Custom sigmas for the denoising process. Mutually exclusive with <code>timesteps</code>.`,name:"sigmas"},{anchor:"diffusers.JoyImageEditPipeline.__call__.guidance_scale",description:`<strong>guidance_scale</strong> (<em>float</em>, <em>optional</em>, defaults to 4.0) — | |
| Classifier-free guidance scale.`,name:"guidance_scale"},{anchor:"diffusers.JoyImageEditPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<em>str</em> or <em>List[str]</em>, <em>optional</em>) — | |
| Negative prompt(s) used to suppress undesired content.`,name:"negative_prompt"},{anchor:"diffusers.JoyImageEditPipeline.__call__.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<em>int</em>, <em>optional</em>, defaults to 1) — | |
| Number of generated samples per prompt.`,name:"num_images_per_prompt"},{anchor:"diffusers.JoyImageEditPipeline.__call__.generator",description:`<strong>generator</strong> (<em>torch.Generator</em> or <em>List[torch.Generator]</em>, <em>optional</em>) — | |
| RNG generator(s) for deterministic sampling.`,name:"generator"},{anchor:"diffusers.JoyImageEditPipeline.__call__.latents",description:`<strong>latents</strong> (<em>torch.Tensor</em>, <em>optional</em>) — | |
| Pre-generated noisy latents for the target slot. Sampled from a Gaussian distribution when not | |
| provided. Can be used to seed generation from a specific starting noise tensor.`,name:"latents"},{anchor:"diffusers.JoyImageEditPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<em>torch.Tensor</em>, <em>optional</em>) — | |
| Pre-computed prompt embeddings. When provided <code>prompt</code> can be omitted.`,name:"prompt_embeds"},{anchor:"diffusers.JoyImageEditPipeline.__call__.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<em>torch.Tensor</em>, <em>optional</em>) — | |
| Attention mask for <code>prompt_embeds</code>.`,name:"prompt_embeds_mask"},{anchor:"diffusers.JoyImageEditPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<em>torch.Tensor</em>, <em>optional</em>) — | |
| Pre-computed negative prompt embeddings.`,name:"negative_prompt_embeds"},{anchor:"diffusers.JoyImageEditPipeline.__call__.negative_prompt_embeds_mask",description:`<strong>negative_prompt_embeds_mask</strong> (<em>torch.Tensor</em>, <em>optional</em>) — | |
| Attention mask for <code>negative_prompt_embeds</code>.`,name:"negative_prompt_embeds_mask"},{anchor:"diffusers.JoyImageEditPipeline.__call__.output_type",description:`<strong>output_type</strong> (<em>str</em>, <em>optional</em>, defaults to <code>"pil"</code>) — | |
| Output format. Pass <code>"latent"</code> to return raw latents.`,name:"output_type"},{anchor:"diffusers.JoyImageEditPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<em>bool</em>, <em>optional</em>, defaults to <em>True</em>) — | |
| Whether to return a <a href="/docs/diffusers/pr_14340/en/api/pipelines/joyimage_edit#diffusers.JoyImageEditPipelineOutput">JoyImageEditPipelineOutput</a> or a plain tensor.`,name:"return_dict"},{anchor:"diffusers.JoyImageEditPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<em>Callable</em>, <em>PipelineCallback</em>, <em>MultiPipelineCallbacks</em>, <em>optional</em>) — | |
| Callback invoked at the end of each denoising step with signature <code>(self, step: int, timestep: int, callback_kwargs: Dict)</code>.`,name:"callback_on_step_end"},{anchor:"diffusers.JoyImageEditPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<em>List[str]</em>, <em>optional</em>, defaults to <code>["latents"]</code>) — | |
| Tensor keys included in <code>callback_kwargs</code> for <code>callback_on_step_end</code>.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.JoyImageEditPipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<em>int</em>, <em>optional</em>, defaults to 4096) — | |
| Maximum sequence length for prompt encoding.`,name:"max_sequence_length"},{anchor:"diffusers.JoyImageEditPipeline.__call__.enable_denormalization",description:`<strong>enable_denormalization</strong> (<em>bool</em>, <em>optional</em>, defaults to <em>True</em>) — | |
| Denormalise latents before VAE decoding.`,name:"enable_denormalization"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is <code>True</code>, returns a pipeline output object containing the generated image(s). | |
| Otherwise returns the image tensor directly.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>[<em>~pipelines.joyimage.JoyImageEditPipelineOutput</em>] or <em>torch.Tensor</em></p> | |
| `});var R=e(k,4);ae(R,{anchor:"diffusers.JoyImageEditPipeline.__call__.example",children:(r,h)=>{var G=se(),O=e(C(G),2);s(O,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSm95SW1hZ2VFZGl0UGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwbG9hZF9pbWFnZSUwQSUwQW1vZGVsX2lkJTIwJTNEJTIwJTIyamRvcGVuc291cmNlJTJGSm95QUktSW1hZ2UtRWRpdC1EaWZmdXNlcnMlMjIlMEFwaXBlJTIwJTNEJTIwSm95SW1hZ2VFZGl0UGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBaW1hZ2UlMjAlM0QlMjBsb2FkX2ltYWdlKCUyMmh0dHBzJTNBJTJGJTJGaHVnZ2luZ2ZhY2UuY28lMkZkYXRhc2V0cyUyRmRpZmZ1c2VycyUyRmRvY3MtaW1hZ2VzJTJGcmVzb2x2ZSUyRm1haW4lMkZhc3Ryb25hdXQuanBnJTIyKSUwQW91dHB1dCUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwaW1hZ2UlM0RpbWFnZSUyQyUyMCUyMCUyMyUyMHBhc3MlMjBhbiUyMGltYWdlJTIwZm9yJTIwZWRpdGluZyUzQiUyMG9taXQlMjBmb3IlMjB0ZXh0LXRvLWltYWdlJTIwZ2VuZXJhdGlvbiUwQSUyMCUyMCUyMCUyMHByb21wdCUzRCUyMkFkZCUyMHdpbmdzJTIwdG8lMjB0aGUlMjBhc3Ryb25hdXQuJTIyJTJDJTBBJTIwJTIwJTIwJTIwbnVtX2luZmVyZW5jZV9zdGVwcyUzRDQwJTJDJTBBJTIwJTIwJTIwJTIwZ3VpZGFuY2Vfc2NhbGUlM0Q0LjAlMkMlMEElMjAlMjAlMjAlMjBnZW5lcmF0b3IlM0R0b3JjaC5tYW51YWxfc2VlZCgwKSUyQyUwQSklMEFvdXRwdXQuaW1hZ2VzJTVCMCU1RC5zYXZlKCUyMmpveWltYWdlX2VkaXQucG5nJTIyKQ==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> JoyImageEditPipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> load_image | |
| <span class="hljs-meta">>>> </span>model_id = <span class="hljs-string">"jdopensource/JoyAI-Image-Edit-Diffusers"</span> | |
| <span class="hljs-meta">>>> </span>pipe = JoyImageEditPipeline.from_pretrained(model_id, torch_dtype=torch.bfloat16) | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>image = load_image(<span class="hljs-string">"https://huggingface.co/datasets/diffusers/docs-images/resolve/main/astronaut.jpg"</span>) | |
| <span class="hljs-meta">>>> </span>output = pipe( | |
| <span class="hljs-meta">... </span> image=image, <span class="hljs-comment"># pass an image for editing; omit for text-to-image generation</span> | |
| <span class="hljs-meta">... </span> prompt=<span class="hljs-string">"Add wings to the astronaut."</span>, | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">40</span>, | |
| <span class="hljs-meta">... </span> guidance_scale=<span class="hljs-number">4.0</span>, | |
| <span class="hljs-meta">... </span> generator=torch.manual_seed(<span class="hljs-number">0</span>), | |
| <span class="hljs-meta">... </span>) | |
| <span class="hljs-meta">>>> </span>output.images[<span class="hljs-number">0</span>].save(<span class="hljs-string">"joyimage_edit.png"</span>)`,lang:"python",wrap:!1}),f(r,G)},$$slots:{default:!0}}),a(l);var m=e(l,2),z=n(m);t(z,{name:"check_inputs",anchor:"diffusers.JoyImageEditPipeline.check_inputs",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L409",parameters:[{name:"prompt",val:""},{name:"height",val:""},{name:"width",val:""},{name:"negative_prompt",val:" = None"},{name:"prompt_embeds",val:" = None"},{name:"negative_prompt_embeds",val:" = None"},{name:"prompt_embeds_mask",val:" = None"},{name:"negative_prompt_embeds_mask",val:" = None"},{name:"callback_on_step_end_tensor_inputs",val:" = None"}],raiseDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <ul> | |
| <li><code>ValueError</code> — On any invalid combination of arguments.</li> | |
| </ul> | |
| `,raiseType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>ValueError</code></p> | |
| `}),o(2),a(m);var d=e(m,2),S=n(d);t(S,{name:"denormalize_latents",anchor:"diffusers.JoyImageEditPipeline.denormalize_latents",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L476",parameters:[{name:"latent",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPipeline.denormalize_latents.latent",description:"<strong>latent</strong> — Normalised latent tensor.",name:"latent"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Latent tensor in the scale expected by <code>vae.decode</code>.</p> | |
| `}),o(2),a(d);var c=e(d,2),L=n(c);t(L,{name:"encode_prompt",anchor:"diffusers.JoyImageEditPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L364",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"num_images_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"max_sequence_length",val:": int = 1024"},{name:"template_type",val:": str = 'image'"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.prompt",description:"<strong>prompt</strong> — Prompt string or list of prompt strings.",name:"prompt"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.device",description:"<strong>device</strong> — Target device.",name:"device"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.num_images_per_prompt",description:"<strong>num_images_per_prompt</strong> — Number of outputs to generate per prompt.",name:"num_images_per_prompt"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.prompt_embeds",description:"<strong>prompt_embeds</strong> — Pre-computed prompt embeddings.",name:"prompt_embeds"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.prompt_embeds_mask",description:"<strong>prompt_embeds_mask</strong> — Attention mask for pre-computed embeddings.",name:"prompt_embeds_mask"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.max_sequence_length",description:"<strong>max_sequence_length</strong> — Maximum output sequence length.",name:"max_sequence_length"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt.template_type",description:"<strong>template_type</strong> — Prompt template key (<code>"image"</code> or <code>"multiple_images"</code>).",name:"template_type"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Tuple of (prompt_embeds, prompt_embeds_mask).</p> | |
| `}),o(4),a(c);var g=e(c,2),Q=n(g);t(Q,{name:"encode_prompt_multiple_images",anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L286",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"num_images_per_prompt",val:": int = 1"},{name:"images",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"template_type",val:": typing.Optional[str] = 'multiple_images'"},{name:"max_sequence_length",val:": typing.Optional[int] = None"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.prompt",description:"<strong>prompt</strong> — Prompt string(s), optionally containing <code><image>\\n</code> tokens.",name:"prompt"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.device",description:"<strong>device</strong> — Target device.",name:"device"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.num_images_per_prompt",description:"<strong>num_images_per_prompt</strong> — Number of outputs to generate per prompt.",name:"num_images_per_prompt"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.images",description:"<strong>images</strong> — Pixel tensors corresponding to the inline image tokens.",name:"images"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.prompt_embeds",description:"<strong>prompt_embeds</strong> — Pre-computed prompt embeddings.",name:"prompt_embeds"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.prompt_embeds_mask",description:"<strong>prompt_embeds_mask</strong> — Attention mask for pre-computed embeddings.",name:"prompt_embeds_mask"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.template_type",description:"<strong>template_type</strong> — Must be <code>"multiple_images"</code>.",name:"template_type"},{anchor:"diffusers.JoyImageEditPipeline.encode_prompt_multiple_images.max_sequence_length",description:`<strong>max_sequence_length</strong> — If set, truncate the output to this length | |
| (keeping the last <code>max_sequence_length</code> tokens).`,name:"max_sequence_length"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Tuple of (prompt_embeds, prompt_embeds_mask).</p> | |
| `}),o(4),a(g);var _=e(g,2),X=n(_);t(X,{name:"normalize_latents",anchor:"diffusers.JoyImageEditPipeline.normalize_latents",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L447",parameters:[{name:"latent",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPipeline.normalize_latents.latent",description:"<strong>latent</strong> — Raw latent tensor from <code>vae.encode</code>.",name:"latent"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Normalised latent tensor.</p> | |
| `}),o(4),a(_);var N=e(_,2),q=n(N);t(q,{name:"prepare_latents",anchor:"diffusers.JoyImageEditPipeline.prepare_latents",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_joyimage_edit.py#L502",parameters:[{name:"batch_size",val:": int"},{name:"num_channels_latents",val:": int"},{name:"height",val:": int"},{name:"width",val:": int"},{name:"video_length",val:": int"},{name:"dtype",val:": dtype"},{name:"device",val:": device"},{name:"generator",val:": typing.Union[torch.Generator, typing.List[torch.Generator], NoneType]"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"image",val:": typing.Optional[typing.List[PIL.Image.Image]] = None"},{name:"enable_denormalization",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.batch_size",description:"<strong>batch_size</strong> — Number of samples in the batch.",name:"batch_size"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.num_channels_latents",description:"<strong>num_channels_latents</strong> — Latent channel dimension from the transformer config.",name:"num_channels_latents"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.height",description:"<strong>height</strong> — Spatial height in pixels.",name:"height"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.width",description:"<strong>width</strong> — Spatial width in pixels.",name:"width"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.video_length",description:"<strong>video_length</strong> — Number of frames (1 for image inference).",name:"video_length"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.dtype",description:"<strong>dtype</strong> — Floating-point dtype for the latent tensor.",name:"dtype"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.device",description:"<strong>device</strong> — Target device.",name:"device"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.generator",description:"<strong>generator</strong> — RNG generator(s) for reproducible sampling.",name:"generator"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.latents",description:"<strong>latents</strong> — Optional user-provided initial noise for the target slot. When <code>None</code> random noise is sampled.",name:"latents"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.image",description:"<strong>image</strong> — Optional list of PIL reference images to VAE-encode as conditioning slots.",name:"image"},{anchor:"diffusers.JoyImageEditPipeline.prepare_latents.enable_denormalization",description:"<strong>enable_denormalization</strong> — Whether to normalise encoded reference latents.",name:"enable_denormalization"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Tuple of <code>(latents, image_latents)</code> where <code>latents</code> has shape <code>(B, 1, C, T, H', W')</code> and | |
| <code>image_latents</code> has shape <code>(B, N_ref, C, T, H', W')</code> or <code>None</code> when no reference images are given.</p> | |
| `,raiseDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <ul> | |
| <li><code>ValueError</code> — If <code>generator</code> is a list whose length differs from <code>batch_size</code>.</li> | |
| </ul> | |
| `,raiseType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>ValueError</code></p> | |
| `}),o(2),a(N),a(p);var W=e(p,2);i(W,{title:"JoyImageEditPipelineOutput",local:"diffusers.JoyImageEditPipelineOutput",headingTag:"h2"});var u=e(W,2),D=n(u);t(D,{name:"class diffusers.JoyImageEditPipelineOutput",anchor:"diffusers.JoyImageEditPipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14340/src/diffusers/pipelines/joyimage/pipeline_output.py#L11",parameters:[{name:"images",val:": typing.Union[typing.List[PIL.Image.Image], numpy.ndarray]"}]}),o(2),a(u);var F=e(u,2);K(F,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/joyimage_edit.md"}),o(2),f(V,b),ne()}export{ce as component}; | |
Xet Storage Details
- Size:
- 32.4 kB
- Xet hash:
- 974b4d53a7d2859ed783f3a543e31712fcd9def195b6e6e71cbc9dd45270695d
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.