Buckets:
| import"../chunks/DsnmJJEf.js";import{i as Y,h as X,H as n,a as g,D as a,E as F,s as H}from"../chunks/BtE7mKSK.js";import{p as S,o as E,s as e,f as B,a as u,b as Q,c as t,d as h,r as o,n as i}from"../chunks/jDjavuwI.js";import{E as A}from"../chunks/SrSJA0zO.js";const L='{"title":"GLM-Image","local":"glm-image","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"Usage examples","local":"usage-examples","sections":[{"title":"Text to Image Generation","local":"text-to-image-generation","sections":[],"depth":3},{"title":"Image to Image Generation","local":"image-to-image-generation","sections":[],"depth":3}],"depth":2},{"title":"GlmImagePipeline","local":"diffusers.GlmImagePipeline","sections":[],"depth":2},{"title":"GlmImagePipelineOutput","local":"diffusers.pipelines.glm_image.pipeline_output.GlmImagePipelineOutput","sections":[],"depth":2}],"depth":1}';var q=h('<meta name="hf:doc:metadata"/>'),D=h("<p>Examples:</p> <!>",1),O=h(`<p></p> <!> <!> <p>GLM-Image is an image generation model adopts a hybrid autoregressive + diffusion decoder architecture, effectively pushing the upper bound of visual fidelity and fine-grained details. In general image generation quality, it aligns with industry-standard LDM-based approaches, while demonstrating significant advantages in knowledge-intensive image generation scenarios.</p> <p>Model architecture: a hybrid autoregressive + diffusion decoder design、</p> <ul><li>Autoregressive generator: a 9B-parameter model initialized from <a href="https://huggingface.co/zai-org/GLM-4-9B-0414" rel="nofollow">GLM-4-9B-0414</a>, with an expanded vocabulary to incorporate visual tokens. The model first generates a compact encoding of approximately 256 tokens, then expands to 1K–4K tokens, corresponding to 1K–2K high-resolution image outputs. You can check AR model in class <code>GlmImageForConditionalGeneration</code> of <code>transformers</code> library.</li> <li>Diffusion Decoder: a 7B-parameter decoder based on a single-stream DiT architecture for latent-space image decoding. It is equipped with a Glyph Encoder text module, significantly improving accurate text rendering within images.</li></ul> <p>Post-training with decoupled reinforcement learning: the model introduces a fine-grained, modular feedback strategy using the GRPO algorithm, substantially enhancing both semantic understanding and visual detail quality.</p> <ul><li>Autoregressive module: provides low-frequency feedback signals focused on aesthetics and semantic alignment, improving instruction following and artistic expressiveness.</li> <li>Decoder module: delivers high-frequency feedback targeting detail fidelity and text accuracy, resulting in highly realistic textures, lighting, and color reproduction, as well as more precise text rendering.</li></ul> <p>GLM-Image supports both text-to-image and image-to-image generation within a single model</p> <ul><li>Text-to-image: generates high-detail images from textual descriptions, with particularly strong performance in information-dense scenarios.</li> <li>Image-to-image: supports a wide range of tasks, including image editing, style transfer, multi-subject consistency, and identity-preserving generation for people and objects.</li></ul> <p>This pipeline was contributed by <a href="https://github.com/zRzRzRzRzRzRzR" rel="nofollow">zRzRzRzRzRzRzR</a>. The codebase can be found <a href="https://huggingface.co/zai-org/GLM-Image" rel="nofollow">here</a>.</p> <!> <!> <!> <!> <!> <ul><li>Since the AR model used in GLM-Image is configured with <code>do_sample=True</code> and a temperature of <code>0.95</code> by default, the generated images can vary significantly across runs. We do not recommend setting do_sample=False, as this may lead to incorrect or degenerate outputs from the AR model.</li></ul> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for text-to-image generation using GLM-Image.</p> <p>This pipeline integrates both the AR (autoregressive) model for token generation and the DiT (diffusion | |
| transformer) model for image decoding.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Function invoked when calling the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encodes the prompt into text encoder hidden states.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Generate prior tokens for the DiT model using the AR model.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Extract glyph texts from prompt(s). Returns a list of lists for batch processing.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for CogView3 pipelines.</p></div> <!> <p></p>`,1);function ae(Z,k){S(k,!1),E(()=>{new URLSearchParams(window.location.search).get("fw")}),Y();var _=O();X("cjx4w8",s=>{var c=q();H(c,"content",L),u(s,c)});var f=e(B(_),2);n(f,{title:"GLM-Image",local:"glm-image",headingTag:"h1"});var M=e(f,2);n(M,{title:"Overview",local:"overview",headingTag:"h2"});var y=e(M,18);n(y,{title:"Usage examples",local:"usage-examples",headingTag:"h2"});var b=e(y,2);n(b,{title:"Text to Image Generation",local:"text-to-image-generation",headingTag:"h3"});var G=e(b,2);g(G,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzLnBpcGVsaW5lcy5nbG1faW1hZ2UlMjBpbXBvcnQlMjBHbG1JbWFnZVBpcGVsaW5lJTBBJTBBcGlwZSUyMCUzRCUyMEdsbUltYWdlUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUyMnphaS1vcmclMkZHTE0tSW1hZ2UlMjIlMkN0b3JjaF9kdHlwZSUzRHRvcmNoLmJmbG9hdDE2JTJDZGV2aWNlX21hcCUzRCUyMmN1ZGElMjIpJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMGJlYXV0aWZ1bGx5JTIwZGVzaWduZWQlMjBtb2Rlcm4lMjBmb29kJTIwbWFnYXppbmUlMjBzdHlsZSUyMGRlc3NlcnQlMjByZWNpcGUlMjBpbGx1c3RyYXRpb24lMkMlMjB0aGVtZWQlMjBhcm91bmQlMjBhJTIwcmFzcGJlcnJ5JTIwbW91c3NlJTIwY2FrZS4lMjBUaGUlMjBvdmVyYWxsJTIwbGF5b3V0JTIwaXMlMjBjbGVhbiUyMGFuZCUyMGJyaWdodCUyQyUyMGRpdmlkZWQlMjBpbnRvJTIwZm91ciUyMG1haW4lMjBhcmVhcyUzQSUyMHRoZSUyMHRvcCUyMGxlZnQlMjBmZWF0dXJlcyUyMGElMjBib2xkJTIwYmxhY2slMjB0aXRsZSUyMCdSYXNwYmVycnklMjBNb3Vzc2UlMjBDYWtlJTIwUmVjaXBlJTIwR3VpZGUnJTJDJTIwd2l0aCUyMGElMjBzb2Z0LWxpdCUyMGNsb3NlLXVwJTIwcGhvdG8lMjBvZiUyMHRoZSUyMGZpbmlzaGVkJTIwY2FrZSUyMG9uJTIwdGhlJTIwcmlnaHQlMkMlMjBzaG93Y2FzaW5nJTIwYSUyMGxpZ2h0JTIwcGluayUyMGNha2UlMjBhZG9ybmVkJTIwd2l0aCUyMGZyZXNoJTIwcmFzcGJlcnJpZXMlMjBhbmQlMjBtaW50JTIwbGVhdmVzJTNCJTIwdGhlJTIwYm90dG9tJTIwbGVmdCUyMGNvbnRhaW5zJTIwYW4lMjBpbmdyZWRpZW50JTIwbGlzdCUyMHNlY3Rpb24lMkMlMjB0aXRsZWQlMjAnSW5ncmVkaWVudHMnJTIwaW4lMjBhJTIwc2ltcGxlJTIwZm9udCUyQyUyMGxpc3RpbmclMjAnRmxvdXIlMjAxNTBnJyUyQyUyMCdFZ2dzJTIwMyclMkMlMjAnU3VnYXIlMjAxMjBnJyUyQyUyMCdSYXNwYmVycnklMjBwdXJlZSUyMDIwMGcnJTJDJTIwJ0dlbGF0aW4lMjBzaGVldHMlMjAxMGcnJTJDJTIwJ1doaXBwaW5nJTIwY3JlYW0lMjAzMDBtbCclMkMlMjBhbmQlMjAnRnJlc2glMjByYXNwYmVycmllcyclMkMlMjBlYWNoJTIwYWNjb21wYW5pZWQlMjBieSUyMG1pbmltYWxpc3QlMjBsaW5lJTIwaWNvbnMlMjAobGlrZSUyMGElMjBmbG91ciUyMGJhZyUyQyUyMGVnZ3MlMkMlMjBzdWdhciUyMGphciUyQyUyMGV0Yy4pJTNCJTIwdGhlJTIwYm90dG9tJTIwcmlnaHQlMjBkaXNwbGF5cyUyMGZvdXIlMjBlcXVhbGx5JTIwc2l6ZWQlMjBzdGVwJTIwYm94ZXMlMkMlMjBlYWNoJTIwY29udGFpbmluZyUyMGhpZ2gtZGVmaW5pdGlvbiUyMG1hY3JvJTIwcGhvdG9zJTIwYW5kJTIwY29ycmVzcG9uZGluZyUyMGluc3RydWN0aW9ucyUyQyUyMGFycmFuZ2VkJTIwZnJvbSUyMHRvcCUyMHRvJTIwYm90dG9tJTIwYXMlMjBmb2xsb3dzJTNBJTIwU3RlcCUyMDElMjBzaG93cyUyMGElMjB3aGlzayUyMHdoaXBwaW5nJTIwd2hpdGUlMjBmb2FtJTIwKHdpdGglMjB0aGUlMjBpbnN0cnVjdGlvbiUyMCdXaGlwJTIwZWdnJTIwd2hpdGVzJTIwdG8lMjBzdGlmZiUyMHBlYWtzJyklMkMlMjBTdGVwJTIwMiUyMHNob3dzJTIwYSUyMHJlZC1hbmQtd2hpdGUlMjBtaXh0dXJlJTIwYmVpbmclMjBmb2xkZWQlMjB3aXRoJTIwYSUyMHNwYXR1bGElMjAod2l0aCUyMHRoZSUyMGluc3RydWN0aW9uJTIwJ0dlbnRseSUyMGZvbGQlMjBpbiUyMHRoZSUyMHB1cmVlJTIwYW5kJTIwYmF0dGVyJyklMkMlMjBTdGVwJTIwMyUyMHNob3dzJTIwcGluayUyMGxpcXVpZCUyMGJlaW5nJTIwcG91cmVkJTIwaW50byUyMGElMjByb3VuZCUyMG1vbGQlMjAod2l0aCUyMHRoZSUyMGluc3RydWN0aW9uJTIwJ1BvdXIlMjBpbnRvJTIwbW9sZCUyMGFuZCUyMGNoaWxsJTIwZm9yJTIwNCUyMGhvdXJzJyklMkMlMjBTdGVwJTIwNCUyMHNob3dzJTIwdGhlJTIwZmluaXNoZWQlMjBjYWtlJTIwZGVjb3JhdGVkJTIwd2l0aCUyMHJhc3BiZXJyaWVzJTIwYW5kJTIwbWludCUyMGxlYXZlcyUyMCh3aXRoJTIwdGhlJTIwaW5zdHJ1Y3Rpb24lMjAnRGVjb3JhdGUlMjB3aXRoJTIwcmFzcGJlcnJpZXMlMjBhbmQlMjBtaW50JyklM0IlMjBhJTIwbGlnaHQlMjBicm93biUyMGluZm9ybWF0aW9uJTIwYmFyJTIwcnVucyUyMGFsb25nJTIwdGhlJTIwYm90dG9tJTIwZWRnZSUyQyUyMHdpdGglMjBpY29ucyUyMG9uJTIwdGhlJTIwbGVmdCUyMHJlcHJlc2VudGluZyUyMCdQcmVwYXJhdGlvbiUyMHRpbWUlM0ElMjAzMCUyMG1pbnV0ZXMnJTJDJTIwJ0Nvb2tpbmclMjB0aW1lJTNBJTIwMjAlMjBtaW51dGVzJyUyQyUyMGFuZCUyMCdTZXJ2aW5ncyUzQSUyMDgnLiUyMFRoZSUyMG92ZXJhbGwlMjBjb2xvciUyMHNjaGVtZSUyMGlzJTIwZG9taW5hdGVkJTIwYnklMjBjcmVhbXklMjB3aGl0ZSUyMGFuZCUyMGxpZ2h0JTIwcGluayUyQyUyMHdpdGglMjBhJTIwc3VidGxlJTIwcGFwZXIlMjB0ZXh0dXJlJTIwaW4lMjB0aGUlMjBiYWNrZ3JvdW5kJTJDJTIwZmVhdHVyaW5nJTIwY29tcGFjdCUyMGFuZCUyMG9yZGVybHklMjB0ZXh0JTIwYW5kJTIwaW1hZ2UlMjBsYXlvdXQlMjB3aXRoJTIwY2xlYXIlMjBpbmZvcm1hdGlvbiUyMGhpZXJhcmNoeS4lMjIlMEFpbWFnZSUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwcHJvbXB0JTNEcHJvbXB0JTJDJTBBJTIwJTIwJTIwJTIwaGVpZ2h0JTNEMzIlMjAqJTIwMzIlMkMlMEElMjAlMjAlMjAlMjB3aWR0aCUzRDM2JTIwKiUyMDMyJTJDJTBBJTIwJTIwJTIwJTIwbnVtX2luZmVyZW5jZV9zdGVwcyUzRDMwJTJDJTBBJTIwJTIwJTIwJTIwZ3VpZGFuY2Vfc2NhbGUlM0QxLjUlMkMlMEElMjAlMjAlMjAlMjBnZW5lcmF0b3IlM0R0b3JjaC5HZW5lcmF0b3IoZGV2aWNlJTNEJTIyY3VkYSUyMikubWFudWFsX3NlZWQoNDIpJTJDJTBBKS5pbWFnZXMlNUIwJTVEJTBBJTBBaW1hZ2Uuc2F2ZSglMjJvdXRwdXRfdDJpLnBuZyUyMik=",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers.pipelines.glm_image <span class="hljs-keyword">import</span> GlmImagePipeline | |
| pipe = GlmImagePipeline.from_pretrained(<span class="hljs-string">"zai-org/GLM-Image"</span>,torch_dtype=torch.bfloat16,device_map=<span class="hljs-string">"cuda"</span>) | |
| prompt = <span class="hljs-string">"A beautifully designed modern food magazine style dessert recipe illustration, themed around a raspberry mousse cake. The overall layout is clean and bright, divided into four main areas: the top left features a bold black title 'Raspberry Mousse Cake Recipe Guide', with a soft-lit close-up photo of the finished cake on the right, showcasing a light pink cake adorned with fresh raspberries and mint leaves; the bottom left contains an ingredient list section, titled 'Ingredients' in a simple font, listing 'Flour 150g', 'Eggs 3', 'Sugar 120g', 'Raspberry puree 200g', 'Gelatin sheets 10g', 'Whipping cream 300ml', and 'Fresh raspberries', each accompanied by minimalist line icons (like a flour bag, eggs, sugar jar, etc.); the bottom right displays four equally sized step boxes, each containing high-definition macro photos and corresponding instructions, arranged from top to bottom as follows: Step 1 shows a whisk whipping white foam (with the instruction 'Whip egg whites to stiff peaks'), Step 2 shows a red-and-white mixture being folded with a spatula (with the instruction 'Gently fold in the puree and batter'), Step 3 shows pink liquid being poured into a round mold (with the instruction 'Pour into mold and chill for 4 hours'), Step 4 shows the finished cake decorated with raspberries and mint leaves (with the instruction 'Decorate with raspberries and mint'); a light brown information bar runs along the bottom edge, with icons on the left representing 'Preparation time: 30 minutes', 'Cooking time: 20 minutes', and 'Servings: 8'. The overall color scheme is dominated by creamy white and light pink, with a subtle paper texture in the background, featuring compact and orderly text and image layout with clear information hierarchy."</span> | |
| image = pipe( | |
| prompt=prompt, | |
| height=<span class="hljs-number">32</span> * <span class="hljs-number">32</span>, | |
| width=<span class="hljs-number">36</span> * <span class="hljs-number">32</span>, | |
| num_inference_steps=<span class="hljs-number">30</span>, | |
| guidance_scale=<span class="hljs-number">1.5</span>, | |
| generator=torch.Generator(device=<span class="hljs-string">"cuda"</span>).manual_seed(<span class="hljs-number">42</span>), | |
| ).images[<span class="hljs-number">0</span>] | |
| image.save(<span class="hljs-string">"output_t2i.png"</span>)`,lang:"python",wrap:!1});var I=e(G,2);n(I,{title:"Image to Image Generation",local:"image-to-image-generation",headingTag:"h3"});var w=e(I,2);g(w,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzLnBpcGVsaW5lcy5nbG1faW1hZ2UlMjBpbXBvcnQlMjBHbG1JbWFnZVBpcGVsaW5lJTBBZnJvbSUyMFBJTCUyMGltcG9ydCUyMEltYWdlJTBBJTBBcGlwZSUyMCUzRCUyMEdsbUltYWdlUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUyMnphaS1vcmclMkZHTE0tSW1hZ2UlMjIlMkN0b3JjaF9kdHlwZSUzRHRvcmNoLmJmbG9hdDE2JTJDZGV2aWNlX21hcCUzRCUyMmN1ZGElMjIpJTBBaW1hZ2VfcGF0aCUyMCUzRCUyMCUyMmNvbmQuanBnJTIyJTIwJTBBcHJvbXB0JTIwJTNEJTIwJTIyUmVwbGFjZSUyMHRoZSUyMGJhY2tncm91bmQlMjBvZiUyMHRoZSUyMHNub3clMjBmb3Jlc3QlMjB3aXRoJTIwYW4lMjB1bmRlcmdyb3VuZCUyMHN0YXRpb24lMjBmZWF0dXJpbmclMjBhbiUyMGF1dG9tYXRpYyUyMGVzY2FsYXRvci4lMjIlMEFpbWFnZSUyMCUzRCUyMEltYWdlLm9wZW4oaW1hZ2VfcGF0aCkuY29udmVydCglMjJSR0IlMjIpJTBBaW1hZ2UlMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMGltYWdlJTNEJTVCaW1hZ2UlNUQlMkMlMjAlMjMlMjBjYW4lMjBpbnB1dCUyMG11bHRpcGxlJTIwaW1hZ2VzJTIwZm9yJTIwbXVsdGktaW1hZ2UtdG8taW1hZ2UlMjBnZW5lcmF0aW9uJTIwc3VjaCUyMGFzJTIwJTVCaW1hZ2UlMkMlMjBpbWFnZTElNUQlMEElMjAlMjAlMjAlMjBoZWlnaHQlM0QzMyUyMColMjAzMiUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMzIlMjAqJTIwMzIlMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNEMzAlMkMlMEElMjAlMjAlMjAlMjBndWlkYW5jZV9zY2FsZSUzRDEuNSUyQyUwQSUyMCUyMCUyMCUyMGdlbmVyYXRvciUzRHRvcmNoLkdlbmVyYXRvcihkZXZpY2UlM0QlMjJjdWRhJTIyKS5tYW51YWxfc2VlZCg0MiklMkMlMEEpLmltYWdlcyU1QjAlNUQlMEElMEFpbWFnZS5zYXZlKCUyMm91dHB1dF9pMmkucG5nJTIyKQ==",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers.pipelines.glm_image <span class="hljs-keyword">import</span> GlmImagePipeline | |
| <span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| pipe = GlmImagePipeline.from_pretrained(<span class="hljs-string">"zai-org/GLM-Image"</span>,torch_dtype=torch.bfloat16,device_map=<span class="hljs-string">"cuda"</span>) | |
| image_path = <span class="hljs-string">"cond.jpg"</span> | |
| prompt = <span class="hljs-string">"Replace the background of the snow forest with an underground station featuring an automatic escalator."</span> | |
| image = Image.<span class="hljs-built_in">open</span>(image_path).convert(<span class="hljs-string">"RGB"</span>) | |
| image = pipe( | |
| prompt=prompt, | |
| image=[image], <span class="hljs-comment"># can input multiple images for multi-image-to-image generation such as [image, image1]</span> | |
| height=<span class="hljs-number">33</span> * <span class="hljs-number">32</span>, | |
| width=<span class="hljs-number">32</span> * <span class="hljs-number">32</span>, | |
| num_inference_steps=<span class="hljs-number">30</span>, | |
| guidance_scale=<span class="hljs-number">1.5</span>, | |
| generator=torch.Generator(device=<span class="hljs-string">"cuda"</span>).manual_seed(<span class="hljs-number">42</span>), | |
| ).images[<span class="hljs-number">0</span>] | |
| image.save(<span class="hljs-string">"output_i2i.png"</span>)`,lang:"python",wrap:!1});var J=e(w,4);n(J,{title:"GlmImagePipeline",local:"diffusers.GlmImagePipeline",headingTag:"h2"});var r=e(J,2),T=t(r);a(T,{name:"class diffusers.GlmImagePipeline",anchor:"diffusers.GlmImagePipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/glm_image/pipeline_glm_image.py#L161",parameters:[{name:"tokenizer",val:": ByT5Tokenizer"},{name:"processor",val:": GlmImageProcessor"},{name:"text_encoder",val:": T5EncoderModel"},{name:"vision_language_encoder",val:": GlmImageForConditionalGeneration"},{name:"vae",val:": AutoencoderKL"},{name:"transformer",val:": GlmImageTransformer2DModel"},{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"}],parametersDescription:[{anchor:"diffusers.GlmImagePipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>PreTrainedTokenizer</code>) — | |
| Tokenizer for the text encoder.`,name:"tokenizer"},{anchor:"diffusers.GlmImagePipeline.processor",description:`<strong>processor</strong> (<code>AutoProcessor</code>) — | |
| Processor for the AR model to handle chat templates and tokenization.`,name:"processor"},{anchor:"diffusers.GlmImagePipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>T5EncoderModel</code>) — | |
| Frozen text-encoder for glyph embeddings.`,name:"text_encoder"},{anchor:"diffusers.GlmImagePipeline.vision_language_encoder",description:`<strong>vision_language_encoder</strong> (<code>GlmImageForConditionalGeneration</code>) — | |
| The AR model that generates image tokens from text prompts.`,name:"vision_language_encoder"},{anchor:"diffusers.GlmImagePipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14229/en/api/models/autoencoderkl#diffusers.AutoencoderKL">AutoencoderKL</a>) — | |
| Variational Auto-Encoder (VAE) Model to encode and decode images to and from latent representations.`,name:"vae"},{anchor:"diffusers.GlmImagePipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14229/en/api/models/glm_image_transformer2d#diffusers.GlmImageTransformer2DModel">GlmImageTransformer2DModel</a>) — | |
| A text conditioned transformer to denoise the encoded image latents (DiT).`,name:"transformer"},{anchor:"diffusers.GlmImagePipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14229/en/api/schedulers/overview#diffusers.SchedulerMixin">SchedulerMixin</a>) — | |
| A scheduler to be used in combination with <code>transformer</code> to denoise the encoded image latents.`,name:"scheduler"}]});var l=e(T,6),v=t(l);a(v,{name:"__call__",anchor:"diffusers.GlmImagePipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/glm_image/pipeline_glm_image.py#L719",parameters:[{name:"prompt",val:": str | list[str] | None = None"},{name:"image",val:": typing.Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, list[PIL.Image.Image], list[numpy.ndarray], list[torch.Tensor], NoneType] = None"},{name:"height",val:": int | None = None"},{name:"width",val:": int | None = None"},{name:"num_inference_steps",val:": int = 50"},{name:"timesteps",val:": list[int] | None = None"},{name:"sigmas",val:": list[float] | None = None"},{name:"guidance_scale",val:": float = 1.5"},{name:"num_images_per_prompt",val:": int = 1"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prior_token_ids",val:": typing.Optional[torch.Tensor] = None"},{name:"prior_token_image_ids",val:": list[torch.Tensor] | None = None"},{name:"source_image_grid_thw",val:": list[torch.Tensor] | None = None"},{name:"crops_coords_top_left",val:": tuple = (0, 0)"},{name:"output_type",val:": str = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": dict[str, typing.Any] | None = None"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": list = ['latents']"},{name:"max_sequence_length",val:": int = 2048"}],parametersDescription:[{anchor:"diffusers.GlmImagePipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| The prompt or prompts to guide the image generation. Must contain shape info in the format ’<sop>H | |
| W<eop>’ where H and W are token dimensions (d32). Example: “A beautiful sunset<sop>36 24<eop>” | |
| generates a 1152x768 image.</eop></sop></eop></sop>`,name:"prompt"},{anchor:"diffusers.GlmImagePipeline.__call__.image",description:"<strong>image</strong> — Optional condition images for image-to-image generation.",name:"image"},{anchor:"diffusers.GlmImagePipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, <em>optional</em>) — | |
| The height in pixels. If not provided, derived from prompt shape info.`,name:"height"},{anchor:"diffusers.GlmImagePipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, <em>optional</em>) — | |
| The width in pixels. If not provided, derived from prompt shape info.`,name:"width"},{anchor:"diffusers.GlmImagePipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to <code>50</code>) — | |
| The number of denoising steps for DiT.`,name:"num_inference_steps"},{anchor:"diffusers.GlmImagePipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>list[int]</code>, <em>optional</em>) — | |
| Custom timesteps to use for the denoising process. If not defined, the scheduler’s default schedule for | |
| <code>num_inference_steps</code> is used.`,name:"timesteps"},{anchor:"diffusers.GlmImagePipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<code>list[float]</code>, <em>optional</em>) — | |
| Custom sigmas to use for the denoising process. If not defined, the scheduler’s default schedule is | |
| used.`,name:"sigmas"},{anchor:"diffusers.GlmImagePipeline.__call__.guidance_scale",description:`<strong>guidance_scale</strong> (<code>float</code>, <em>optional</em>, defaults to <code>1.5</code>) — | |
| Guidance scale for classifier-free guidance.`,name:"guidance_scale"},{anchor:"diffusers.GlmImagePipeline.__call__.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to <code>1</code>) — | |
| The number of images to generate per prompt.`,name:"num_images_per_prompt"},{anchor:"diffusers.GlmImagePipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code>, <em>optional</em>) — | |
| Random generator for reproducibility.`,name:"generator"},{anchor:"diffusers.GlmImagePipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated noisy latents to be used as inputs for image generation.`,name:"latents"},{anchor:"diffusers.GlmImagePipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. If not provided, embeddings are generated from <code>prompt</code>.`,name:"prompt_embeds"},{anchor:"diffusers.GlmImagePipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings. Used when classifier-free guidance is enabled.`,name:"negative_prompt_embeds"},{anchor:"diffusers.GlmImagePipeline.__call__.prior_token_ids",description:`<strong>prior_token_ids</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated prior token ids from <code>generate_prior_tokens</code>. If supplied, prior generation is skipped.`,name:"prior_token_ids"},{anchor:"diffusers.GlmImagePipeline.__call__.prior_token_image_ids",description:`<strong>prior_token_image_ids</strong> (<code>list[torch.Tensor]</code>, <em>optional</em>) — | |
| Image token ids associated with <code>prior_token_ids</code>.`,name:"prior_token_image_ids"},{anchor:"diffusers.GlmImagePipeline.__call__.source_image_grid_thw",description:`<strong>source_image_grid_thw</strong> (<code>list[torch.Tensor]</code>, <em>optional</em>) — | |
| Per-sample THW grid information for the source image tokens.`,name:"source_image_grid_thw"},{anchor:"diffusers.GlmImagePipeline.__call__.crops_coords_top_left",description:`<strong>crops_coords_top_left</strong> (<code>tuple[int, int]</code>, <em>optional</em>, defaults to <code>(0, 0)</code>) — | |
| The top-left coordinates of the crop used for conditioning embeddings.`,name:"crops_coords_top_left"},{anchor:"diffusers.GlmImagePipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"pil"</code>) — | |
| Output format: “pil”, “np”, or “latent”.`,name:"output_type"},{anchor:"diffusers.GlmImagePipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <code>GlmImagePipelineOutput</code> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.GlmImagePipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| A kwargs dictionary that if specified is passed along to the <code>AttentionProcessor</code>.`,name:"attention_kwargs"},{anchor:"diffusers.GlmImagePipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <code>PipelineCallback</code>, <code>MultiPipelineCallbacks</code>, <em>optional</em>) — | |
| A function called at the end of each denoising step.`,name:"callback_on_step_end"},{anchor:"diffusers.GlmImagePipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>list[str]</code>, <em>optional</em>) — | |
| Tensor inputs passed to <code>callback_on_step_end</code>.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.GlmImagePipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2048</code>) — | |
| Maximum sequence length for the text encoder.`,name:"max_sequence_length"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Generated images.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>GlmImagePipelineOutput</code> or <code>tuple</code></p> | |
| `});var W=e(v,4);A(W,{anchor:"diffusers.GlmImagePipeline.__call__.example",children:(s,c)=>{var x=D(),P=e(B(x),2);g(P,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwR2xtSW1hZ2VQaXBlbGluZSUwQSUwQXBpcGUlMjAlM0QlMjBHbG1JbWFnZVBpcGVsaW5lLmZyb21fcHJldHJhaW5lZCglMjJ6YWktb3JnJTJGR0xNLUltYWdlJTIyJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHBob3RvJTIwb2YlMjBhbiUyMGFzdHJvbmF1dCUyMHJpZGluZyUyMGElMjBob3JzZSUyMG9uJTIwbWFycyUyMiUwQWltYWdlJTIwJTNEJTIwcGlwZShwcm9tcHQpLmltYWdlcyU1QjAlNUQlMEFpbWFnZS5zYXZlKCUyMm91dHB1dC5wbmclMjIp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> GlmImagePipeline | |
| <span class="hljs-meta">>>> </span>pipe = GlmImagePipeline.from_pretrained(<span class="hljs-string">"zai-org/GLM-Image"</span>, torch_dtype=torch.bfloat16) | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"A photo of an astronaut riding a horse on mars"</span> | |
| <span class="hljs-meta">>>> </span>image = pipe(prompt).images[<span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>image.save(<span class="hljs-string">"output.png"</span>)`,lang:"python",wrap:!1}),u(s,x)},$$slots:{default:!0}}),o(l);var d=e(l,2),N=t(d);a(N,{name:"encode_prompt",anchor:"diffusers.GlmImagePipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/glm_image/pipeline_glm_image.py#L545",parameters:[{name:"prompt",val:": str | list[str]"},{name:"do_classifier_free_guidance",val:": bool = True"},{name:"num_images_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"},{name:"max_sequence_length",val:": int = 2048"}],parametersDescription:[{anchor:"diffusers.GlmImagePipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| prompt to be encoded`,name:"prompt"},{anchor:"diffusers.GlmImagePipeline.encode_prompt.do_classifier_free_guidance",description:`<strong>do_classifier_free_guidance</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to use classifier free guidance or not.`,name:"do_classifier_free_guidance"},{anchor:"diffusers.GlmImagePipeline.encode_prompt.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of images that should be generated per prompt. torch device to place the resulting embeddings on`,name:"num_images_per_prompt"},{anchor:"diffusers.GlmImagePipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt weighting. If not | |
| provided, text embeddings will be generated from <code>prompt</code> input argument.`,name:"prompt_embeds"},{anchor:"diffusers.GlmImagePipeline.encode_prompt.device",description:`<strong>device</strong> — (<code>torch.device</code>, <em>optional</em>): | |
| torch device`,name:"device"},{anchor:"diffusers.GlmImagePipeline.encode_prompt.dtype",description:`<strong>dtype</strong> — (<code>torch.dtype</code>, <em>optional</em>): | |
| torch dtype`,name:"dtype"},{anchor:"diffusers.GlmImagePipeline.encode_prompt.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to <code>2048</code>) — | |
| Maximum sequence length in encoded prompt. Can be set to other values but may lead to poorer results.`,name:"max_sequence_length"}]}),i(2),o(d);var p=e(d,2),C=t(p);a(C,{name:"generate_prior_tokens",anchor:"diffusers.GlmImagePipeline.generate_prior_tokens",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/glm_image/pipeline_glm_image.py#L321",parameters:[{name:"prompt",val:": str | list[str]"},{name:"height",val:": int"},{name:"width",val:": int"},{name:"image",val:": list[list[PIL.Image.Image]] | None = None"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"generator",val:": typing.Optional[torch.Generator] = None"}],parametersDescription:[{anchor:"diffusers.GlmImagePipeline.generate_prior_tokens.prompt",description:"<strong>prompt</strong> — Single prompt or list of prompts",name:"prompt"},{anchor:"diffusers.GlmImagePipeline.generate_prior_tokens.height",description:"<strong>height</strong> — Target image height",name:"height"},{anchor:"diffusers.GlmImagePipeline.generate_prior_tokens.width",description:"<strong>width</strong> — Target image width",name:"width"},{anchor:"diffusers.GlmImagePipeline.generate_prior_tokens.image",description:`<strong>image</strong> — Normalized image input as List[List[PIL.Image]]. Should be pre-validated | |
| using _validate_and_normalize_images() before calling this method.`,name:"image"},{anchor:"diffusers.GlmImagePipeline.generate_prior_tokens.device",description:"<strong>device</strong> — Target device",name:"device"},{anchor:"diffusers.GlmImagePipeline.generate_prior_tokens.generator",description:"<strong>generator</strong> — Random generator for reproducibility",name:"generator"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <ul> | |
| <li>prior_token_ids: Tensor of shape (batch_size, num_tokens) with upsampled prior tokens</li> | |
| <li>prior_token_image_ids_per_sample: List of tensors, one per sample. Each tensor contains | |
| the upsampled prior token ids for all condition images in that sample. None for t2i.</li> | |
| <li>source_image_grid_thw_per_sample: List of tensors, one per sample. Each tensor has shape | |
| (num_condition_images, 3) with upsampled grid info. None for t2i.</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Tuple of</p> | |
| `}),i(2),o(p);var U=e(p,2),z=t(U);a(z,{name:"get_glyph_texts",anchor:"diffusers.GlmImagePipeline.get_glyph_texts",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/glm_image/pipeline_glm_image.py#L476",parameters:[{name:"prompt",val:""}]}),i(2),o(U),o(r);var j=e(r,2);n(j,{title:"GlmImagePipelineOutput",local:"diffusers.pipelines.glm_image.pipeline_output.GlmImagePipelineOutput",headingTag:"h2"});var m=e(j,2),R=t(m);a(R,{name:"class diffusers.pipelines.glm_image.pipeline_output.GlmImagePipelineOutput",anchor:"diffusers.pipelines.glm_image.pipeline_output.GlmImagePipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14229/src/diffusers/pipelines/glm_image/pipeline_output.py#L10",parameters:[{name:"images",val:": list[PIL.Image.Image] | numpy.ndarray"}],parametersDescription:[{anchor:"diffusers.pipelines.glm_image.pipeline_output.GlmImagePipelineOutput.images",description:`<strong>images</strong> (<code>List[PIL.Image.Image]</code> or <code>np.ndarray</code>) — | |
| List of denoised PIL images of length <code>batch_size</code> or numpy array of shape <code>(batch_size, height, width, num_channels)</code>. PIL images or numpy array present the denoised images of the diffusion pipeline.`,name:"images"}]}),i(2),o(m);var V=e(m,2);F(V,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/glm_image.md"}),i(2),u(Z,_),Q()}export{ae as component}; | |
Xet Storage Details
- Size:
- 34.4 kB
- Xet hash:
- 35738aee2006025aa04631391868877d1d84f3b0c34102ebb080240e4e914de5
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.