ideogram4_custom_blocks / ideogram4_unified.py
OzzyGT's picture
OzzyGT HF Staff
Upload 7 files
0a66847 verified
Raw
History Blame Contribute Delete
5.24 kB
# Copyright 2026 The HuggingFace Team. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""Unified custom modular-diffusers AutoBlocks for Ideogram4.
Folds text-to-image, image-to-image, and Differential Diffusion into a single `AutoBlocks`, selected by which
inputs are present (the canonical `QwenImageAutoBlocks` shape). Pure assembly: every leaf/core block is reused
from the in-repo Ideogram4 blocks and the img2img / differential modules; nothing new is computed here.
prompt -> text2image
prompt + image -> image2image
prompt + image + diffdiff_map -> differential diffusion
"""
from diffusers.modular_pipelines.ideogram4.decoders import Ideogram4DecodeStep
from diffusers.modular_pipelines.ideogram4.encoders import Ideogram4PromptUpsampleStep, Ideogram4TextEncoderStep
from diffusers.modular_pipelines.ideogram4.modular_blocks_ideogram4 import Ideogram4CoreDenoiseStep
from diffusers.modular_pipelines.modular_pipeline import ConditionalPipelineBlocks, SequentialPipelineBlocks
from diffusers.modular_pipelines.modular_pipeline_utils import InsertableDict, OutputParam
from diffusers.utils import logging
from .ideogram4_differential import Ideogram4DiffDiffCoreDenoiseStep
from .ideogram4_img2img import Ideogram4AutoVaeEncoderStep, Ideogram4Img2ImgCoreDenoiseStep
logger = logging.get_logger(__name__) # pylint: disable=invalid-name
class Ideogram4UnifiedCoreDenoiseStep(ConditionalPipelineBlocks):
block_classes = [Ideogram4CoreDenoiseStep, Ideogram4Img2ImgCoreDenoiseStep, Ideogram4DiffDiffCoreDenoiseStep]
block_names = ["text2image", "img2img", "differential"]
block_trigger_inputs = ["image_latents", "diffdiff_map"]
default_block_name = "text2image"
def select_block(self, image_latents=None, diffdiff_map=None):
# `diffdiff_map` must be checked before `image_latents`: differential diffusion also produces `image_latents`
# (it VAE-encodes the reference), so the map is the discriminator between it and plain img2img.
if diffdiff_map is not None:
return "differential"
if image_latents is not None:
return "img2img"
return "text2image"
@property
def description(self) -> str:
return (
"Core denoising step. \n"
" - `Ideogram4DiffDiffCoreDenoiseStep` (differential) is used when `diffdiff_map` is provided.\n"
" - `Ideogram4Img2ImgCoreDenoiseStep` (img2img) is used when `image_latents` is provided (no map).\n"
" - `Ideogram4CoreDenoiseStep` (text2image) is used otherwise."
)
@property
def outputs(self) -> list[OutputParam]:
return [OutputParam.template("latents", description="Unpatchified (B, ae_channels, H, W) latents.")]
UNIFIED_AUTO_BLOCKS = InsertableDict(
[
("prompt_upsample", Ideogram4PromptUpsampleStep()),
("text_encoder", Ideogram4TextEncoderStep()),
("vae_encoder", Ideogram4AutoVaeEncoderStep()),
("denoise", Ideogram4UnifiedCoreDenoiseStep()),
("decode", Ideogram4DecodeStep()),
]
)
# auto_docstring
class Ideogram4AutoBlocks(SequentialPipelineBlocks):
"""
Unified Auto Modular pipeline for Ideogram4: text-to-image, image-to-image, and Differential Diffusion in one,
selected by which inputs are present. (optional) prompt upsampling -> encode text -> VAE-encode the reference
(when `image` is given) -> core denoise (asymmetric CFG over two transformers) -> decode.
Supported workflows:
- `text2image`: requires `prompt`
- `image2image`: requires `prompt`, `image` (optional `strength`)
- `differential`: requires `prompt`, `image`, `diffdiff_map`
Note: `strength` applies only to image-to-image; Differential Diffusion uses the full schedule and ignores it.
"""
model_name = "ideogram4"
block_classes = list(UNIFIED_AUTO_BLOCKS.values())
block_names = list(UNIFIED_AUTO_BLOCKS.keys())
_workflow_map = {
"text2image": {"prompt": True},
"image2image": {"prompt": True, "image": True},
"differential": {"prompt": True, "image": True, "diffdiff_map": True},
}
@property
def description(self) -> str:
return (
"Unified Auto Modular pipeline for Ideogram4 (text-to-image / image-to-image / Differential Diffusion), "
"selected by which inputs are present: (optional) prompt upsampling -> encode text -> VAE-encode the "
"reference when `image` is given -> core denoise (asymmetric CFG over two transformers) -> decode."
)
@property
def outputs(self) -> list[OutputParam]:
return [OutputParam.template("images")]