| import os |
| from dataclasses import dataclass |
|
|
| import torch |
| from einops import rearrange |
| from huggingface_hub import hf_hub_download |
| |
| from safetensors.torch import load_file as load_sft |
|
|
| from flux.model import Flux, FluxParams |
| from flux.modules.autoencoder import AutoEncoder, AutoEncoderParams |
| from flux.modules.conditioner import HFEmbedder |
| from transformers import (CLIPTextModel, CLIPTokenizer, T5EncoderModel, |
| T5Tokenizer, BitsAndBytesConfig) |
|
|
|
|
| @dataclass |
| class ModelSpec: |
| params: FluxParams |
| ae_params: AutoEncoderParams |
| ckpt_path: str | None |
| ae_path: str | None |
| repo_id: str | None |
| repo_flow: str | None |
| repo_ae: str | None |
|
|
| configs = { |
| "flux-dev": ModelSpec( |
| repo_id="black-forest-labs/FLUX.1-dev", |
| repo_flow="flux1-dev.safetensors", |
| repo_ae=None, |
| ckpt_path=os.getenv("FLUX_DEV"), |
| params=FluxParams( |
| in_channels=64, |
| out_channels=64, |
| vec_in_dim=768, |
| context_in_dim=4096, |
| hidden_size=3072, |
| mlp_ratio=4.0, |
| num_heads=24, |
| depth=19, |
| depth_single_blocks=38, |
| axes_dim=[16, 56, 56], |
| theta=10_000, |
| qkv_bias=True, |
| guidance_embed=True, |
| ), |
| ae_path=os.getenv("AE"), |
| ae_params=AutoEncoderParams( |
| resolution=256, |
| in_channels=3, |
| ch=128, |
| out_ch=3, |
| ch_mult=[1, 2, 4, 4], |
| num_res_blocks=2, |
| z_channels=16, |
| scale_factor=0.3611, |
| shift_factor=0.1159, |
| ), |
| ), |
| "flux-fill-dev": ModelSpec( |
| repo_id="black-forest-labs/FLUX.1-Fill-dev", |
| repo_flow="flux1-fill-dev.safetensors", |
| repo_ae="ae.safetensors", |
| ckpt_path=os.getenv("FLUX_FILL_DEV"), |
| params=FluxParams( |
| in_channels=64, |
| out_channels=384, |
| vec_in_dim=768, |
| context_in_dim=4096, |
| hidden_size=3072, |
| mlp_ratio=4.0, |
| num_heads=24, |
| depth=19, |
| depth_single_blocks=38, |
| axes_dim=[16, 56, 56], |
| theta=10_000, |
| qkv_bias=True, |
| guidance_embed=True, |
| ), |
| ae_path=os.getenv("AE"), |
| ae_params=AutoEncoderParams( |
| resolution=256, |
| in_channels=3, |
| ch=128, |
| out_ch=3, |
| ch_mult=[1, 2, 4, 4], |
| num_res_blocks=2, |
| z_channels=16, |
| scale_factor=0.3611, |
| shift_factor=0.1159, |
| ), |
| ), |
| "flux-kontext-dev": ModelSpec( |
| repo_id="black-forest-labs/FLUX.1-Kontext-dev", |
| repo_flow="flux1-kontext-dev.safetensors", |
| repo_ae="ae.safetensors", |
| ckpt_path=os.getenv("FLUX_FILL_DEV"), |
| params=FluxParams( |
| in_channels=64, |
| out_channels=64, |
| vec_in_dim=768, |
| context_in_dim=4096, |
| hidden_size=3072, |
| mlp_ratio=4.0, |
| num_heads=24, |
| depth=19, |
| depth_single_blocks=38, |
| axes_dim=[16, 56, 56], |
| theta=10_000, |
| qkv_bias=True, |
| guidance_embed=True, |
| ), |
| ae_path=os.getenv("AE"), |
| ae_params=AutoEncoderParams( |
| resolution=256, |
| in_channels=3, |
| ch=128, |
| out_ch=3, |
| ch_mult=[1, 2, 4, 4], |
| num_res_blocks=2, |
| z_channels=16, |
| scale_factor=0.3611, |
| shift_factor=0.1159, |
| ), |
| ), |
| "flux-schnell": ModelSpec( |
| repo_id="black-forest-labs/FLUX.1-schnell", |
| repo_flow="flux1-schnell.safetensors", |
| repo_ae="black-forest-labs/FLUX.1-schnell", |
| ckpt_path=os.getenv("FLUX_SCHNELL"), |
| params=FluxParams( |
| in_channels=64, |
| out_channels=64, |
| vec_in_dim=768, |
| context_in_dim=4096, |
| hidden_size=3072, |
| mlp_ratio=4.0, |
| num_heads=24, |
| depth=19, |
| depth_single_blocks=38, |
| axes_dim=[16, 56, 56], |
| theta=10000.0, |
| qkv_bias=True, |
| guidance_embed=False, |
| ), |
| ae_path="ae.safetensors", |
| ae_params=AutoEncoderParams( |
| resolution=256, |
| in_channels=3, |
| ch=128, |
| out_ch=3, |
| ch_mult=[1, 2, 4, 4], |
| num_res_blocks=2, |
| z_channels=16, |
| scale_factor=0.3611, |
| shift_factor=0.1159, |
| ), |
| ), |
| } |
|
|
|
|
| def print_load_warning(missing: list[str], unexpected: list[str]) -> None: |
| if len(missing) > 0 and len(unexpected) > 0: |
| print(f"Got {len(missing)} missing keys:\n\t" + "\n\t".join(missing)) |
| print("\n" + "-" * 79 + "\n") |
| print(f"Got {len(unexpected)} unexpected keys:\n\t" + "\n\t".join(unexpected)) |
| elif len(missing) > 0: |
| print(f"Got {len(missing)} missing keys:\n\t" + "\n\t".join(missing)) |
| elif len(unexpected) > 0: |
| print(f"Got {len(unexpected)} unexpected keys:\n\t" + "\n\t".join(unexpected)) |
|
|
| def _replace_linear_with_4bit(module, compute_dtype=torch.bfloat16): |
| """Recursively replace all nn.Linear with bitsandbytes 4-bit layers""" |
| import bitsandbytes as bnb |
| for name, child in module.named_children(): |
| if name == "img_in": |
| continue |
| if isinstance(child, torch.nn.Linear): |
| has_bias = child.bias is not None |
| new_layer = bnb.nn.Linear4bit( |
| child.in_features, |
| child.out_features, |
| bias=has_bias, |
| compute_dtype=compute_dtype, |
| compress_statistics=True, |
| quant_type="nf4", |
| ) |
| new_layer.weight = bnb.nn.Params4bit( |
| child.weight.data, |
| requires_grad=False, |
| quant_type="nf4", |
| ) |
| if has_bias: |
| new_layer.bias = torch.nn.Parameter(child.bias.data) |
| setattr(module, name, new_layer) |
| else: |
| _replace_linear_with_4bit(child, compute_dtype) |
|
|
| def load_flow_model(name: str, device: str | torch.device = "cuda", hf_download: bool = True): |
| |
| print("Init model") |
| |
| ckpt_path = configs[name].ckpt_path |
| if ( |
| ckpt_path is None |
| and configs[name].repo_id is not None |
| and configs[name].repo_flow is not None |
| and hf_download |
| ): |
| ckpt_path = hf_hub_download(configs[name].repo_id, configs[name].repo_flow) |
|
|
| |
| target_device = torch.device(device) |
| model = Flux(configs[name].params).to(dtype=torch.bfloat16) |
|
|
| if ckpt_path is not None: |
| print("Loading checkpoint") |
| |
| sd = load_sft(ckpt_path, device="cpu") |
|
|
| |
| sd = {k.replace("model.diffusion_model.", ""): v for k, v in sd.items()} |
| |
|
|
| |
| img_in_weight = sd.pop("img_in.weight", None) |
| img_in_bias = sd.pop("img_in.bias", None) |
|
|
| |
| missing, unexpected = model.load_state_dict(sd, strict=False, assign=True) |
| print_load_warning(missing, unexpected) |
|
|
| |
| |
|
|
| if img_in_weight is not None: |
| with torch.no_grad(): |
| w = img_in_weight.to(device=device, dtype=torch.bfloat16) |
| |
| |
| if model.img_in.weight.shape[1] != w.shape[1]: |
| |
| model.img_in.weight = torch.nn.Parameter(model.img_in.weight[:, :w.shape[1]]) |
| |
| model.img_in.weight.copy_(w) |
|
|
| if img_in_bias is not None and getattr(model.img_in, "bias", None) is not None: |
| with torch.no_grad(): |
| b = img_in_bias.to(device=device, dtype=torch.bfloat16) |
| model.img_in.bias.copy_(b) |
|
|
| |
| print("Quantizing model to NF4 (this may take a minute)...") |
| _replace_linear_with_4bit(model, compute_dtype=torch.bfloat16) |
| print("NF4 quantization complete.") |
|
|
| |
| model = model.to(target_device) |
| return model |
|
|
|
|
| def load_t5(device: str | torch.device = "cuda", max_length: int = 512) -> HFEmbedder: |
| |
| return HFEmbedder( |
| "google/t5-v1_1-xxl", |
| max_length=max_length, |
| is_clip=False, |
| torch_dtype=torch.bfloat16, |
| device_map="cpu" |
| ) |
|
|
|
|
| def load_clip(device: str | torch.device = "cuda") -> HFEmbedder: |
| |
| return HFEmbedder("openai/clip-vit-large-patch14", max_length=77, is_clip=True, torch_dtype=torch.bfloat16) |
|
|
|
|
| def load_ae(name: str, device: str | torch.device = "cuda", hf_download: bool = True) -> AutoEncoder: |
| ckpt_path = configs[name].ae_path |
|
|
| |
| if ckpt_path is not None and not os.path.exists(ckpt_path) and hf_download: |
| repo_id = configs[name].repo_ae or configs[name].repo_id |
| ckpt_path = hf_hub_download(repo_id, ckpt_path) |
| elif ckpt_path is None and configs[name].repo_id is not None and hf_download: |
| repo_id = configs[name].repo_ae or configs[name].repo_id |
| ckpt_path = hf_hub_download(repo_id, "ae.safetensors") |
|
|
| |
| print("Init AE") |
|
|
| |
| ae = AutoEncoder(configs[name].ae_params) |
|
|
| if ckpt_path is not None: |
| sd = load_sft(ckpt_path, device=str(device)) |
| missing, unexpected = ae.load_state_dict(sd, strict=False, assign=True) |
| print_load_warning(missing, unexpected) |
|
|
| ae = ae.to(device) |
| return ae |
|
|
|
|
| |
| |
| |
| |
| |
| |
|
|
| |
| |
| |
|
|
| |
| |
|
|
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
|
|
|
|
| |
| |
| |
| |
| |
|
|