x

Files changed (4) hide show

README.md +1 -0
mapping_t5.json +0 -1
src/pipeline.py +44 -76
transformer_int8.json +0 -0

README.md ADDED Viewed

	@@ -0,0 +1 @@


1	+ # FLUX OPT

mapping_t5.json DELETED Viewed

@@ -1 +0,0 @@

- {"encoder.block.0.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.0.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.0.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.0.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.0.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.0.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.0.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.1.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.2.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.3.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.4.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.5.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.6.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.7.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.8.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.9.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.10.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.11.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.12.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.13.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.14.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.15.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.16.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.17.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.18.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.19.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.20.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.21.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.22.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.0.SelfAttention.q": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.0.SelfAttention.k": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.0.SelfAttention.v": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.0.SelfAttention.o": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.1.DenseReluDense.wi_0": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.1.DenseReluDense.wi_1": {"weights": "qint8", "activations": "none"}, "encoder.block.23.layer.1.DenseReluDense.wo": {"weights": "qint8", "activations": "none"}}

src/pipeline.py CHANGED Viewed

@@ -1,34 +1,37 @@
-# FLux Optimization Pipeline
 import os
 import torch
 import torch._dynamo
 import gc
 from huggingface_hub.constants import HF_HUB_CACHE
-from transformers import T5EncoderModel, T5TokenizerFast, CLIPTokenizer, CLIPTextModel
 from torchao.quantization import quantize_, int8_weight_only, fpx_weight_only
 from torch import Generator
 from diffusers import FluxTransformer2DModel, DiffusionPipeline
 from PIL.Image import Image
-from diffusers import FluxPipeline, AutoencoderKL, AutoencoderTiny
 from pipelines.models import TextToImageRequest
-from optimum.quanto import requantize
-import json
-import transformers
 torch._dynamo.config.suppress_errors = True
 os.environ['PYTORCH_CUDA_ALLOC_CONF']="expandable_segments:True"
 os.environ["TOKENIZERS_PARALLELISM"] = "True"
-CHECKPOINT = "black-forest-labs/FLUX.1-schnell"
-REVISION = "741f7c3ce8b383c54771c7003378a50191e9efe9"
 Pipeline = None
-apply_quanto=1
 import torch
 import gc
@@ -36,88 +39,52 @@ import os
 import json
 import transformers
-def perform_memory_maintenance():
-    """A convoluted way of handling memory management for CUDA."""
-    [fn() for fn in [
-        torch.cuda.empty_cache,
-        torch.cuda.reset_max_memory_allocated,
-        torch.cuda.reset_peak_memory_stats,
-        gc.collect
-    ]]
-def obscurely_load_encoder(repo_path):
-    """
-    Loads a T5 encoder with multiple layers of abstraction and complexity.
-    Args:
-        repo_path (str): The cryptic location of the repository files.
-    Returns:
-        An enigmatic, quantized T5 encoder model.
-    """
-    # Hidden mechanism to load JSON data
-    def load_json(file_path):
-        with open(file_path, "r") as f:
-            return json.load(f)
-    # Fetch quantization map
-    quant_map = load_json("mapping_t5.json")
-    # Acquire the mysterious T5 configuration
-    t5_config = transformers.T5Config(**load_json(os.path.join(repo_path, "config.json")))
-    # Cloak the model instantiation in an unfamiliar syntax
-    device_context = torch.device("cuda")
-    encoder = transformers.T5EncoderModel(t5_config).to(torch.bfloat16) if device_context.type == "meta" else None
-    # A vacuous state_dict waiting for purpose
-    model_weights = None
-    # Perform the shadowy act of quantization
-    requantize(
-        model=encoder,
-        state_dict=model_weights,
-        quantization_map=quant_map,
-        device=torch.device("cuda")
-    )
-    return encoder
-def load_pipeline() -> Pipeline:
-    try:
-        origin_t5_path = os.path.join(HF_HUB_CACHE, "models--RichardWilliam--XULF_T5_bf16/snapshots/63a3d9ef7b586655600ac9bd4e4747d038237761")
-        text_encoder_2 = obscurely_load_encoder(_path=origin_t5_path)
-    except:
-        text_encoder_2 =  T5EncoderModel.from_pretrained("RichardWilliam/XULF_T5_bf16",
-                    revision = "63a3d9ef7b586655600ac9bd4e4747d038237761",
-                    torch_dtype=torch.bfloat16).to(memory_format=torch.channels_last)
-    origin_vae = AutoencoderTiny.from_pretrained("RichardWilliam/XULF_Vae",
                     revision="3ee225c539465c27adadec45c6e8af50a7397b7d",
                     torch_dtype=torch.bfloat16)
     trans_path = os.path.join(HF_HUB_CACHE, "models--RichardWilliam--XULF_Transfomer/snapshots/6860c51af40329808f270e159a0d018559a1204f")
-    origin_trans = FluxTransformer2DModel.from_pretrained(trans_path,
                         torch_dtype=torch.bfloat16,
                         use_safetensors=False).to(memory_format=torch.channels_last)
-    transformer = origin_trans
-    pipeline = DiffusionPipeline.from_pretrained(CHECKPOINT,
-                        revision=REVISION,
-                        vae=origin_vae,
                         transformer=transformer,
                         text_encoder_2=text_encoder_2,
                         torch_dtype=torch.bfloat16)
     pipeline.to("cuda")
     try:
-        # pipeline.enable_sequential_cpu_offload()
-        pipeline.vae.enable_slicing()
     except:
-        pass
-    for __ in range(3):
-        pipeline(prompt="sweet, subordinative, gender, mormyre, arteriolosclerosis, positivism, Antiochianism, palmerite",
                         width=1024,
                         height=1024,
                         guidance_scale=0.0,
@@ -128,7 +95,8 @@ def load_pipeline() -> Pipeline:
 @torch.no_grad()
 def infer(request: TextToImageRequest, pipeline: Pipeline) -> Image:
-    perform_memory_maintenance()
     generator = Generator(pipeline.device).manual_seed(request.seed)

+# Quanto optimization, unique
 import os
 import torch
 import torch._dynamo
 import gc
+import json
+import transformers
 from huggingface_hub.constants import HF_HUB_CACHE
+from transformers import T5EncoderModel
+import diffusers
 from torchao.quantization import quantize_, int8_weight_only, fpx_weight_only
 from torch import Generator
 from diffusers import FluxTransformer2DModel, DiffusionPipeline
 from PIL.Image import Image
+from diffusers import AutoencoderTiny
 from pipelines.models import TextToImageRequest
+from optimum.quanto import requantize as optimum_quant
+try:
+    from huggingface_hub import hf_hub_download
+except:
+    pass
 torch._dynamo.config.suppress_errors = True
 os.environ['PYTORCH_CUDA_ALLOC_CONF']="expandable_segments:True"
 os.environ["TOKENIZERS_PARALLELISM"] = "True"
+ckpt_main = "black-forest-labs/FLUX.1-schnell"
+revision_main = "741f7c3ce8b383c54771c7003378a50191e9efe9"
 Pipeline = None
+apply_transformer_tag = 1
 import torch
 import gc
 import json
 import transformers
+def convert_transformer_to_int8(repo_path):
+    with open("transformer_int8.json", "r") as f:
+        quantization_map = json.load(f)
+    with torch.device("meta"):
+        transformer_config_path = os.path.join(repo_path, "config.json")
+        transformer = diffusers.FluxTransformer2DModel.from_config(transformer_config_path).to(torch.bfloat16)
+    state_dict = hf_hub_download(repo_path, "diffusion_pytorch_models.safetensors")
+    optimum_quant(transformer, state_dict, quantization_map, device=torch.device("cuda"))
+    return transformer
+def load_pipeline() -> Pipeline:
+    original_vae = AutoencoderTiny.from_pretrained("RichardWilliam/XULF_Vae",
                     revision="3ee225c539465c27adadec45c6e8af50a7397b7d",
                     torch_dtype=torch.bfloat16)
+    text_encoder_2 =  T5EncoderModel.from_pretrained("RichardWilliam/XULF_T5_bf16",
+                revision = "63a3d9ef7b586655600ac9bd4e4747d038237761",
+                torch_dtype=torch.bfloat16).to(memory_format=torch.channels_last)
     trans_path = os.path.join(HF_HUB_CACHE, "models--RichardWilliam--XULF_Transfomer/snapshots/6860c51af40329808f270e159a0d018559a1204f")
+    pre_quanted_trans = FluxTransformer2DModel.from_pretrained(trans_path,
                         torch_dtype=torch.bfloat16,
                         use_safetensors=False).to(memory_format=torch.channels_last)
+    transformer = pre_quanted_trans
+    pipeline = DiffusionPipeline.from_pretrained(ckpt_main,
+                        revision=revision_main,
+                        vae=original_vae,
                         transformer=transformer,
                         text_encoder_2=text_encoder_2,
                         torch_dtype=torch.bfloat16)
     pipeline.to("cuda")
     try:
+        pipeline.enable_int8()
+        pipeline.transformer = convert_transformer_to_int8(trans_path)
     except:
+        print("Use origin pipeline")
+    for warm_up_prompt in range(3):
+        pipeline(prompt="puffer, cutie, buttinsky, prototrophic, betulinamaric, quintet, tunesome, decaspermous",
                         width=1024,
                         height=1024,
                         guidance_scale=0.0,
 @torch.no_grad()
 def infer(request: TextToImageRequest, pipeline: Pipeline) -> Image:
+    gc.collect()
+    torch.cuda.empty_cache()
     generator = Generator(pipeline.device).manual_seed(request.seed)

transformer_int8.json ADDED Viewed

File without changes