MyApricity
/

transquant_2

Model card Files Files and versions

xet

Community

Your Name commited on Jan 26, 2025

Commit

24f9b3f

1 Parent(s): 4fcd1d5

yy

Browse files

Files changed (2) hide show

pyproject.toml +1 -1
src/pipeline.py +141 -8

pyproject.toml CHANGED Viewed

@@ -16,7 +16,7 @@ dependencies = [
   "protobuf==5.28.3",
   "sentencepiece==0.2.0",
   "torchao==0.6.1",
-  "optimum-quanto",
   "hf_transfer==0.1.8",
   "setuptools==75.2.0",
   "edge-maxxing-pipelines @ git+https://github.com/womboai/edge-maxxing@7c760ac54f6052803dadb3ade8ebfc9679a94589#subdirectory=pipelines",

   "protobuf==5.28.3",
   "sentencepiece==0.2.0",
   "torchao==0.6.1",
+  "bitsandbytes",
   "hf_transfer==0.1.8",
   "setuptools==75.2.0",
   "edge-maxxing-pipelines @ git+https://github.com/womboai/edge-maxxing@7c760ac54f6052803dadb3ade8ebfc9679a94589#subdirectory=pipelines",

src/pipeline.py CHANGED Viewed

@@ -2,7 +2,8 @@ import os
 import torch
 import torch._dynamo
 import gc
 import json
 import transformers
 from huggingface_hub.constants import HF_HUB_CACHE
@@ -15,7 +16,6 @@ from diffusers import FluxTransformer2DModel, DiffusionPipeline
 from PIL.Image import Image
 from diffusers import FluxPipeline, AutoencoderKL, AutoencoderTiny
 from pipelines.models import TextToImageRequest
-from optimum.quanto import requantize
 import json
@@ -40,6 +40,138 @@ def remove_cache():
     torch.cuda.reset_max_memory_allocated()
     torch.cuda.reset_peak_memory_stats()
 class InitModel:
@@ -93,19 +225,20 @@ def load_pipeline() -> Pipeline:
                         torch_dtype=torch.bfloat16)
     pipeline.to("cuda")
     try:
-        pipeline.disable_vae_slice()
     except:
         print("Using origin pipeline")
-    promts_listing = [
-        "melanogen, endosome",
         "buffer, cutie, buttinsky, prototrophic",
-        "puzzlehead, fistical, must return non duplicate",
-        "apical, polymyodous, tiptilt"
     ]
-    for p in promts_listing:
         pipeline(prompt=p,
                         width=1024,
                         height=1024,

 import torch
 import torch._dynamo
 import gc
+import bitsandbytes as bnb
+from bitsandbytes.nn.modules import Params4bit, QuantState
 import json
 import transformers
 from huggingface_hub.constants import HF_HUB_CACHE
 from PIL.Image import Image
 from diffusers import FluxPipeline, AutoencoderKL, AutoencoderTiny
 from pipelines.models import TextToImageRequest
 import json
     torch.cuda.reset_max_memory_allocated()
     torch.cuda.reset_peak_memory_stats()
+# ---------------- NF4 ----------------
+def functional_linear_4bits(x, weight, bias):
+    out = bnb.matmul_4bit(x, weight.t(), bias=bias, quant_state=weight.quant_state)
+    out = out.to(x)
+    return out
+def copy_quant_state(state, device=None):
+    if state is None:
+        return None
+    device = device or state.absmax.device
+    state2 = (
+        QuantState(
+            absmax=state.state2.absmax.to(device),
+            shape=state.state2.shape,
+            code=state.state2.code.to(device),
+            blocksize=state.state2.blocksize,
+            quant_type=state.state2.quant_type,
+            dtype=state.state2.dtype,
+        )
+        if state.nested
+        else None
+    )
+    return QuantState(
+        absmax=state.absmax.to(device),
+        shape=state.shape,
+        code=state.code,
+        blocksize=state.blocksize,
+        quant_type=state.quant_type,
+        dtype=state.dtype,
+        offset=state.offset.to(device) if state.nested else None,
+        state2=state2,
+    )
+class ForgeParams4bit(Params4bit):
+    def to(self, *args, **kwargs):
+        device, dtype, non_blocking, convert_to_format = torch._C._nn._parse_to(*args, **kwargs)
+        if device is not None and device.type == "cuda" and not self.bnb_quantized:
+            return self._quantize(device)
+        else:
+            n = ForgeParams4bit(
+                torch.nn.Parameter.to(self, device=device, dtype=dtype, non_blocking=non_blocking),
+                requires_grad=self.requires_grad,
+                quant_state=copy_quant_state(self.quant_state, device),
+                compress_statistics=False,
+                blocksize=64,
+                quant_type=self.quant_type,
+                quant_storage=self.quant_storage,
+                bnb_quantized=self.bnb_quantized,
+                module=self.module
+            )
+            self.module.quant_state = n.quant_state
+            self.data = n.data
+            self.quant_state = n.quant_state
+            return n
+class ForgeLoader4Bit(torch.nn.Module):
+    def __init__(self, *, device, dtype, quant_type, **kwargs):
+        super().__init__()
+        self.dummy = torch.nn.Parameter(torch.empty(1, device=device, dtype=dtype))
+        self.weight = None
+        self.quant_state = None
+        self.bias = None
+        self.quant_type = quant_type
+    def _save_to_state_dict(self, destination, prefix, keep_vars):
+        super()._save_to_state_dict(destination, prefix, keep_vars)
+        quant_state = getattr(self.weight, "quant_state", None)
+        if quant_state is not None:
+            for k, v in quant_state.as_dict(packed=True).items():
+                destination[prefix + "weight." + k] = v if keep_vars else v.detach()
+        return
+    def _load_from_state_dict(self, state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs):
+        quant_state_keys = {k[len(prefix + "weight."):] for k in state_dict.keys() if k.startswith(prefix + "weight.")}
+        if any('bitsandbytes' in k for k in quant_state_keys):
+            quant_state_dict = {k: state_dict[prefix + "weight." + k] for k in quant_state_keys}
+            self.weight = ForgeParams4bit.from_prequantized(
+                data=state_dict[prefix + 'weight'],
+                quantized_stats=quant_state_dict,
+                requires_grad=False,
+                device=torch.device('cuda'),
+                module=self
+            )
+            self.quant_state = self.weight.quant_state
+            if prefix + 'bias' in state_dict:
+                self.bias = torch.nn.Parameter(state_dict[prefix + 'bias'].to(self.dummy))
+            del self.dummy
+        elif hasattr(self, 'dummy'):
+            if prefix + 'weight' in state_dict:
+                self.weight = ForgeParams4bit(
+                    state_dict[prefix + 'weight'].to(self.dummy),
+                    requires_grad=False,
+                    compress_statistics=True,
+                    quant_type=self.quant_type,
+                    quant_storage=torch.uint8,
+                    module=self,
+                )
+                self.quant_state = self.weight.quant_state
+            if prefix + 'bias' in state_dict:
+                self.bias = torch.nn.Parameter(state_dict[prefix + 'bias'].to(self.dummy))
+            del self.dummy
+        else:
+            super()._load_from_state_dict(state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs)
+class Linear(ForgeLoader4Bit):
+    def __init__(self, *args, device=None, dtype=None, **kwargs):
+        super().__init__(device=device, dtype=dtype, quant_type='nf4')
+    def forward(self, x):
+        self.weight.quant_state = self.quant_state
+        if self.bias is not None and self.bias.dtype != x.dtype:
+            self.bias.data = self.bias.data.to(x.dtype)
+        return functional_linear_4bits(x, self.weight, self.bias)
+# Replace nn.Linear with the 4-bit quantized Linear
+# torch.nn.Linear = Linear
 class InitModel:
                         torch_dtype=torch.bfloat16)
     pipeline.to("cuda")
     try:
+        pipeline.enable_vae_slicing()
+        torch.nn.LinearLayer = Linear
     except:
         print("Using origin pipeline")
+    prms = [
+        "melanogen, tiptilt",
+        "melanogen, endosome, apical, polymyodous, ",
         "buffer, cutie, buttinsky, prototrophic",
+        "puzzlehead",
     ]
+    for __ in prms:
         pipeline(prompt=p,
                         width=1024,
                         height=1024,