Instructions to use toxzak/gemma4-e2b-exp-quant with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use toxzak/gemma4-e2b-exp-quant with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="toxzak/gemma4-e2b-exp-quant")# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("toxzak/gemma4-e2b-exp-quant", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use toxzak/gemma4-e2b-exp-quant with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "toxzak/gemma4-e2b-exp-quant" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "toxzak/gemma4-e2b-exp-quant", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/toxzak/gemma4-e2b-exp-quant
- SGLang
How to use toxzak/gemma4-e2b-exp-quant with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "toxzak/gemma4-e2b-exp-quant" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "toxzak/gemma4-e2b-exp-quant", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "toxzak/gemma4-e2b-exp-quant" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "toxzak/gemma4-e2b-exp-quant", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use toxzak/gemma4-e2b-exp-quant with Docker Model Runner:
docker model run hf.co/toxzak/gemma4-e2b-exp-quant
File size: 4,834 Bytes
9c41926 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 | from __future__ import annotations
from math import ceil
from typing import Sequence
import torch
INT4_QMAX = 7
def _validate_signed_int4(values: torch.Tensor) -> torch.Tensor:
flat = values.to(torch.int8).flatten()
if flat.numel() == 0:
return flat
if flat.min().item() < -8 or flat.max().item() > 7:
raise ValueError("signed int4 values must be in [-8, 7]")
return flat
def pack_signed_int4(values: torch.Tensor) -> torch.Tensor:
"""Pack signed int4 values into uint8 bytes, two values per byte."""
flat = _validate_signed_int4(values)
encoded = (flat + 8).to(torch.uint8)
if encoded.numel() % 2:
encoded = torch.cat([encoded, torch.full((1,), 8, dtype=torch.uint8, device=encoded.device)])
low = encoded[0::2]
high = torch.bitwise_left_shift(encoded[1::2], 4)
return torch.bitwise_or(low, high).contiguous()
def unpack_signed_int4(packed: torch.Tensor, count: int) -> torch.Tensor:
"""Unpack uint8 bytes produced by pack_signed_int4."""
if count < 0:
raise ValueError("count must be non-negative")
packed = packed.to(torch.uint8).flatten()
low = torch.bitwise_and(packed, 0x0F)
high = torch.bitwise_right_shift(packed, 4)
encoded = torch.stack((low, high), dim=1).flatten()[:count]
return encoded.to(torch.int16).sub(8).to(torch.int8)
def estimate_groupwise_int4_bpw(
shape: Sequence[int],
group_size: int = 128,
scale_bits: int = 16,
) -> float:
"""Estimate bits per weight for row-wise group INT4 plus per-group scales."""
if len(shape) != 2:
raise ValueError("shape must be a 2D weight matrix shape")
if group_size <= 0:
raise ValueError("group_size must be positive")
rows, cols = int(shape[0]), int(shape[1])
if rows <= 0 or cols <= 0:
raise ValueError("shape dimensions must be positive")
groups_per_row = ceil(cols / group_size)
weight_bits = rows * cols * 4
scale_overhead_bits = rows * groups_per_row * scale_bits
return (weight_bits + scale_overhead_bits) / (rows * cols)
def quantize_groupwise_int4(
weight: torch.Tensor,
group_size: int = 128,
scale_dtype: torch.dtype = torch.float16,
) -> dict:
"""Symmetric row-wise group INT4 quantization for 2D weight matrices.
This is an inference-oriented storage format: the packed nibbles and group
scales can be consumed by an INT4 kernel later, while still being easy to
dequantize for correctness/perplexity evaluation today.
"""
if weight.ndim != 2:
raise ValueError("weight must be a 2D tensor")
if group_size <= 0:
raise ValueError("group_size must be positive")
source = weight.detach().to(torch.float32)
rows, cols = source.shape
groups_per_row = ceil(cols / group_size)
padded_cols = groups_per_row * group_size
if padded_cols != cols:
padded = torch.zeros((rows, padded_cols), dtype=source.dtype, device=source.device)
padded[:, :cols] = source
source = padded
blocks = source.reshape(rows, groups_per_row, group_size)
max_abs = blocks.abs().amax(dim=2)
scales = torch.where(
max_abs > 0,
max_abs / INT4_QMAX,
torch.ones_like(max_abs),
)
stored_scales = scales.to(scale_dtype).to(torch.float32)
q = torch.round(blocks / stored_scales.unsqueeze(-1)).clamp(-INT4_QMAX, INT4_QMAX).to(torch.int8)
return {
"format": "groupwise_int4",
"bits": 4,
"group_size": int(group_size),
"orig_shape": [int(rows), int(cols)],
"padded_shape": [int(rows), int(padded_cols)],
"scales": stored_scales.to(scale_dtype).cpu(),
"packed_int4": pack_signed_int4(q.cpu()),
"bpw": estimate_groupwise_int4_bpw((rows, cols), group_size=group_size, scale_bits=16),
}
def dequantize_groupwise_int4(entry: dict, device: str | torch.device = "cpu") -> torch.Tensor:
"""Dequantize an entry produced by quantize_groupwise_int4."""
rows, cols = [int(dim) for dim in entry["orig_shape"]]
padded_rows, padded_cols = [int(dim) for dim in entry.get("padded_shape", entry["orig_shape"])]
group_size = int(entry["group_size"])
if padded_rows != rows:
raise ValueError("padded row count must match original row count")
if padded_cols % group_size != 0:
raise ValueError("padded columns must be divisible by group_size")
groups_per_row = padded_cols // group_size
count = rows * groups_per_row * group_size
q = unpack_signed_int4(entry["packed_int4"].to(device), count).to(torch.float32)
q = q.reshape(rows, groups_per_row, group_size)
scales = entry["scales"].to(device=device, dtype=torch.float32).reshape(rows, groups_per_row, 1)
restored = (q * scales).reshape(rows, padded_cols)
return restored[:, :cols].contiguous()
|