"""Portable Triton unpacking for E2M1 + E4M3 block16 scaled FP4 weights.""" import torch import triton import triton.language as tl @triton.jit def _unpack(W, S, G, O, SIZE:tl.constexpr, BLOCK:tl.constexpr): i=tl.program_id(0)*BLOCK+tl.arange(0,BLOCK) packed=tl.load(W+i//2, i>4) magnitude=code&7 value=tl.where(magnitude<4,magnitude.to(tl.float32)*.5, tl.where(magnitude<6,magnitude.to(tl.float32)-2,(magnitude.to(tl.float32)-4)*2)) value=tl.where((code&8)!=0,-value,value) scale_byte=tl.load(S+i//16,i>3)&15 power=((exponent+120)<<23).to(tl.float32,bitcast=True) scale=tl.where(exponent==0,mantissa.to(tl.float32)*.001953125,(1+mantissa.to(tl.float32)*.125)*power) result=value*(scale*tl.load(G)) tl.store(O+i,result,i