Spaces:
Sleeping
Sleeping
| from functools import partial | |
| from itertools import repeat | |
| from typing import Iterable | |
| import math | |
| import torch.nn as nn | |
| import torch.nn.functional as F | |
| __all__ = [ | |
| "get_abs_pos", | |
| "PatchEmbed", | |
| "Mlp", | |
| "DropPath", | |
| ] | |
| def to_2tuple(x): | |
| if isinstance(x, Iterable) and not isinstance(x, str): | |
| return tuple(x) | |
| return tuple(repeat(x, 2)) | |
| def get_abs_pos(abs_pos, has_cls_token, hw): | |
| """ | |
| Calculate absolute positional embeddings. If needed, resize embeddings and remove cls_token | |
| dimension for the original embeddings. | |
| Args: | |
| abs_pos (Tensor): absolute positional embeddings with (1, num_position, C). | |
| has_cls_token (bool): If true, has 1 embedding in abs_pos for cls token. | |
| hw (Tuple): size of input image tokens. | |
| Returns: | |
| Absolute positional embeddings after processing with shape (1, H, W, C) | |
| """ | |
| h, w = hw | |
| if has_cls_token: | |
| abs_pos = abs_pos[:, 1:] | |
| xy_num = abs_pos.shape[1] | |
| size = int(math.sqrt(xy_num)) | |
| assert size * size == xy_num | |
| if size != h or size != w: | |
| new_abs_pos = F.interpolate( | |
| abs_pos.reshape(1, size, size, -1).permute(0, 3, 1, 2), | |
| size=(h, w), | |
| mode="bicubic", | |
| align_corners=False, | |
| ) | |
| return new_abs_pos.permute(0, 2, 3, 1) | |
| else: | |
| return abs_pos.reshape(1, h, w, -1) | |
| def drop_path( | |
| x, drop_prob: float = 0.0, training: bool = False, scale_by_keep: bool = True | |
| ): | |
| if drop_prob == 0.0 or not training: | |
| return x | |
| keep_prob = 1 - drop_prob | |
| shape = (x.shape[0],) + (1,) * ( | |
| x.ndim - 1 | |
| ) # work with diff dim tensors, not just 2D ConvNets | |
| random_tensor = x.new_empty(shape).bernoulli_(keep_prob) | |
| if keep_prob > 0.0 and scale_by_keep: | |
| random_tensor.div_(keep_prob) | |
| return x * random_tensor | |
| class PatchEmbed(nn.Module): | |
| """ | |
| Image to Patch Embedding. | |
| """ | |
| def __init__( | |
| self, | |
| kernel_size=(16, 16), | |
| stride=(16, 16), | |
| padding=(0, 0), | |
| in_chans=3, | |
| embed_dim=768, | |
| ): | |
| """ | |
| Args: | |
| kernel_size (Tuple): kernel size of the projection layer. | |
| stride (Tuple): stride of the projection layer. | |
| padding (Tuple): padding size of the projection layer. | |
| in_chans (int): Number of input image channels. | |
| embed_dim (int): embed_dim (int): Patch embedding dimension. | |
| """ | |
| super().__init__() | |
| self.proj = nn.Conv2d( | |
| in_chans, embed_dim, kernel_size=kernel_size, stride=stride, padding=padding | |
| ) | |
| def forward(self, x): | |
| x = self.proj(x) | |
| # B C H W -> B H W C | |
| x = x.permute(0, 2, 3, 1) | |
| return x | |
| class DropPath(nn.Module): | |
| """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).""" | |
| def __init__(self, drop_prob: float = 0.0, scale_by_keep: bool = True): | |
| super(DropPath, self).__init__() | |
| self.drop_prob = drop_prob | |
| self.scale_by_keep = scale_by_keep | |
| def forward(self, x): | |
| return drop_path(x, self.drop_prob, self.training, self.scale_by_keep) | |
| def extra_repr(self): | |
| return f"drop_prob={round(self.drop_prob,3):0.3f}" | |
| class Mlp(nn.Module): | |
| def __init__( | |
| self, | |
| in_features, | |
| hidden_features=None, | |
| out_features=None, | |
| act_layer=nn.GELU, | |
| norm_layer=None, | |
| bias=True, | |
| drop=0.0, | |
| use_conv=False, | |
| ): | |
| super().__init__() | |
| out_features = out_features or in_features | |
| hidden_features = hidden_features or in_features | |
| bias = to_2tuple(bias) | |
| drop_probs = to_2tuple(drop) | |
| linear_layer = partial(nn.Conv2d, kernel_size=1) if use_conv else nn.Linear | |
| self.fc1 = linear_layer(in_features, hidden_features, bias=bias[0]) | |
| self.act = act_layer() | |
| self.drop1 = nn.Dropout(drop_probs[0]) | |
| self.norm = ( | |
| norm_layer(hidden_features) if norm_layer is not None else nn.Identity() | |
| ) | |
| self.fc2 = linear_layer(hidden_features, out_features, bias=bias[1]) | |
| self.drop2 = nn.Dropout(drop_probs[1]) | |
| def forward(self, x): | |
| x = self.fc1(x) | |
| x = self.act(x) | |
| x = self.drop1(x) | |
| x = self.norm(x) | |
| x = self.fc2(x) | |
| x = self.drop2(x) | |
| return x | |