| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
|
|
| import math |
| from dataclasses import dataclass |
| from typing import Any |
|
|
| import torch |
| from torch import nn |
| from transformers.configuration_utils import PretrainedConfig |
| from transformers.modeling_utils import PreTrainedModel |
| from transformers.utils import ModelOutput, logging |
| from transformers.utils.deprecation import deprecate_kwarg |
|
|
| from fla.layers.mamba import Mamba |
| from fla.models.mamba.configuration_mamba import MambaConfig |
| from fla.models.utils import FLAGenerationMixin |
| from fla.modules import FusedCrossEntropyLoss, FusedLinearCrossEntropyLoss, RMSNorm |
| from fla.modules.l2warp import l2_warp |
|
|
| try: |
| from transformers.modeling_layers import GradientCheckpointingLayer |
| except ImportError: |
| from fla.models.modeling_layers import GradientCheckpointingLayer |
|
|
| logger = logging.get_logger(__name__) |
|
|
|
|
| class MambaCache: |
| """ |
| Cache for mamba model which does not have attention mechanism and key value states. |
| |
| Arguments: |
| config (`PretrainedConfig): |
| The configuration file defining the shape-related attributes required to initialize the static cache. |
| batch_size (`int`): |
| The batch size with which the model will be used. Note that a new instance must be instantiated if a |
| smaller batch size is used. |
| dtype (`torch.dtype`, *optional*, defaults to `torch.float16`): |
| The default `dtype` to use when initializing the layer. |
| device (`torch.device` or `str`, *optional*): |
| The device on which the cache should be initialized. Should be the same as the layer. |
| |
| Attributes: |
| dtype: (`torch.dtype`): |
| The default `dtype` used to initializing the cache. |
| intermediate_size: (`int`): |
| Model's intermediate_size taken from config. |
| ssm_state_size: (`int`): |
| Model's state_size taken from config. |
| conv_kernel_size: (`int`): |
| Model's convolution kernel size taken from config |
| conv_states: (`torch.Tensor`): |
| A tensor of shape `[layer_idx, batch_size, intermediate_size, conv_kernel_size]` that holds convolutional states. |
| ssm_states: (`torch.Tensor`): |
| A tensor of shape `[layer_idx, batch_size, intermediate_size, ssm_state_size]` that holds ssm states |
| |
| Example: |
| |
| ```python |
| >>> from transformers import AutoTokenizer, MambaForCausalLM, MambaCache |
| |
| >>> model = MambaForCausalLM.from_pretrained("state-spaces/mamba-130m-hf") |
| >>> tokenizer = AutoTokenizer.from_pretrained("state-spaces/mamba-130m-hf") |
| |
| >>> inputs = tokenizer(text="My name is Mamba", return_tensors="pt") |
| |
| >>> # Prepare a cache class and pass it to model's forward |
| >>> past_key_values = MambaCache(config=model.config, batch_size=1, device=model.device, dtype=model.dtype) |
| >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True) |
| >>> outputs.past_key_values |
| MambaCache() |
| ``` |
| """ |
|
|
| |
| def __init__( |
| self, |
| config: PretrainedConfig, |
| batch_size: int = None, |
| dtype: torch.dtype = torch.float16, |
| device: torch.device | str | None = None, |
| max_batch_size: int | None = None, |
| ): |
| if max_batch_size is not None: |
| logger.warning_once( |
| f"The 'max_batch_size' argument of {self.__class__.__name__} is deprecated and will be removed in " |
| "v4.46. Use the more precisely named 'batch_size' argument instead.", |
| ) |
| self.dtype = dtype |
| self.batch_size = batch_size or max_batch_size |
| self.intermediate_size = config.intermediate_size |
| self.ssm_state_size = config.state_size |
| self.conv_kernel_size = config.conv_kernel |
|
|
| self.conv_states: torch.Tensor = torch.zeros( |
| config.num_hidden_layers, |
| self.batch_size, |
| self.intermediate_size, |
| self.conv_kernel_size, |
| device=device, |
| dtype=dtype, |
| ) |
| self.ssm_states: torch.Tensor = torch.zeros( |
| config.num_hidden_layers, |
| self.batch_size, |
| self.intermediate_size, |
| self.ssm_state_size, |
| device=device, |
| dtype=dtype, |
| ) |
|
|
| torch._dynamo.mark_static_address(self.conv_states) |
| torch._dynamo.mark_static_address(self.ssm_states) |
|
|
| def update_conv_state( |
| self, layer_idx: int, new_conv_state: torch.Tensor, cache_position: torch.LongTensor, |
| ) -> torch.Tensor: |
| conv_state = self.conv_states[layer_idx] |
| cache_position = cache_position.clamp(0, self.conv_kernel_size - 1) |
|
|
| conv_state = conv_state.roll(shifts=-1, dims=-1) |
| conv_state[:, :, cache_position] = new_conv_state.to(conv_state.device) |
| self.conv_states[layer_idx].zero_() |
| self.conv_states[layer_idx] += conv_state |
| return self.conv_states[layer_idx] |
|
|
| def update_ssm_state(self, layer_idx: int, new_ssm_state: torch.Tensor): |
| self.ssm_states[layer_idx] = new_ssm_state.to(self.ssm_states.device) |
| return self.ssm_states[layer_idx] |
|
|
| def reset(self): |
| self.conv_states.zero_() |
| self.ssm_states.zero_() |
|
|
|
|
| class MambaBlock(GradientCheckpointingLayer): |
|
|
| def __init__(self, config, layer_idx): |
| super().__init__() |
| self.config = config |
| self.layer_idx = layer_idx |
| self.residual_in_fp32 = config.residual_in_fp32 |
| self.norm = RMSNorm(config.hidden_size, eps=config.norm_eps) |
| self.mixer = Mamba( |
| hidden_size=config.hidden_size, |
| state_size=config.state_size, |
| conv_kernel=config.conv_kernel, |
| intermediate_size=config.intermediate_size, |
| time_step_rank=config.time_step_rank, |
| use_bias=config.use_bias, |
| layer_idx=layer_idx, |
| ) |
|
|
| def forward( |
| self, |
| hidden_states, |
| cache_params: MambaCache | None = None, |
| cache_position: torch.LongTensor | None = None, |
| attention_mask: torch.LongTensor | None = None, |
| ): |
| residual = hidden_states |
| hidden_states = self.norm(hidden_states) |
| if self.residual_in_fp32: |
| residual = residual.to(torch.float32) |
|
|
| hidden_states = self.mixer( |
| hidden_states, cache_params=cache_params, cache_position=cache_position, attention_mask=attention_mask, |
| ) |
| hidden_states = residual + hidden_states |
| if self.residual_in_fp32: |
| hidden_states = hidden_states.to(dtype=self.norm.weight.dtype) |
| return hidden_states |
|
|
|
|
| class MambaPreTrainedModel(PreTrainedModel): |
| """ |
| An abstract class to handle weights initialization and a simple interface for downloading and loading pretrained |
| models. |
| """ |
|
|
| config_class = MambaConfig |
| base_model_prefix = 'backbone' |
| _no_split_modules = ['Mamba', 'MambaBlock'] |
| supports_gradient_checkpointing = True |
| _is_stateful = True |
|
|
| def _init_weights(self, module): |
| """Initialize the weights.""" |
| if isinstance(module, nn.Linear): |
| nn.init.normal_(module.weight, mean=0.0, std=self.config.initializer_range) |
| if module.bias is not None: |
| if not getattr(module.bias, "_no_reinit", False): |
| nn.init.zeros_(module.bias) |
| elif isinstance(module, Mamba): |
| module.A_log._no_weight_decay = True |
| module.D._no_weight_decay = True |
|
|
| dt_init_std = self.config.time_step_rank**-0.5 * self.config.time_step_scale |
| if self.config.time_step_init_scheme == "constant": |
| nn.init.constant_(module.dt_proj.weight, dt_init_std) |
| elif self.config.time_step_init_scheme == "random": |
| nn.init.uniform_(module.dt_proj.weight, -dt_init_std, dt_init_std) |
|
|
| dt = torch.exp( |
| torch.rand(self.config.intermediate_size) |
| * (math.log(self.config.time_step_max) - math.log(self.config.time_step_min)) |
| + math.log(self.config.time_step_min), |
| ).clamp(min=self.config.time_step_floor) |
| |
| inv_dt = dt + torch.log(-torch.expm1(-dt)) |
| with torch.no_grad(): |
| module.dt_proj.bias.data = nn.Parameter(inv_dt.to(module.dt_proj.bias.device)) |
| module.dt_proj.bias._no_reinit = True |
| elif isinstance(module, nn.Embedding): |
| nn.init.normal_(module.weight, std=self.config.initializer_range) |
| elif hasattr(module, 'reset_parameters'): |
| module.reset_parameters() |
|
|
| if self.config.rescale_prenorm_residual: |
| |
| |
| |
| |
| |
| |
| for name, p in module.named_parameters(): |
| if name in ["out_proj.weight"]: |
| |
| |
| |
| |
| nn.init.kaiming_uniform_(p, a=math.sqrt(5)) |
| with torch.no_grad(): |
| p /= math.sqrt(self.config.num_hidden_layers) |
|
|
|
|
| @dataclass |
| class MambaOutput(ModelOutput): |
| """ |
| Class for the MAMBA model outputs. |
| |
| Args: |
| last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`): |
| Sequence of hidden-states at the output of the last layer of the model. |
| cache_params (`MambaCache`): |
| The state of the model at the last time step. Can be used in a forward method with the next `input_ids` to |
| avoid providing the old `input_ids`. |
| |
| Includes both the State space model state matrices after the selective scan, and the Convolutional states |
| hidden_states (`tuple(torch.FloatTensor)`, *optional*, |
| returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`): |
| Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, + |
| one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`. |
| |
| Hidden-states of the model at the output of each layer plus the optional initial embedding outputs. |
| """ |
|
|
| last_hidden_state: torch.FloatTensor | None = None |
| cache_params: MambaCache | None = None |
| hidden_states: tuple[torch.FloatTensor] | None = None |
|
|
|
|
| @dataclass |
| class MambaCausalLMOutput(ModelOutput): |
| """ |
| Base class for causal language model (or autoregressive) outputs. |
| |
| Args: |
| loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided): |
| Language modeling loss (for next-token prediction). |
| logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`): |
| Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax). |
| cache_params (`MambaCache`): |
| The state of the model at the last time step. Can be used in a forward method with the next `input_ids` to |
| avoid providing the old `input_ids`. |
| |
| Includes both the State space model state matrices after the selective scan, and the Convolutional states |
| hidden_states (`tuple(torch.FloatTensor)`, *optional*, |
| returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`): |
| Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, + |
| one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`. |
| |
| Hidden-states of the model at the output of each layer plus the optional initial embedding outputs. |
| """ |
|
|
| loss: torch.FloatTensor | None = None |
| logits: torch.FloatTensor | None = None |
| cache_params: MambaCache | None = None |
| hidden_states: tuple[torch.FloatTensor] | None = None |
|
|
|
|
| class MambaModel(MambaPreTrainedModel): |
| def __init__(self, config): |
| super().__init__(config) |
|
|
| self.embeddings = nn.Embedding(config.vocab_size, config.hidden_size) |
| self.layers = nn.ModuleList([MambaBlock(config, layer_idx=idx) for idx in range(config.num_hidden_layers)]) |
|
|
| self.gradient_checkpointing = False |
| self.norm_f = RMSNorm(config.hidden_size, eps=config.norm_eps) |
| |
| self._register_load_state_dict_pre_hook(self.load_hook) |
| self.post_init() |
|
|
| def load_hook(self, state_dict, prefix, *args): |
| for k in state_dict: |
| if "embedding." in k: |
| state_dict[k.replace("embedding.", "embeddings.")] = state_dict.pop(k) |
| break |
|
|
| def get_input_embeddings(self): |
| return self.embeddings |
|
|
| def set_input_embeddings(self, new_embeddings): |
| self.embeddings = new_embeddings |
|
|
| def forward( |
| self, |
| input_ids: torch.LongTensor | None = None, |
| inputs_embeds: torch.LongTensor | None = None, |
| cache_params: MambaCache | None = None, |
| use_cache: bool | None = None, |
| output_hidden_states: bool | None = None, |
| return_dict: bool | None = None, |
| cache_position: torch.LongTensor | None = None, |
| attention_mask: torch.LongTensor | None = None, |
| ) -> tuple | MambaOutput: |
| output_hidden_states = ( |
| output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states |
| ) |
| use_cache = use_cache if use_cache is not None else (self.config.use_cache if not self.training else False) |
| return_dict = return_dict if return_dict is not None else self.config.use_return_dict |
|
|
| if (input_ids is None) ^ (inputs_embeds is not None): |
| raise ValueError( |
| "You cannot specify both input_ids and inputs_embeds at the same time, and must specify either one", |
| ) |
|
|
| if inputs_embeds is None: |
| inputs_embeds = self.embeddings(input_ids) |
|
|
| if use_cache: |
| if cache_params is None: |
| cache_params = MambaCache( |
| self.config, inputs_embeds.size(0), device=inputs_embeds.device, dtype=inputs_embeds.dtype, |
| ) |
| cache_position = torch.arange(0, self.config.conv_kernel, device=inputs_embeds.device) |
| elif cache_position is None: |
| |
| |
| |
| raise ValueError( |
| "You have to specify the `cache_position` manually when `use_cache=True` and `cache_params` is passed, " |
| "you don't have to pass a `cache_params` if you are in prefilling stage because in that case it will " |
| "be initialized for you automatically", |
| ) |
| else: |
| cache_params = None |
|
|
| hidden_states = inputs_embeds |
| all_hidden_states = () if output_hidden_states else None |
| for mixer_block in self.layers: |
| hidden_states = mixer_block( |
| hidden_states, |
| cache_params=cache_params, |
| cache_position=cache_position, |
| attention_mask=attention_mask, |
| ) |
|
|
| if output_hidden_states: |
| all_hidden_states = all_hidden_states + (hidden_states,) |
|
|
| hidden_states = self.norm_f(hidden_states) |
|
|
| if output_hidden_states: |
| all_hidden_states = all_hidden_states + (hidden_states,) |
|
|
| if not return_dict: |
| return tuple(v for v in [hidden_states, cache_params, all_hidden_states] if v is not None) |
|
|
| return MambaOutput( |
| last_hidden_state=hidden_states, |
| cache_params=cache_params if use_cache else None, |
| hidden_states=all_hidden_states, |
| ) |
|
|
|
|
| class MambaForCausalLM(MambaPreTrainedModel, FLAGenerationMixin): |
|
|
| _tied_weights_keys = ["lm_head.weight"] |
|
|
| def __init__(self, config): |
| super().__init__(config) |
| self.backbone = MambaModel(config) |
| self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False) |
| self.criterion = None |
|
|
| |
| self.post_init() |
|
|
| def get_output_embeddings(self): |
| return self.lm_head |
|
|
| def set_output_embeddings(self, new_embeddings): |
| self.lm_head = new_embeddings |
|
|
| def get_input_embeddings(self): |
| return self.backbone.get_input_embeddings() |
|
|
| def set_input_embeddings(self, new_embeddings): |
| return self.backbone.set_input_embeddings(new_embeddings) |
|
|
| def _update_model_kwargs_for_generation( |
| self, outputs: ModelOutput, |
| model_kwargs: dict[str, Any], |
| num_new_tokens: int = 1, |
| **kwargs, |
| ) -> dict[str, Any]: |
| model_kwargs["cache_params"] = outputs.get("cache_params", None) |
| if ( |
| model_kwargs.get("use_cache", True) |
| and "cache_position" in model_kwargs |
| and model_kwargs["cache_position"] is not None |
| ): |
| model_kwargs["cache_position"] = model_kwargs["cache_position"][-1:] + num_new_tokens |
|
|
| if "attention_mask" in model_kwargs: |
| attention_mask = model_kwargs["attention_mask"] |
| model_kwargs["attention_mask"] = torch.cat( |
| [attention_mask, attention_mask.new_ones((attention_mask.shape[0], 1))], dim=-1, |
| ) |
|
|
| return model_kwargs |
|
|
| @deprecate_kwarg("num_logits_to_keep", version="4.50", new_name="logits_to_keep") |
| def forward( |
| self, |
| input_ids: torch.LongTensor | None = None, |
| attention_mask: torch.LongTensor | None = None, |
| inputs_embeds: torch.FloatTensor | None = None, |
| cache_params: MambaCache | None = None, |
| labels: torch.LongTensor | None = None, |
| output_hidden_states: bool | None = None, |
| return_dict: bool | None = None, |
| use_cache: bool | None = None, |
| cache_position: torch.Tensor | None = None, |
| logits_to_keep: int | None = 0, |
| **kwargs, |
| ) -> tuple | MambaCausalLMOutput: |
| r""" |
| labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*): |
| Labels for language modeling. Note that the labels **are shifted** inside the model, i.e. you can set |
| `labels = input_ids` Indices are selected in `[-100, 0, ..., config.vocab_size]` All labels set to `-100` |
| are ignored (masked), the loss is only computed for labels in `[0, ..., config.vocab_size]` |
| """ |
| return_dict = return_dict if return_dict is not None else self.config.use_return_dict |
|
|
| mamba_outputs = self.backbone( |
| input_ids, |
| cache_params=cache_params, |
| inputs_embeds=inputs_embeds, |
| output_hidden_states=output_hidden_states, |
| return_dict=return_dict, |
| use_cache=use_cache, |
| cache_position=cache_position, |
| attention_mask=attention_mask, |
| ) |
| hidden_states = mamba_outputs[0] |
|
|
| loss, logits = None, None |
| if not self.config.fuse_linear_cross_entropy or labels is None: |
| logits = self.lm_head(hidden_states if logits_to_keep is None else hidden_states[:, -logits_to_keep:]) |
| if labels is not None: |
| if getattr(self, 'criterion', None) is None: |
| if self.config.fuse_linear_cross_entropy: |
| criterion = FusedLinearCrossEntropyLoss(use_l2warp=self.config.use_l2warp) |
| elif self.config.fuse_cross_entropy: |
| criterion = FusedCrossEntropyLoss(inplace_backward=True) |
| else: |
| criterion = nn.CrossEntropyLoss() |
| else: |
| criterion = self.criterion |
| |
| labels = labels.to(hidden_states.device) |
| labels = torch.cat((labels[..., 1:], torch.full_like(labels[:, :1], criterion.ignore_index)), 1) |
| if self.config.fuse_linear_cross_entropy: |
| loss = criterion(hidden_states, labels, self.lm_head.weight, self.lm_head.bias) |
| else: |
| loss = criterion(logits.view(labels.numel(), -1), labels.view(-1)) |
| loss = l2_warp(loss, logits) if self.config.use_l2warp else loss |
|
|
| if not return_dict: |
| output = (logits,) + mamba_outputs[1:] |
| return (loss,) + output if loss is not None else output |
|
|
| return MambaCausalLMOutput( |
| loss=loss, |
| logits=logits, |
| cache_params=mamba_outputs.cache_params, |
| hidden_states=mamba_outputs.hidden_states, |
| ) |
|
|