{ "architecture": { "attention_variant": "MHA", "family": "masked_lm", "is_moe": false, "mixer": "attention", "positional": "rope", "tie_word_embeddings": false, "view": "encoder" }, "capabilities": { "attention_backends": [ "eager", "sdpa", "flash_attention", "flex_attention" ], "attention_patterns": [ "bidirectional" ], "attention_schedule": null, "kernels": { "RMSNorm": [ "kernels-community/liger-kernels", "kernels-community/rmsnorm", "kernels-community/mlx_rmsnorm" ] }, "task_heads": [ "masked_lm", "sequence_classification", "token_classification" ], "tensor_parallel": true }, "components": [ { "children": [ "embed_tokens", "decoder_layers", "norm", "rotary_emb" ], "class_name": "EuroBertModel", "id": "model", "kind": "model", "path_pattern": "model" }, { "attributes": { "embedding_dim": "config.hidden_size", "num_embeddings": "config.vocab_size", "tied_lm_head": false }, "class_name": "Embedding", "id": "embed_tokens", "kind": "embedding", "path_pattern": "model.embed_tokens" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "EuroBertRMSNorm", "id": "norm", "kind": "normalization", "path_pattern": "model.norm" }, { "attributes": { "head_dim": "config.head_dim", "rope_theta": 10000.0, "scheme": "rope" }, "class_name": "EuroBertRotaryEmbedding", "id": "rotary_emb", "kind": "position", "path_pattern": "model.rotary_emb" } ], "config": { "class_name": "EuroBertConfig", "model_type": "eurobert", "module": "transformers.models.eurobert.configuration_eurobert", "referenced_fields": { "num_hidden_layers": 12 }, "salient_fields": { "head_dim": 64, "hidden_act": "silu", "hidden_size": 768, "intermediate_size": 3072, "is_encoder_decoder": false, "max_position_embeddings": 8192, "num_attention_heads": 12, "num_hidden_layers": 12, "num_key_value_heads": 12, "tie_word_embeddings": false, "vocab_size": 128256 } }, "dataflow": { "input": { "name": "input_ids", "shape": [ "B", "S" ] }, "output": { "shape": [ "B", "S", "config.hidden_size" ] }, "shapes": { "decoder_layer.input_layernorm": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layer.mlp": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layer.post_attention_layernorm": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layer.self_attn": { "in": null, "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layers": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "embed_tokens": { "in": [ "B", "S" ], "out": [ "B", "S", "config.hidden_size" ] }, "norm": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "rotary_emb": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.head_dim" ] } }, "source": "observed_forward_meta" }, "edges": [ { "kind": "data", "source": "embed_tokens", "target": "decoder_layers" }, { "kind": "data", "source": "decoder_layers", "target": "norm" }, { "kind": "mask", "source": "input:attention_mask", "target": "decoder_layer.self_attn" }, { "kind": "residual", "source": "decoder_layer", "target": "decoder_layer.self_attn" }, { "kind": "residual", "source": "decoder_layer", "target": "decoder_layer.mlp" }, { "kind": "position", "source": "rotary_emb", "target": "decoder_layer.self_attn" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.q_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.k_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.v_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.q_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.k_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.v_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp", "target": "decoder_layer.mlp.gate_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp", "target": "decoder_layer.mlp.up_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp.gate_proj", "target": "decoder_layer.mlp.down_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp.up_proj", "target": "decoder_layer.mlp.down_proj" }, { "kind": "data", "provenance": "observed_forward", "source": "decoder_layer.input_layernorm", "target": "decoder_layer.self_attn" }, { "kind": "data", "provenance": "observed_forward", "source": "decoder_layer.self_attn", "target": "decoder_layer.post_attention_layernorm" }, { "kind": "data", "provenance": "observed_forward", "source": "decoder_layer.post_attention_layernorm", "target": "decoder_layer.mlp" } ], "extends": "llama", "model_type": "eurobert", "patches": [ { "added": { "attrs": [ "attention_bias", "attention_dropout", "bos_token_id", "classifier_pooling", "eos_token_id", "head_dim", "hidden_act", "hidden_size", "initializer_range", "intermediate_size", "mask_token_id", "max_position_embeddings", "mlp_bias", "model_type", "num_attention_heads", "num_hidden_layers", "num_key_value_heads", "pad_token_id", "pretraining_tp", "rms_norm_eps", "rope_parameters", "tie_word_embeddings", "vocab_size" ], "methods": [ "__post_init__" ] }, "component_kind": "config", "parent_class": "LlamaConfig", "relation": "new", "target_class": "EuroBertConfig" }, { "component_kind": "normalization", "overridden": { "methods": [ "__init__" ] }, "parent_class": "LlamaRMSNorm", "relation": "inherits", "target_class": "EuroBertRMSNorm" }, { "component_kind": "attention", "overridden": { "methods": [ "__init__" ] }, "parent_class": "LlamaAttention", "relation": "inherits", "target_class": "EuroBertAttention" }, { "component_kind": "model", "overridden": { "methods": [ "forward" ] }, "parent_class": "LlamaModel", "relation": "inherits", "target_class": "EuroBertModel" }, { "added": { "attrs": [ "_pp_plan", "_tied_weights_keys", "_tp_plan" ], "methods": [ "__init__", "forward" ] }, "component_kind": null, "parent_class": "EuroBertPreTrainedModel", "relation": "new", "target_class": "EuroBertForMaskedLM" }, { "added": { "methods": [ "__init__", "forward" ] }, "component_kind": "head", "parent_class": "EuroBertPreTrainedModel", "relation": "new", "target_class": "EuroBertForSequenceClassification" }, { "added": { "methods": [ "__init__", "forward", "get_input_embeddings", "set_input_embeddings" ] }, "component_kind": "head", "parent_class": "EuroBertPreTrainedModel", "relation": "new", "target_class": "EuroBertForTokenClassification" } ], "provenance": { "config_class": "EuroBertConfig", "config_module": "transformers.models.eurobert.configuration_eurobert", "model_class": "EuroBertModel", "model_module": "transformers.models.eurobert.modeling_eurobert" }, "repeats": [ { "body": "decoder_layer", "container_path_pattern": "model.layers", "count": 12, "count_expr": "config.num_hidden_layers", "count_source": "config", "id": "decoder_layers", "index_symbol": "i", "item_path_pattern": "model.layers.{i}", "kind": "symbolic_repeat", "provenance": { "class_name": "EuroBertDecoderLayer", "container_path_pattern": "model.layers", "item_path_pattern": "model.layers.{i}", "source": "module_tree_repeat_collapse" }, "repeated_class_name": "EuroBertDecoderLayer" } ], "schema_version": "architecture-template-v0", "templates": [ { "children": [ "decoder_layer.self_attn", "decoder_layer.mlp", "decoder_layer.input_layernorm", "decoder_layer.post_attention_layernorm" ], "class_name": "EuroBertDecoderLayer", "id": "decoder_layer", "kind": "transformer_block", "path_pattern": "model.layers.{i}" }, { "attributes": { "head_dim": 64, "n_heads": 12, "n_kv_heads": 12, "pattern": "bidirectional", "rope": true, "variant": "MHA" }, "children": [ "decoder_layer.self_attn.q_proj", "decoder_layer.self_attn.k_proj", "decoder_layer.self_attn.v_proj", "decoder_layer.self_attn.o_proj" ], "class_name": "EuroBertAttention", "id": "decoder_layer.self_attn", "kind": "attention", "path_pattern": "model.layers.{i}.self_attn" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.q_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.q_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.k_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.k_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.v_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.v_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "rowwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.o_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.o_proj" }, { "attributes": { "activation": "silu", "hidden_size": 768, "intermediate_size": 3072 }, "children": [ "decoder_layer.mlp.gate_proj", "decoder_layer.mlp.up_proj", "decoder_layer.mlp.down_proj" ], "class_name": "EuroBertMLP", "id": "decoder_layer.mlp", "kind": "feed_forward", "path_pattern": "model.layers.{i}.mlp" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.intermediate_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.gate_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.gate_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.intermediate_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.up_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.up_proj" }, { "attributes": { "in_features": "config.intermediate_size", "out_features": "config.hidden_size", "tp": "rowwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.down_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.down_proj" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "EuroBertRMSNorm", "id": "decoder_layer.input_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.input_layernorm" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "EuroBertRMSNorm", "id": "decoder_layer.post_attention_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.post_attention_layernorm" } ] }