{ "architecture": { "attention_variant": "MLA", "family": "causal_lm", "is_moe": true, "mixer": "attention", "moe": { "experts_per_token": 8, "num_experts": 256, "num_shared_experts": 1 }, "positional": "rope", "tie_word_embeddings": false, "view": "decoder" }, "capabilities": { "attention_backends": [ "eager", "sdpa", "flash_attention", "flex_attention" ], "attention_patterns": [ "causal" ], "attention_schedule": null, "kernels": { "RMSNorm": [ "kernels-community/liger-kernels", "kernels-community/rmsnorm", "kernels-community/mlx_rmsnorm" ] }, "task_heads": [ "causal_lm", "sequence_classification", "token_classification" ], "tensor_parallel": true }, "components": [ { "children": [ "embed_tokens", "decoder_layers", "norm", "rotary_emb" ], "class_name": "DeepseekV3Model", "id": "model", "kind": "model", "path_pattern": "model" }, { "attributes": { "embedding_dim": "config.hidden_size", "num_embeddings": "config.vocab_size", "tied_lm_head": false }, "class_name": "Embedding", "id": "embed_tokens", "kind": "embedding", "path_pattern": "model.embed_tokens" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "DeepseekV3RMSNorm", "id": "norm", "kind": "normalization", "path_pattern": "model.norm" }, { "attributes": { "head_dim": "config.head_dim", "rope_theta": 10000.0, "scheme": "rope" }, "class_name": "DeepseekV3RotaryEmbedding", "id": "rotary_emb", "kind": "position", "path_pattern": "model.rotary_emb" } ], "config": { "class_name": "DeepseekV3Config", "model_type": "deepseek_v3", "module": "transformers.models.deepseek_v3.configuration_deepseek_v3", "referenced_fields": { "num_hidden_layers": 61 }, "salient_fields": { "head_dim": 64, "hidden_act": "silu", "hidden_size": 7168, "intermediate_size": 18432, "is_encoder_decoder": false, "max_position_embeddings": 4096, "num_attention_heads": 128, "num_hidden_layers": 61, "num_key_value_heads": 128, "tie_word_embeddings": false, "vocab_size": 129280 } }, "edges": [ { "kind": "data", "source": "embed_tokens", "target": "decoder_layers" }, { "kind": "data", "source": "decoder_layers", "target": "norm" }, { "kind": "data", "source": "decoder_layer.self_attn", "target": "decoder_layer.mlp" }, { "kind": "data", "source": "decoder_layer.mlp", "target": "decoder_layer.input_layernorm" }, { "kind": "data", "source": "decoder_layer.input_layernorm", "target": "decoder_layer.post_attention_layernorm" }, { "kind": "mask", "source": "input:attention_mask", "target": "decoder_layer.self_attn" }, { "kind": "residual", "source": "decoder_layer", "target": "decoder_layer.self_attn" }, { "kind": "residual", "source": "decoder_layer", "target": "decoder_layer.mlp" }, { "kind": "position", "source": "rotary_emb", "target": "decoder_layer.self_attn" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.q_a_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.q_b_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.kv_a_proj_with_mqa" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.kv_b_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.q_a_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.q_b_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.kv_a_proj_with_mqa", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.kv_b_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp", "target": "decoder_layer.mlp.gate_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp", "target": "decoder_layer.mlp.up_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp.gate_proj", "target": "decoder_layer.mlp.down_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp.up_proj", "target": "decoder_layer.mlp.down_proj" }, { "kind": "cache_read", "source": "state:kv_cache", "target": "decoder_layer.self_attn" }, { "kind": "cache_write", "source": "decoder_layer.self_attn", "target": "state:kv_cache" } ], "extends": "llama", "model_type": "deepseek_v3", "patches": [ { "component_kind": "moe", "overridden": { "methods": [ "__init__", "forward" ] }, "parent_class": "DeepseekV2TopkRouter", "parent_model": "deepseek_v2", "relation": "inherits", "target_class": "DeepseekV3TopkRouter" }, { "component_kind": "moe", "overridden": { "methods": [ "__init__" ] }, "parent_class": "Qwen2MoeExperts", "parent_model": "qwen2_moe", "relation": "inherits", "target_class": "DeepseekV3Experts" }, { "added": { "methods": [ "__init__", "forward" ] }, "component_kind": "attention", "parent_class": "Module", "relation": "new", "target_class": "DeepseekV3Attention" }, { "component_kind": "transformer_block", "overridden": { "methods": [ "__init__" ] }, "parent_class": "LlamaDecoderLayer", "relation": "inherits", "target_class": "DeepseekV3DecoderLayer" }, { "added": { "attrs": [ "_keep_in_fp32_modules_strict", "_keys_to_ignore_on_load_unexpected" ], "methods": [ "_init_weights" ] }, "component_kind": "model", "parent_class": "LlamaPreTrainedModel", "relation": "inherits", "target_class": "DeepseekV3PreTrainedModel" }, { "component_kind": "head", "parent_class": "GenericForSequenceClassification", "relation": "new", "target_class": "DeepseekV3ForSequenceClassification" }, { "component_kind": "head", "parent_class": "GenericForTokenClassification", "relation": "new", "target_class": "DeepseekV3ForTokenClassification" } ], "provenance": { "config_class": "DeepseekV3Config", "config_module": "transformers.models.deepseek_v3.configuration_deepseek_v3", "model_class": "DeepseekV3Model", "model_module": "transformers.models.deepseek_v3.modeling_deepseek_v3" }, "repeats": [ { "body": "decoder_layer", "container_path_pattern": "model.layers", "count": 61, "count_expr": "config.num_hidden_layers", "count_source": "config", "id": "decoder_layers", "index_symbol": "i", "item_path_pattern": "model.layers.{i}", "kind": "symbolic_repeat", "provenance": { "class_name": "DeepseekV3DecoderLayer", "container_path_pattern": "model.layers", "item_path_pattern": "model.layers.{i}", "source": "module_tree_repeat_collapse" }, "repeated_class_name": "DeepseekV3DecoderLayer" } ], "schema_version": "architecture-template-v0", "templates": [ { "children": [ "decoder_layer.self_attn", "decoder_layer.mlp", "decoder_layer.input_layernorm", "decoder_layer.post_attention_layernorm" ], "class_name": "DeepseekV3DecoderLayer", "id": "decoder_layer", "kind": "transformer_block", "path_pattern": "model.layers.{i}" }, { "attributes": { "head_dim": 64, "n_heads": 128, "n_kv_heads": 128, "pattern": "causal", "rope": true, "variant": "MLA" }, "children": [ "decoder_layer.self_attn.q_a_proj", "decoder_layer.self_attn.q_a_layernorm", "decoder_layer.self_attn.q_b_proj", "decoder_layer.self_attn.kv_a_proj_with_mqa", "decoder_layer.self_attn.kv_a_layernorm", "decoder_layer.self_attn.kv_b_proj", "decoder_layer.self_attn.o_proj" ], "class_name": "DeepseekV3Attention", "id": "decoder_layer.self_attn", "kind": "attention", "path_pattern": "model.layers.{i}.self_attn" }, { "attributes": { "in_features": "config.hidden_size", "out_features": 1536 }, "class_name": "Linear", "id": "decoder_layer.self_attn.q_a_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.q_a_proj" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "DeepseekV3RMSNorm", "id": "decoder_layer.self_attn.q_a_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.self_attn.q_a_layernorm" }, { "attributes": { "in_features": 1536, "out_features": 24576 }, "class_name": "Linear", "id": "decoder_layer.self_attn.q_b_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.q_b_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": 576 }, "class_name": "Linear", "id": "decoder_layer.self_attn.kv_a_proj_with_mqa", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.kv_a_proj_with_mqa" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "DeepseekV3RMSNorm", "id": "decoder_layer.self_attn.kv_a_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.self_attn.kv_a_layernorm" }, { "attributes": { "in_features": 512, "out_features": 32768 }, "class_name": "Linear", "id": "decoder_layer.self_attn.kv_b_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.kv_b_proj" }, { "attributes": { "in_features": 16384, "out_features": "config.hidden_size" }, "class_name": "Linear", "id": "decoder_layer.self_attn.o_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.o_proj" }, { "attributes": { "activation": "silu", "hidden_size": 7168, "intermediate_size": 18432 }, "children": [ "decoder_layer.mlp.gate_proj", "decoder_layer.mlp.up_proj", "decoder_layer.mlp.down_proj" ], "class_name": "DeepseekV3MLP", "id": "decoder_layer.mlp", "kind": "feed_forward", "path_pattern": "model.layers.{i}.mlp" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.intermediate_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.gate_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.gate_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.intermediate_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.up_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.up_proj" }, { "attributes": { "in_features": "config.intermediate_size", "out_features": "config.hidden_size", "tp": "rowwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.down_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.down_proj" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "DeepseekV3RMSNorm", "id": "decoder_layer.input_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.input_layernorm" }, { "attributes": { "kernel": "RMSNorm", "norm_type": "rms" }, "class_name": "DeepseekV3RMSNorm", "id": "decoder_layer.post_attention_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.post_attention_layernorm" } ] }