{ "architecture": { "attention_variant": "MHA", "family": "causal_lm", "is_moe": false, "mixer": "attention", "positional": "rope", "tie_word_embeddings": true, "view": "decoder" }, "capabilities": { "attention_backends": [ "eager", "sdpa", "flash_attention", "flex_attention" ], "attention_patterns": [ "causal" ], "attention_schedule": null, "task_heads": [ "causal_lm" ], "tensor_parallel": true }, "components": [ { "children": [ "embed_tokens", "decoder_layers", "norm", "rotary_emb" ], "class_name": "CohereModel", "id": "model", "kind": "model", "path_pattern": "model" }, { "attributes": { "embedding_dim": "config.hidden_size", "num_embeddings": "config.vocab_size", "tied_lm_head": true }, "class_name": "Embedding", "id": "embed_tokens", "kind": "embedding", "path_pattern": "model.embed_tokens" }, { "attributes": { "norm_type": "layer" }, "class_name": "CohereLayerNorm", "id": "norm", "kind": "normalization", "path_pattern": "model.norm" }, { "attributes": { "rope_theta": 500000.0, "scheme": "rope" }, "class_name": "CohereRotaryEmbedding", "id": "rotary_emb", "kind": "position", "path_pattern": "model.rotary_emb" } ], "config": { "class_name": "CohereConfig", "model_type": "cohere", "module": "transformers.models.cohere.configuration_cohere", "referenced_fields": { "num_hidden_layers": 40 }, "salient_fields": { "hidden_act": "silu", "hidden_size": 8192, "intermediate_size": 22528, "is_encoder_decoder": false, "max_position_embeddings": 8192, "num_attention_heads": 64, "num_hidden_layers": 40, "num_key_value_heads": 64, "tie_word_embeddings": true, "vocab_size": 256000 } }, "dataflow": { "input": { "name": "input_ids", "shape": [ "B", "S" ] }, "output": { "shape": [ "B", "S", "config.hidden_size" ] }, "shapes": { "decoder_layer.input_layernorm": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layer.mlp": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layer.self_attn": { "in": null, "out": [ "B", "S", "config.hidden_size" ] }, "decoder_layers": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "embed_tokens": { "in": [ "B", "S" ], "out": [ "B", "S", "config.hidden_size" ] }, "norm": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", "config.hidden_size" ] }, "rotary_emb": { "in": [ "B", "S", "config.hidden_size" ], "out": [ "B", "S", 128 ] } }, "source": "observed_forward_meta" }, "edges": [ { "kind": "data", "source": "embed_tokens", "target": "decoder_layers" }, { "kind": "data", "source": "decoder_layers", "target": "norm" }, { "kind": "mask", "source": "input:attention_mask", "target": "decoder_layer.self_attn" }, { "kind": "residual", "source": "decoder_layer", "target": "decoder_layer.self_attn" }, { "kind": "residual", "source": "decoder_layer", "target": "decoder_layer.mlp" }, { "kind": "position", "source": "rotary_emb", "target": "decoder_layer.self_attn" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.q_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.k_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn", "target": "decoder_layer.self_attn.v_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.q_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.k_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.self_attn.v_proj", "target": "decoder_layer.self_attn.o_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp", "target": "decoder_layer.mlp.gate_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp", "target": "decoder_layer.mlp.up_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp.gate_proj", "target": "decoder_layer.mlp.down_proj" }, { "kind": "data", "provenance": "intra_module", "source": "decoder_layer.mlp.up_proj", "target": "decoder_layer.mlp.down_proj" }, { "kind": "cache_read", "source": "state:kv_cache", "target": "decoder_layer.self_attn" }, { "kind": "cache_write", "source": "decoder_layer.self_attn", "target": "state:kv_cache" }, { "kind": "data", "provenance": "observed_forward", "source": "decoder_layer.input_layernorm", "target": "decoder_layer.self_attn" }, { "kind": "data", "provenance": "observed_forward", "source": "decoder_layer.self_attn", "target": "decoder_layer.mlp" } ], "extends": "llama", "model_type": "cohere", "patches": [ { "added": { "methods": [ "__init__", "forward" ] }, "component_kind": "normalization", "parent_class": "Module", "relation": "new", "target_class": "CohereLayerNorm" }, { "component_kind": "position", "overridden": { "methods": [ "forward" ] }, "parent_class": "LlamaRotaryEmbedding", "relation": "inherits", "target_class": "CohereRotaryEmbedding" }, { "component_kind": "feed_forward", "overridden": { "methods": [ "__init__" ] }, "parent_class": "LlamaMLP", "relation": "inherits", "target_class": "CohereMLP" }, { "component_kind": "attention", "overridden": { "methods": [ "__init__", "forward" ] }, "parent_class": "LlamaAttention", "relation": "inherits", "target_class": "CohereAttention" }, { "added": { "methods": [ "__init__", "forward" ] }, "component_kind": "transformer_block", "parent_class": "GradientCheckpointingLayer", "relation": "new", "target_class": "CohereDecoderLayer" }, { "component_kind": "model", "overridden": { "methods": [ "__init__" ] }, "parent_class": "LlamaModel", "relation": "inherits", "target_class": "CohereModel" }, { "component_kind": "lm_head", "overridden": { "methods": [ "__init__", "forward" ] }, "parent_class": "LlamaForCausalLM", "relation": "inherits", "target_class": "CohereForCausalLM" } ], "provenance": { "config_class": "CohereConfig", "config_module": "transformers.models.cohere.configuration_cohere", "model_class": "CohereModel", "model_module": "transformers.models.cohere.modeling_cohere" }, "repeats": [ { "body": "decoder_layer", "container_path_pattern": "model.layers", "count": 40, "count_expr": "config.num_hidden_layers", "count_source": "config", "id": "decoder_layers", "index_symbol": "i", "item_path_pattern": "model.layers.{i}", "kind": "symbolic_repeat", "provenance": { "class_name": "CohereDecoderLayer", "container_path_pattern": "model.layers", "item_path_pattern": "model.layers.{i}", "source": "module_tree_repeat_collapse" }, "repeated_class_name": "CohereDecoderLayer" } ], "schema_version": "architecture-template-v0", "templates": [ { "children": [ "decoder_layer.self_attn", "decoder_layer.mlp", "decoder_layer.input_layernorm" ], "class_name": "CohereDecoderLayer", "id": "decoder_layer", "kind": "transformer_block", "path_pattern": "model.layers.{i}" }, { "attributes": { "head_dim": 128, "n_heads": 64, "n_kv_heads": 64, "pattern": "causal", "rope": true, "variant": "MHA" }, "children": [ "decoder_layer.self_attn.q_proj", "decoder_layer.self_attn.k_proj", "decoder_layer.self_attn.v_proj", "decoder_layer.self_attn.o_proj" ], "class_name": "CohereAttention", "id": "decoder_layer.self_attn", "kind": "attention", "path_pattern": "model.layers.{i}.self_attn" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.q_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.q_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.k_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.k_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.v_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.v_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.hidden_size", "tp": "rowwise" }, "class_name": "Linear", "id": "decoder_layer.self_attn.o_proj", "kind": "projection", "path_pattern": "model.layers.{i}.self_attn.o_proj" }, { "attributes": { "activation": "silu", "hidden_size": 8192, "intermediate_size": 22528 }, "children": [ "decoder_layer.mlp.gate_proj", "decoder_layer.mlp.up_proj", "decoder_layer.mlp.down_proj" ], "class_name": "CohereMLP", "id": "decoder_layer.mlp", "kind": "feed_forward", "path_pattern": "model.layers.{i}.mlp" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.intermediate_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.gate_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.gate_proj" }, { "attributes": { "in_features": "config.hidden_size", "out_features": "config.intermediate_size", "tp": "colwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.up_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.up_proj" }, { "attributes": { "in_features": "config.intermediate_size", "out_features": "config.hidden_size", "tp": "rowwise" }, "class_name": "Linear", "id": "decoder_layer.mlp.down_proj", "kind": "projection", "path_pattern": "model.layers.{i}.mlp.down_proj" }, { "attributes": { "norm_type": "layer" }, "class_name": "CohereLayerNorm", "id": "decoder_layer.input_layernorm", "kind": "normalization", "path_pattern": "model.layers.{i}.input_layernorm" } ] }