| { |
| "01-ai/Yi-1.5-6B-Chat": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 4096, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 4, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 6.0 |
| }, |
| "01-ai/Yi-1.5-9B-Chat": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 4096, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 4, |
| "num_layers": 48, |
| "torch_dtype": null, |
| "total_b_params": 9.0 |
| }, |
| "EleutherAI/gpt-j-6b": { |
| "architectures": [ |
| "GPTJForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": null, |
| "model_type": "gptj", |
| "num_attention_heads": 16, |
| "num_key_value_heads": 16, |
| "num_layers": 28, |
| "torch_dtype": null, |
| "total_b_params": 6.05 |
| }, |
| "EleutherAI/gpt-neox-20b": { |
| "architectures": [ |
| "GPTNeoXForCausalLM" |
| ], |
| "hidden_size": 6144, |
| "max_position_embeddings": 2048, |
| "model_type": "gpt_neox", |
| "num_attention_heads": 64, |
| "num_key_value_heads": 64, |
| "num_layers": 44, |
| "torch_dtype": null, |
| "total_b_params": 20.0 |
| }, |
| "HuggingFaceTB/SmolLM2-1.7B-Instruct": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 2048, |
| "max_position_embeddings": 8192, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 32, |
| "num_layers": 24, |
| "torch_dtype": null, |
| "total_b_params": 1.7 |
| }, |
| "Qwen/Qwen2.5-0.5B-Instruct": { |
| "architectures": [ |
| "Qwen2ForCausalLM" |
| ], |
| "hidden_size": 896, |
| "max_position_embeddings": 32768, |
| "model_type": "qwen2", |
| "num_attention_heads": 14, |
| "num_key_value_heads": 2, |
| "num_layers": 24, |
| "torch_dtype": null, |
| "total_b_params": 0.49 |
| }, |
| "Qwen/Qwen2.5-1.5B-Instruct": { |
| "architectures": [ |
| "Qwen2ForCausalLM" |
| ], |
| "hidden_size": 1536, |
| "max_position_embeddings": 32768, |
| "model_type": "qwen2", |
| "num_attention_heads": 12, |
| "num_key_value_heads": 2, |
| "num_layers": 28, |
| "torch_dtype": null, |
| "total_b_params": 1.54 |
| }, |
| "Qwen/Qwen2.5-3B-Instruct": { |
| "architectures": [ |
| "Qwen2ForCausalLM" |
| ], |
| "hidden_size": 2048, |
| "max_position_embeddings": 32768, |
| "model_type": "qwen2", |
| "num_attention_heads": 16, |
| "num_key_value_heads": 2, |
| "num_layers": 36, |
| "torch_dtype": null, |
| "total_b_params": 3.09 |
| }, |
| "Qwen/Qwen2.5-7B-Instruct": { |
| "architectures": [ |
| "Qwen2ForCausalLM" |
| ], |
| "hidden_size": 3584, |
| "max_position_embeddings": 32768, |
| "model_type": "qwen2", |
| "num_attention_heads": 28, |
| "num_key_value_heads": 4, |
| "num_layers": 28, |
| "torch_dtype": null, |
| "total_b_params": 7.62 |
| }, |
| "Qwen/Qwen3-14B": { |
| "architectures": [ |
| "Qwen3ForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 40960, |
| "model_type": "qwen3", |
| "num_attention_heads": 40, |
| "num_key_value_heads": 8, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 14.0 |
| }, |
| "allenai/OLMo-2-1124-13B-Instruct": { |
| "architectures": [ |
| "Olmo2ForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 4096, |
| "model_type": "olmo2", |
| "num_attention_heads": 40, |
| "num_key_value_heads": 40, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 14.0 |
| }, |
| "allenai/OLMo-2-1124-7B-Instruct": { |
| "architectures": [ |
| "Olmo2ForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 4096, |
| "model_type": "olmo2", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 32, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "berkeley-nest/Starling-LM-7B-alpha": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 8192, |
| "model_type": "mistral", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "deepseek-ai/deepseek-llm-7b-chat": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 4096, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 32, |
| "num_layers": 30, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "google/gemma-2-2b-it": { |
| "architectures": [ |
| "Gemma2ForCausalLM" |
| ], |
| "hidden_size": 2304, |
| "max_position_embeddings": 8192, |
| "model_type": "gemma2", |
| "num_attention_heads": 8, |
| "num_key_value_heads": 4, |
| "num_layers": 26, |
| "torch_dtype": null, |
| "total_b_params": 2.61 |
| }, |
| "google/gemma-3-12b-it": { |
| "architectures": [ |
| "Gemma3ForConditionalGeneration" |
| ], |
| "hidden_size": 3840, |
| "max_position_embeddings": null, |
| "model_type": "gemma3", |
| "num_attention_heads": 16, |
| "num_key_value_heads": 8, |
| "num_layers": 48, |
| "torch_dtype": null, |
| "total_b_params": 12.0 |
| }, |
| "google/gemma-3-27b-it": { |
| "architectures": [ |
| "Gemma3ForConditionalGeneration" |
| ], |
| "hidden_size": 5376, |
| "max_position_embeddings": null, |
| "model_type": "gemma3", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 16, |
| "num_layers": 62, |
| "torch_dtype": null, |
| "total_b_params": 27.0 |
| }, |
| "ibm-granite/granite-3.0-2b-instruct": { |
| "architectures": [ |
| "GraniteForCausalLM" |
| ], |
| "hidden_size": 2048, |
| "max_position_embeddings": 4096, |
| "model_type": "granite", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 2.0 |
| }, |
| "ibm-granite/granite-3.0-8b-instruct": { |
| "architectures": [ |
| "GraniteForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 4096, |
| "model_type": "granite", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 8.0 |
| }, |
| "internlm/internlm2-7b": { |
| "architectures": [ |
| "InternLM2ForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 32768, |
| "model_type": "internlm2", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "internlm/internlm2_5-20b-chat": { |
| "architectures": [ |
| "InternLM2ForCausalLM" |
| ], |
| "hidden_size": 6144, |
| "max_position_embeddings": 32768, |
| "model_type": "internlm2", |
| "num_attention_heads": 48, |
| "num_key_value_heads": 8, |
| "num_layers": 48, |
| "torch_dtype": null, |
| "total_b_params": 20.0 |
| }, |
| "meta-llama/Llama-2-13b-chat-hf": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 4096, |
| "model_type": "llama", |
| "num_attention_heads": 40, |
| "num_key_value_heads": 40, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 13.0 |
| }, |
| "meta-llama/Llama-3.1-8B-Instruct": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 131072, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 8.03 |
| }, |
| "meta-llama/Llama-3.2-3B-Instruct": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 3072, |
| "max_position_embeddings": 131072, |
| "model_type": "llama", |
| "num_attention_heads": 24, |
| "num_key_value_heads": 8, |
| "num_layers": 28, |
| "torch_dtype": null, |
| "total_b_params": 3.21 |
| }, |
| "microsoft/Phi-3-medium-4k-instruct": { |
| "architectures": [ |
| "Phi3ForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 4096, |
| "model_type": "phi3", |
| "num_attention_heads": 40, |
| "num_key_value_heads": 10, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 14.0 |
| }, |
| "microsoft/Phi-3-mini-4k-instruct": { |
| "architectures": [ |
| "Phi3ForCausalLM" |
| ], |
| "hidden_size": 3072, |
| "max_position_embeddings": 4096, |
| "model_type": "phi3", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 32, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 3.8 |
| }, |
| "microsoft/phi-2": { |
| "architectures": [ |
| "PhiForCausalLM" |
| ], |
| "hidden_size": 2560, |
| "max_position_embeddings": 2048, |
| "model_type": "phi", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 32, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 2.7 |
| }, |
| "microsoft/phi-4": { |
| "architectures": [ |
| "Phi3ForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 16384, |
| "model_type": "phi3", |
| "num_attention_heads": 40, |
| "num_key_value_heads": 10, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 14.7 |
| }, |
| "mistralai/Mistral-7B-Instruct-v0.3": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 32768, |
| "model_type": "mistral", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 7.25 |
| }, |
| "mistralai/Mistral-Small-24B-Instruct-2501": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 32768, |
| "model_type": "mistral", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 24.0 |
| }, |
| "mistralai/Mistral-Small-Instruct-2409": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 6144, |
| "max_position_embeddings": 32768, |
| "model_type": "mistral", |
| "num_attention_heads": 48, |
| "num_key_value_heads": 8, |
| "num_layers": 56, |
| "torch_dtype": null, |
| "total_b_params": 24.0 |
| }, |
| "mlabonne/AlphaMonarch-7B": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 32768, |
| "model_type": "mistral", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "mlabonne/NeuralBeagle14-7B": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 32768, |
| "model_type": "mistral", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "nvidia/AceMath-1.5B-Instruct": { |
| "architectures": [ |
| "Qwen2ForCausalLM" |
| ], |
| "hidden_size": 1536, |
| "max_position_embeddings": 4096, |
| "model_type": "qwen2", |
| "num_attention_heads": 12, |
| "num_key_value_heads": 2, |
| "num_layers": 28, |
| "torch_dtype": null, |
| "total_b_params": 1.5 |
| }, |
| "nvidia/AceMath-7B-Instruct": { |
| "architectures": [ |
| "Qwen2ForCausalLM" |
| ], |
| "hidden_size": 3584, |
| "max_position_embeddings": 4096, |
| "model_type": "qwen2", |
| "num_attention_heads": 28, |
| "num_key_value_heads": 4, |
| "num_layers": 28, |
| "torch_dtype": null, |
| "total_b_params": 7.0 |
| }, |
| "nvidia/Llama-3.1-Minitron-4B-Width-Base": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 3072, |
| "max_position_embeddings": 131072, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 5.0 |
| }, |
| "nvidia/Mistral-NeMo-Minitron-8B-Instruct": { |
| "architectures": [ |
| "MistralForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 8192, |
| "model_type": "mistral", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 40, |
| "torch_dtype": null, |
| "total_b_params": 8.0 |
| }, |
| "nvidia/Nemotron-Mini-4B-Instruct": { |
| "architectures": [ |
| "NemotronForCausalLM" |
| ], |
| "hidden_size": 3072, |
| "max_position_embeddings": 4096, |
| "model_type": "nemotron", |
| "num_attention_heads": 24, |
| "num_key_value_heads": 8, |
| "num_layers": 32, |
| "torch_dtype": null, |
| "total_b_params": 4.0 |
| }, |
| "openai-community/gpt2": { |
| "architectures": [ |
| "GPT2LMHeadModel" |
| ], |
| "hidden_size": 768, |
| "max_position_embeddings": null, |
| "model_type": "gpt2", |
| "num_attention_heads": 12, |
| "num_key_value_heads": 12, |
| "num_layers": 12, |
| "torch_dtype": null, |
| "total_b_params": 0.124 |
| }, |
| "tiiuae/Falcon3-3B-Instruct": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 3072, |
| "max_position_embeddings": 32768, |
| "model_type": "llama", |
| "num_attention_heads": 12, |
| "num_key_value_heads": 4, |
| "num_layers": 22, |
| "torch_dtype": null, |
| "total_b_params": 3.0 |
| }, |
| "upstage/SOLAR-10.7B-Instruct-v1.0": { |
| "architectures": [ |
| "LlamaForCausalLM" |
| ], |
| "hidden_size": 4096, |
| "max_position_embeddings": 4096, |
| "model_type": "llama", |
| "num_attention_heads": 32, |
| "num_key_value_heads": 8, |
| "num_layers": 48, |
| "torch_dtype": null, |
| "total_b_params": 10.7 |
| }, |
| "upstage/solar-pro-preview-instruct": { |
| "architectures": [ |
| "SolarForCausalLM" |
| ], |
| "hidden_size": 5120, |
| "max_position_embeddings": 4096, |
| "model_type": "solar", |
| "num_attention_heads": 40, |
| "num_key_value_heads": 10, |
| "num_layers": 64, |
| "torch_dtype": null, |
| "total_b_params": 22.1 |
| } |
| } |