{ "model_type": "alpha-er", "architectures": [ "AlphaErForCausalLM" ], "auto_map": { "AutoModelForCausalLM": "modeling_alpha.AlphaErForCausalLM" }, "vocab_size": 12288, "block_size": 512, "n_layer": 2, "n_embd": 1024, "n_head": 8, "ffn_dim": 20480, "mlp_experts": 64, "lm_head_rank": 128, "attn_rank": 128, "attn_soft_cap": 30.0, "layer_norm_eps": 1e-05, "torch_dtype": "float32", "trained_step": 20000 }