Research / config.json
jpllm's picture
Adiciona arquitetura Hadamard-GS, config.json, checkpoints (10M a 90M) e documentação técnica
ecf9a39 verified
Raw History Blame Contribute Delete
1.31 kB
{
"architectures": [
"SupraMiniWithGate"
],
"model_type": "supra_mini_hadamard_gs",
"vocab_size": 4096,
"d_model": 128,
"num_layers": 6,
"num_heads": 4,
"d_head": 16,
"max_len": 256,
"inner_dim": 64,
"intermediate_dim": 64,
"hadamard_dim": 128,
"total_parameters": 812800,
"tied_embeddings": true,
"dynamic_gram_schmidt": {
"enabled": true,
"gate_projection": "Dense(128, bias=True) -> Sigmoid",
"formula": "h_orth = h_final - (sigmoid(W_gate @ h_final + b) * proj_{x_in}(h_final))"
},
"sublayer_diffusion": {
"projection": "Pad zeros (64 -> 128)",
"permutation_seed": 2026,
"isometry": "Sylvester Hadamard 128 (normalized 1/sqrt(128))"
},
"base_tokenizer": "SupraLabs/Supra-Mini-v6-1M",
"training_history": {
"stage_1_tinystories": {
"tokens": 10000000,
"final_loss": 2.7381,
"optimizer": "AdamW 1e-3 -> 5e-4"
},
"stage_2_general_mix": {
"tokens": 20000000,
"final_loss": 4.7582,
"optimizer": "AdamW 1e-3 -> 1e-4"
},
"stage_3_annealing": {
"tokens": 50000000,
"final_loss": 4.25,
"optimizer": "AdamW 5e-4 -> 5e-5"
},
"stage_4_self_distill": {
"tokens": 20000000,
"loss_type": "0.65 One-Hot + 0.35 Top-5 Soft-CE",
"lr": 0.0001
}
}
}