# Nexus Coder Configuration - Small version v0.3 # ~125M params, fine-tune on 1 GPU # Author: Hieu Louis (2026) model: name: "Nexus Coder Small" agent_name: "Nexus" author: "Hieu Louis" version: "0.3.0-small" github: "mhieuhonda" year: "2026" architecture: vocab_size: 16000 hidden_size: 768 num_hidden_layers: 12 num_attention_heads: 12 num_kv_heads: 4 head_dim: 64 intermediate_size: 2048 hidden_act: "silu" norm_type: "rmsnorm" moe: num_experts: 8 num_active_experts: 2 router_aux_loss_coef: 0.001 context: max_position_embeddings: 8192 rotary_emb_base: 10000.0 attention: use_flash_attention: true use_flash_attention_2: false use_alibi: false use_sliding_window: false use_qk_norm: true mlp_parallel: true compute: use_kv_cache: true kv_cache_quantization: null gradient_checkpointing: false params: total: "~125M" active: "~45M" expert_utilization: "25%" training: learning_rate: 3.0e-4 weight_decay: 0.01 warmup_steps: 50 max_steps: 1000 per_device_batch_size: 8 gradient_accumulation_steps: 2 logging_steps: 10 save_steps: 200 max_grad_norm: 1.0 seed: 42 inference: max_new_tokens: 200 temperature: 0.8 top_k: 50 top_p: 0.9 do_sample: true personality: type: "humorous" language: "bilingual" environment: python_version: "3.12.13" pytorch_version: ">=2.0" cuda_required: false