NexusCoder / configs /nexus_coder_tiny.yaml
AdminReal's picture
Import NexusCoder from github.com/mhieuhonda/NexusCoder
eca5751 verified
Raw
History Blame Contribute Delete
1.61 kB
# Nexus Coder Configuration - Tiny version v0.3
# Used for quick verification on CPU (~5M params)
# Author: Hieu Louis (2026)
model:
name: "Nexus Coder Tiny"
agent_name: "Nexus"
author: "Hieu Louis"
version: "0.3.0-tiny"
github: "mhieuhonda"
year: "2026"
architecture:
vocab_size: 2000
hidden_size: 256
num_hidden_layers: 4
num_attention_heads: 8
num_kv_heads: 2
head_dim: 32
intermediate_size: 512
hidden_act: "silu"
norm_type: "rmsnorm"
moe:
num_experts: 4
num_active_experts: 2
router_aux_loss_coef: 0.001
router_jitter_noise: 0.0
context:
max_position_embeddings: 512
rotary_emb_base: 10000.0
rope_scaling_type: null
rope_scaling_factor: 1.0
# v0.3 NEW architecture features (most OFF for tiny — too small to benefit)
attention:
use_flash_attention: false
use_flash_attention_2: false
use_alibi: false
use_sliding_window: false
sliding_window_size: 256
use_qk_norm: false
mlp_parallel: true
compute:
use_kv_cache: true
kv_cache_quantization: null
gradient_checkpointing: false
params:
total: "~8M (demo only)"
active: "~5M"
note: "For testing only. Use nexus_coder_10b.yaml for the real model."
training:
learning_rate: 5.0e-4
weight_decay: 0.01
warmup_steps: 10
max_steps: 30
per_device_batch_size: 2
gradient_accumulation_steps: 1
logging_steps: 5
save_steps: 30
inference:
max_new_tokens: 50
temperature: 0.8
top_k: 50
top_p: 0.9
do_sample: true
personality:
type: "humorous"
language: "bilingual"
environment:
python_version: "3.12.13"
pytorch_version: ">=2.0"
cuda_required: false