File size: 1,611 Bytes
eca5751
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
# Nexus Coder Configuration - Tiny version v0.3
# Used for quick verification on CPU (~5M params)
# Author: Hieu Louis (2026)

model:
  name: "Nexus Coder Tiny"
  agent_name: "Nexus"
  author: "Hieu Louis"
  version: "0.3.0-tiny"
  github: "mhieuhonda"
  year: "2026"

architecture:
  vocab_size: 2000
  hidden_size: 256
  num_hidden_layers: 4
  num_attention_heads: 8
  num_kv_heads: 2
  head_dim: 32
  intermediate_size: 512
  hidden_act: "silu"
  norm_type: "rmsnorm"

moe:
  num_experts: 4
  num_active_experts: 2
  router_aux_loss_coef: 0.001
  router_jitter_noise: 0.0

context:
  max_position_embeddings: 512
  rotary_emb_base: 10000.0
  rope_scaling_type: null
  rope_scaling_factor: 1.0

# v0.3 NEW architecture features (most OFF for tiny — too small to benefit)
attention:
  use_flash_attention: false
  use_flash_attention_2: false
  use_alibi: false
  use_sliding_window: false
  sliding_window_size: 256
  use_qk_norm: false
  mlp_parallel: true

compute:
  use_kv_cache: true
  kv_cache_quantization: null
  gradient_checkpointing: false

params:
  total: "~8M (demo only)"
  active: "~5M"
  note: "For testing only. Use nexus_coder_10b.yaml for the real model."

training:
  learning_rate: 5.0e-4
  weight_decay: 0.01
  warmup_steps: 10
  max_steps: 30
  per_device_batch_size: 2
  gradient_accumulation_steps: 1
  logging_steps: 5
  save_steps: 30

inference:
  max_new_tokens: 50
  temperature: 0.8
  top_k: 50
  top_p: 0.9
  do_sample: true

personality:
  type: "humorous"
  language: "bilingual"

environment:
  python_version: "3.12.13"
  pytorch_version: ">=2.0"
  cuda_required: false