FP8 DFlash checkpoint weights update

#4
Files changed (2) hide show
  1. config.json +13 -7
  2. model.safetensors +1 -1
config.json CHANGED
@@ -4,7 +4,7 @@
4
  "hidden_act": "silu",
5
  "hidden_size": 3072,
6
  "intermediate_size": 12288,
7
- "max_position_embeddings": 1048576,
8
  "model_type": "laguna",
9
  "num_attention_heads": 72,
10
  "num_hidden_layers": 6,
@@ -20,6 +20,18 @@
20
  "sliding_attention",
21
  "sliding_attention"
22
  ],
 
 
 
 
 
 
 
 
 
 
 
 
23
  "sliding_windows": [
24
  512,
25
  512,
@@ -28,12 +40,6 @@
28
  512,
29
  512
30
  ],
31
- "rope_theta": 500000.0,
32
- "gating": "per-head",
33
- "architectures": [
34
- "DFlashLagunaForCausalLM"
35
- ],
36
- "num_experts": 0,
37
  "draft_vocab_size": 100352,
38
  "torch_dtype": "bfloat16",
39
  "eagle_aux_hidden_state_layer_ids": [
 
4
  "hidden_act": "silu",
5
  "hidden_size": 3072,
6
  "intermediate_size": 12288,
7
+ "max_position_embeddings": 262144,
8
  "model_type": "laguna",
9
  "num_attention_heads": 72,
10
  "num_hidden_layers": 6,
 
20
  "sliding_attention",
21
  "sliding_attention"
22
  ],
23
+ "rope_theta": 500000.0,
24
+ "partial_rotary_factor": 0.5,
25
+ "rope_parameters": {
26
+ "rope_type": "yarn",
27
+ "factor": 32.0,
28
+ "original_max_position_embeddings": 8192
29
+ },
30
+ "gating": "per-head",
31
+ "architectures": [
32
+ "DFlashLagunaForCausalLM"
33
+ ],
34
+ "num_experts": 0,
35
  "sliding_windows": [
36
  512,
37
  512,
 
40
  512,
41
  512
42
  ],
 
 
 
 
 
 
43
  "draft_vocab_size": 100352,
44
  "torch_dtype": "bfloat16",
45
  "eagle_aux_hidden_state_layer_ids": [
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3026487dc6a6b22a78f15ca1433eae69e85b77b19629daf135681683de768f6e
3
  size 2229962896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:933dcd552289da3f7489f358964a97160c8175ee109d1b18543fdcd2156651b9
3
  size 2229962896