w-ahmad commited on
Commit
d1fd160
·
verified ·
1 Parent(s): 09a5bdd

Auto upload zain 2026-08-12T21:24:09.274426

Browse files
Files changed (43) hide show
  1. zain/Activation/out/glu-situglu_low-150L_run/training_log.jsonl +2 -0
  2. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/config.json +36 -0
  3. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/model.safetensors +3 -0
  4. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/optimizer.pt +3 -0
  5. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/rng_state.pth +3 -0
  6. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/scheduler.pt +3 -0
  7. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/tokenizer.json +0 -0
  8. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/tokenizer_config.json +13 -0
  9. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/trainer_state.json +136 -0
  10. zain/Activation/out/glu-waleed-150L_run/checkpoint-200/training_args.bin +3 -0
  11. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/config.json +36 -0
  12. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/model.safetensors +3 -0
  13. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/optimizer.pt +3 -0
  14. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/rng_state.pth +3 -0
  15. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/scheduler.pt +3 -0
  16. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/tokenizer.json +0 -0
  17. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/tokenizer_config.json +13 -0
  18. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/trainer_state.json +187 -0
  19. zain/Activation/out/glu-waleed-150L_run/checkpoint-300/training_args.bin +3 -0
  20. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/config.json +36 -0
  21. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/model.safetensors +3 -0
  22. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/optimizer.pt +3 -0
  23. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/rng_state.pth +3 -0
  24. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/scheduler.pt +3 -0
  25. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/tokenizer.json +0 -0
  26. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/tokenizer_config.json +13 -0
  27. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/trainer_state.json +238 -0
  28. zain/Activation/out/glu-waleed-150L_run/checkpoint-400/training_args.bin +3 -0
  29. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/config.json +36 -0
  30. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/model.safetensors +3 -0
  31. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/optimizer.pt +3 -0
  32. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/rng_state.pth +3 -0
  33. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/scheduler.pt +3 -0
  34. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/tokenizer.json +0 -0
  35. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/tokenizer_config.json +13 -0
  36. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/trainer_state.json +289 -0
  37. zain/Activation/out/glu-waleed-150L_run/checkpoint-500/training_args.bin +3 -0
  38. zain/Activation/out/glu-waleed-150L_run/config.json +36 -0
  39. zain/Activation/out/glu-waleed-150L_run/model.safetensors +3 -0
  40. zain/Activation/out/glu-waleed-150L_run/tokenizer.json +0 -0
  41. zain/Activation/out/glu-waleed-150L_run/tokenizer_config.json +13 -0
  42. zain/Activation/out/glu-waleed-150L_run/training_args.bin +3 -0
  43. zain/Activation/out/glu-waleed-150L_run/training_log.jsonl +24 -0
zain/Activation/out/glu-situglu_low-150L_run/training_log.jsonl ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ {"step": 20, "epoch": 0.0006740815638692282, "timestamp": 1786569824.8138995, "loss": 8.23435287475586, "grad_norm": 1.4453125, "learning_rate": 6.65e-05, "train/total_time_seconds": 7.173142466694117, "train/time_per_step_avg": 0.35865712333470584, "train/epoch_time_elapsed": 7.740738794207573, "train/estimated_remaining_minutes": 2.8692569866776467}
2
+ {"step": 40, "epoch": 0.0013481631277384564, "timestamp": 1786569832.2752674, "loss": 7.900041961669922, "grad_norm": 1.375, "learning_rate": 0.0001365, "train/total_time_seconds": 14.130188584327698, "train/time_per_step_avg": 0.35325471460819247, "train/epoch_time_elapsed": 15.202106323093176, "train/estimated_remaining_minutes": 2.7082861453294753}
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "waleed",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "glu",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 150,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:498fd09549d49fd93deb0f300fc27f403b2b9a41bcafed1b8ae7404b33ee4371
3
+ size 50427120
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:61f681059f6321e75996f6c19bf7e545c579c01845296d8cc718d9f57c604dd9
3
+ size 101708778
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
3
+ size 14244
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:65adfad786ae2b224643009c3ac2998e8fc0229311e4b753becffd2fb9569fe9
3
+ size 1064
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/trainer_state.json ADDED
@@ -0,0 +1,136 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.006740815638692282,
6
+ "eval_steps": 50,
7
+ "global_step": 200,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0006740815638692282,
14
+ "grad_norm": 1.3203125,
15
+ "learning_rate": 6.65e-05,
16
+ "loss": 8.226606750488282,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.3125,
22
+ "learning_rate": 0.0001365,
23
+ "loss": 7.892790985107422,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.0016852039096730705,
28
+ "eval_loss": 7.459151268005371,
29
+ "eval_runtime": 20.6744,
30
+ "eval_samples_per_second": 460.811,
31
+ "eval_steps_per_second": 0.484,
32
+ "step": 50
33
+ },
34
+ {
35
+ "epoch": 0.0020222446916076846,
36
+ "grad_norm": 1.234375,
37
+ "learning_rate": 0.00020649999999999998,
38
+ "loss": 7.4655517578125,
39
+ "step": 60
40
+ },
41
+ {
42
+ "epoch": 0.002696326255476913,
43
+ "grad_norm": 1.1171875,
44
+ "learning_rate": 0.0002765,
45
+ "loss": 6.900899505615234,
46
+ "step": 80
47
+ },
48
+ {
49
+ "epoch": 0.003370407819346141,
50
+ "grad_norm": 0.8046875,
51
+ "learning_rate": 0.0003465,
52
+ "loss": 6.384212875366211,
53
+ "step": 100
54
+ },
55
+ {
56
+ "epoch": 0.003370407819346141,
57
+ "eval_loss": 6.164856433868408,
58
+ "eval_runtime": 20.6327,
59
+ "eval_samples_per_second": 461.743,
60
+ "eval_steps_per_second": 0.485,
61
+ "step": 100
62
+ },
63
+ {
64
+ "epoch": 0.004044489383215369,
65
+ "grad_norm": 0.5234375,
66
+ "learning_rate": 0.0004165,
67
+ "loss": 6.051737594604492,
68
+ "step": 120
69
+ },
70
+ {
71
+ "epoch": 0.0047185709470845974,
72
+ "grad_norm": 0.4609375,
73
+ "learning_rate": 0.00048649999999999995,
74
+ "loss": 5.780973052978515,
75
+ "step": 140
76
+ },
77
+ {
78
+ "epoch": 0.005055611729019211,
79
+ "eval_loss": 5.543824195861816,
80
+ "eval_runtime": 20.6322,
81
+ "eval_samples_per_second": 461.755,
82
+ "eval_steps_per_second": 0.485,
83
+ "step": 150
84
+ },
85
+ {
86
+ "epoch": 0.005392652510953826,
87
+ "grad_norm": 0.98046875,
88
+ "learning_rate": 0.0005565,
89
+ "loss": 5.545684051513672,
90
+ "step": 160
91
+ },
92
+ {
93
+ "epoch": 0.006066734074823054,
94
+ "grad_norm": 0.62109375,
95
+ "learning_rate": 0.0006265,
96
+ "loss": 5.289410018920899,
97
+ "step": 180
98
+ },
99
+ {
100
+ "epoch": 0.006740815638692282,
101
+ "grad_norm": 0.5390625,
102
+ "learning_rate": 0.0006965,
103
+ "loss": 4.994114685058594,
104
+ "step": 200
105
+ },
106
+ {
107
+ "epoch": 0.006740815638692282,
108
+ "eval_loss": 4.839845180511475,
109
+ "eval_runtime": 20.6673,
110
+ "eval_samples_per_second": 460.971,
111
+ "eval_steps_per_second": 0.484,
112
+ "step": 200
113
+ }
114
+ ],
115
+ "logging_steps": 20,
116
+ "max_steps": 500,
117
+ "num_input_tokens_seen": 0,
118
+ "num_train_epochs": 1,
119
+ "save_steps": 100,
120
+ "stateful_callbacks": {
121
+ "TrainerControl": {
122
+ "args": {
123
+ "should_epoch_stop": false,
124
+ "should_evaluate": false,
125
+ "should_log": false,
126
+ "should_save": true,
127
+ "should_training_stop": false
128
+ },
129
+ "attributes": {}
130
+ }
131
+ },
132
+ "total_flos": 483941312102400.0,
133
+ "train_batch_size": 32,
134
+ "trial_name": null,
135
+ "trial_params": null
136
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-200/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b827f60f983e3f5ad52047a134946ee6cd902cd1de8cdbd39e8e9aa340a1aef1
3
+ size 4920
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "waleed",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "glu",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 150,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b6ed98e05fae7de919d1730d84d7b1acfa96565be7a3dd2368178203c47718d5
3
+ size 50427120
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:600ab3f5dfd1a0ec0bdf1c9e4dac80193572d54ead651deb0c54f9b4ae16e591
3
+ size 101708778
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:068fbd993087219c15b8c0baa13fc39644a4dcdfe92d8be3fa6434deece90371
3
+ size 14244
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7275519db1cfbc1c3b3ff4cc789ae435d90674550d32700685019af96bdd8863
3
+ size 1064
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/trainer_state.json ADDED
@@ -0,0 +1,187 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.010111223458038422,
6
+ "eval_steps": 50,
7
+ "global_step": 300,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0006740815638692282,
14
+ "grad_norm": 1.3203125,
15
+ "learning_rate": 6.65e-05,
16
+ "loss": 8.226606750488282,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.3125,
22
+ "learning_rate": 0.0001365,
23
+ "loss": 7.892790985107422,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.0016852039096730705,
28
+ "eval_loss": 7.459151268005371,
29
+ "eval_runtime": 20.6744,
30
+ "eval_samples_per_second": 460.811,
31
+ "eval_steps_per_second": 0.484,
32
+ "step": 50
33
+ },
34
+ {
35
+ "epoch": 0.0020222446916076846,
36
+ "grad_norm": 1.234375,
37
+ "learning_rate": 0.00020649999999999998,
38
+ "loss": 7.4655517578125,
39
+ "step": 60
40
+ },
41
+ {
42
+ "epoch": 0.002696326255476913,
43
+ "grad_norm": 1.1171875,
44
+ "learning_rate": 0.0002765,
45
+ "loss": 6.900899505615234,
46
+ "step": 80
47
+ },
48
+ {
49
+ "epoch": 0.003370407819346141,
50
+ "grad_norm": 0.8046875,
51
+ "learning_rate": 0.0003465,
52
+ "loss": 6.384212875366211,
53
+ "step": 100
54
+ },
55
+ {
56
+ "epoch": 0.003370407819346141,
57
+ "eval_loss": 6.164856433868408,
58
+ "eval_runtime": 20.6327,
59
+ "eval_samples_per_second": 461.743,
60
+ "eval_steps_per_second": 0.485,
61
+ "step": 100
62
+ },
63
+ {
64
+ "epoch": 0.004044489383215369,
65
+ "grad_norm": 0.5234375,
66
+ "learning_rate": 0.0004165,
67
+ "loss": 6.051737594604492,
68
+ "step": 120
69
+ },
70
+ {
71
+ "epoch": 0.0047185709470845974,
72
+ "grad_norm": 0.4609375,
73
+ "learning_rate": 0.00048649999999999995,
74
+ "loss": 5.780973052978515,
75
+ "step": 140
76
+ },
77
+ {
78
+ "epoch": 0.005055611729019211,
79
+ "eval_loss": 5.543824195861816,
80
+ "eval_runtime": 20.6322,
81
+ "eval_samples_per_second": 461.755,
82
+ "eval_steps_per_second": 0.485,
83
+ "step": 150
84
+ },
85
+ {
86
+ "epoch": 0.005392652510953826,
87
+ "grad_norm": 0.98046875,
88
+ "learning_rate": 0.0005565,
89
+ "loss": 5.545684051513672,
90
+ "step": 160
91
+ },
92
+ {
93
+ "epoch": 0.006066734074823054,
94
+ "grad_norm": 0.62109375,
95
+ "learning_rate": 0.0006265,
96
+ "loss": 5.289410018920899,
97
+ "step": 180
98
+ },
99
+ {
100
+ "epoch": 0.006740815638692282,
101
+ "grad_norm": 0.5390625,
102
+ "learning_rate": 0.0006965,
103
+ "loss": 4.994114685058594,
104
+ "step": 200
105
+ },
106
+ {
107
+ "epoch": 0.006740815638692282,
108
+ "eval_loss": 4.839845180511475,
109
+ "eval_runtime": 20.6673,
110
+ "eval_samples_per_second": 460.971,
111
+ "eval_steps_per_second": 0.484,
112
+ "step": 200
113
+ },
114
+ {
115
+ "epoch": 0.00741489720256151,
116
+ "grad_norm": 1.0703125,
117
+ "learning_rate": 0.0007,
118
+ "loss": 4.748640823364258,
119
+ "step": 220
120
+ },
121
+ {
122
+ "epoch": 0.008088978766430738,
123
+ "grad_norm": 0.6796875,
124
+ "learning_rate": 0.0007,
125
+ "loss": 4.502994537353516,
126
+ "step": 240
127
+ },
128
+ {
129
+ "epoch": 0.008426019548365353,
130
+ "eval_loss": 4.319539546966553,
131
+ "eval_runtime": 20.6786,
132
+ "eval_samples_per_second": 460.717,
133
+ "eval_steps_per_second": 0.484,
134
+ "step": 250
135
+ },
136
+ {
137
+ "epoch": 0.008763060330299966,
138
+ "grad_norm": 0.455078125,
139
+ "learning_rate": 0.0007,
140
+ "loss": 4.330507278442383,
141
+ "step": 260
142
+ },
143
+ {
144
+ "epoch": 0.009437141894169195,
145
+ "grad_norm": 0.59765625,
146
+ "learning_rate": 0.0007,
147
+ "loss": 4.185513687133789,
148
+ "step": 280
149
+ },
150
+ {
151
+ "epoch": 0.010111223458038422,
152
+ "grad_norm": 0.62890625,
153
+ "learning_rate": 0.0007,
154
+ "loss": 4.058372116088867,
155
+ "step": 300
156
+ },
157
+ {
158
+ "epoch": 0.010111223458038422,
159
+ "eval_loss": 4.001541614532471,
160
+ "eval_runtime": 20.746,
161
+ "eval_samples_per_second": 459.22,
162
+ "eval_steps_per_second": 0.482,
163
+ "step": 300
164
+ }
165
+ ],
166
+ "logging_steps": 20,
167
+ "max_steps": 500,
168
+ "num_input_tokens_seen": 0,
169
+ "num_train_epochs": 1,
170
+ "save_steps": 100,
171
+ "stateful_callbacks": {
172
+ "TrainerControl": {
173
+ "args": {
174
+ "should_epoch_stop": false,
175
+ "should_evaluate": false,
176
+ "should_log": false,
177
+ "should_save": true,
178
+ "should_training_stop": false
179
+ },
180
+ "attributes": {}
181
+ }
182
+ },
183
+ "total_flos": 725911968153600.0,
184
+ "train_batch_size": 32,
185
+ "trial_name": null,
186
+ "trial_params": null
187
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-300/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b827f60f983e3f5ad52047a134946ee6cd902cd1de8cdbd39e8e9aa340a1aef1
3
+ size 4920
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "waleed",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "glu",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 150,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:acf1ce7f138d292235a0cd10eafffd59a9aa9ca60d99858356f0394236830d2b
3
+ size 50427120
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1a7fe06aeac9ea87d94f255285ddbece675611441794defb92668c91d36e1c5d
3
+ size 101708778
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f9a6944c405a002fce05f295d08ea6650e2e2ad6dbf5d6da1e9053f7bf7f5827
3
+ size 14244
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:59985ace3cc91dc071d22964a9d7ec324bcb84a61c5666f3264564df2f5797e4
3
+ size 1064
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/trainer_state.json ADDED
@@ -0,0 +1,238 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.013481631277384564,
6
+ "eval_steps": 50,
7
+ "global_step": 400,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0006740815638692282,
14
+ "grad_norm": 1.3203125,
15
+ "learning_rate": 6.65e-05,
16
+ "loss": 8.226606750488282,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.3125,
22
+ "learning_rate": 0.0001365,
23
+ "loss": 7.892790985107422,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.0016852039096730705,
28
+ "eval_loss": 7.459151268005371,
29
+ "eval_runtime": 20.6744,
30
+ "eval_samples_per_second": 460.811,
31
+ "eval_steps_per_second": 0.484,
32
+ "step": 50
33
+ },
34
+ {
35
+ "epoch": 0.0020222446916076846,
36
+ "grad_norm": 1.234375,
37
+ "learning_rate": 0.00020649999999999998,
38
+ "loss": 7.4655517578125,
39
+ "step": 60
40
+ },
41
+ {
42
+ "epoch": 0.002696326255476913,
43
+ "grad_norm": 1.1171875,
44
+ "learning_rate": 0.0002765,
45
+ "loss": 6.900899505615234,
46
+ "step": 80
47
+ },
48
+ {
49
+ "epoch": 0.003370407819346141,
50
+ "grad_norm": 0.8046875,
51
+ "learning_rate": 0.0003465,
52
+ "loss": 6.384212875366211,
53
+ "step": 100
54
+ },
55
+ {
56
+ "epoch": 0.003370407819346141,
57
+ "eval_loss": 6.164856433868408,
58
+ "eval_runtime": 20.6327,
59
+ "eval_samples_per_second": 461.743,
60
+ "eval_steps_per_second": 0.485,
61
+ "step": 100
62
+ },
63
+ {
64
+ "epoch": 0.004044489383215369,
65
+ "grad_norm": 0.5234375,
66
+ "learning_rate": 0.0004165,
67
+ "loss": 6.051737594604492,
68
+ "step": 120
69
+ },
70
+ {
71
+ "epoch": 0.0047185709470845974,
72
+ "grad_norm": 0.4609375,
73
+ "learning_rate": 0.00048649999999999995,
74
+ "loss": 5.780973052978515,
75
+ "step": 140
76
+ },
77
+ {
78
+ "epoch": 0.005055611729019211,
79
+ "eval_loss": 5.543824195861816,
80
+ "eval_runtime": 20.6322,
81
+ "eval_samples_per_second": 461.755,
82
+ "eval_steps_per_second": 0.485,
83
+ "step": 150
84
+ },
85
+ {
86
+ "epoch": 0.005392652510953826,
87
+ "grad_norm": 0.98046875,
88
+ "learning_rate": 0.0005565,
89
+ "loss": 5.545684051513672,
90
+ "step": 160
91
+ },
92
+ {
93
+ "epoch": 0.006066734074823054,
94
+ "grad_norm": 0.62109375,
95
+ "learning_rate": 0.0006265,
96
+ "loss": 5.289410018920899,
97
+ "step": 180
98
+ },
99
+ {
100
+ "epoch": 0.006740815638692282,
101
+ "grad_norm": 0.5390625,
102
+ "learning_rate": 0.0006965,
103
+ "loss": 4.994114685058594,
104
+ "step": 200
105
+ },
106
+ {
107
+ "epoch": 0.006740815638692282,
108
+ "eval_loss": 4.839845180511475,
109
+ "eval_runtime": 20.6673,
110
+ "eval_samples_per_second": 460.971,
111
+ "eval_steps_per_second": 0.484,
112
+ "step": 200
113
+ },
114
+ {
115
+ "epoch": 0.00741489720256151,
116
+ "grad_norm": 1.0703125,
117
+ "learning_rate": 0.0007,
118
+ "loss": 4.748640823364258,
119
+ "step": 220
120
+ },
121
+ {
122
+ "epoch": 0.008088978766430738,
123
+ "grad_norm": 0.6796875,
124
+ "learning_rate": 0.0007,
125
+ "loss": 4.502994537353516,
126
+ "step": 240
127
+ },
128
+ {
129
+ "epoch": 0.008426019548365353,
130
+ "eval_loss": 4.319539546966553,
131
+ "eval_runtime": 20.6786,
132
+ "eval_samples_per_second": 460.717,
133
+ "eval_steps_per_second": 0.484,
134
+ "step": 250
135
+ },
136
+ {
137
+ "epoch": 0.008763060330299966,
138
+ "grad_norm": 0.455078125,
139
+ "learning_rate": 0.0007,
140
+ "loss": 4.330507278442383,
141
+ "step": 260
142
+ },
143
+ {
144
+ "epoch": 0.009437141894169195,
145
+ "grad_norm": 0.59765625,
146
+ "learning_rate": 0.0007,
147
+ "loss": 4.185513687133789,
148
+ "step": 280
149
+ },
150
+ {
151
+ "epoch": 0.010111223458038422,
152
+ "grad_norm": 0.62890625,
153
+ "learning_rate": 0.0007,
154
+ "loss": 4.058372116088867,
155
+ "step": 300
156
+ },
157
+ {
158
+ "epoch": 0.010111223458038422,
159
+ "eval_loss": 4.001541614532471,
160
+ "eval_runtime": 20.746,
161
+ "eval_samples_per_second": 459.22,
162
+ "eval_steps_per_second": 0.482,
163
+ "step": 300
164
+ },
165
+ {
166
+ "epoch": 0.010785305021907651,
167
+ "grad_norm": 0.54296875,
168
+ "learning_rate": 0.0007,
169
+ "loss": 3.9491451263427733,
170
+ "step": 320
171
+ },
172
+ {
173
+ "epoch": 0.011459386585776879,
174
+ "grad_norm": 0.69921875,
175
+ "learning_rate": 0.0007,
176
+ "loss": 3.8855216979980467,
177
+ "step": 340
178
+ },
179
+ {
180
+ "epoch": 0.011796427367711493,
181
+ "eval_loss": 3.792919874191284,
182
+ "eval_runtime": 20.7384,
183
+ "eval_samples_per_second": 459.39,
184
+ "eval_steps_per_second": 0.482,
185
+ "step": 350
186
+ },
187
+ {
188
+ "epoch": 0.012133468149646108,
189
+ "grad_norm": 0.53515625,
190
+ "learning_rate": 0.0007,
191
+ "loss": 3.78214111328125,
192
+ "step": 360
193
+ },
194
+ {
195
+ "epoch": 0.012807549713515335,
196
+ "grad_norm": 0.478515625,
197
+ "learning_rate": 0.0007,
198
+ "loss": 3.7168258666992187,
199
+ "step": 380
200
+ },
201
+ {
202
+ "epoch": 0.013481631277384564,
203
+ "grad_norm": 0.48046875,
204
+ "learning_rate": 0.0007,
205
+ "loss": 3.660007858276367,
206
+ "step": 400
207
+ },
208
+ {
209
+ "epoch": 0.013481631277384564,
210
+ "eval_loss": 3.6307969093322754,
211
+ "eval_runtime": 20.7141,
212
+ "eval_samples_per_second": 459.927,
213
+ "eval_steps_per_second": 0.483,
214
+ "step": 400
215
+ }
216
+ ],
217
+ "logging_steps": 20,
218
+ "max_steps": 500,
219
+ "num_input_tokens_seen": 0,
220
+ "num_train_epochs": 1,
221
+ "save_steps": 100,
222
+ "stateful_callbacks": {
223
+ "TrainerControl": {
224
+ "args": {
225
+ "should_epoch_stop": false,
226
+ "should_evaluate": false,
227
+ "should_log": false,
228
+ "should_save": true,
229
+ "should_training_stop": false
230
+ },
231
+ "attributes": {}
232
+ }
233
+ },
234
+ "total_flos": 967882624204800.0,
235
+ "train_batch_size": 32,
236
+ "trial_name": null,
237
+ "trial_params": null
238
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-400/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b827f60f983e3f5ad52047a134946ee6cd902cd1de8cdbd39e8e9aa340a1aef1
3
+ size 4920
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "waleed",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "glu",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 150,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee8cdea3b2881be1f0bf35b491fcbc952c771b9e744b1d72e647cde565cb331a
3
+ size 50427120
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a16e55a641d7ec7ed033f9c9abce7806443b9c4e9a5c6a80989dec3c2bfa07ee
3
+ size 101708778
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b1a97db8e41139aa1239ba7fb79ddeb0af5998c6305a440c1fe182e6ad02f2f5
3
+ size 14244
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ad40e275c0c46c262ecd103385af8fb68cf3f382102b4f63937b0dbd1c69cfc
3
+ size 1064
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/trainer_state.json ADDED
@@ -0,0 +1,289 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.016852039096730706,
6
+ "eval_steps": 50,
7
+ "global_step": 500,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0006740815638692282,
14
+ "grad_norm": 1.3203125,
15
+ "learning_rate": 6.65e-05,
16
+ "loss": 8.226606750488282,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.3125,
22
+ "learning_rate": 0.0001365,
23
+ "loss": 7.892790985107422,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.0016852039096730705,
28
+ "eval_loss": 7.459151268005371,
29
+ "eval_runtime": 20.6744,
30
+ "eval_samples_per_second": 460.811,
31
+ "eval_steps_per_second": 0.484,
32
+ "step": 50
33
+ },
34
+ {
35
+ "epoch": 0.0020222446916076846,
36
+ "grad_norm": 1.234375,
37
+ "learning_rate": 0.00020649999999999998,
38
+ "loss": 7.4655517578125,
39
+ "step": 60
40
+ },
41
+ {
42
+ "epoch": 0.002696326255476913,
43
+ "grad_norm": 1.1171875,
44
+ "learning_rate": 0.0002765,
45
+ "loss": 6.900899505615234,
46
+ "step": 80
47
+ },
48
+ {
49
+ "epoch": 0.003370407819346141,
50
+ "grad_norm": 0.8046875,
51
+ "learning_rate": 0.0003465,
52
+ "loss": 6.384212875366211,
53
+ "step": 100
54
+ },
55
+ {
56
+ "epoch": 0.003370407819346141,
57
+ "eval_loss": 6.164856433868408,
58
+ "eval_runtime": 20.6327,
59
+ "eval_samples_per_second": 461.743,
60
+ "eval_steps_per_second": 0.485,
61
+ "step": 100
62
+ },
63
+ {
64
+ "epoch": 0.004044489383215369,
65
+ "grad_norm": 0.5234375,
66
+ "learning_rate": 0.0004165,
67
+ "loss": 6.051737594604492,
68
+ "step": 120
69
+ },
70
+ {
71
+ "epoch": 0.0047185709470845974,
72
+ "grad_norm": 0.4609375,
73
+ "learning_rate": 0.00048649999999999995,
74
+ "loss": 5.780973052978515,
75
+ "step": 140
76
+ },
77
+ {
78
+ "epoch": 0.005055611729019211,
79
+ "eval_loss": 5.543824195861816,
80
+ "eval_runtime": 20.6322,
81
+ "eval_samples_per_second": 461.755,
82
+ "eval_steps_per_second": 0.485,
83
+ "step": 150
84
+ },
85
+ {
86
+ "epoch": 0.005392652510953826,
87
+ "grad_norm": 0.98046875,
88
+ "learning_rate": 0.0005565,
89
+ "loss": 5.545684051513672,
90
+ "step": 160
91
+ },
92
+ {
93
+ "epoch": 0.006066734074823054,
94
+ "grad_norm": 0.62109375,
95
+ "learning_rate": 0.0006265,
96
+ "loss": 5.289410018920899,
97
+ "step": 180
98
+ },
99
+ {
100
+ "epoch": 0.006740815638692282,
101
+ "grad_norm": 0.5390625,
102
+ "learning_rate": 0.0006965,
103
+ "loss": 4.994114685058594,
104
+ "step": 200
105
+ },
106
+ {
107
+ "epoch": 0.006740815638692282,
108
+ "eval_loss": 4.839845180511475,
109
+ "eval_runtime": 20.6673,
110
+ "eval_samples_per_second": 460.971,
111
+ "eval_steps_per_second": 0.484,
112
+ "step": 200
113
+ },
114
+ {
115
+ "epoch": 0.00741489720256151,
116
+ "grad_norm": 1.0703125,
117
+ "learning_rate": 0.0007,
118
+ "loss": 4.748640823364258,
119
+ "step": 220
120
+ },
121
+ {
122
+ "epoch": 0.008088978766430738,
123
+ "grad_norm": 0.6796875,
124
+ "learning_rate": 0.0007,
125
+ "loss": 4.502994537353516,
126
+ "step": 240
127
+ },
128
+ {
129
+ "epoch": 0.008426019548365353,
130
+ "eval_loss": 4.319539546966553,
131
+ "eval_runtime": 20.6786,
132
+ "eval_samples_per_second": 460.717,
133
+ "eval_steps_per_second": 0.484,
134
+ "step": 250
135
+ },
136
+ {
137
+ "epoch": 0.008763060330299966,
138
+ "grad_norm": 0.455078125,
139
+ "learning_rate": 0.0007,
140
+ "loss": 4.330507278442383,
141
+ "step": 260
142
+ },
143
+ {
144
+ "epoch": 0.009437141894169195,
145
+ "grad_norm": 0.59765625,
146
+ "learning_rate": 0.0007,
147
+ "loss": 4.185513687133789,
148
+ "step": 280
149
+ },
150
+ {
151
+ "epoch": 0.010111223458038422,
152
+ "grad_norm": 0.62890625,
153
+ "learning_rate": 0.0007,
154
+ "loss": 4.058372116088867,
155
+ "step": 300
156
+ },
157
+ {
158
+ "epoch": 0.010111223458038422,
159
+ "eval_loss": 4.001541614532471,
160
+ "eval_runtime": 20.746,
161
+ "eval_samples_per_second": 459.22,
162
+ "eval_steps_per_second": 0.482,
163
+ "step": 300
164
+ },
165
+ {
166
+ "epoch": 0.010785305021907651,
167
+ "grad_norm": 0.54296875,
168
+ "learning_rate": 0.0007,
169
+ "loss": 3.9491451263427733,
170
+ "step": 320
171
+ },
172
+ {
173
+ "epoch": 0.011459386585776879,
174
+ "grad_norm": 0.69921875,
175
+ "learning_rate": 0.0007,
176
+ "loss": 3.8855216979980467,
177
+ "step": 340
178
+ },
179
+ {
180
+ "epoch": 0.011796427367711493,
181
+ "eval_loss": 3.792919874191284,
182
+ "eval_runtime": 20.7384,
183
+ "eval_samples_per_second": 459.39,
184
+ "eval_steps_per_second": 0.482,
185
+ "step": 350
186
+ },
187
+ {
188
+ "epoch": 0.012133468149646108,
189
+ "grad_norm": 0.53515625,
190
+ "learning_rate": 0.0007,
191
+ "loss": 3.78214111328125,
192
+ "step": 360
193
+ },
194
+ {
195
+ "epoch": 0.012807549713515335,
196
+ "grad_norm": 0.478515625,
197
+ "learning_rate": 0.0007,
198
+ "loss": 3.7168258666992187,
199
+ "step": 380
200
+ },
201
+ {
202
+ "epoch": 0.013481631277384564,
203
+ "grad_norm": 0.48046875,
204
+ "learning_rate": 0.0007,
205
+ "loss": 3.660007858276367,
206
+ "step": 400
207
+ },
208
+ {
209
+ "epoch": 0.013481631277384564,
210
+ "eval_loss": 3.6307969093322754,
211
+ "eval_runtime": 20.7141,
212
+ "eval_samples_per_second": 459.927,
213
+ "eval_steps_per_second": 0.483,
214
+ "step": 400
215
+ },
216
+ {
217
+ "epoch": 0.014155712841253791,
218
+ "grad_norm": 0.55078125,
219
+ "learning_rate": 0.0007,
220
+ "loss": 3.5877830505371096,
221
+ "step": 420
222
+ },
223
+ {
224
+ "epoch": 0.01482979440512302,
225
+ "grad_norm": 0.51953125,
226
+ "learning_rate": 0.0007,
227
+ "loss": 3.560054397583008,
228
+ "step": 440
229
+ },
230
+ {
231
+ "epoch": 0.015166835187057633,
232
+ "eval_loss": 3.5101380348205566,
233
+ "eval_runtime": 20.5828,
234
+ "eval_samples_per_second": 462.863,
235
+ "eval_steps_per_second": 0.486,
236
+ "step": 450
237
+ },
238
+ {
239
+ "epoch": 0.015503875968992248,
240
+ "grad_norm": 0.52734375,
241
+ "learning_rate": 0.0007,
242
+ "loss": 3.5127296447753906,
243
+ "step": 460
244
+ },
245
+ {
246
+ "epoch": 0.016177957532861477,
247
+ "grad_norm": 0.43359375,
248
+ "learning_rate": 0.0007,
249
+ "loss": 3.4453048706054688,
250
+ "step": 480
251
+ },
252
+ {
253
+ "epoch": 0.016852039096730706,
254
+ "grad_norm": 0.431640625,
255
+ "learning_rate": 0.0007,
256
+ "loss": 3.415237045288086,
257
+ "step": 500
258
+ },
259
+ {
260
+ "epoch": 0.016852039096730706,
261
+ "eval_loss": 3.402008295059204,
262
+ "eval_runtime": 20.7141,
263
+ "eval_samples_per_second": 459.929,
264
+ "eval_steps_per_second": 0.483,
265
+ "step": 500
266
+ }
267
+ ],
268
+ "logging_steps": 20,
269
+ "max_steps": 500,
270
+ "num_input_tokens_seen": 0,
271
+ "num_train_epochs": 1,
272
+ "save_steps": 100,
273
+ "stateful_callbacks": {
274
+ "TrainerControl": {
275
+ "args": {
276
+ "should_epoch_stop": false,
277
+ "should_evaluate": false,
278
+ "should_log": false,
279
+ "should_save": true,
280
+ "should_training_stop": true
281
+ },
282
+ "attributes": {}
283
+ }
284
+ },
285
+ "total_flos": 1209853280256000.0,
286
+ "train_batch_size": 32,
287
+ "trial_name": null,
288
+ "trial_params": null
289
+ }
zain/Activation/out/glu-waleed-150L_run/checkpoint-500/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b827f60f983e3f5ad52047a134946ee6cd902cd1de8cdbd39e8e9aa340a1aef1
3
+ size 4920
zain/Activation/out/glu-waleed-150L_run/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "waleed",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "glu",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 150,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/glu-waleed-150L_run/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee8cdea3b2881be1f0bf35b491fcbc952c771b9e744b1d72e647cde565cb331a
3
+ size 50427120
zain/Activation/out/glu-waleed-150L_run/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/glu-waleed-150L_run/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/glu-waleed-150L_run/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b827f60f983e3f5ad52047a134946ee6cd902cd1de8cdbd39e8e9aa340a1aef1
3
+ size 4920
zain/Activation/out/glu-waleed-150L_run/training_log.jsonl CHANGED
@@ -11,3 +11,27 @@
11
  {"step": 160, "epoch": 0.005392652510953826, "timestamp": 1786569522.1553469, "loss": 5.545684051513672, "grad_norm": 0.98046875, "learning_rate": 0.0005565, "train/total_time_seconds": 55.305282179266214, "train/time_per_step_avg": 0.34286197159439324, "train/epoch_time_elapsed": 122.13275729864836, "train/estimated_remaining_minutes": 1.9587287438490115}
12
  {"step": 180, "epoch": 0.006066734074823054, "timestamp": 1786569529.4016442, "loss": 5.289410018920899, "grad_norm": 0.62109375, "learning_rate": 0.0006265, "train/total_time_seconds": 62.03357548639178, "train/time_per_step_avg": 0.3401971862465143, "train/epoch_time_elapsed": 129.37905456498265, "train/estimated_remaining_minutes": 1.8380318662634603}
13
  {"step": 200, "epoch": 0.006740815638692282, "timestamp": 1786569536.7232873, "loss": 4.994114685058594, "grad_norm": 0.5390625, "learning_rate": 0.0006965, "train/total_time_seconds": 68.81566143780947, "train/time_per_step_avg": 0.33934433944523335, "train/epoch_time_elapsed": 136.7006975747645, "train/estimated_remaining_minutes": 1.7203915359452369}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11
  {"step": 160, "epoch": 0.005392652510953826, "timestamp": 1786569522.1553469, "loss": 5.545684051513672, "grad_norm": 0.98046875, "learning_rate": 0.0005565, "train/total_time_seconds": 55.305282179266214, "train/time_per_step_avg": 0.34286197159439324, "train/epoch_time_elapsed": 122.13275729864836, "train/estimated_remaining_minutes": 1.9587287438490115}
12
  {"step": 180, "epoch": 0.006066734074823054, "timestamp": 1786569529.4016442, "loss": 5.289410018920899, "grad_norm": 0.62109375, "learning_rate": 0.0006265, "train/total_time_seconds": 62.03357548639178, "train/time_per_step_avg": 0.3401971862465143, "train/epoch_time_elapsed": 129.37905456498265, "train/estimated_remaining_minutes": 1.8380318662634603}
13
  {"step": 200, "epoch": 0.006740815638692282, "timestamp": 1786569536.7232873, "loss": 4.994114685058594, "grad_norm": 0.5390625, "learning_rate": 0.0006965, "train/total_time_seconds": 68.81566143780947, "train/time_per_step_avg": 0.33934433944523335, "train/epoch_time_elapsed": 136.7006975747645, "train/estimated_remaining_minutes": 1.7203915359452369}
14
+ {"step": 200, "epoch": 0.006740815638692282, "timestamp": 1786569557.3919575, "eval_loss": 4.839845180511475, "eval_runtime": 20.6673, "eval_samples_per_second": 460.971, "eval_steps_per_second": 0.484, "train/total_time_seconds": 68.81566143780947, "train/time_per_step_avg": 0.33934433944523335, "train/epoch_time_elapsed": 157.3693672977388, "train/estimated_remaining_minutes": 1.7203915359452369}
15
+ {"step": 220, "epoch": 0.00741489720256151, "timestamp": 1786569565.3659294, "loss": 4.748640823364258, "grad_norm": 1.0703125, "learning_rate": 0.0007, "train/total_time_seconds": 75.62983629480004, "train/time_per_step_avg": 0.3397757708281279, "train/epoch_time_elapsed": 165.343339856714, "train/estimated_remaining_minutes": 1.6042692547381827}
16
+ {"step": 240, "epoch": 0.008088978766430738, "timestamp": 1786569572.7260487, "loss": 4.502994537353516, "grad_norm": 0.6796875, "learning_rate": 0.0007, "train/total_time_seconds": 82.4601291641593, "train/time_per_step_avg": 0.34021866325289013, "train/epoch_time_elapsed": 172.70345924049616, "train/estimated_remaining_minutes": 1.4888634432417651}
17
+ {"step": 250, "epoch": 0.008426019548365353, "timestamp": 1786569597.0672214, "eval_loss": 4.319539546966553, "eval_runtime": 20.6786, "eval_samples_per_second": 460.717, "eval_steps_per_second": 0.484, "train/total_time_seconds": 85.84956888481975, "train/time_per_step_avg": 0.3395453579351306, "train/epoch_time_elapsed": 197.04463040456176, "train/estimated_remaining_minutes": 1.430826148080329}
18
+ {"step": 260, "epoch": 0.008763060330299966, "timestamp": 1786569600.7534935, "loss": 4.330507278442383, "grad_norm": 0.455078125, "learning_rate": 0.0007, "train/total_time_seconds": 89.27321784198284, "train/time_per_step_avg": 0.33967935662716625, "train/epoch_time_elapsed": 200.73090418428183, "train/estimated_remaining_minutes": 1.3734341206458898}
19
+ {"step": 280, "epoch": 0.009437141894169195, "timestamp": 1786569608.106018, "loss": 4.185513687133789, "grad_norm": 0.59765625, "learning_rate": 0.0007, "train/total_time_seconds": 96.10496650263667, "train/time_per_step_avg": 0.3407139101624489, "train/epoch_time_elapsed": 208.0834284685552, "train/estimated_remaining_minutes": 1.2585174184869088}
20
+ {"step": 300, "epoch": 0.010111223458038422, "timestamp": 1786569615.3244562, "loss": 4.058372116088867, "grad_norm": 0.62890625, "learning_rate": 0.0007, "train/total_time_seconds": 102.80648648738861, "train/time_per_step_avg": 0.33990825049579143, "train/epoch_time_elapsed": 215.30186684429646, "train/estimated_remaining_minutes": 1.142294294304318}
21
+ {"step": 300, "epoch": 0.010111223458038422, "timestamp": 1786569636.0719028, "eval_loss": 4.001541614532471, "eval_runtime": 20.746, "eval_samples_per_second": 459.22, "eval_steps_per_second": 0.482, "train/total_time_seconds": 102.80648648738861, "train/time_per_step_avg": 0.33990825049579143, "train/epoch_time_elapsed": 236.0493126027286, "train/estimated_remaining_minutes": 1.142294294304318}
22
+ {"step": 320, "epoch": 0.010785305021907651, "timestamp": 1786569644.0682247, "loss": 3.9491451263427733, "grad_norm": 0.54296875, "learning_rate": 0.0007, "train/total_time_seconds": 109.62384182959795, "train/time_per_step_avg": 0.33994005534797905, "train/epoch_time_elapsed": 244.04563404619694, "train/estimated_remaining_minutes": 1.0277235171524808}
23
+ {"step": 340, "epoch": 0.011459386585776879, "timestamp": 1786569651.4826317, "loss": 3.8855216979980467, "grad_norm": 0.69921875, "learning_rate": 0.0007, "train/total_time_seconds": 116.5046098344028, "train/time_per_step_avg": 0.34044480670243504, "train/epoch_time_elapsed": 251.46004226058722, "train/estimated_remaining_minutes": 0.913761645760022}
24
+ {"step": 350, "epoch": 0.011796427367711493, "timestamp": 1786569675.8987064, "eval_loss": 3.792919874191284, "eval_runtime": 20.7384, "eval_samples_per_second": 459.39, "eval_steps_per_second": 0.482, "train/total_time_seconds": 119.91959270089865, "train/time_per_step_avg": 0.34070023816078904, "train/epoch_time_elapsed": 275.87611592933536, "train/estimated_remaining_minutes": 0.8565685192921333}
25
+ {"step": 360, "epoch": 0.012133468149646108, "timestamp": 1786569679.5336637, "loss": 3.78214111328125, "grad_norm": 0.53515625, "learning_rate": 0.0007, "train/total_time_seconds": 123.29239882901311, "train/time_per_step_avg": 0.3401918098703027, "train/epoch_time_elapsed": 279.5110743716359, "train/estimated_remaining_minutes": 0.7991173998176776}
26
+ {"step": 380, "epoch": 0.012807549713515335, "timestamp": 1786569686.8399236, "loss": 3.7168258666992187, "grad_norm": 0.478515625, "learning_rate": 0.0007, "train/total_time_seconds": 130.06856789067388, "train/time_per_step_avg": 0.33963601388037207, "train/epoch_time_elapsed": 286.8173341713846, "train/estimated_remaining_minutes": 0.6845714099509151}
27
+ {"step": 400, "epoch": 0.013481631277384564, "timestamp": 1786569694.158065, "loss": 3.660007858276367, "grad_norm": 0.48046875, "learning_rate": 0.0007, "train/total_time_seconds": 136.86465257406235, "train/time_per_step_avg": 0.34058166086673736, "train/epoch_time_elapsed": 294.1354754194617, "train/estimated_remaining_minutes": 0.5702693857252598}
28
+ {"step": 400, "epoch": 0.013481631277384564, "timestamp": 1786569714.873715, "eval_loss": 3.6307969093322754, "eval_runtime": 20.7141, "eval_samples_per_second": 459.927, "eval_steps_per_second": 0.483, "train/total_time_seconds": 136.86465257406235, "train/time_per_step_avg": 0.34058166086673736, "train/epoch_time_elapsed": 314.8511252440512, "train/estimated_remaining_minutes": 0.5702693857252598}
29
+ {"step": 420, "epoch": 0.014155712841253791, "timestamp": 1786569722.7422597, "loss": 3.5877830505371096, "grad_norm": 0.55078125, "learning_rate": 0.0007, "train/total_time_seconds": 143.5808736383915, "train/time_per_step_avg": 0.3395703180879355, "train/epoch_time_elapsed": 322.7196694277227, "train/estimated_remaining_minutes": 0.4558122972647349}
30
+ {"step": 440, "epoch": 0.01482979440512302, "timestamp": 1786569730.1023037, "loss": 3.560054397583008, "grad_norm": 0.51953125, "learning_rate": 0.0007, "train/total_time_seconds": 150.4077126979828, "train/time_per_step_avg": 0.33903102863579987, "train/epoch_time_elapsed": 330.07971423864365, "train/estimated_remaining_minutes": 0.34183571067723356}
31
+ {"step": 450, "epoch": 0.015166835187057633, "timestamp": 1786569754.3255029, "eval_loss": 3.5101380348205566, "eval_runtime": 20.5828, "eval_samples_per_second": 462.863, "eval_steps_per_second": 0.486, "train/total_time_seconds": 153.7891417145729, "train/time_per_step_avg": 0.3386954901367426, "train/epoch_time_elapsed": 354.30291237682104, "train/estimated_remaining_minutes": 0.28479470687883873}
32
+ {"step": 460, "epoch": 0.015503875968992248, "timestamp": 1786569758.0804605, "loss": 3.5127296447753906, "grad_norm": 0.52734375, "learning_rate": 0.0007, "train/total_time_seconds": 157.28131246194243, "train/time_per_step_avg": 0.33988913632929324, "train/epoch_time_elapsed": 358.0578713901341, "train/estimated_remaining_minutes": 0.22794393110426442}
33
+ {"step": 480, "epoch": 0.016177957532861477, "timestamp": 1786569765.378189, "loss": 3.4453048706054688, "grad_norm": 0.43359375, "learning_rate": 0.0007, "train/total_time_seconds": 164.06114525720477, "train/time_per_step_avg": 0.33992577366530896, "train/epoch_time_elapsed": 365.35559993982315, "train/estimated_remaining_minutes": 0.11393135087305888}
34
+ {"step": 500, "epoch": 0.016852039096730706, "timestamp": 1786569772.603253, "loss": 3.415237045288086, "grad_norm": 0.431640625, "learning_rate": 0.0007, "train/total_time_seconds": 170.76810985431075, "train/time_per_step_avg": 0.33903457280248406, "train/epoch_time_elapsed": 372.5806630514562, "train/estimated_remaining_minutes": 0.0}
35
+ {"step": 500, "epoch": 0.016852039096730706, "timestamp": 1786569793.3190465, "eval_loss": 3.402008295059204, "eval_runtime": 20.7141, "eval_samples_per_second": 459.929, "eval_steps_per_second": 0.483, "train/total_time_seconds": 170.76810985431075, "train/time_per_step_avg": 0.33903457280248406, "train/epoch_time_elapsed": 393.29645624384284, "train/estimated_remaining_minutes": 0.0}
36
+ {"step": 500, "epoch": 0.016852039096730706, "timestamp": 1786569793.739768, "train_runtime": 394.5028, "train_samples_per_second": 40.557, "train_steps_per_second": 1.267, "total_flos": 1209853280256000.0, "train_loss": 4.914910415649414, "train/total_time_seconds": 170.76810985431075, "train/time_per_step_avg": 0.33903457280248406, "train/epoch_time_elapsed": 393.71717697381973, "train/estimated_remaining_minutes": 0.0}
37
+ {"step": 500, "epoch": 0.016852039096730706, "timestamp": 1786569814.3469703, "eval_loss": 3.402008295059204, "eval_runtime": 20.6042, "eval_samples_per_second": 462.382, "eval_steps_per_second": 0.485, "train/total_time_seconds": 170.76810985431075, "train/time_per_step_avg": 0.33903457280248406, "train/epoch_time_elapsed": 414.324380222708, "train/estimated_remaining_minutes": 0.0}