w-ahmad commited on
Commit
da74604
·
verified ·
1 Parent(s): dd8818a

Auto upload zain 2026-08-19T16:33:38.712916 (part 2)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/optimizer.pt +3 -0
  3. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/rng_state.pth +3 -0
  4. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/scheduler.pt +3 -0
  5. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer.json +0 -0
  6. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer_config.json +13 -0
  7. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/trainer_state.json +1360 -0
  8. zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/training_args.bin +3 -0
  9. zain/Activation/out/mlp-linear-3L_run/checkpoint-400/model.safetensors +1 -1
  10. zain/Activation/out/mlp-linear-3L_run/checkpoint-400/optimizer.pt +1 -1
  11. zain/Activation/out/mlp-linear-3L_run/checkpoint-400/rng_state.pth +1 -1
  12. zain/Activation/out/mlp-linear-3L_run/checkpoint-400/scheduler.pt +1 -1
  13. zain/Activation/out/mlp-linear-3L_run/checkpoint-400/trainer_state.json +95 -111
  14. zain/Activation/out/mlp-linear-3L_run/checkpoint-400/training_args.bin +1 -1
  15. zain/Activation/out/mlp-linear-3L_run/checkpoint-500/model.safetensors +1 -1
  16. zain/Activation/out/mlp-linear-3L_run/checkpoint-500/optimizer.pt +1 -1
  17. zain/Activation/out/mlp-linear-3L_run/checkpoint-500/rng_state.pth +1 -1
  18. zain/Activation/out/mlp-linear-3L_run/checkpoint-500/scheduler.pt +1 -1
  19. zain/Activation/out/mlp-linear-3L_run/checkpoint-500/trainer_state.json +115 -139
  20. zain/Activation/out/mlp-linear-3L_run/checkpoint-500/training_args.bin +1 -1
  21. zain/Activation/out/mlp-linear-3L_run/checkpoint-600/model.safetensors +1 -1
  22. zain/Activation/out/mlp-linear-3L_run/checkpoint-600/optimizer.pt +1 -1
  23. zain/Activation/out/mlp-linear-3L_run/checkpoint-600/rng_state.pth +1 -1
  24. zain/Activation/out/mlp-linear-3L_run/checkpoint-600/scheduler.pt +1 -1
  25. zain/Activation/out/mlp-linear-3L_run/checkpoint-600/trainer_state.json +140 -164
  26. zain/Activation/out/mlp-linear-3L_run/checkpoint-600/training_args.bin +1 -1
  27. zain/Activation/out/mlp-linear-3L_run/checkpoint-700/model.safetensors +1 -1
  28. zain/Activation/out/mlp-linear-3L_run/checkpoint-700/optimizer.pt +1 -1
  29. zain/Activation/out/mlp-linear-3L_run/checkpoint-700/rng_state.pth +1 -1
  30. zain/Activation/out/mlp-linear-3L_run/checkpoint-700/scheduler.pt +1 -1
  31. zain/Activation/out/mlp-linear-3L_run/checkpoint-700/trainer_state.json +160 -192
  32. zain/Activation/out/mlp-linear-3L_run/checkpoint-700/training_args.bin +1 -1
  33. zain/Activation/out/mlp-linear-3L_run/checkpoint-800/model.safetensors +1 -1
  34. zain/Activation/out/mlp-linear-3L_run/checkpoint-800/optimizer.pt +1 -1
  35. zain/Activation/out/mlp-linear-3L_run/checkpoint-800/rng_state.pth +1 -1
  36. zain/Activation/out/mlp-linear-3L_run/checkpoint-800/scheduler.pt +1 -1
  37. zain/Activation/out/mlp-linear-3L_run/checkpoint-800/trainer_state.json +185 -217
  38. zain/Activation/out/mlp-linear-3L_run/checkpoint-800/training_args.bin +1 -1
  39. zain/Activation/out/mlp-linear-3L_run/checkpoint-900/model.safetensors +1 -1
  40. zain/Activation/out/mlp-linear-3L_run/checkpoint-900/optimizer.pt +1 -1
  41. zain/Activation/out/mlp-linear-3L_run/checkpoint-900/rng_state.pth +1 -1
  42. zain/Activation/out/mlp-linear-3L_run/checkpoint-900/scheduler.pt +1 -1
  43. zain/Activation/out/mlp-linear-3L_run/checkpoint-900/trainer_state.json +205 -245
  44. zain/Activation/out/mlp-linear-3L_run/checkpoint-900/training_args.bin +1 -1
  45. zain/Activation/out/mlp-linear-3L_run/training_log.jsonl +0 -0
  46. zain/Activation/wandb/debug-internal.log +35 -63
  47. zain/Activation/wandb/debug.log +22 -24
  48. zain/Activation/wandb/run-20260819_163014-smtfejrp/files/output.log +364 -0
  49. zain/Activation/wandb/run-20260819_163014-smtfejrp/files/requirements.txt +149 -0
  50. zain/Activation/wandb/run-20260819_163014-smtfejrp/files/wandb-metadata.json +118 -0
.gitattributes CHANGED
@@ -106,3 +106,4 @@ zain/Activation/wandb/run-20260818_234902-aoa8wkuw/run-aoa8wkuw.wandb filter=lfs
106
  zain/Activation/wandb/run-20260818_235450-ckmqoyjn/run-ckmqoyjn.wandb filter=lfs diff=lfs merge=lfs -text
107
  zain/Activation/wandb/run-20260819_000034-awq28nwd/run-awq28nwd.wandb filter=lfs diff=lfs merge=lfs -text
108
  zain/Activation/wandb/run-20260819_000621-6zbmve2d/run-6zbmve2d.wandb filter=lfs diff=lfs merge=lfs -text
 
 
106
  zain/Activation/wandb/run-20260818_235450-ckmqoyjn/run-ckmqoyjn.wandb filter=lfs diff=lfs merge=lfs -text
107
  zain/Activation/wandb/run-20260819_000034-awq28nwd/run-awq28nwd.wandb filter=lfs diff=lfs merge=lfs -text
108
  zain/Activation/wandb/run-20260819_000621-6zbmve2d/run-6zbmve2d.wandb filter=lfs diff=lfs merge=lfs -text
109
+ zain/Activation/wandb/run-20260819_163014-smtfejrp/run-smtfejrp.wandb filter=lfs diff=lfs merge=lfs -text
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ce76c8d713a0e55e136bdef2c6d778e39cc2c72664541e823421cce9f2b5e5e1
3
+ size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9213080fe2b45399b87036ca9ff9164533abe6b368e5c828136ee184486749d4
3
+ size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:84a9254a847978e7f5aa3da452546981aa184489b48bcdca370761da60268da8
3
+ size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/trainer_state.json ADDED
@@ -0,0 +1,1360 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.057296932928884395,
6
+ "eval_steps": 200,
7
+ "global_step": 3400,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
+ "step": 60
32
+ },
33
+ {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
+ "step": 80
39
+ },
40
+ {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
45
+ "step": 100
46
+ },
47
+ {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
+ "step": 120
53
+ },
54
+ {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
+ "step": 140
60
+ },
61
+ {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
+ "step": 160
67
+ },
68
+ {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
+ "step": 180
74
+ },
75
+ {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
+ "step": 200
81
+ },
82
+ {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
+ "step": 200
89
+ },
90
+ {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
+ "step": 220
96
+ },
97
+ {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
+ "step": 240
103
+ },
104
+ {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
+ "step": 260
110
+ },
111
+ {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
+ "step": 280
117
+ },
118
+ {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
123
+ "step": 300
124
+ },
125
+ {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
130
+ "step": 320
131
+ },
132
+ {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
+ "step": 340
138
+ },
139
+ {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
+ "step": 360
145
+ },
146
+ {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
+ "step": 380
152
+ },
153
+ {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
+ "step": 400
159
+ },
160
+ {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
+ "step": 400
167
+ },
168
+ {
169
+ "epoch": 0.007077856420626896,
170
+ "grad_norm": 0.98046875,
171
+ "learning_rate": 0.0001257,
172
+ "loss": 5.9373779296875,
173
+ "step": 420
174
+ },
175
+ {
176
+ "epoch": 0.00741489720256151,
177
+ "grad_norm": 1.6328125,
178
+ "learning_rate": 0.00013169999999999998,
179
+ "loss": 5.839211273193359,
180
+ "step": 440
181
+ },
182
+ {
183
+ "epoch": 0.007751937984496124,
184
+ "grad_norm": 0.9609375,
185
+ "learning_rate": 0.00013769999999999999,
186
+ "loss": 5.740922927856445,
187
+ "step": 460
188
+ },
189
+ {
190
+ "epoch": 0.008088978766430738,
191
+ "grad_norm": 0.90234375,
192
+ "learning_rate": 0.00014369999999999997,
193
+ "loss": 5.6399181365966795,
194
+ "step": 480
195
+ },
196
+ {
197
+ "epoch": 0.008426019548365353,
198
+ "grad_norm": 1.1796875,
199
+ "learning_rate": 0.00014969999999999998,
200
+ "loss": 5.560699081420898,
201
+ "step": 500
202
+ },
203
+ {
204
+ "epoch": 0.008763060330299966,
205
+ "grad_norm": 2.328125,
206
+ "learning_rate": 0.0001557,
207
+ "loss": 5.474863433837891,
208
+ "step": 520
209
+ },
210
+ {
211
+ "epoch": 0.00910010111223458,
212
+ "grad_norm": 1.125,
213
+ "learning_rate": 0.0001617,
214
+ "loss": 5.396588516235352,
215
+ "step": 540
216
+ },
217
+ {
218
+ "epoch": 0.009437141894169195,
219
+ "grad_norm": 1.6484375,
220
+ "learning_rate": 0.0001677,
221
+ "loss": 5.331023406982422,
222
+ "step": 560
223
+ },
224
+ {
225
+ "epoch": 0.00977418267610381,
226
+ "grad_norm": 1.0703125,
227
+ "learning_rate": 0.00017369999999999997,
228
+ "loss": 5.257175445556641,
229
+ "step": 580
230
+ },
231
+ {
232
+ "epoch": 0.010111223458038422,
233
+ "grad_norm": 2.359375,
234
+ "learning_rate": 0.00017969999999999998,
235
+ "loss": 5.152382659912109,
236
+ "step": 600
237
+ },
238
+ {
239
+ "epoch": 0.010111223458038422,
240
+ "eval_loss": 5.1175713539123535,
241
+ "eval_runtime": 7.457,
242
+ "eval_samples_per_second": 1277.585,
243
+ "eval_steps_per_second": 0.939,
244
+ "step": 600
245
+ },
246
+ {
247
+ "epoch": 0.010448264239973037,
248
+ "grad_norm": 1.375,
249
+ "learning_rate": 0.0001857,
250
+ "loss": 5.096985244750977,
251
+ "step": 620
252
+ },
253
+ {
254
+ "epoch": 0.010785305021907651,
255
+ "grad_norm": 1.421875,
256
+ "learning_rate": 0.0001917,
257
+ "loss": 5.019801330566406,
258
+ "step": 640
259
+ },
260
+ {
261
+ "epoch": 0.011122345803842264,
262
+ "grad_norm": 2.125,
263
+ "learning_rate": 0.00019769999999999998,
264
+ "loss": 4.979957962036133,
265
+ "step": 660
266
+ },
267
+ {
268
+ "epoch": 0.011459386585776879,
269
+ "grad_norm": 2.203125,
270
+ "learning_rate": 0.0002037,
271
+ "loss": 4.945652008056641,
272
+ "step": 680
273
+ },
274
+ {
275
+ "epoch": 0.011796427367711493,
276
+ "grad_norm": 1.8515625,
277
+ "learning_rate": 0.00020969999999999997,
278
+ "loss": 4.866051483154297,
279
+ "step": 700
280
+ },
281
+ {
282
+ "epoch": 0.012133468149646108,
283
+ "grad_norm": 1.625,
284
+ "learning_rate": 0.00021569999999999998,
285
+ "loss": 4.841766357421875,
286
+ "step": 720
287
+ },
288
+ {
289
+ "epoch": 0.01247050893158072,
290
+ "grad_norm": 3.40625,
291
+ "learning_rate": 0.00022169999999999997,
292
+ "loss": 4.798672103881836,
293
+ "step": 740
294
+ },
295
+ {
296
+ "epoch": 0.012807549713515335,
297
+ "grad_norm": 2.21875,
298
+ "learning_rate": 0.00022769999999999998,
299
+ "loss": 4.768531036376953,
300
+ "step": 760
301
+ },
302
+ {
303
+ "epoch": 0.01314459049544995,
304
+ "grad_norm": 1.640625,
305
+ "learning_rate": 0.0002337,
306
+ "loss": 4.73585319519043,
307
+ "step": 780
308
+ },
309
+ {
310
+ "epoch": 0.013481631277384564,
311
+ "grad_norm": 1.6875,
312
+ "learning_rate": 0.0002397,
313
+ "loss": 4.697021865844727,
314
+ "step": 800
315
+ },
316
+ {
317
+ "epoch": 0.013481631277384564,
318
+ "eval_loss": 4.676848411560059,
319
+ "eval_runtime": 7.4394,
320
+ "eval_samples_per_second": 1280.62,
321
+ "eval_steps_per_second": 0.941,
322
+ "step": 800
323
+ },
324
+ {
325
+ "epoch": 0.013818672059319177,
326
+ "grad_norm": 1.6484375,
327
+ "learning_rate": 0.00024569999999999995,
328
+ "loss": 4.658943176269531,
329
+ "step": 820
330
+ },
331
+ {
332
+ "epoch": 0.014155712841253791,
333
+ "grad_norm": 3.234375,
334
+ "learning_rate": 0.0002517,
335
+ "loss": 4.5957294464111325,
336
+ "step": 840
337
+ },
338
+ {
339
+ "epoch": 0.014492753623188406,
340
+ "grad_norm": 1.578125,
341
+ "learning_rate": 0.0002577,
342
+ "loss": 4.5967552185058596,
343
+ "step": 860
344
+ },
345
+ {
346
+ "epoch": 0.01482979440512302,
347
+ "grad_norm": 1.3203125,
348
+ "learning_rate": 0.00026369999999999996,
349
+ "loss": 4.549871444702148,
350
+ "step": 880
351
+ },
352
+ {
353
+ "epoch": 0.015166835187057633,
354
+ "grad_norm": 2.28125,
355
+ "learning_rate": 0.0002697,
356
+ "loss": 4.512709045410157,
357
+ "step": 900
358
+ },
359
+ {
360
+ "epoch": 0.015503875968992248,
361
+ "grad_norm": 1.6875,
362
+ "learning_rate": 0.0002757,
363
+ "loss": 4.475643920898437,
364
+ "step": 920
365
+ },
366
+ {
367
+ "epoch": 0.015840916750926862,
368
+ "grad_norm": 1.6015625,
369
+ "learning_rate": 0.00028169999999999996,
370
+ "loss": 4.43329086303711,
371
+ "step": 940
372
+ },
373
+ {
374
+ "epoch": 0.016177957532861477,
375
+ "grad_norm": 2.03125,
376
+ "learning_rate": 0.00028769999999999995,
377
+ "loss": 4.402384567260742,
378
+ "step": 960
379
+ },
380
+ {
381
+ "epoch": 0.01651499831479609,
382
+ "grad_norm": 1.71875,
383
+ "learning_rate": 0.0002937,
384
+ "loss": 4.382797622680664,
385
+ "step": 980
386
+ },
387
+ {
388
+ "epoch": 0.016852039096730706,
389
+ "grad_norm": 1.21875,
390
+ "learning_rate": 0.00029969999999999997,
391
+ "loss": 4.368251037597656,
392
+ "step": 1000
393
+ },
394
+ {
395
+ "epoch": 0.016852039096730706,
396
+ "eval_loss": 4.342709064483643,
397
+ "eval_runtime": 7.4513,
398
+ "eval_samples_per_second": 1278.567,
399
+ "eval_steps_per_second": 0.939,
400
+ "step": 1000
401
+ },
402
+ {
403
+ "epoch": 0.017189079878665317,
404
+ "grad_norm": 1.640625,
405
+ "learning_rate": 0.0003,
406
+ "loss": 4.33197135925293,
407
+ "step": 1020
408
+ },
409
+ {
410
+ "epoch": 0.01752612066059993,
411
+ "grad_norm": 1.8984375,
412
+ "learning_rate": 0.0003,
413
+ "loss": 4.315201950073242,
414
+ "step": 1040
415
+ },
416
+ {
417
+ "epoch": 0.017863161442534546,
418
+ "grad_norm": 1.8828125,
419
+ "learning_rate": 0.0003,
420
+ "loss": 4.262122344970703,
421
+ "step": 1060
422
+ },
423
+ {
424
+ "epoch": 0.01820020222446916,
425
+ "grad_norm": 2.3125,
426
+ "learning_rate": 0.0003,
427
+ "loss": 4.262465286254883,
428
+ "step": 1080
429
+ },
430
+ {
431
+ "epoch": 0.018537243006403775,
432
+ "grad_norm": 2.25,
433
+ "learning_rate": 0.0003,
434
+ "loss": 4.2220817565917965,
435
+ "step": 1100
436
+ },
437
+ {
438
+ "epoch": 0.01887428378833839,
439
+ "grad_norm": 1.6328125,
440
+ "learning_rate": 0.0003,
441
+ "loss": 4.21630973815918,
442
+ "step": 1120
443
+ },
444
+ {
445
+ "epoch": 0.019211324570273004,
446
+ "grad_norm": 1.84375,
447
+ "learning_rate": 0.0003,
448
+ "loss": 4.189186859130859,
449
+ "step": 1140
450
+ },
451
+ {
452
+ "epoch": 0.01954836535220762,
453
+ "grad_norm": 2.546875,
454
+ "learning_rate": 0.0003,
455
+ "loss": 4.164257431030274,
456
+ "step": 1160
457
+ },
458
+ {
459
+ "epoch": 0.01988540613414223,
460
+ "grad_norm": 1.4609375,
461
+ "learning_rate": 0.0003,
462
+ "loss": 4.161908721923828,
463
+ "step": 1180
464
+ },
465
+ {
466
+ "epoch": 0.020222446916076844,
467
+ "grad_norm": 1.5625,
468
+ "learning_rate": 0.0003,
469
+ "loss": 4.131203460693359,
470
+ "step": 1200
471
+ },
472
+ {
473
+ "epoch": 0.020222446916076844,
474
+ "eval_loss": 4.1340837478637695,
475
+ "eval_runtime": 7.4658,
476
+ "eval_samples_per_second": 1276.087,
477
+ "eval_steps_per_second": 0.938,
478
+ "step": 1200
479
+ },
480
+ {
481
+ "epoch": 0.02055948769801146,
482
+ "grad_norm": 1.2890625,
483
+ "learning_rate": 0.0003,
484
+ "loss": 4.109837341308594,
485
+ "step": 1220
486
+ },
487
+ {
488
+ "epoch": 0.020896528479946073,
489
+ "grad_norm": 2.109375,
490
+ "learning_rate": 0.0003,
491
+ "loss": 4.08643684387207,
492
+ "step": 1240
493
+ },
494
+ {
495
+ "epoch": 0.021233569261880688,
496
+ "grad_norm": 1.5859375,
497
+ "learning_rate": 0.0003,
498
+ "loss": 4.069121932983398,
499
+ "step": 1260
500
+ },
501
+ {
502
+ "epoch": 0.021570610043815303,
503
+ "grad_norm": 1.7890625,
504
+ "learning_rate": 0.0003,
505
+ "loss": 4.091477966308593,
506
+ "step": 1280
507
+ },
508
+ {
509
+ "epoch": 0.021907650825749917,
510
+ "grad_norm": 1.484375,
511
+ "learning_rate": 0.0003,
512
+ "loss": 4.061969757080078,
513
+ "step": 1300
514
+ },
515
+ {
516
+ "epoch": 0.022244691607684528,
517
+ "grad_norm": 1.5625,
518
+ "learning_rate": 0.0003,
519
+ "loss": 4.03832893371582,
520
+ "step": 1320
521
+ },
522
+ {
523
+ "epoch": 0.022581732389619143,
524
+ "grad_norm": 1.359375,
525
+ "learning_rate": 0.0003,
526
+ "loss": 4.031367492675781,
527
+ "step": 1340
528
+ },
529
+ {
530
+ "epoch": 0.022918773171553757,
531
+ "grad_norm": 1.4375,
532
+ "learning_rate": 0.0003,
533
+ "loss": 4.001505661010742,
534
+ "step": 1360
535
+ },
536
+ {
537
+ "epoch": 0.023255813953488372,
538
+ "grad_norm": 1.7734375,
539
+ "learning_rate": 0.0003,
540
+ "loss": 4.031853866577149,
541
+ "step": 1380
542
+ },
543
+ {
544
+ "epoch": 0.023592854735422986,
545
+ "grad_norm": 1.4609375,
546
+ "learning_rate": 0.0003,
547
+ "loss": 4.0036476135253904,
548
+ "step": 1400
549
+ },
550
+ {
551
+ "epoch": 0.023592854735422986,
552
+ "eval_loss": 3.999983072280884,
553
+ "eval_runtime": 7.5012,
554
+ "eval_samples_per_second": 1270.062,
555
+ "eval_steps_per_second": 0.933,
556
+ "step": 1400
557
+ },
558
+ {
559
+ "epoch": 0.0239298955173576,
560
+ "grad_norm": 1.53125,
561
+ "learning_rate": 0.0003,
562
+ "loss": 4.006050109863281,
563
+ "step": 1420
564
+ },
565
+ {
566
+ "epoch": 0.024266936299292215,
567
+ "grad_norm": 1.359375,
568
+ "learning_rate": 0.0003,
569
+ "loss": 3.9725921630859373,
570
+ "step": 1440
571
+ },
572
+ {
573
+ "epoch": 0.02460397708122683,
574
+ "grad_norm": 1.375,
575
+ "learning_rate": 0.0003,
576
+ "loss": 4.006849670410157,
577
+ "step": 1460
578
+ },
579
+ {
580
+ "epoch": 0.02494101786316144,
581
+ "grad_norm": 2.03125,
582
+ "learning_rate": 0.0003,
583
+ "loss": 3.955257797241211,
584
+ "step": 1480
585
+ },
586
+ {
587
+ "epoch": 0.025278058645096056,
588
+ "grad_norm": 1.359375,
589
+ "learning_rate": 0.0003,
590
+ "loss": 3.959492874145508,
591
+ "step": 1500
592
+ },
593
+ {
594
+ "epoch": 0.02561509942703067,
595
+ "grad_norm": 1.6484375,
596
+ "learning_rate": 0.0003,
597
+ "loss": 3.924856185913086,
598
+ "step": 1520
599
+ },
600
+ {
601
+ "epoch": 0.025952140208965285,
602
+ "grad_norm": 1.53125,
603
+ "learning_rate": 0.0003,
604
+ "loss": 3.915934753417969,
605
+ "step": 1540
606
+ },
607
+ {
608
+ "epoch": 0.0262891809908999,
609
+ "grad_norm": 1.46875,
610
+ "learning_rate": 0.0003,
611
+ "loss": 3.9002685546875,
612
+ "step": 1560
613
+ },
614
+ {
615
+ "epoch": 0.026626221772834514,
616
+ "grad_norm": 1.5078125,
617
+ "learning_rate": 0.0003,
618
+ "loss": 3.895922088623047,
619
+ "step": 1580
620
+ },
621
+ {
622
+ "epoch": 0.026963262554769128,
623
+ "grad_norm": 1.4453125,
624
+ "learning_rate": 0.0003,
625
+ "loss": 3.9393074035644533,
626
+ "step": 1600
627
+ },
628
+ {
629
+ "epoch": 0.026963262554769128,
630
+ "eval_loss": 3.9100229740142822,
631
+ "eval_runtime": 7.7821,
632
+ "eval_samples_per_second": 1224.215,
633
+ "eval_steps_per_second": 0.899,
634
+ "step": 1600
635
+ },
636
+ {
637
+ "epoch": 0.027300303336703743,
638
+ "grad_norm": 1.5234375,
639
+ "learning_rate": 0.0003,
640
+ "loss": 3.8935520172119142,
641
+ "step": 1620
642
+ },
643
+ {
644
+ "epoch": 0.027637344118638354,
645
+ "grad_norm": 1.65625,
646
+ "learning_rate": 0.0003,
647
+ "loss": 3.8741275787353517,
648
+ "step": 1640
649
+ },
650
+ {
651
+ "epoch": 0.02797438490057297,
652
+ "grad_norm": 1.8046875,
653
+ "learning_rate": 0.0003,
654
+ "loss": 3.8751964569091797,
655
+ "step": 1660
656
+ },
657
+ {
658
+ "epoch": 0.028311425682507583,
659
+ "grad_norm": 1.515625,
660
+ "learning_rate": 0.0003,
661
+ "loss": 3.8847869873046874,
662
+ "step": 1680
663
+ },
664
+ {
665
+ "epoch": 0.028648466464442197,
666
+ "grad_norm": 1.640625,
667
+ "learning_rate": 0.0003,
668
+ "loss": 3.900635528564453,
669
+ "step": 1700
670
+ },
671
+ {
672
+ "epoch": 0.028985507246376812,
673
+ "grad_norm": 1.359375,
674
+ "learning_rate": 0.0003,
675
+ "loss": 3.913836669921875,
676
+ "step": 1720
677
+ },
678
+ {
679
+ "epoch": 0.029322548028311426,
680
+ "grad_norm": 1.4921875,
681
+ "learning_rate": 0.0003,
682
+ "loss": 3.8652713775634764,
683
+ "step": 1740
684
+ },
685
+ {
686
+ "epoch": 0.02965958881024604,
687
+ "grad_norm": 1.921875,
688
+ "learning_rate": 0.0003,
689
+ "loss": 3.865024185180664,
690
+ "step": 1760
691
+ },
692
+ {
693
+ "epoch": 0.029996629592180656,
694
+ "grad_norm": 1.78125,
695
+ "learning_rate": 0.0003,
696
+ "loss": 3.825577163696289,
697
+ "step": 1780
698
+ },
699
+ {
700
+ "epoch": 0.030333670374115267,
701
+ "grad_norm": 1.3984375,
702
+ "learning_rate": 0.0003,
703
+ "loss": 3.842971420288086,
704
+ "step": 1800
705
+ },
706
+ {
707
+ "epoch": 0.030333670374115267,
708
+ "eval_loss": 3.8462002277374268,
709
+ "eval_runtime": 7.4378,
710
+ "eval_samples_per_second": 1280.896,
711
+ "eval_steps_per_second": 0.941,
712
+ "step": 1800
713
+ },
714
+ {
715
+ "epoch": 0.03067071115604988,
716
+ "grad_norm": 1.2890625,
717
+ "learning_rate": 0.0003,
718
+ "loss": 3.8543495178222655,
719
+ "step": 1820
720
+ },
721
+ {
722
+ "epoch": 0.031007751937984496,
723
+ "grad_norm": 1.53125,
724
+ "learning_rate": 0.0003,
725
+ "loss": 3.843314361572266,
726
+ "step": 1840
727
+ },
728
+ {
729
+ "epoch": 0.03134479271991911,
730
+ "grad_norm": 1.640625,
731
+ "learning_rate": 0.0003,
732
+ "loss": 3.841319274902344,
733
+ "step": 1860
734
+ },
735
+ {
736
+ "epoch": 0.031681833501853725,
737
+ "grad_norm": 1.109375,
738
+ "learning_rate": 0.0003,
739
+ "loss": 3.831389617919922,
740
+ "step": 1880
741
+ },
742
+ {
743
+ "epoch": 0.032018874283788336,
744
+ "grad_norm": 1.4140625,
745
+ "learning_rate": 0.0003,
746
+ "loss": 3.8035839080810545,
747
+ "step": 1900
748
+ },
749
+ {
750
+ "epoch": 0.032355915065722954,
751
+ "grad_norm": 1.8359375,
752
+ "learning_rate": 0.0003,
753
+ "loss": 3.8156478881835936,
754
+ "step": 1920
755
+ },
756
+ {
757
+ "epoch": 0.032692955847657565,
758
+ "grad_norm": 1.5,
759
+ "learning_rate": 0.0003,
760
+ "loss": 3.7988842010498045,
761
+ "step": 1940
762
+ },
763
+ {
764
+ "epoch": 0.03302999662959218,
765
+ "grad_norm": 2.078125,
766
+ "learning_rate": 0.0003,
767
+ "loss": 3.8014480590820314,
768
+ "step": 1960
769
+ },
770
+ {
771
+ "epoch": 0.033367037411526794,
772
+ "grad_norm": 1.640625,
773
+ "learning_rate": 0.0003,
774
+ "loss": 3.7902145385742188,
775
+ "step": 1980
776
+ },
777
+ {
778
+ "epoch": 0.03370407819346141,
779
+ "grad_norm": 1.6953125,
780
+ "learning_rate": 0.0003,
781
+ "loss": 3.8097972869873047,
782
+ "step": 2000
783
+ },
784
+ {
785
+ "epoch": 0.03370407819346141,
786
+ "eval_loss": 3.7975757122039795,
787
+ "eval_runtime": 7.4856,
788
+ "eval_samples_per_second": 1272.718,
789
+ "eval_steps_per_second": 0.935,
790
+ "step": 2000
791
+ },
792
+ {
793
+ "epoch": 0.03404111897539602,
794
+ "grad_norm": 1.7734375,
795
+ "learning_rate": 0.0003,
796
+ "loss": 3.801420211791992,
797
+ "step": 2020
798
+ },
799
+ {
800
+ "epoch": 0.034378159757330634,
801
+ "grad_norm": 1.5234375,
802
+ "learning_rate": 0.0003,
803
+ "loss": 3.787317657470703,
804
+ "step": 2040
805
+ },
806
+ {
807
+ "epoch": 0.03471520053926525,
808
+ "grad_norm": 1.46875,
809
+ "learning_rate": 0.0003,
810
+ "loss": 3.797407531738281,
811
+ "step": 2060
812
+ },
813
+ {
814
+ "epoch": 0.03505224132119986,
815
+ "grad_norm": 1.734375,
816
+ "learning_rate": 0.0003,
817
+ "loss": 3.759593963623047,
818
+ "step": 2080
819
+ },
820
+ {
821
+ "epoch": 0.03538928210313448,
822
+ "grad_norm": 1.4765625,
823
+ "learning_rate": 0.0003,
824
+ "loss": 3.7659400939941405,
825
+ "step": 2100
826
+ },
827
+ {
828
+ "epoch": 0.03572632288506909,
829
+ "grad_norm": 1.5,
830
+ "learning_rate": 0.0003,
831
+ "loss": 3.766071319580078,
832
+ "step": 2120
833
+ },
834
+ {
835
+ "epoch": 0.03606336366700371,
836
+ "grad_norm": 1.5078125,
837
+ "learning_rate": 0.0003,
838
+ "loss": 3.770547866821289,
839
+ "step": 2140
840
+ },
841
+ {
842
+ "epoch": 0.03640040444893832,
843
+ "grad_norm": 1.9453125,
844
+ "learning_rate": 0.0003,
845
+ "loss": 3.7436546325683593,
846
+ "step": 2160
847
+ },
848
+ {
849
+ "epoch": 0.03673744523087293,
850
+ "grad_norm": 1.390625,
851
+ "learning_rate": 0.0003,
852
+ "loss": 3.752705764770508,
853
+ "step": 2180
854
+ },
855
+ {
856
+ "epoch": 0.03707448601280755,
857
+ "grad_norm": 1.515625,
858
+ "learning_rate": 0.0003,
859
+ "loss": 3.7526622772216798,
860
+ "step": 2200
861
+ },
862
+ {
863
+ "epoch": 0.03707448601280755,
864
+ "eval_loss": 3.7623653411865234,
865
+ "eval_runtime": 7.4294,
866
+ "eval_samples_per_second": 1282.342,
867
+ "eval_steps_per_second": 0.942,
868
+ "step": 2200
869
+ },
870
+ {
871
+ "epoch": 0.03741152679474216,
872
+ "grad_norm": 1.765625,
873
+ "learning_rate": 0.0003,
874
+ "loss": 3.7532962799072265,
875
+ "step": 2220
876
+ },
877
+ {
878
+ "epoch": 0.03774856757667678,
879
+ "grad_norm": 1.765625,
880
+ "learning_rate": 0.0003,
881
+ "loss": 3.7590545654296874,
882
+ "step": 2240
883
+ },
884
+ {
885
+ "epoch": 0.03808560835861139,
886
+ "grad_norm": 1.71875,
887
+ "learning_rate": 0.0003,
888
+ "loss": 3.7438419342041014,
889
+ "step": 2260
890
+ },
891
+ {
892
+ "epoch": 0.03842264914054601,
893
+ "grad_norm": 1.859375,
894
+ "learning_rate": 0.0003,
895
+ "loss": 3.7286632537841795,
896
+ "step": 2280
897
+ },
898
+ {
899
+ "epoch": 0.03875968992248062,
900
+ "grad_norm": 1.609375,
901
+ "learning_rate": 0.0003,
902
+ "loss": 3.749654006958008,
903
+ "step": 2300
904
+ },
905
+ {
906
+ "epoch": 0.03909673070441524,
907
+ "grad_norm": 1.515625,
908
+ "learning_rate": 0.0003,
909
+ "loss": 3.716779327392578,
910
+ "step": 2320
911
+ },
912
+ {
913
+ "epoch": 0.03943377148634985,
914
+ "grad_norm": 1.34375,
915
+ "learning_rate": 0.0003,
916
+ "loss": 3.7487411499023438,
917
+ "step": 2340
918
+ },
919
+ {
920
+ "epoch": 0.03977081226828446,
921
+ "grad_norm": 1.5859375,
922
+ "learning_rate": 0.0003,
923
+ "loss": 3.748704528808594,
924
+ "step": 2360
925
+ },
926
+ {
927
+ "epoch": 0.04010785305021908,
928
+ "grad_norm": 1.421875,
929
+ "learning_rate": 0.0003,
930
+ "loss": 3.721243667602539,
931
+ "step": 2380
932
+ },
933
+ {
934
+ "epoch": 0.04044489383215369,
935
+ "grad_norm": 1.7109375,
936
+ "learning_rate": 0.0003,
937
+ "loss": 3.723830795288086,
938
+ "step": 2400
939
+ },
940
+ {
941
+ "epoch": 0.04044489383215369,
942
+ "eval_loss": 3.7325456142425537,
943
+ "eval_runtime": 7.4184,
944
+ "eval_samples_per_second": 1284.24,
945
+ "eval_steps_per_second": 0.944,
946
+ "step": 2400
947
+ },
948
+ {
949
+ "epoch": 0.04078193461408831,
950
+ "grad_norm": 1.546875,
951
+ "learning_rate": 0.0003,
952
+ "loss": 3.7124366760253906,
953
+ "step": 2420
954
+ },
955
+ {
956
+ "epoch": 0.04111897539602292,
957
+ "grad_norm": 1.546875,
958
+ "learning_rate": 0.0003,
959
+ "loss": 3.695796585083008,
960
+ "step": 2440
961
+ },
962
+ {
963
+ "epoch": 0.041456016177957536,
964
+ "grad_norm": 1.5078125,
965
+ "learning_rate": 0.0003,
966
+ "loss": 3.700525665283203,
967
+ "step": 2460
968
+ },
969
+ {
970
+ "epoch": 0.04179305695989215,
971
+ "grad_norm": 1.328125,
972
+ "learning_rate": 0.0003,
973
+ "loss": 3.7314002990722654,
974
+ "step": 2480
975
+ },
976
+ {
977
+ "epoch": 0.04213009774182676,
978
+ "grad_norm": 1.7421875,
979
+ "learning_rate": 0.0003,
980
+ "loss": 3.728242874145508,
981
+ "step": 2500
982
+ },
983
+ {
984
+ "epoch": 0.042467138523761376,
985
+ "grad_norm": 1.3125,
986
+ "learning_rate": 0.0003,
987
+ "loss": 3.7244899749755858,
988
+ "step": 2520
989
+ },
990
+ {
991
+ "epoch": 0.04280417930569599,
992
+ "grad_norm": 1.7265625,
993
+ "learning_rate": 0.0003,
994
+ "loss": 3.7204193115234374,
995
+ "step": 2540
996
+ },
997
+ {
998
+ "epoch": 0.043141220087630605,
999
+ "grad_norm": 2.09375,
1000
+ "learning_rate": 0.0003,
1001
+ "loss": 3.7023361206054686,
1002
+ "step": 2560
1003
+ },
1004
+ {
1005
+ "epoch": 0.043478260869565216,
1006
+ "grad_norm": 1.7890625,
1007
+ "learning_rate": 0.0003,
1008
+ "loss": 3.712936019897461,
1009
+ "step": 2580
1010
+ },
1011
+ {
1012
+ "epoch": 0.043815301651499834,
1013
+ "grad_norm": 1.640625,
1014
+ "learning_rate": 0.0003,
1015
+ "loss": 3.6954383850097656,
1016
+ "step": 2600
1017
+ },
1018
+ {
1019
+ "epoch": 0.043815301651499834,
1020
+ "eval_loss": 3.698262929916382,
1021
+ "eval_runtime": 7.6077,
1022
+ "eval_samples_per_second": 1252.285,
1023
+ "eval_steps_per_second": 0.92,
1024
+ "step": 2600
1025
+ },
1026
+ {
1027
+ "epoch": 0.044152342433434445,
1028
+ "grad_norm": 1.2890625,
1029
+ "learning_rate": 0.0003,
1030
+ "loss": 3.6608612060546877,
1031
+ "step": 2620
1032
+ },
1033
+ {
1034
+ "epoch": 0.044489383215369056,
1035
+ "grad_norm": 1.703125,
1036
+ "learning_rate": 0.0003,
1037
+ "loss": 3.68970947265625,
1038
+ "step": 2640
1039
+ },
1040
+ {
1041
+ "epoch": 0.044826423997303674,
1042
+ "grad_norm": 1.5234375,
1043
+ "learning_rate": 0.0003,
1044
+ "loss": 3.728267288208008,
1045
+ "step": 2660
1046
+ },
1047
+ {
1048
+ "epoch": 0.045163464779238285,
1049
+ "grad_norm": 1.59375,
1050
+ "learning_rate": 0.0003,
1051
+ "loss": 3.6732406616210938,
1052
+ "step": 2680
1053
+ },
1054
+ {
1055
+ "epoch": 0.0455005055611729,
1056
+ "grad_norm": 1.515625,
1057
+ "learning_rate": 0.0003,
1058
+ "loss": 3.679242706298828,
1059
+ "step": 2700
1060
+ },
1061
+ {
1062
+ "epoch": 0.045837546343107514,
1063
+ "grad_norm": 1.4921875,
1064
+ "learning_rate": 0.0003,
1065
+ "loss": 3.689365768432617,
1066
+ "step": 2720
1067
+ },
1068
+ {
1069
+ "epoch": 0.04617458712504213,
1070
+ "grad_norm": 1.3828125,
1071
+ "learning_rate": 0.0003,
1072
+ "loss": 3.6569408416748046,
1073
+ "step": 2740
1074
+ },
1075
+ {
1076
+ "epoch": 0.046511627906976744,
1077
+ "grad_norm": 1.8125,
1078
+ "learning_rate": 0.0003,
1079
+ "loss": 3.707513427734375,
1080
+ "step": 2760
1081
+ },
1082
+ {
1083
+ "epoch": 0.04684866868891136,
1084
+ "grad_norm": 1.828125,
1085
+ "learning_rate": 0.0003,
1086
+ "loss": 3.669821929931641,
1087
+ "step": 2780
1088
+ },
1089
+ {
1090
+ "epoch": 0.04718570947084597,
1091
+ "grad_norm": 1.4765625,
1092
+ "learning_rate": 0.0003,
1093
+ "loss": 3.652143859863281,
1094
+ "step": 2800
1095
+ },
1096
+ {
1097
+ "epoch": 0.04718570947084597,
1098
+ "eval_loss": 3.671437978744507,
1099
+ "eval_runtime": 7.4168,
1100
+ "eval_samples_per_second": 1284.522,
1101
+ "eval_steps_per_second": 0.944,
1102
+ "step": 2800
1103
+ },
1104
+ {
1105
+ "epoch": 0.047522750252780584,
1106
+ "grad_norm": 1.171875,
1107
+ "learning_rate": 0.0003,
1108
+ "loss": 3.684069061279297,
1109
+ "step": 2820
1110
+ },
1111
+ {
1112
+ "epoch": 0.0478597910347152,
1113
+ "grad_norm": 1.671875,
1114
+ "learning_rate": 0.0003,
1115
+ "loss": 3.6512351989746095,
1116
+ "step": 2840
1117
+ },
1118
+ {
1119
+ "epoch": 0.04819683181664981,
1120
+ "grad_norm": 1.8984375,
1121
+ "learning_rate": 0.0003,
1122
+ "loss": 3.6539031982421877,
1123
+ "step": 2860
1124
+ },
1125
+ {
1126
+ "epoch": 0.04853387259858443,
1127
+ "grad_norm": 1.46875,
1128
+ "learning_rate": 0.0003,
1129
+ "loss": 3.6580196380615235,
1130
+ "step": 2880
1131
+ },
1132
+ {
1133
+ "epoch": 0.04887091338051904,
1134
+ "grad_norm": 1.375,
1135
+ "learning_rate": 0.0003,
1136
+ "loss": 3.643080139160156,
1137
+ "step": 2900
1138
+ },
1139
+ {
1140
+ "epoch": 0.04920795416245366,
1141
+ "grad_norm": 1.8125,
1142
+ "learning_rate": 0.0003,
1143
+ "loss": 3.6346275329589846,
1144
+ "step": 2920
1145
+ },
1146
+ {
1147
+ "epoch": 0.04954499494438827,
1148
+ "grad_norm": 1.4921875,
1149
+ "learning_rate": 0.0003,
1150
+ "loss": 3.6610347747802736,
1151
+ "step": 2940
1152
+ },
1153
+ {
1154
+ "epoch": 0.04988203572632288,
1155
+ "grad_norm": 1.4453125,
1156
+ "learning_rate": 0.0003,
1157
+ "loss": 3.6628185272216798,
1158
+ "step": 2960
1159
+ },
1160
+ {
1161
+ "epoch": 0.0502190765082575,
1162
+ "grad_norm": 1.546875,
1163
+ "learning_rate": 0.0003,
1164
+ "loss": 3.6714599609375,
1165
+ "step": 2980
1166
+ },
1167
+ {
1168
+ "epoch": 0.05055611729019211,
1169
+ "grad_norm": 1.6796875,
1170
+ "learning_rate": 0.0003,
1171
+ "loss": 3.6391124725341797,
1172
+ "step": 3000
1173
+ },
1174
+ {
1175
+ "epoch": 0.05055611729019211,
1176
+ "eval_loss": 3.6535143852233887,
1177
+ "eval_runtime": 7.4227,
1178
+ "eval_samples_per_second": 1283.487,
1179
+ "eval_steps_per_second": 0.943,
1180
+ "step": 3000
1181
+ },
1182
+ {
1183
+ "epoch": 0.05089315807212673,
1184
+ "grad_norm": 1.453125,
1185
+ "learning_rate": 0.0003,
1186
+ "loss": 3.659154510498047,
1187
+ "step": 3020
1188
+ },
1189
+ {
1190
+ "epoch": 0.05123019885406134,
1191
+ "grad_norm": 1.6484375,
1192
+ "learning_rate": 0.0003,
1193
+ "loss": 3.6401947021484373,
1194
+ "step": 3040
1195
+ },
1196
+ {
1197
+ "epoch": 0.05156723963599596,
1198
+ "grad_norm": 1.4765625,
1199
+ "learning_rate": 0.0003,
1200
+ "loss": 3.67232666015625,
1201
+ "step": 3060
1202
+ },
1203
+ {
1204
+ "epoch": 0.05190428041793057,
1205
+ "grad_norm": 1.4296875,
1206
+ "learning_rate": 0.0003,
1207
+ "loss": 3.649647521972656,
1208
+ "step": 3080
1209
+ },
1210
+ {
1211
+ "epoch": 0.05224132119986518,
1212
+ "grad_norm": 1.796875,
1213
+ "learning_rate": 0.0003,
1214
+ "loss": 3.6890209197998045,
1215
+ "step": 3100
1216
+ },
1217
+ {
1218
+ "epoch": 0.0525783619817998,
1219
+ "grad_norm": 1.7109375,
1220
+ "learning_rate": 0.0003,
1221
+ "loss": 3.639459991455078,
1222
+ "step": 3120
1223
+ },
1224
+ {
1225
+ "epoch": 0.05291540276373441,
1226
+ "grad_norm": 1.828125,
1227
+ "learning_rate": 0.0003,
1228
+ "loss": 3.6333686828613283,
1229
+ "step": 3140
1230
+ },
1231
+ {
1232
+ "epoch": 0.05325244354566903,
1233
+ "grad_norm": 1.4609375,
1234
+ "learning_rate": 0.0003,
1235
+ "loss": 3.6446548461914063,
1236
+ "step": 3160
1237
+ },
1238
+ {
1239
+ "epoch": 0.05358948432760364,
1240
+ "grad_norm": 1.75,
1241
+ "learning_rate": 0.0003,
1242
+ "loss": 3.639226531982422,
1243
+ "step": 3180
1244
+ },
1245
+ {
1246
+ "epoch": 0.053926525109538256,
1247
+ "grad_norm": 1.3984375,
1248
+ "learning_rate": 0.0003,
1249
+ "loss": 3.6387962341308593,
1250
+ "step": 3200
1251
+ },
1252
+ {
1253
+ "epoch": 0.053926525109538256,
1254
+ "eval_loss": 3.6315455436706543,
1255
+ "eval_runtime": 7.6341,
1256
+ "eval_samples_per_second": 1247.949,
1257
+ "eval_steps_per_second": 0.917,
1258
+ "step": 3200
1259
+ },
1260
+ {
1261
+ "epoch": 0.05426356589147287,
1262
+ "grad_norm": 1.4296875,
1263
+ "learning_rate": 0.0003,
1264
+ "loss": 3.6562938690185547,
1265
+ "step": 3220
1266
+ },
1267
+ {
1268
+ "epoch": 0.054600606673407485,
1269
+ "grad_norm": 1.6640625,
1270
+ "learning_rate": 0.0003,
1271
+ "loss": 3.64786376953125,
1272
+ "step": 3240
1273
+ },
1274
+ {
1275
+ "epoch": 0.0549376474553421,
1276
+ "grad_norm": 1.890625,
1277
+ "learning_rate": 0.0003,
1278
+ "loss": 3.6485218048095702,
1279
+ "step": 3260
1280
+ },
1281
+ {
1282
+ "epoch": 0.05527468823727671,
1283
+ "grad_norm": 1.5,
1284
+ "learning_rate": 0.0003,
1285
+ "loss": 3.6148853302001953,
1286
+ "step": 3280
1287
+ },
1288
+ {
1289
+ "epoch": 0.055611729019211326,
1290
+ "grad_norm": 1.453125,
1291
+ "learning_rate": 0.0003,
1292
+ "loss": 3.6581153869628906,
1293
+ "step": 3300
1294
+ },
1295
+ {
1296
+ "epoch": 0.05594876980114594,
1297
+ "grad_norm": 1.5,
1298
+ "learning_rate": 0.0003,
1299
+ "loss": 3.6328590393066404,
1300
+ "step": 3320
1301
+ },
1302
+ {
1303
+ "epoch": 0.056285810583080555,
1304
+ "grad_norm": 1.8984375,
1305
+ "learning_rate": 0.0003,
1306
+ "loss": 3.608339309692383,
1307
+ "step": 3340
1308
+ },
1309
+ {
1310
+ "epoch": 0.056622851365015166,
1311
+ "grad_norm": 1.546875,
1312
+ "learning_rate": 0.0003,
1313
+ "loss": 3.6275489807128904,
1314
+ "step": 3360
1315
+ },
1316
+ {
1317
+ "epoch": 0.056959892146949784,
1318
+ "grad_norm": 1.4296875,
1319
+ "learning_rate": 0.0003,
1320
+ "loss": 3.5989017486572266,
1321
+ "step": 3380
1322
+ },
1323
+ {
1324
+ "epoch": 0.057296932928884395,
1325
+ "grad_norm": 1.2734375,
1326
+ "learning_rate": 0.0003,
1327
+ "loss": 3.6258113861083983,
1328
+ "step": 3400
1329
+ },
1330
+ {
1331
+ "epoch": 0.057296932928884395,
1332
+ "eval_loss": 3.6147406101226807,
1333
+ "eval_runtime": 7.5189,
1334
+ "eval_samples_per_second": 1267.079,
1335
+ "eval_steps_per_second": 0.931,
1336
+ "step": 3400
1337
+ }
1338
+ ],
1339
+ "logging_steps": 20,
1340
+ "max_steps": 5000,
1341
+ "num_input_tokens_seen": 0,
1342
+ "num_train_epochs": 1,
1343
+ "save_steps": 100,
1344
+ "stateful_callbacks": {
1345
+ "TrainerControl": {
1346
+ "args": {
1347
+ "should_epoch_stop": false,
1348
+ "should_evaluate": false,
1349
+ "should_log": false,
1350
+ "should_save": true,
1351
+ "should_training_stop": false
1352
+ },
1353
+ "attributes": {}
1354
+ }
1355
+ },
1356
+ "total_flos": 82290986188800.0,
1357
+ "train_batch_size": 16,
1358
+ "trial_name": null,
1359
+ "trial_params": null
1360
+ }
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
+ size 4920
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d0b9caf5aef6a40ddfc8b439193dcdf1f09e366ea4a9d962e91615e5601beacc
3
  size 2036216
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cdc56d71e15d5e56d0de962a47a3aead834194bc0b3094ae1086120c923a424a
3
  size 2036216
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ff081781d005f5468b429863b16218d3d6c579e354919efcd3aff94c2d14c009
3
  size 4089360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2c19e49647f886d882d4653e3b6276951ba8cf04c7c237042f3b55328aa31393
3
  size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3cf9097d4513154245c48236b6ec5137b7ee2a21c9f58f2cba798ea275c6026f
3
  size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f768f3dc231ac5aded23c6f019b283714352126180577c2c811e2d205bc7704f
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9cd73bda6f3e39fc4545a8946a8d87b16f2d59ac45ad626e25c0e3ac41ad3e7e
3
  size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/trainer_state.json CHANGED
@@ -2,188 +2,172 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.013481631277384564,
6
- "eval_steps": 100,
7
  "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.4921875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.323189544677735,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.609375,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.292604064941406,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.5234375,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.182575988769532,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.296875,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.984093475341797,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.265625,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.79156265258789,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.687252998352051,
50
- "eval_runtime": 7.4625,
51
- "eval_samples_per_second": 1276.648,
52
- "eval_steps_per_second": 0.938,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.25,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.586382293701172,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.356954193115234,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.21875,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.125580596923828,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.1953125,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.8821556091308596,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.125,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.638446807861328,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.517786502838135,
93
- "eval_runtime": 7.4364,
94
- "eval_samples_per_second": 1281.124,
95
- "eval_steps_per_second": 0.941,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.1171875,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.419033813476562,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 0.99609375,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.19798698425293,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 0.85546875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.01253662109375,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 2.4375,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.827148818969727,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.1171875,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.6793663024902346,
131
- "step": 300
132
- },
133
- {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.60944938659668,
136
- "eval_runtime": 7.4311,
137
- "eval_samples_per_second": 1282.036,
138
- "eval_steps_per_second": 0.942,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 0.9375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.544954299926758,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.2734375,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.418224334716797,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.890625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.2604835510253904,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.265625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.145714187622071,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.6328125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.026620101928711,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 4.965348720550537,
179
- "eval_runtime": 7.4478,
180
- "eval_samples_per_second": 1279.165,
181
- "eval_steps_per_second": 0.94,
182
  "step": 400
183
  }
184
  ],
185
  "logging_steps": 20,
186
- "max_steps": 2500,
187
  "num_input_tokens_seen": 0,
188
  "num_train_epochs": 1,
189
  "save_steps": 100,
@@ -199,8 +183,8 @@
199
  "attributes": {}
200
  }
201
  },
202
- "total_flos": 19362584985600.0,
203
- "train_batch_size": 32,
204
  "trial_name": null,
205
  "trial_params": null
206
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.006740815638692282,
6
+ "eval_steps": 200,
7
  "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
 
 
 
 
 
 
 
 
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
  "step": 120
53
  },
54
  {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
  "step": 140
60
  },
61
  {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
  "step": 160
67
  },
68
  {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
  "step": 180
74
  },
75
  {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
  "step": 200
81
  },
82
  {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
  "step": 220
96
  },
97
  {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
  "step": 240
103
  },
104
  {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
  "step": 260
110
  },
111
  {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
  "step": 280
117
  },
118
  {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
 
 
 
 
 
 
 
 
123
  "step": 300
124
  },
125
  {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
130
  "step": 320
131
  },
132
  {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
  "step": 340
138
  },
139
  {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
  "step": 360
145
  },
146
  {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
  "step": 380
152
  },
153
  {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
  "step": 400
159
  },
160
  {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
  "step": 400
167
  }
168
  ],
169
  "logging_steps": 20,
170
+ "max_steps": 5000,
171
  "num_input_tokens_seen": 0,
172
  "num_train_epochs": 1,
173
  "save_steps": 100,
 
183
  "attributes": {}
184
  }
185
  },
186
+ "total_flos": 9681292492800.0,
187
+ "train_batch_size": 16,
188
  "trial_name": null,
189
  "trial_params": null
190
  }
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e2e1d330457822673a5ced171f375887b3e5542845ed4294676608daea03e08
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
  size 4920
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e73b2c3bc8f988935078161f373d02138a5db1821bc1738752c255eb080e88a
3
  size 2036216
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8268e70c42413ec7b95cd060155f41891df1ef6759c56990157f3f55c74f345a
3
  size 2036216
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ac22ed2b649796ed9e0bf0c02e1f42772e98eaf4a3523f019df0fd3959e496bc
3
  size 4089360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fa6f3c8b6645828936657c806f68d113379205778239766bd5d55e26ef1fda85
3
  size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:95b6047bd8cc6f4cdf7c46dea47edb8e542435510070c6cd1e0a7d9ccf5fd7da
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3cf9097d4513154245c48236b6ec5137b7ee2a21c9f58f2cba798ea275c6026f
3
  size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7aade0a5cf98168c48557061a263a0ce94127833f42effbb1795ad03b5bd88a5
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2627ccec0bb9a51b7d9d753a9441035aec88f305994eb3b5ccbb3e0571f519d6
3
  size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/trainer_state.json CHANGED
@@ -2,231 +2,207 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.016852039096730706,
6
- "eval_steps": 100,
7
  "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.4921875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.323189544677735,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.609375,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.292604064941406,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.5234375,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.182575988769532,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.296875,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.984093475341797,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.265625,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.79156265258789,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.687252998352051,
50
- "eval_runtime": 7.4625,
51
- "eval_samples_per_second": 1276.648,
52
- "eval_steps_per_second": 0.938,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.25,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.586382293701172,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.356954193115234,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.21875,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.125580596923828,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.1953125,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.8821556091308596,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.125,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.638446807861328,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.517786502838135,
93
- "eval_runtime": 7.4364,
94
- "eval_samples_per_second": 1281.124,
95
- "eval_steps_per_second": 0.941,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.1171875,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.419033813476562,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 0.99609375,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.19798698425293,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 0.85546875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.01253662109375,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 2.4375,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.827148818969727,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.1171875,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.6793663024902346,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.60944938659668,
136
- "eval_runtime": 7.4311,
137
- "eval_samples_per_second": 1282.036,
138
- "eval_steps_per_second": 0.942,
139
- "step": 300
140
- },
141
- {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 0.9375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.544954299926758,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.2734375,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.418224334716797,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.890625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.2604835510253904,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.265625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.145714187622071,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.6328125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.026620101928711,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 4.965348720550537,
179
- "eval_runtime": 7.4478,
180
- "eval_samples_per_second": 1279.165,
181
- "eval_steps_per_second": 0.94,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 0.96484375,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.905144500732422,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.953125,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.836254501342774,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.90625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.767382431030273,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.6875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.681509017944336,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.609375,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.624195098876953,
217
- "step": 500
218
- },
219
- {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.591919898986816,
222
- "eval_runtime": 7.3895,
223
- "eval_samples_per_second": 1289.263,
224
- "eval_steps_per_second": 0.947,
225
  "step": 500
226
  }
227
  ],
228
  "logging_steps": 20,
229
- "max_steps": 2500,
230
  "num_input_tokens_seen": 0,
231
  "num_train_epochs": 1,
232
  "save_steps": 100,
@@ -242,8 +218,8 @@
242
  "attributes": {}
243
  }
244
  },
245
- "total_flos": 24203231232000.0,
246
- "train_batch_size": 32,
247
  "trial_name": null,
248
  "trial_params": null
249
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.008426019548365353,
6
+ "eval_steps": 200,
7
  "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
 
 
 
 
 
 
 
 
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
  "step": 120
53
  },
54
  {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
  "step": 140
60
  },
61
  {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
  "step": 160
67
  },
68
  {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
  "step": 180
74
  },
75
  {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
  "step": 200
81
  },
82
  {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
  "step": 220
96
  },
97
  {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
  "step": 240
103
  },
104
  {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
  "step": 260
110
  },
111
  {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
  "step": 280
117
  },
118
  {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
123
  "step": 300
124
  },
125
  {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
 
 
 
 
 
 
 
 
130
  "step": 320
131
  },
132
  {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
  "step": 340
138
  },
139
  {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
  "step": 360
145
  },
146
  {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
  "step": 380
152
  },
153
  {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
  "step": 400
159
  },
160
  {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
  "step": 400
167
  },
168
  {
169
+ "epoch": 0.007077856420626896,
170
+ "grad_norm": 0.98046875,
171
+ "learning_rate": 0.0001257,
172
+ "loss": 5.9373779296875,
173
  "step": 420
174
  },
175
  {
176
+ "epoch": 0.00741489720256151,
177
+ "grad_norm": 1.6328125,
178
+ "learning_rate": 0.00013169999999999998,
179
+ "loss": 5.839211273193359,
180
  "step": 440
181
  },
182
  {
183
+ "epoch": 0.007751937984496124,
184
+ "grad_norm": 0.9609375,
185
+ "learning_rate": 0.00013769999999999999,
186
+ "loss": 5.740922927856445,
187
  "step": 460
188
  },
189
  {
190
+ "epoch": 0.008088978766430738,
191
+ "grad_norm": 0.90234375,
192
+ "learning_rate": 0.00014369999999999997,
193
+ "loss": 5.6399181365966795,
194
  "step": 480
195
  },
196
  {
197
+ "epoch": 0.008426019548365353,
198
+ "grad_norm": 1.1796875,
199
+ "learning_rate": 0.00014969999999999998,
200
+ "loss": 5.560699081420898,
 
 
 
 
 
 
 
 
201
  "step": 500
202
  }
203
  ],
204
  "logging_steps": 20,
205
+ "max_steps": 5000,
206
  "num_input_tokens_seen": 0,
207
  "num_train_epochs": 1,
208
  "save_steps": 100,
 
218
  "attributes": {}
219
  }
220
  },
221
+ "total_flos": 12101615616000.0,
222
+ "train_batch_size": 16,
223
  "trial_name": null,
224
  "trial_params": null
225
  }
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e2e1d330457822673a5ced171f375887b3e5542845ed4294676608daea03e08
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
  size 4920
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:052e01594a5efbc459df4083494e99db95a4271d0f71e51553a63936fea9e719
3
  size 2036216
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ab2e2cfacb38b4d1367fa6d371a767c6bb149e9ca0f5bc2e0a9bd8afc9ba2f9
3
  size 2036216
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b7fe3387c7955d6ca268acb4e8e5b16749a1a9ac5d9987c50f9a99fb42e20b64
3
  size 4089360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:973f768bdaf58bb9c10b90ad211f2d0f7fb6bec95950be6f045639f355cb72cb
3
  size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:068fbd993087219c15b8c0baa13fc39644a4dcdfe92d8be3fa6434deece90371
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f37c40ce327861a7ca13b719d3aa37510a143368b6e74358bdb14becb3899e1e
3
  size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:295d5fdc0e0bbae04d3db91b9c6051a7cccac49c004495947e95dcd6c59adda3
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:44937bb66e00fa484302cf89e685ffa903e6b6f2eac5fdae0a7e91e0560ce411
3
  size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/trainer_state.json CHANGED
@@ -2,274 +2,250 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.020222446916076844,
6
- "eval_steps": 100,
7
  "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.4921875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.323189544677735,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.609375,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.292604064941406,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.5234375,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.182575988769532,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.296875,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.984093475341797,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.265625,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.79156265258789,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.687252998352051,
50
- "eval_runtime": 7.4625,
51
- "eval_samples_per_second": 1276.648,
52
- "eval_steps_per_second": 0.938,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.25,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.586382293701172,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.356954193115234,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.21875,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.125580596923828,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.1953125,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.8821556091308596,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.125,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.638446807861328,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.517786502838135,
93
- "eval_runtime": 7.4364,
94
- "eval_samples_per_second": 1281.124,
95
- "eval_steps_per_second": 0.941,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.1171875,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.419033813476562,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 0.99609375,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.19798698425293,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 0.85546875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.01253662109375,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 2.4375,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.827148818969727,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.1171875,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.6793663024902346,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.60944938659668,
136
- "eval_runtime": 7.4311,
137
- "eval_samples_per_second": 1282.036,
138
- "eval_steps_per_second": 0.942,
139
- "step": 300
140
- },
141
- {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 0.9375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.544954299926758,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.2734375,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.418224334716797,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.890625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.2604835510253904,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.265625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.145714187622071,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.6328125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.026620101928711,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 4.965348720550537,
179
- "eval_runtime": 7.4478,
180
- "eval_samples_per_second": 1279.165,
181
- "eval_steps_per_second": 0.94,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 0.96484375,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.905144500732422,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.953125,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.836254501342774,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.90625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.767382431030273,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.6875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.681509017944336,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.609375,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.624195098876953,
217
- "step": 500
218
- },
219
- {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.591919898986816,
222
- "eval_runtime": 7.3895,
223
- "eval_samples_per_second": 1289.263,
224
- "eval_steps_per_second": 0.947,
225
  "step": 500
226
  },
227
  {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.359375,
230
- "learning_rate": 0.0003,
231
- "loss": 4.573441696166992,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.890625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.496570587158203,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.6796875,
244
- "learning_rate": 0.0003,
245
- "loss": 4.437310409545899,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.953125,
251
- "learning_rate": 0.0003,
252
- "loss": 4.387492752075195,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.375,
258
- "learning_rate": 0.0003,
259
- "loss": 4.346551513671875,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.322638034820557,
265
- "eval_runtime": 7.5459,
266
- "eval_samples_per_second": 1262.548,
267
- "eval_steps_per_second": 0.928,
268
  "step": 600
269
  }
270
  ],
271
  "logging_steps": 20,
272
- "max_steps": 2500,
273
  "num_input_tokens_seen": 0,
274
  "num_train_epochs": 1,
275
  "save_steps": 100,
@@ -285,8 +261,8 @@
285
  "attributes": {}
286
  }
287
  },
288
- "total_flos": 29043877478400.0,
289
- "train_batch_size": 32,
290
  "trial_name": null,
291
  "trial_params": null
292
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.010111223458038422,
6
+ "eval_steps": 200,
7
  "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
 
 
 
 
 
 
 
 
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
  "step": 120
53
  },
54
  {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
  "step": 140
60
  },
61
  {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
  "step": 160
67
  },
68
  {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
  "step": 180
74
  },
75
  {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
  "step": 200
81
  },
82
  {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
  "step": 220
96
  },
97
  {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
  "step": 240
103
  },
104
  {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
  "step": 260
110
  },
111
  {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
  "step": 280
117
  },
118
  {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
123
  "step": 300
124
  },
125
  {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
 
 
 
 
 
 
 
 
130
  "step": 320
131
  },
132
  {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
  "step": 340
138
  },
139
  {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
  "step": 360
145
  },
146
  {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
  "step": 380
152
  },
153
  {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
  "step": 400
159
  },
160
  {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
  "step": 400
167
  },
168
  {
169
+ "epoch": 0.007077856420626896,
170
+ "grad_norm": 0.98046875,
171
+ "learning_rate": 0.0001257,
172
+ "loss": 5.9373779296875,
173
  "step": 420
174
  },
175
  {
176
+ "epoch": 0.00741489720256151,
177
+ "grad_norm": 1.6328125,
178
+ "learning_rate": 0.00013169999999999998,
179
+ "loss": 5.839211273193359,
180
  "step": 440
181
  },
182
  {
183
+ "epoch": 0.007751937984496124,
184
+ "grad_norm": 0.9609375,
185
+ "learning_rate": 0.00013769999999999999,
186
+ "loss": 5.740922927856445,
187
  "step": 460
188
  },
189
  {
190
+ "epoch": 0.008088978766430738,
191
+ "grad_norm": 0.90234375,
192
+ "learning_rate": 0.00014369999999999997,
193
+ "loss": 5.6399181365966795,
194
  "step": 480
195
  },
196
  {
197
+ "epoch": 0.008426019548365353,
198
+ "grad_norm": 1.1796875,
199
+ "learning_rate": 0.00014969999999999998,
200
+ "loss": 5.560699081420898,
 
 
 
 
 
 
 
 
201
  "step": 500
202
  },
203
  {
204
+ "epoch": 0.008763060330299966,
205
+ "grad_norm": 2.328125,
206
+ "learning_rate": 0.0001557,
207
+ "loss": 5.474863433837891,
208
  "step": 520
209
  },
210
  {
211
+ "epoch": 0.00910010111223458,
212
+ "grad_norm": 1.125,
213
+ "learning_rate": 0.0001617,
214
+ "loss": 5.396588516235352,
215
  "step": 540
216
  },
217
  {
218
+ "epoch": 0.009437141894169195,
219
+ "grad_norm": 1.6484375,
220
+ "learning_rate": 0.0001677,
221
+ "loss": 5.331023406982422,
222
  "step": 560
223
  },
224
  {
225
+ "epoch": 0.00977418267610381,
226
+ "grad_norm": 1.0703125,
227
+ "learning_rate": 0.00017369999999999997,
228
+ "loss": 5.257175445556641,
229
  "step": 580
230
  },
231
  {
232
+ "epoch": 0.010111223458038422,
233
+ "grad_norm": 2.359375,
234
+ "learning_rate": 0.00017969999999999998,
235
+ "loss": 5.152382659912109,
236
  "step": 600
237
  },
238
  {
239
+ "epoch": 0.010111223458038422,
240
+ "eval_loss": 5.1175713539123535,
241
+ "eval_runtime": 7.457,
242
+ "eval_samples_per_second": 1277.585,
243
+ "eval_steps_per_second": 0.939,
244
  "step": 600
245
  }
246
  ],
247
  "logging_steps": 20,
248
+ "max_steps": 5000,
249
  "num_input_tokens_seen": 0,
250
  "num_train_epochs": 1,
251
  "save_steps": 100,
 
261
  "attributes": {}
262
  }
263
  },
264
+ "total_flos": 14521938739200.0,
265
+ "train_batch_size": 16,
266
  "trial_name": null,
267
  "trial_params": null
268
  }
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e2e1d330457822673a5ced171f375887b3e5542845ed4294676608daea03e08
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
  size 4920
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3b46169a4694deeba0ed3f14f47e1c4783069b469ab882c14eda5d272d0302cf
3
  size 2036216
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d88313a11f98633534da3abe872a0a8c638d686b7802d4bc680c383765235f92
3
  size 2036216
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c538c2c5a635326bdb569159284adf0f6769a5ee266d2068f80cf0cbed282562
3
  size 4089360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:171503e70341506b7b06a73f6ad11a065570acb4569cdd00504cb10a5893085b
3
  size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2768285b45b2a0c05f6f50bbb8c0287fca6f62a8cde6d1b1f02151ac72ee8dc
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f37c40ce327861a7ca13b719d3aa37510a143368b6e74358bdb14becb3899e1e
3
  size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d6c38b4e6128ebabd81a22bafe942090015968e75f90e5e9ef25cd5ebdef8357
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:00ed284532aedfadb0830a4afe3ecb31332daef31ec310763e778ba54396ad39
3
  size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/trainer_state.json CHANGED
@@ -2,317 +2,285 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.023592854735422986,
6
- "eval_steps": 100,
7
  "global_step": 700,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.4921875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.323189544677735,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.609375,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.292604064941406,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.5234375,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.182575988769532,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.296875,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.984093475341797,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.265625,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.79156265258789,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.687252998352051,
50
- "eval_runtime": 7.4625,
51
- "eval_samples_per_second": 1276.648,
52
- "eval_steps_per_second": 0.938,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.25,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.586382293701172,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.356954193115234,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.21875,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.125580596923828,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.1953125,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.8821556091308596,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.125,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.638446807861328,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.517786502838135,
93
- "eval_runtime": 7.4364,
94
- "eval_samples_per_second": 1281.124,
95
- "eval_steps_per_second": 0.941,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.1171875,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.419033813476562,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 0.99609375,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.19798698425293,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 0.85546875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.01253662109375,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 2.4375,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.827148818969727,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.1171875,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.6793663024902346,
131
- "step": 300
132
- },
133
- {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.60944938659668,
136
- "eval_runtime": 7.4311,
137
- "eval_samples_per_second": 1282.036,
138
- "eval_steps_per_second": 0.942,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 0.9375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.544954299926758,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.2734375,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.418224334716797,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.890625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.2604835510253904,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.265625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.145714187622071,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.6328125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.026620101928711,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 4.965348720550537,
179
- "eval_runtime": 7.4478,
180
- "eval_samples_per_second": 1279.165,
181
- "eval_steps_per_second": 0.94,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 0.96484375,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.905144500732422,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.953125,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.836254501342774,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.90625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.767382431030273,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.6875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.681509017944336,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.609375,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.624195098876953,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.591919898986816,
222
- "eval_runtime": 7.3895,
223
- "eval_samples_per_second": 1289.263,
224
- "eval_steps_per_second": 0.947,
225
- "step": 500
226
- },
227
- {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.359375,
230
- "learning_rate": 0.0003,
231
- "loss": 4.573441696166992,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.890625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.496570587158203,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.6796875,
244
- "learning_rate": 0.0003,
245
- "loss": 4.437310409545899,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.953125,
251
- "learning_rate": 0.0003,
252
- "loss": 4.387492752075195,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.375,
258
- "learning_rate": 0.0003,
259
- "loss": 4.346551513671875,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.322638034820557,
265
- "eval_runtime": 7.5459,
266
- "eval_samples_per_second": 1262.548,
267
- "eval_steps_per_second": 0.928,
268
  "step": 600
269
  },
270
  {
271
- "epoch": 0.020896528479946073,
272
- "grad_norm": 1.5078125,
273
- "learning_rate": 0.0003,
274
- "loss": 4.285432815551758,
275
  "step": 620
276
  },
277
  {
278
- "epoch": 0.021570610043815303,
279
- "grad_norm": 1.40625,
280
- "learning_rate": 0.0003,
281
- "loss": 4.2501472473144535,
282
  "step": 640
283
  },
284
  {
285
- "epoch": 0.022244691607684528,
286
- "grad_norm": 1.7421875,
287
- "learning_rate": 0.0003,
288
- "loss": 4.211347579956055,
289
  "step": 660
290
  },
291
  {
292
- "epoch": 0.022918773171553757,
293
- "grad_norm": 2.09375,
294
- "learning_rate": 0.0003,
295
- "loss": 4.168280410766601,
296
  "step": 680
297
  },
298
  {
299
- "epoch": 0.023592854735422986,
300
- "grad_norm": 1.734375,
301
- "learning_rate": 0.0003,
302
- "loss": 4.165232086181641,
303
- "step": 700
304
- },
305
- {
306
- "epoch": 0.023592854735422986,
307
- "eval_loss": 4.144440650939941,
308
- "eval_runtime": 7.5426,
309
- "eval_samples_per_second": 1263.095,
310
- "eval_steps_per_second": 0.928,
311
  "step": 700
312
  }
313
  ],
314
  "logging_steps": 20,
315
- "max_steps": 2500,
316
  "num_input_tokens_seen": 0,
317
  "num_train_epochs": 1,
318
  "save_steps": 100,
@@ -328,8 +296,8 @@
328
  "attributes": {}
329
  }
330
  },
331
- "total_flos": 33884523724800.0,
332
- "train_batch_size": 32,
333
  "trial_name": null,
334
  "trial_params": null
335
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.011796427367711493,
6
+ "eval_steps": 200,
7
  "global_step": 700,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
 
 
 
 
 
 
 
 
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
  "step": 120
53
  },
54
  {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
  "step": 140
60
  },
61
  {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
  "step": 160
67
  },
68
  {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
  "step": 180
74
  },
75
  {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
  "step": 200
81
  },
82
  {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
  "step": 220
96
  },
97
  {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
  "step": 240
103
  },
104
  {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
  "step": 260
110
  },
111
  {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
  "step": 280
117
  },
118
  {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
 
 
 
 
 
 
 
 
123
  "step": 300
124
  },
125
  {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
130
  "step": 320
131
  },
132
  {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
  "step": 340
138
  },
139
  {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
  "step": 360
145
  },
146
  {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
  "step": 380
152
  },
153
  {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
  "step": 400
159
  },
160
  {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
  "step": 400
167
  },
168
  {
169
+ "epoch": 0.007077856420626896,
170
+ "grad_norm": 0.98046875,
171
+ "learning_rate": 0.0001257,
172
+ "loss": 5.9373779296875,
173
  "step": 420
174
  },
175
  {
176
+ "epoch": 0.00741489720256151,
177
+ "grad_norm": 1.6328125,
178
+ "learning_rate": 0.00013169999999999998,
179
+ "loss": 5.839211273193359,
180
  "step": 440
181
  },
182
  {
183
+ "epoch": 0.007751937984496124,
184
+ "grad_norm": 0.9609375,
185
+ "learning_rate": 0.00013769999999999999,
186
+ "loss": 5.740922927856445,
187
  "step": 460
188
  },
189
  {
190
+ "epoch": 0.008088978766430738,
191
+ "grad_norm": 0.90234375,
192
+ "learning_rate": 0.00014369999999999997,
193
+ "loss": 5.6399181365966795,
194
  "step": 480
195
  },
196
  {
197
+ "epoch": 0.008426019548365353,
198
+ "grad_norm": 1.1796875,
199
+ "learning_rate": 0.00014969999999999998,
200
+ "loss": 5.560699081420898,
201
  "step": 500
202
  },
203
  {
204
+ "epoch": 0.008763060330299966,
205
+ "grad_norm": 2.328125,
206
+ "learning_rate": 0.0001557,
207
+ "loss": 5.474863433837891,
 
 
 
 
 
 
 
 
208
  "step": 520
209
  },
210
  {
211
+ "epoch": 0.00910010111223458,
212
+ "grad_norm": 1.125,
213
+ "learning_rate": 0.0001617,
214
+ "loss": 5.396588516235352,
215
  "step": 540
216
  },
217
  {
218
+ "epoch": 0.009437141894169195,
219
+ "grad_norm": 1.6484375,
220
+ "learning_rate": 0.0001677,
221
+ "loss": 5.331023406982422,
222
  "step": 560
223
  },
224
  {
225
+ "epoch": 0.00977418267610381,
226
+ "grad_norm": 1.0703125,
227
+ "learning_rate": 0.00017369999999999997,
228
+ "loss": 5.257175445556641,
229
  "step": 580
230
  },
231
  {
232
+ "epoch": 0.010111223458038422,
233
+ "grad_norm": 2.359375,
234
+ "learning_rate": 0.00017969999999999998,
235
+ "loss": 5.152382659912109,
236
  "step": 600
237
  },
238
  {
239
+ "epoch": 0.010111223458038422,
240
+ "eval_loss": 5.1175713539123535,
241
+ "eval_runtime": 7.457,
242
+ "eval_samples_per_second": 1277.585,
243
+ "eval_steps_per_second": 0.939,
244
  "step": 600
245
  },
246
  {
247
+ "epoch": 0.010448264239973037,
248
+ "grad_norm": 1.375,
249
+ "learning_rate": 0.0001857,
250
+ "loss": 5.096985244750977,
251
  "step": 620
252
  },
253
  {
254
+ "epoch": 0.010785305021907651,
255
+ "grad_norm": 1.421875,
256
+ "learning_rate": 0.0001917,
257
+ "loss": 5.019801330566406,
258
  "step": 640
259
  },
260
  {
261
+ "epoch": 0.011122345803842264,
262
+ "grad_norm": 2.125,
263
+ "learning_rate": 0.00019769999999999998,
264
+ "loss": 4.979957962036133,
265
  "step": 660
266
  },
267
  {
268
+ "epoch": 0.011459386585776879,
269
+ "grad_norm": 2.203125,
270
+ "learning_rate": 0.0002037,
271
+ "loss": 4.945652008056641,
272
  "step": 680
273
  },
274
  {
275
+ "epoch": 0.011796427367711493,
276
+ "grad_norm": 1.8515625,
277
+ "learning_rate": 0.00020969999999999997,
278
+ "loss": 4.866051483154297,
 
 
 
 
 
 
 
 
279
  "step": 700
280
  }
281
  ],
282
  "logging_steps": 20,
283
+ "max_steps": 5000,
284
  "num_input_tokens_seen": 0,
285
  "num_train_epochs": 1,
286
  "save_steps": 100,
 
296
  "attributes": {}
297
  }
298
  },
299
+ "total_flos": 16942261862400.0,
300
+ "train_batch_size": 16,
301
  "trial_name": null,
302
  "trial_params": null
303
  }
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e2e1d330457822673a5ced171f375887b3e5542845ed4294676608daea03e08
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
  size 4920
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0b7bef8bb399a03156a1b06f6ececb151c8acdf62e104b42a72dccb0a276a4fc
3
  size 2036216
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0edb3ed95464c61de67c9cffe7938ede900e73d8fafa0983797e3d5ec92328f7
3
  size 2036216
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:07e8ade78fde459dc7325659078da9e3dd59c4f5fb4577e30567d577d0a4185d
3
  size 4089360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:522169e9d2be0062e3eda5870ee10162fec7ef76f98244b4288d242b0044761a
3
  size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f9a6944c405a002fce05f295d08ea6650e2e2ad6dbf5d6da1e9053f7bf7f5827
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
3
  size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2224470b783e3f05273b6a4ac609fa6ba76b1099ae64a843a85fea2d2bf1e0dd
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c39c131425d66759f6276fee23b3a54e5b3da37f6f8ee3949f69449c73bf15ee
3
  size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/trainer_state.json CHANGED
@@ -2,360 +2,328 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.026963262554769128,
6
- "eval_steps": 100,
7
  "global_step": 800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.4921875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.323189544677735,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.609375,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.292604064941406,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.5234375,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.182575988769532,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.296875,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.984093475341797,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.265625,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.79156265258789,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.687252998352051,
50
- "eval_runtime": 7.4625,
51
- "eval_samples_per_second": 1276.648,
52
- "eval_steps_per_second": 0.938,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.25,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.586382293701172,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.356954193115234,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.21875,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.125580596923828,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.1953125,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.8821556091308596,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.125,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.638446807861328,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.517786502838135,
93
- "eval_runtime": 7.4364,
94
- "eval_samples_per_second": 1281.124,
95
- "eval_steps_per_second": 0.941,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.1171875,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.419033813476562,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 0.99609375,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.19798698425293,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 0.85546875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.01253662109375,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 2.4375,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.827148818969727,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.1171875,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.6793663024902346,
131
- "step": 300
132
- },
133
- {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.60944938659668,
136
- "eval_runtime": 7.4311,
137
- "eval_samples_per_second": 1282.036,
138
- "eval_steps_per_second": 0.942,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 0.9375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.544954299926758,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.2734375,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.418224334716797,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.890625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.2604835510253904,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.265625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.145714187622071,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.6328125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.026620101928711,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 4.965348720550537,
179
- "eval_runtime": 7.4478,
180
- "eval_samples_per_second": 1279.165,
181
- "eval_steps_per_second": 0.94,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 0.96484375,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.905144500732422,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.953125,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.836254501342774,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.90625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.767382431030273,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.6875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.681509017944336,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.609375,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.624195098876953,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.591919898986816,
222
- "eval_runtime": 7.3895,
223
- "eval_samples_per_second": 1289.263,
224
- "eval_steps_per_second": 0.947,
225
- "step": 500
226
- },
227
- {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.359375,
230
- "learning_rate": 0.0003,
231
- "loss": 4.573441696166992,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.890625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.496570587158203,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.6796875,
244
- "learning_rate": 0.0003,
245
- "loss": 4.437310409545899,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.953125,
251
- "learning_rate": 0.0003,
252
- "loss": 4.387492752075195,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.375,
258
- "learning_rate": 0.0003,
259
- "loss": 4.346551513671875,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.322638034820557,
265
- "eval_runtime": 7.5459,
266
- "eval_samples_per_second": 1262.548,
267
- "eval_steps_per_second": 0.928,
268
  "step": 600
269
  },
270
  {
271
- "epoch": 0.020896528479946073,
272
- "grad_norm": 1.5078125,
273
- "learning_rate": 0.0003,
274
- "loss": 4.285432815551758,
275
  "step": 620
276
  },
277
  {
278
- "epoch": 0.021570610043815303,
279
- "grad_norm": 1.40625,
280
- "learning_rate": 0.0003,
281
- "loss": 4.2501472473144535,
282
  "step": 640
283
  },
284
  {
285
- "epoch": 0.022244691607684528,
286
- "grad_norm": 1.7421875,
287
- "learning_rate": 0.0003,
288
- "loss": 4.211347579956055,
289
  "step": 660
290
  },
291
  {
292
- "epoch": 0.022918773171553757,
293
- "grad_norm": 2.09375,
294
- "learning_rate": 0.0003,
295
- "loss": 4.168280410766601,
296
  "step": 680
297
  },
298
  {
299
- "epoch": 0.023592854735422986,
300
- "grad_norm": 1.734375,
301
- "learning_rate": 0.0003,
302
- "loss": 4.165232086181641,
303
  "step": 700
304
  },
305
  {
306
- "epoch": 0.023592854735422986,
307
- "eval_loss": 4.144440650939941,
308
- "eval_runtime": 7.5426,
309
- "eval_samples_per_second": 1263.095,
310
- "eval_steps_per_second": 0.928,
311
- "step": 700
312
- },
313
- {
314
- "epoch": 0.024266936299292215,
315
- "grad_norm": 0.98828125,
316
- "learning_rate": 0.0003,
317
- "loss": 4.132168579101562,
318
  "step": 720
319
  },
320
  {
321
- "epoch": 0.02494101786316144,
322
- "grad_norm": 1.390625,
323
- "learning_rate": 0.0003,
324
- "loss": 4.118199157714844,
325
  "step": 740
326
  },
327
  {
328
- "epoch": 0.02561509942703067,
329
- "grad_norm": 2.0625,
330
- "learning_rate": 0.0003,
331
- "loss": 4.074006271362305,
332
  "step": 760
333
  },
334
  {
335
- "epoch": 0.0262891809908999,
336
- "grad_norm": 1.46875,
337
- "learning_rate": 0.0003,
338
- "loss": 4.034224319458008,
339
  "step": 780
340
  },
341
  {
342
- "epoch": 0.026963262554769128,
343
- "grad_norm": 1.28125,
344
- "learning_rate": 0.0003,
345
- "loss": 4.037029266357422,
346
  "step": 800
347
  },
348
  {
349
- "epoch": 0.026963262554769128,
350
- "eval_loss": 4.028254508972168,
351
- "eval_runtime": 7.524,
352
- "eval_samples_per_second": 1266.218,
353
- "eval_steps_per_second": 0.93,
354
  "step": 800
355
  }
356
  ],
357
  "logging_steps": 20,
358
- "max_steps": 2500,
359
  "num_input_tokens_seen": 0,
360
  "num_train_epochs": 1,
361
  "save_steps": 100,
@@ -371,8 +339,8 @@
371
  "attributes": {}
372
  }
373
  },
374
- "total_flos": 38725169971200.0,
375
- "train_batch_size": 32,
376
  "trial_name": null,
377
  "trial_params": null
378
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.013481631277384564,
6
+ "eval_steps": 200,
7
  "global_step": 800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
 
 
 
 
 
 
 
 
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
  "step": 120
53
  },
54
  {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
  "step": 140
60
  },
61
  {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
  "step": 160
67
  },
68
  {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
  "step": 180
74
  },
75
  {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
  "step": 200
81
  },
82
  {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
  "step": 220
96
  },
97
  {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
  "step": 240
103
  },
104
  {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
  "step": 260
110
  },
111
  {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
  "step": 280
117
  },
118
  {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
 
 
 
 
 
 
 
 
123
  "step": 300
124
  },
125
  {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
130
  "step": 320
131
  },
132
  {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
  "step": 340
138
  },
139
  {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
  "step": 360
145
  },
146
  {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
  "step": 380
152
  },
153
  {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
  "step": 400
159
  },
160
  {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
  "step": 400
167
  },
168
  {
169
+ "epoch": 0.007077856420626896,
170
+ "grad_norm": 0.98046875,
171
+ "learning_rate": 0.0001257,
172
+ "loss": 5.9373779296875,
173
  "step": 420
174
  },
175
  {
176
+ "epoch": 0.00741489720256151,
177
+ "grad_norm": 1.6328125,
178
+ "learning_rate": 0.00013169999999999998,
179
+ "loss": 5.839211273193359,
180
  "step": 440
181
  },
182
  {
183
+ "epoch": 0.007751937984496124,
184
+ "grad_norm": 0.9609375,
185
+ "learning_rate": 0.00013769999999999999,
186
+ "loss": 5.740922927856445,
187
  "step": 460
188
  },
189
  {
190
+ "epoch": 0.008088978766430738,
191
+ "grad_norm": 0.90234375,
192
+ "learning_rate": 0.00014369999999999997,
193
+ "loss": 5.6399181365966795,
194
  "step": 480
195
  },
196
  {
197
+ "epoch": 0.008426019548365353,
198
+ "grad_norm": 1.1796875,
199
+ "learning_rate": 0.00014969999999999998,
200
+ "loss": 5.560699081420898,
201
  "step": 500
202
  },
203
  {
204
+ "epoch": 0.008763060330299966,
205
+ "grad_norm": 2.328125,
206
+ "learning_rate": 0.0001557,
207
+ "loss": 5.474863433837891,
 
 
 
 
 
 
 
 
208
  "step": 520
209
  },
210
  {
211
+ "epoch": 0.00910010111223458,
212
+ "grad_norm": 1.125,
213
+ "learning_rate": 0.0001617,
214
+ "loss": 5.396588516235352,
215
  "step": 540
216
  },
217
  {
218
+ "epoch": 0.009437141894169195,
219
+ "grad_norm": 1.6484375,
220
+ "learning_rate": 0.0001677,
221
+ "loss": 5.331023406982422,
222
  "step": 560
223
  },
224
  {
225
+ "epoch": 0.00977418267610381,
226
+ "grad_norm": 1.0703125,
227
+ "learning_rate": 0.00017369999999999997,
228
+ "loss": 5.257175445556641,
229
  "step": 580
230
  },
231
  {
232
+ "epoch": 0.010111223458038422,
233
+ "grad_norm": 2.359375,
234
+ "learning_rate": 0.00017969999999999998,
235
+ "loss": 5.152382659912109,
236
  "step": 600
237
  },
238
  {
239
+ "epoch": 0.010111223458038422,
240
+ "eval_loss": 5.1175713539123535,
241
+ "eval_runtime": 7.457,
242
+ "eval_samples_per_second": 1277.585,
243
+ "eval_steps_per_second": 0.939,
244
  "step": 600
245
  },
246
  {
247
+ "epoch": 0.010448264239973037,
248
+ "grad_norm": 1.375,
249
+ "learning_rate": 0.0001857,
250
+ "loss": 5.096985244750977,
251
  "step": 620
252
  },
253
  {
254
+ "epoch": 0.010785305021907651,
255
+ "grad_norm": 1.421875,
256
+ "learning_rate": 0.0001917,
257
+ "loss": 5.019801330566406,
258
  "step": 640
259
  },
260
  {
261
+ "epoch": 0.011122345803842264,
262
+ "grad_norm": 2.125,
263
+ "learning_rate": 0.00019769999999999998,
264
+ "loss": 4.979957962036133,
265
  "step": 660
266
  },
267
  {
268
+ "epoch": 0.011459386585776879,
269
+ "grad_norm": 2.203125,
270
+ "learning_rate": 0.0002037,
271
+ "loss": 4.945652008056641,
272
  "step": 680
273
  },
274
  {
275
+ "epoch": 0.011796427367711493,
276
+ "grad_norm": 1.8515625,
277
+ "learning_rate": 0.00020969999999999997,
278
+ "loss": 4.866051483154297,
279
  "step": 700
280
  },
281
  {
282
+ "epoch": 0.012133468149646108,
283
+ "grad_norm": 1.625,
284
+ "learning_rate": 0.00021569999999999998,
285
+ "loss": 4.841766357421875,
 
 
 
 
 
 
 
 
286
  "step": 720
287
  },
288
  {
289
+ "epoch": 0.01247050893158072,
290
+ "grad_norm": 3.40625,
291
+ "learning_rate": 0.00022169999999999997,
292
+ "loss": 4.798672103881836,
293
  "step": 740
294
  },
295
  {
296
+ "epoch": 0.012807549713515335,
297
+ "grad_norm": 2.21875,
298
+ "learning_rate": 0.00022769999999999998,
299
+ "loss": 4.768531036376953,
300
  "step": 760
301
  },
302
  {
303
+ "epoch": 0.01314459049544995,
304
+ "grad_norm": 1.640625,
305
+ "learning_rate": 0.0002337,
306
+ "loss": 4.73585319519043,
307
  "step": 780
308
  },
309
  {
310
+ "epoch": 0.013481631277384564,
311
+ "grad_norm": 1.6875,
312
+ "learning_rate": 0.0002397,
313
+ "loss": 4.697021865844727,
314
  "step": 800
315
  },
316
  {
317
+ "epoch": 0.013481631277384564,
318
+ "eval_loss": 4.676848411560059,
319
+ "eval_runtime": 7.4394,
320
+ "eval_samples_per_second": 1280.62,
321
+ "eval_steps_per_second": 0.941,
322
  "step": 800
323
  }
324
  ],
325
  "logging_steps": 20,
326
+ "max_steps": 5000,
327
  "num_input_tokens_seen": 0,
328
  "num_train_epochs": 1,
329
  "save_steps": 100,
 
339
  "attributes": {}
340
  }
341
  },
342
+ "total_flos": 19362584985600.0,
343
+ "train_batch_size": 16,
344
  "trial_name": null,
345
  "trial_params": null
346
  }
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e2e1d330457822673a5ced171f375887b3e5542845ed4294676608daea03e08
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
  size 4920
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b98e3d3479511fdf20d344f73ed00ad6f6caedcb3d08a494ecd42cedbd61d4f1
3
  size 2036216
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8f12b34cc7fdfd828fc5af59ab51663686e06ecf81a5c0bcaa560f7f1e344f6f
3
  size 2036216
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fc055980aa96a5a10d9ca69e20cda8dca139beb6414cb5b8cbcd6aef459b9a17
3
  size 4089360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d775d3b8c1660cfecccbcb10103788b0d29af15ab335fb8688ec578610c2811
3
  size 4089360
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4b11a10749bfb1630d95eef94125f4590e8610c579d217d1f158e71ce518d72b
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
3
  size 14244
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:389b3459cdd290ee62a7be41d029a930e2587531b43c7e5ade84614c6ebc3507
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5bfb7b9c091ab83b2f167c61a3ae9a0c249f254498c8844e764454e070c9b77a
3
  size 1064
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/trainer_state.json CHANGED
@@ -2,403 +2,363 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.030333670374115267,
6
- "eval_steps": 100,
7
  "global_step": 900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.4921875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.323189544677735,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.609375,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.292604064941406,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.5234375,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.182575988769532,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.296875,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.984093475341797,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.265625,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.79156265258789,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.687252998352051,
50
- "eval_runtime": 7.4625,
51
- "eval_samples_per_second": 1276.648,
52
- "eval_steps_per_second": 0.938,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.25,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.586382293701172,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.356954193115234,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.21875,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.125580596923828,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.1953125,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.8821556091308596,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.125,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.638446807861328,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.517786502838135,
93
- "eval_runtime": 7.4364,
94
- "eval_samples_per_second": 1281.124,
95
- "eval_steps_per_second": 0.941,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.1171875,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.419033813476562,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 0.99609375,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.19798698425293,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 0.85546875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.01253662109375,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 2.4375,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.827148818969727,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.1171875,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.6793663024902346,
131
- "step": 300
132
- },
133
- {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.60944938659668,
136
- "eval_runtime": 7.4311,
137
- "eval_samples_per_second": 1282.036,
138
- "eval_steps_per_second": 0.942,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 0.9375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.544954299926758,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.2734375,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.418224334716797,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.890625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.2604835510253904,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.265625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.145714187622071,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.6328125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.026620101928711,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 4.965348720550537,
179
- "eval_runtime": 7.4478,
180
- "eval_samples_per_second": 1279.165,
181
- "eval_steps_per_second": 0.94,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 0.96484375,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.905144500732422,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.953125,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.836254501342774,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.90625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.767382431030273,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.6875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.681509017944336,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.609375,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.624195098876953,
217
- "step": 500
218
- },
219
- {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.591919898986816,
222
- "eval_runtime": 7.3895,
223
- "eval_samples_per_second": 1289.263,
224
- "eval_steps_per_second": 0.947,
225
  "step": 500
226
  },
227
  {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.359375,
230
- "learning_rate": 0.0003,
231
- "loss": 4.573441696166992,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.890625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.496570587158203,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.6796875,
244
- "learning_rate": 0.0003,
245
- "loss": 4.437310409545899,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.953125,
251
- "learning_rate": 0.0003,
252
- "loss": 4.387492752075195,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.375,
258
- "learning_rate": 0.0003,
259
- "loss": 4.346551513671875,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.322638034820557,
265
- "eval_runtime": 7.5459,
266
- "eval_samples_per_second": 1262.548,
267
- "eval_steps_per_second": 0.928,
268
  "step": 600
269
  },
270
  {
271
- "epoch": 0.020896528479946073,
272
- "grad_norm": 1.5078125,
273
- "learning_rate": 0.0003,
274
- "loss": 4.285432815551758,
275
  "step": 620
276
  },
277
  {
278
- "epoch": 0.021570610043815303,
279
- "grad_norm": 1.40625,
280
- "learning_rate": 0.0003,
281
- "loss": 4.2501472473144535,
282
  "step": 640
283
  },
284
  {
285
- "epoch": 0.022244691607684528,
286
- "grad_norm": 1.7421875,
287
- "learning_rate": 0.0003,
288
- "loss": 4.211347579956055,
289
  "step": 660
290
  },
291
  {
292
- "epoch": 0.022918773171553757,
293
- "grad_norm": 2.09375,
294
- "learning_rate": 0.0003,
295
- "loss": 4.168280410766601,
296
  "step": 680
297
  },
298
  {
299
- "epoch": 0.023592854735422986,
300
- "grad_norm": 1.734375,
301
- "learning_rate": 0.0003,
302
- "loss": 4.165232086181641,
303
- "step": 700
304
- },
305
- {
306
- "epoch": 0.023592854735422986,
307
- "eval_loss": 4.144440650939941,
308
- "eval_runtime": 7.5426,
309
- "eval_samples_per_second": 1263.095,
310
- "eval_steps_per_second": 0.928,
311
  "step": 700
312
  },
313
  {
314
- "epoch": 0.024266936299292215,
315
- "grad_norm": 0.98828125,
316
- "learning_rate": 0.0003,
317
- "loss": 4.132168579101562,
318
  "step": 720
319
  },
320
  {
321
- "epoch": 0.02494101786316144,
322
- "grad_norm": 1.390625,
323
- "learning_rate": 0.0003,
324
- "loss": 4.118199157714844,
325
  "step": 740
326
  },
327
  {
328
- "epoch": 0.02561509942703067,
329
- "grad_norm": 2.0625,
330
- "learning_rate": 0.0003,
331
- "loss": 4.074006271362305,
332
  "step": 760
333
  },
334
  {
335
- "epoch": 0.0262891809908999,
336
- "grad_norm": 1.46875,
337
- "learning_rate": 0.0003,
338
- "loss": 4.034224319458008,
339
  "step": 780
340
  },
341
  {
342
- "epoch": 0.026963262554769128,
343
- "grad_norm": 1.28125,
344
- "learning_rate": 0.0003,
345
- "loss": 4.037029266357422,
346
  "step": 800
347
  },
348
  {
349
- "epoch": 0.026963262554769128,
350
- "eval_loss": 4.028254508972168,
351
- "eval_runtime": 7.524,
352
- "eval_samples_per_second": 1266.218,
353
- "eval_steps_per_second": 0.93,
354
  "step": 800
355
  },
356
  {
357
- "epoch": 0.027637344118638354,
358
- "grad_norm": 1.28125,
359
- "learning_rate": 0.0003,
360
- "loss": 3.9971729278564454,
361
  "step": 820
362
  },
363
  {
364
- "epoch": 0.028311425682507583,
365
- "grad_norm": 1.6875,
366
- "learning_rate": 0.0003,
367
- "loss": 3.988052749633789,
368
  "step": 840
369
  },
370
  {
371
- "epoch": 0.028985507246376812,
372
- "grad_norm": 1.734375,
373
- "learning_rate": 0.0003,
374
- "loss": 4.010898208618164,
375
  "step": 860
376
  },
377
  {
378
- "epoch": 0.02965958881024604,
379
- "grad_norm": 1.5625,
380
- "learning_rate": 0.0003,
381
- "loss": 3.9676307678222655,
382
  "step": 880
383
  },
384
  {
385
- "epoch": 0.030333670374115267,
386
- "grad_norm": 1.7265625,
387
- "learning_rate": 0.0003,
388
- "loss": 3.937486267089844,
389
- "step": 900
390
- },
391
- {
392
- "epoch": 0.030333670374115267,
393
- "eval_loss": 3.954482078552246,
394
- "eval_runtime": 7.9455,
395
- "eval_samples_per_second": 1199.051,
396
- "eval_steps_per_second": 0.881,
397
  "step": 900
398
  }
399
  ],
400
  "logging_steps": 20,
401
- "max_steps": 2500,
402
  "num_input_tokens_seen": 0,
403
  "num_train_epochs": 1,
404
  "save_steps": 100,
@@ -414,8 +374,8 @@
414
  "attributes": {}
415
  }
416
  },
417
- "total_flos": 43565816217600.0,
418
- "train_batch_size": 32,
419
  "trial_name": null,
420
  "trial_params": null
421
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.015166835187057633,
6
+ "eval_steps": 200,
7
  "global_step": 900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0003370407819346141,
14
+ "grad_norm": 1.4140625,
15
+ "learning_rate": 5.7e-06,
16
+ "loss": 8.324227142333985,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.0006740815638692282,
21
+ "grad_norm": 1.5390625,
22
+ "learning_rate": 1.17e-05,
23
+ "loss": 8.318060302734375,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.0010111223458038423,
28
+ "grad_norm": 1.609375,
29
+ "learning_rate": 1.7699999999999997e-05,
30
+ "loss": 8.293100738525391,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.0013481631277384564,
35
+ "grad_norm": 1.7578125,
36
+ "learning_rate": 2.3699999999999997e-05,
37
+ "loss": 8.227334594726562,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.0016852039096730705,
42
+ "grad_norm": 1.5234375,
43
+ "learning_rate": 2.97e-05,
44
+ "loss": 8.111893463134766,
 
 
 
 
 
 
 
 
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.0020222446916076846,
49
+ "grad_norm": 1.3046875,
50
+ "learning_rate": 3.5699999999999994e-05,
51
+ "loss": 7.976696014404297,
52
  "step": 120
53
  },
54
  {
55
+ "epoch": 0.0023592854735422987,
56
+ "grad_norm": 1.296875,
57
+ "learning_rate": 4.17e-05,
58
+ "loss": 7.851339721679688,
59
  "step": 140
60
  },
61
  {
62
+ "epoch": 0.002696326255476913,
63
+ "grad_norm": 1.3046875,
64
+ "learning_rate": 4.7699999999999994e-05,
65
+ "loss": 7.724923706054687,
66
  "step": 160
67
  },
68
  {
69
+ "epoch": 0.003033367037411527,
70
+ "grad_norm": 1.3125,
71
+ "learning_rate": 5.369999999999999e-05,
72
+ "loss": 7.58428726196289,
73
  "step": 180
74
  },
75
  {
76
+ "epoch": 0.003370407819346141,
77
+ "grad_norm": 1.25,
78
+ "learning_rate": 5.97e-05,
79
+ "loss": 7.439914703369141,
80
  "step": 200
81
  },
82
  {
83
+ "epoch": 0.003370407819346141,
84
+ "eval_loss": 7.356490135192871,
85
+ "eval_runtime": 7.5119,
86
+ "eval_samples_per_second": 1268.257,
87
+ "eval_steps_per_second": 0.932,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.003707448601280755,
92
+ "grad_norm": 1.25,
93
+ "learning_rate": 6.57e-05,
94
+ "loss": 7.283377075195313,
95
  "step": 220
96
  },
97
  {
98
+ "epoch": 0.004044489383215369,
99
+ "grad_norm": 1.25,
100
+ "learning_rate": 7.17e-05,
101
+ "loss": 7.127851104736328,
102
  "step": 240
103
  },
104
  {
105
+ "epoch": 0.004381530165149983,
106
+ "grad_norm": 1.2109375,
107
+ "learning_rate": 7.769999999999999e-05,
108
+ "loss": 6.964313507080078,
109
  "step": 260
110
  },
111
  {
112
+ "epoch": 0.0047185709470845974,
113
+ "grad_norm": 1.171875,
114
+ "learning_rate": 8.37e-05,
115
+ "loss": 6.807338714599609,
116
  "step": 280
117
  },
118
  {
119
+ "epoch": 0.005055611729019211,
120
+ "grad_norm": 1.1328125,
121
+ "learning_rate": 8.969999999999998e-05,
122
+ "loss": 6.667655181884766,
 
 
 
 
 
 
 
 
123
  "step": 300
124
  },
125
  {
126
+ "epoch": 0.005392652510953826,
127
+ "grad_norm": 1.109375,
128
+ "learning_rate": 9.57e-05,
129
+ "loss": 6.523377227783203,
130
  "step": 320
131
  },
132
  {
133
+ "epoch": 0.005729693292888439,
134
+ "grad_norm": 1.1171875,
135
+ "learning_rate": 0.00010169999999999999,
136
+ "loss": 6.383005142211914,
137
  "step": 340
138
  },
139
  {
140
+ "epoch": 0.006066734074823054,
141
+ "grad_norm": 1.6875,
142
+ "learning_rate": 0.00010769999999999999,
143
+ "loss": 6.261091232299805,
144
  "step": 360
145
  },
146
  {
147
+ "epoch": 0.0064037748567576675,
148
+ "grad_norm": 1.140625,
149
+ "learning_rate": 0.00011369999999999999,
150
+ "loss": 6.122833251953125,
151
  "step": 380
152
  },
153
  {
154
+ "epoch": 0.006740815638692282,
155
+ "grad_norm": 1.3984375,
156
+ "learning_rate": 0.0001197,
157
+ "loss": 6.019657897949219,
158
  "step": 400
159
  },
160
  {
161
+ "epoch": 0.006740815638692282,
162
+ "eval_loss": 5.966014385223389,
163
+ "eval_runtime": 7.516,
164
+ "eval_samples_per_second": 1267.556,
165
+ "eval_steps_per_second": 0.931,
166
  "step": 400
167
  },
168
  {
169
+ "epoch": 0.007077856420626896,
170
+ "grad_norm": 0.98046875,
171
+ "learning_rate": 0.0001257,
172
+ "loss": 5.9373779296875,
173
  "step": 420
174
  },
175
  {
176
+ "epoch": 0.00741489720256151,
177
+ "grad_norm": 1.6328125,
178
+ "learning_rate": 0.00013169999999999998,
179
+ "loss": 5.839211273193359,
180
  "step": 440
181
  },
182
  {
183
+ "epoch": 0.007751937984496124,
184
+ "grad_norm": 0.9609375,
185
+ "learning_rate": 0.00013769999999999999,
186
+ "loss": 5.740922927856445,
187
  "step": 460
188
  },
189
  {
190
+ "epoch": 0.008088978766430738,
191
+ "grad_norm": 0.90234375,
192
+ "learning_rate": 0.00014369999999999997,
193
+ "loss": 5.6399181365966795,
194
  "step": 480
195
  },
196
  {
197
+ "epoch": 0.008426019548365353,
198
+ "grad_norm": 1.1796875,
199
+ "learning_rate": 0.00014969999999999998,
200
+ "loss": 5.560699081420898,
 
 
 
 
 
 
 
 
201
  "step": 500
202
  },
203
  {
204
+ "epoch": 0.008763060330299966,
205
+ "grad_norm": 2.328125,
206
+ "learning_rate": 0.0001557,
207
+ "loss": 5.474863433837891,
208
  "step": 520
209
  },
210
  {
211
+ "epoch": 0.00910010111223458,
212
+ "grad_norm": 1.125,
213
+ "learning_rate": 0.0001617,
214
+ "loss": 5.396588516235352,
215
  "step": 540
216
  },
217
  {
218
+ "epoch": 0.009437141894169195,
219
+ "grad_norm": 1.6484375,
220
+ "learning_rate": 0.0001677,
221
+ "loss": 5.331023406982422,
222
  "step": 560
223
  },
224
  {
225
+ "epoch": 0.00977418267610381,
226
+ "grad_norm": 1.0703125,
227
+ "learning_rate": 0.00017369999999999997,
228
+ "loss": 5.257175445556641,
229
  "step": 580
230
  },
231
  {
232
+ "epoch": 0.010111223458038422,
233
+ "grad_norm": 2.359375,
234
+ "learning_rate": 0.00017969999999999998,
235
+ "loss": 5.152382659912109,
236
  "step": 600
237
  },
238
  {
239
+ "epoch": 0.010111223458038422,
240
+ "eval_loss": 5.1175713539123535,
241
+ "eval_runtime": 7.457,
242
+ "eval_samples_per_second": 1277.585,
243
+ "eval_steps_per_second": 0.939,
244
  "step": 600
245
  },
246
  {
247
+ "epoch": 0.010448264239973037,
248
+ "grad_norm": 1.375,
249
+ "learning_rate": 0.0001857,
250
+ "loss": 5.096985244750977,
251
  "step": 620
252
  },
253
  {
254
+ "epoch": 0.010785305021907651,
255
+ "grad_norm": 1.421875,
256
+ "learning_rate": 0.0001917,
257
+ "loss": 5.019801330566406,
258
  "step": 640
259
  },
260
  {
261
+ "epoch": 0.011122345803842264,
262
+ "grad_norm": 2.125,
263
+ "learning_rate": 0.00019769999999999998,
264
+ "loss": 4.979957962036133,
265
  "step": 660
266
  },
267
  {
268
+ "epoch": 0.011459386585776879,
269
+ "grad_norm": 2.203125,
270
+ "learning_rate": 0.0002037,
271
+ "loss": 4.945652008056641,
272
  "step": 680
273
  },
274
  {
275
+ "epoch": 0.011796427367711493,
276
+ "grad_norm": 1.8515625,
277
+ "learning_rate": 0.00020969999999999997,
278
+ "loss": 4.866051483154297,
 
 
 
 
 
 
 
 
279
  "step": 700
280
  },
281
  {
282
+ "epoch": 0.012133468149646108,
283
+ "grad_norm": 1.625,
284
+ "learning_rate": 0.00021569999999999998,
285
+ "loss": 4.841766357421875,
286
  "step": 720
287
  },
288
  {
289
+ "epoch": 0.01247050893158072,
290
+ "grad_norm": 3.40625,
291
+ "learning_rate": 0.00022169999999999997,
292
+ "loss": 4.798672103881836,
293
  "step": 740
294
  },
295
  {
296
+ "epoch": 0.012807549713515335,
297
+ "grad_norm": 2.21875,
298
+ "learning_rate": 0.00022769999999999998,
299
+ "loss": 4.768531036376953,
300
  "step": 760
301
  },
302
  {
303
+ "epoch": 0.01314459049544995,
304
+ "grad_norm": 1.640625,
305
+ "learning_rate": 0.0002337,
306
+ "loss": 4.73585319519043,
307
  "step": 780
308
  },
309
  {
310
+ "epoch": 0.013481631277384564,
311
+ "grad_norm": 1.6875,
312
+ "learning_rate": 0.0002397,
313
+ "loss": 4.697021865844727,
314
  "step": 800
315
  },
316
  {
317
+ "epoch": 0.013481631277384564,
318
+ "eval_loss": 4.676848411560059,
319
+ "eval_runtime": 7.4394,
320
+ "eval_samples_per_second": 1280.62,
321
+ "eval_steps_per_second": 0.941,
322
  "step": 800
323
  },
324
  {
325
+ "epoch": 0.013818672059319177,
326
+ "grad_norm": 1.6484375,
327
+ "learning_rate": 0.00024569999999999995,
328
+ "loss": 4.658943176269531,
329
  "step": 820
330
  },
331
  {
332
+ "epoch": 0.014155712841253791,
333
+ "grad_norm": 3.234375,
334
+ "learning_rate": 0.0002517,
335
+ "loss": 4.5957294464111325,
336
  "step": 840
337
  },
338
  {
339
+ "epoch": 0.014492753623188406,
340
+ "grad_norm": 1.578125,
341
+ "learning_rate": 0.0002577,
342
+ "loss": 4.5967552185058596,
343
  "step": 860
344
  },
345
  {
346
+ "epoch": 0.01482979440512302,
347
+ "grad_norm": 1.3203125,
348
+ "learning_rate": 0.00026369999999999996,
349
+ "loss": 4.549871444702148,
350
  "step": 880
351
  },
352
  {
353
+ "epoch": 0.015166835187057633,
354
+ "grad_norm": 2.28125,
355
+ "learning_rate": 0.0002697,
356
+ "loss": 4.512709045410157,
 
 
 
 
 
 
 
 
357
  "step": 900
358
  }
359
  ],
360
  "logging_steps": 20,
361
+ "max_steps": 5000,
362
  "num_input_tokens_seen": 0,
363
  "num_train_epochs": 1,
364
  "save_steps": 100,
 
374
  "attributes": {}
375
  }
376
  },
377
+ "total_flos": 21782908108800.0,
378
+ "train_batch_size": 16,
379
  "trial_name": null,
380
  "trial_params": null
381
  }
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e2e1d330457822673a5ced171f375887b3e5542845ed4294676608daea03e08
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
3
  size 4920
zain/Activation/out/mlp-linear-3L_run/training_log.jsonl CHANGED
The diff for this file is too large to render. See raw diff
 
zain/Activation/wandb/debug-internal.log CHANGED
@@ -1,63 +1,35 @@
1
- {"time":"2026-08-19T00:06:21.305464341Z","level":"INFO","msg":"wandb-core"}
2
- {"time":"2026-08-19T00:06:21.305611275Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
3
- {"time":"2026-08-19T00:06:21.564804232Z","level":"INFO","msg":"stream: created new stream","id":"6zbmve2d"}
4
- {"time":"2026-08-19T00:06:21.564900628Z","level":"INFO","msg":"handler: started"}
5
- {"time":"2026-08-19T00:06:21.565009787Z","level":"INFO","msg":"stream: started"}
6
- {"time":"2026-08-19T00:06:21.565023898Z","level":"INFO","msg":"writer: started","stream_id":"6zbmve2d"}
7
- {"time":"2026-08-19T00:06:21.5650513Z","level":"INFO","msg":"sender: started"}
8
- {"time":"2026-08-19T00:06:21.992222163Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
- {"time":"2026-08-19T00:06:22.075914671Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
10
- {"time":"2026-08-19T00:06:36.992965322Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":7,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":13,"uploaded_len":2}
11
- {"time":"2026-08-19T00:06:37.108332432Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
12
- {"time":"2026-08-19T00:06:51.993186619Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":8,"events_offset":1,"events_lines":2,"console_offset":11,"console_lines":1}
13
- {"time":"2026-08-19T00:06:52.113538002Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
14
- {"time":"2026-08-19T00:07:06.993203978Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":15,"history_lines":7,"events_offset":3,"events_lines":2,"console_offset":13,"console_lines":25}
15
- {"time":"2026-08-19T00:07:07.704084095Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
16
- {"time":"2026-08-19T00:07:21.992750933Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":22,"history_lines":7,"events_offset":5,"events_lines":2,"console_offset":33,"console_lines":1}
17
- {"time":"2026-08-19T00:07:22.142204096Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
18
- {"time":"2026-08-19T00:07:36.992410794Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":29,"history_lines":6,"events_offset":7,"events_lines":2,"console_offset":38,"console_lines":24}
19
- {"time":"2026-08-19T00:07:37.130410236Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
20
- {"time":"2026-08-19T00:07:51.992869883Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":6,"events_offset":9,"events_lines":2,"console_offset":55,"console_lines":1}
21
- {"time":"2026-08-19T00:07:52.124122543Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
22
- {"time":"2026-08-19T00:08:06.993067354Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":11,"events_lines":2,"console_offset":61,"console_lines":23}
23
- {"time":"2026-08-19T00:08:07.370443909Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
24
- {"time":"2026-08-19T00:08:21.993084251Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":77,"console_lines":1}
25
- {"time":"2026-08-19T00:08:22.149421421Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
26
- {"time":"2026-08-19T00:08:36.992788364Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":6,"events_offset":15,"events_lines":2,"console_offset":83,"console_lines":23}
27
- {"time":"2026-08-19T00:08:37.139797176Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
28
- {"time":"2026-08-19T00:08:51.992814158Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":59,"history_lines":8,"events_offset":17,"events_lines":2,"console_offset":99,"console_lines":1}
29
- {"time":"2026-08-19T00:08:52.142919145Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
30
- {"time":"2026-08-19T00:09:06.992976383Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":67,"history_lines":8,"events_offset":19,"events_lines":2,"console_offset":105,"console_lines":31}
31
- {"time":"2026-08-19T00:09:07.140990317Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
32
- {"time":"2026-08-19T00:09:21.993221219Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":75,"history_lines":7,"events_offset":21,"events_lines":2,"console_offset":132,"console_lines":1}
33
- {"time":"2026-08-19T00:09:22.111256159Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
34
- {"time":"2026-08-19T00:09:36.992968586Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":82,"history_lines":7,"events_offset":23,"events_lines":2,"console_offset":136,"console_lines":24}
35
- {"time":"2026-08-19T00:09:37.285303756Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
36
- {"time":"2026-08-19T00:09:51.992822127Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":89,"history_lines":6,"events_offset":25,"events_lines":2,"console_offset":154,"console_lines":1}
37
- {"time":"2026-08-19T00:09:52.109096249Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
38
- {"time":"2026-08-19T00:10:06.992523849Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":95,"history_lines":6,"events_offset":27,"events_lines":2,"console_offset":160,"console_lines":23}
39
- {"time":"2026-08-19T00:10:07.154689008Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
40
- {"time":"2026-08-19T00:10:21.993355847Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":6,"events_offset":29,"events_lines":2,"console_offset":176,"console_lines":1}
41
- {"time":"2026-08-19T00:10:22.178839986Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
42
- {"time":"2026-08-19T00:10:36.992657682Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":107,"history_lines":6,"events_offset":31,"events_lines":2,"console_offset":182,"console_lines":23}
43
- {"time":"2026-08-19T00:10:37.1185939Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
44
- {"time":"2026-08-19T00:10:51.993057513Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":113,"history_lines":7,"events_offset":33,"events_lines":2,"console_offset":198,"console_lines":1}
45
- {"time":"2026-08-19T00:10:52.101588155Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
46
- {"time":"2026-08-19T00:11:06.992973349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":120,"history_lines":7,"events_offset":35,"events_lines":2,"console_offset":204,"console_lines":29}
47
- {"time":"2026-08-19T00:11:07.134793732Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
48
- {"time":"2026-08-19T00:11:21.994425336Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":127,"history_lines":8,"events_offset":37,"events_lines":2,"console_offset":231,"console_lines":1}
49
- {"time":"2026-08-19T00:11:22.134821113Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
50
- {"time":"2026-08-19T00:11:36.993054503Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":135,"history_lines":7,"events_offset":39,"events_lines":2,"console_offset":233,"console_lines":25}
51
- {"time":"2026-08-19T00:11:37.111533939Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
52
- {"time":"2026-08-19T00:11:51.992604078Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":142,"history_lines":7,"events_offset":41,"events_lines":2,"console_offset":253,"console_lines":1}
53
- {"time":"2026-08-19T00:11:52.115538255Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
54
- {"time":"2026-08-19T00:12:06.992994064Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":149,"history_lines":2,"events_offset":43,"events_lines":2,"console_offset":258,"console_lines":20}
55
- {"time":"2026-08-19T00:12:07.157903259Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
56
- {"time":"2026-08-19T00:12:08.630011645Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
57
- {"time":"2026-08-19T00:12:08.630264558Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":151,"history_lines":1,"events_offset":45,"events_lines":1,"console_offset":277,"console_lines":7,"uploaded_len":3,"complete":true,"exit_code":0}
58
- {"time":"2026-08-19T00:12:08.908163631Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
59
- {"time":"2026-08-19T00:12:08.909177309Z","level":"INFO","msg":"handler: operation stats","stats":{}}
60
- {"time":"2026-08-19T00:12:08.911408845Z","level":"INFO","msg":"stream: finishing up"}
61
- {"time":"2026-08-19T00:12:08.911426034Z","level":"INFO","msg":"handler: closed"}
62
- {"time":"2026-08-19T00:12:08.911484689Z","level":"INFO","msg":"sender: closed"}
63
- {"time":"2026-08-19T00:12:08.911488467Z","level":"INFO","msg":"stream: all finished"}
 
1
+ {"time":"2026-08-19T16:30:14.408862805Z","level":"INFO","msg":"wandb-core"}
2
+ {"time":"2026-08-19T16:30:14.409145225Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
3
+ {"time":"2026-08-19T16:30:14.673839961Z","level":"INFO","msg":"stream: created new stream","id":"smtfejrp"}
4
+ {"time":"2026-08-19T16:30:14.673914385Z","level":"INFO","msg":"handler: started"}
5
+ {"time":"2026-08-19T16:30:14.674023743Z","level":"INFO","msg":"stream: started"}
6
+ {"time":"2026-08-19T16:30:14.67404595Z","level":"INFO","msg":"writer: started","stream_id":"smtfejrp"}
7
+ {"time":"2026-08-19T16:30:14.674053413Z","level":"INFO","msg":"sender: started"}
8
+ {"time":"2026-08-19T16:30:15.573430136Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
+ {"time":"2026-08-19T16:30:15.803201108Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
10
+ {"time":"2026-08-19T16:30:30.574169941Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":16,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":34,"uploaded_len":2}
11
+ {"time":"2026-08-19T16:30:30.711458871Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
12
+ {"time":"2026-08-19T16:30:45.574202013Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":16,"events_offset":1,"events_lines":2,"console_offset":33,"console_lines":27}
13
+ {"time":"2026-08-19T16:30:45.755515219Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
14
+ {"time":"2026-08-19T16:31:00.573737519Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":11,"events_offset":3,"events_lines":2,"console_offset":54,"console_lines":1}
15
+ {"time":"2026-08-19T16:31:00.75827199Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
16
+ {"time":"2026-08-19T16:31:15.574343371Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":11,"events_offset":5,"events_lines":2,"console_offset":60,"console_lines":43}
17
+ {"time":"2026-08-19T16:31:15.798424562Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
18
+ {"time":"2026-08-19T16:31:30.573773942Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":54,"history_lines":18,"events_offset":7,"events_lines":2,"console_offset":96,"console_lines":1}
19
+ {"time":"2026-08-19T16:31:30.694320256Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
20
+ {"time":"2026-08-19T16:31:45.574312952Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":72,"history_lines":15,"events_offset":9,"events_lines":2,"console_offset":102,"console_lines":63}
21
+ {"time":"2026-08-19T16:31:45.774027402Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
22
+ {"time":"2026-08-19T16:32:00.574639142Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":87,"history_lines":11,"events_offset":11,"events_lines":2,"console_offset":159,"console_lines":1}
23
+ {"time":"2026-08-19T16:32:00.704830872Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
24
+ {"time":"2026-08-19T16:32:15.573881284Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":11,"events_offset":13,"events_lines":2,"console_offset":165,"console_lines":43}
25
+ {"time":"2026-08-19T16:32:15.785173455Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
26
+ {"time":"2026-08-19T16:32:30.574343755Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":109,"history_lines":19,"events_offset":15,"events_lines":2,"console_offset":201,"console_lines":1}
27
+ {"time":"2026-08-19T16:32:30.715820396Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
28
+ {"time":"2026-08-19T16:32:45.573977636Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":128,"history_lines":14,"events_offset":17,"events_lines":2,"console_offset":207,"console_lines":63}
29
+ {"time":"2026-08-19T16:32:45.688687126Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
30
+ {"time":"2026-08-19T16:33:00.574401218Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":142,"history_lines":11,"events_offset":19,"events_lines":2,"console_offset":264,"console_lines":1}
31
+ {"time":"2026-08-19T16:33:00.726033404Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
32
+ {"time":"2026-08-19T16:33:15.575478736Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":153,"history_lines":13,"events_offset":21,"events_lines":2,"console_offset":270,"console_lines":49}
33
+ {"time":"2026-08-19T16:33:15.746742621Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
34
+ {"time":"2026-08-19T16:33:30.574073819Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":166,"history_lines":18,"events_offset":23,"events_lines":2,"console_offset":317,"console_lines":1}
35
+ {"time":"2026-08-19T16:33:30.725976627Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
zain/Activation/wandb/debug.log CHANGED
@@ -1,25 +1,23 @@
1
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_000621-6zbmve2d/logs/debug.log
2
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_000621-6zbmve2d/logs/debug-internal.log
3
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:init():772] calling init triggers
4
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
 
 
 
5
  config: {'_wandb': {}}
6
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:init():820] starting backend
7
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE
8
- 2026-08-19 00:06:21,304 INFO MainThread:3613580 [wandb_init.py:init():835] sending inform_init request
9
- 2026-08-19 00:06:21,565 INFO MainThread:3613580 [wandb_init.py:init():840] backend started and connected
10
- 2026-08-19 00:06:21,566 INFO MainThread:3613580 [wandb_init.py:init():910] updated telemetry
11
- 2026-08-19 00:06:21,573 INFO MainThread:3613580 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
12
- 2026-08-19 00:06:21,910 INFO MainThread:3613580 [wandb_init.py:init():978] starting run threads in backend
13
- 2026-08-19 00:06:21,987 INFO MainThread:3613580 [wandb_run.py:_console_start():2621] atexit reg
14
- 2026-08-19 00:06:21,987 INFO MainThread:3613580 [wandb_run.py:_redirect():2471] redirect: wrap_raw
15
- 2026-08-19 00:06:21,987 INFO MainThread:3613580 [wandb_run.py:_redirect():2540] Wrapping output streams.
16
- 2026-08-19 00:06:21,987 INFO MainThread:3613580 [wandb_run.py:_redirect():2563] Redirects installed.
17
- 2026-08-19 00:06:21,988 INFO MainThread:3613580 [wandb_init.py:init():1016] run started, returning control to user process
18
- 2026-08-19 00:06:21,989 INFO MainThread:3613580 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'tanh', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-tanh-9L_run', 'per_device_train_batch_size': 32, 'num_train_epochs': 1, 'max_steps': 2500, 'learning_rate': 0.0003, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 500, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-tanh-9L-2.0M-20260819-000620', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-glu-tanh-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
19
- 2026-08-19 00:06:21,991 INFO MainThread:3613580 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x1489c452fe10>>
20
- 2026-08-19 00:06:21,991 INFO MainThread:3613580 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
21
- 2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate/6zbmve2d
22
- 2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
23
- 2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_restore():2570] restore
24
- 2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_restore():2576] restore done
25
- 2026-08-19 00:12:08,910 INFO MainThread:3613580 [wandb_run.py:_footer_sync_info():3993] logging synced files
 
1
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1
2
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_setup.py:_flush():81] Configure stats pid to 3508170
3
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_setup.py:_flush():81] Loading settings from environment variables
4
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_163014-smtfejrp/logs/debug.log
5
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_163014-smtfejrp/logs/debug-internal.log
6
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():772] calling init triggers
7
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
8
  config: {'_wandb': {}}
9
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():820] starting backend
10
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE
11
+ 2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():835] sending inform_init request
12
+ 2026-08-19 16:30:14,674 INFO MainThread:3508170 [wandb_init.py:init():840] backend started and connected
13
+ 2026-08-19 16:30:14,678 INFO MainThread:3508170 [wandb_init.py:init():910] updated telemetry
14
+ 2026-08-19 16:30:14,685 INFO MainThread:3508170 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
15
+ 2026-08-19 16:30:15,492 INFO MainThread:3508170 [wandb_init.py:init():978] starting run threads in backend
16
+ 2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_console_start():2621] atexit reg
17
+ 2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_redirect():2471] redirect: wrap_raw
18
+ 2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_redirect():2540] Wrapping output streams.
19
+ 2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_redirect():2563] Redirects installed.
20
+ 2026-08-19 16:30:15,569 INFO MainThread:3508170 [wandb_init.py:init():1016] run started, returning control to user process
21
+ 2026-08-19 16:30:15,570 INFO MainThread:3508170 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 3, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-3L_run', 'per_device_train_batch_size': 16, 'num_train_epochs': 1, 'max_steps': 5000, 'learning_rate': 0.0003, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 1000, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-3L-1.0M-20260819-163013', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 200, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-3L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
22
+ 2026-08-19 16:30:15,572 INFO MainThread:3508170 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 1016704 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x1495a6a04a90>>
23
+ 2026-08-19 16:30:15,572 INFO MainThread:3508170 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 1016704 None
 
 
 
 
 
zain/Activation/wandb/run-20260819_163014-smtfejrp/files/output.log ADDED
@@ -0,0 +1,364 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 0%| | 0/5000 [00:00<?, ?it/s][transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
2
+ [INFO] Causal mask (float with -inf) applied to all attention layers.
3
+ 2%|▏ | 100/5000 [00:03<01:50, 44.29it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
4
+ {'loss': '8.324', 'grad_norm': '1.414', 'learning_rate': '5.7e-06', 'epoch': '0.000337', 'train/total_time_seconds': '0.7989', 'train/time_per_step_avg': '0.03995', 'train/epoch_time_elapsed': '1.159', 'train/estimated_remaining_minutes': '3.315'}
5
+ {'loss': '8.318', 'grad_norm': '1.539', 'learning_rate': '1.17e-05', 'epoch': '0.0006741', 'train/total_time_seconds': '1.003', 'train/time_per_step_avg': '0.02508', 'train/epoch_time_elapsed': '1.621', 'train/estimated_remaining_minutes': '2.073'}
6
+ {'loss': '8.293', 'grad_norm': '1.609', 'learning_rate': '1.77e-05', 'epoch': '0.001011', 'train/total_time_seconds': '1.202', 'train/time_per_step_avg': '0.02003', 'train/epoch_time_elapsed': '2.112', 'train/estimated_remaining_minutes': '1.649'}
7
+ {'loss': '8.227', 'grad_norm': '1.758', 'learning_rate': '2.37e-05', 'epoch': '0.001348', 'train/total_time_seconds': '1.393', 'train/time_per_step_avg': '0.01741', 'train/epoch_time_elapsed': '2.597', 'train/estimated_remaining_minutes': '1.428'}
8
+ {'loss': '8.112', 'grad_norm': '1.523', 'learning_rate': '2.97e-05', 'epoch': '0.001685', 'train/total_time_seconds': '1.571', 'train/time_per_step_avg': '0.01571', 'train/epoch_time_elapsed': '3.029', 'train/estimated_remaining_minutes': '1.283'}
9
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
10
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
11
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
12
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 227.99it/s]
13
+ 4%|▍ | 200/5000 [00:12<01:44, 46.00[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
14
+ {'loss': '7.977', 'grad_norm': '1.305', 'learning_rate': '3.57e-05', 'epoch': '0.002022', 'train/total_time_seconds': '1.748', 'train/time_per_step_avg': '0.00949', 'train/epoch_time_elapsed': '3.481', 'train/estimated_remaining_minutes': '1.185'}
15
+ {'loss': '7.851', 'grad_norm': '1.297', 'learning_rate': '4.17e-05', 'epoch': '0.002359', 'train/total_time_seconds': '1.929', 'train/time_per_step_avg': '0.009256', 'train/epoch_time_elapsed': '3.92', 'train/estimated_remaining_minutes': '1.116'}
16
+ {'loss': '7.725', 'grad_norm': '1.305', 'learning_rate': '4.77e-05', 'epoch': '0.002696', 'train/total_time_seconds': '2.105', 'train/time_per_step_avg': '0.009031', 'train/epoch_time_elapsed': '4.349', 'train/estimated_remaining_minutes': '1.061'}
17
+ {'loss': '7.584', 'grad_norm': '1.312', 'learning_rate': '5.37e-05', 'epoch': '0.003033', 'train/total_time_seconds': '2.288', 'train/time_per_step_avg': '0.008948', 'train/epoch_time_elapsed': '4.788', 'train/estimated_remaining_minutes': '1.021'}
18
+ {'loss': '7.44', 'grad_norm': '1.25', 'learning_rate': '5.97e-05', 'epoch': '0.00337', 'train/total_time_seconds': '2.468', 'train/time_per_step_avg': '0.008976', 'train/epoch_time_elapsed': '5.223', 'train/estimated_remaining_minutes': '0.9873'}
19
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
20
+ {'eval_loss': '7.356', 'eval_runtime': '7.512', 'eval_samples_per_second': '1268', 'eval_steps_per_second': '0.932', 'epoch': '0.00337', 'train/total_time_seconds': '2.468', 'train/time_per_step_avg': '0.008976', 'train/epoch_time_elapsed': '12.74', 'train/estimated_remaining_minutes': '0.9873'}
21
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
22
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
23
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 359.32it/s]
24
+ 6%|▌ | 300/5000 [00:14<01:43, 45.61it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
25
+ {'loss': '7.283', 'grad_norm': '1.25', 'learning_rate': '6.57e-05', 'epoch': '0.003707', 'train/total_time_seconds': '2.648', 'train/time_per_step_avg': '0.008997', 'train/epoch_time_elapsed': '13.19', 'train/estimated_remaining_minutes': '0.9588'}
26
+ {'loss': '7.128', 'grad_norm': '1.25', 'learning_rate': '7.17e-05', 'epoch': '0.004044', 'train/total_time_seconds': '2.823', 'train/time_per_step_avg': '0.008943', 'train/epoch_time_elapsed': '13.62', 'train/estimated_remaining_minutes': '0.9332'}
27
+ {'loss': '6.964', 'grad_norm': '1.211', 'learning_rate': '7.77e-05', 'epoch': '0.004382', 'train/total_time_seconds': '2.998', 'train/time_per_step_avg': '0.008929', 'train/epoch_time_elapsed': '14.05', 'train/estimated_remaining_minutes': '0.9108'}
28
+ {'loss': '6.807', 'grad_norm': '1.172', 'learning_rate': '8.37e-05', 'epoch': '0.004719', 'train/total_time_seconds': '3.173', 'train/time_per_step_avg': '0.008853', 'train/epoch_time_elapsed': '14.48', 'train/estimated_remaining_minutes': '0.8915'}
29
+ {'loss': '6.668', 'grad_norm': '1.133', 'learning_rate': '8.97e-05', 'epoch': '0.005056', 'train/total_time_seconds': '3.348', 'train/time_per_step_avg': '0.008801', 'train/epoch_time_elapsed': '14.91', 'train/estimated_remaining_minutes': '0.8743'}
30
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
31
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
32
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
33
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 360.06it/s]
34
+ 8%|▊ | 400/5000 [00:24<01:46, 43.02[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
35
+ {'loss': '6.523', 'grad_norm': '1.109', 'learning_rate': '9.57e-05', 'epoch': '0.005393', 'train/total_time_seconds': '3.531', 'train/time_per_step_avg': '0.008839', 'train/epoch_time_elapsed': '15.38', 'train/estimated_remaining_minutes': '0.8608'}
36
+ {'loss': '6.383', 'grad_norm': '1.117', 'learning_rate': '0.0001017', 'epoch': '0.00573', 'train/total_time_seconds': '3.71', 'train/time_per_step_avg': '0.008873', 'train/epoch_time_elapsed': '15.81', 'train/estimated_remaining_minutes': '0.8475'}
37
+ {'loss': '6.261', 'grad_norm': '1.688', 'learning_rate': '0.0001077', 'epoch': '0.006067', 'train/total_time_seconds': '3.889', 'train/time_per_step_avg': '0.008913', 'train/epoch_time_elapsed': '16.24', 'train/estimated_remaining_minutes': '0.8354'}
38
+ {'loss': '6.123', 'grad_norm': '1.141', 'learning_rate': '0.0001137', 'epoch': '0.006404', 'train/total_time_seconds': '4.09', 'train/time_per_step_avg': '0.009175', 'train/epoch_time_elapsed': '16.73', 'train/estimated_remaining_minutes': '0.8289'}
39
+ {'loss': '6.02', 'grad_norm': '1.398', 'learning_rate': '0.0001197', 'epoch': '0.006741', 'train/total_time_seconds': '4.287', 'train/time_per_step_avg': '0.009382', 'train/epoch_time_elapsed': '17.18', 'train/estimated_remaining_minutes': '0.8216'}
40
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
41
+ {'eval_loss': '5.966', 'eval_runtime': '7.516', 'eval_samples_per_second': '1268', 'eval_steps_per_second': '0.931', 'epoch': '0.006741', 'train/total_time_seconds': '4.287', 'train/time_per_step_avg': '0.009382', 'train/epoch_time_elapsed': '24.7', 'train/estimated_remaining_minutes': '0.8216'}
42
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
43
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
44
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 337.95it/s]
45
+ 10%|█ | 500/5000 [00:27<01:55, 39.07it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
46
+ {'loss': '5.937', 'grad_norm': '0.9805', 'learning_rate': '0.0001257', 'epoch': '0.007078', 'train/total_time_seconds': '4.473', 'train/time_per_step_avg': '0.00942', 'train/epoch_time_elapsed': '25.17', 'train/estimated_remaining_minutes': '0.813'}
47
+ {'loss': '5.839', 'grad_norm': '1.633', 'learning_rate': '0.0001317', 'epoch': '0.007415', 'train/total_time_seconds': '4.662', 'train/time_per_step_avg': '0.009518', 'train/epoch_time_elapsed': '25.62', 'train/estimated_remaining_minutes': '0.8053'}
48
+ {'loss': '5.741', 'grad_norm': '0.9609', 'learning_rate': '0.0001377', 'epoch': '0.007752', 'train/total_time_seconds': '4.846', 'train/time_per_step_avg': '0.009575', 'train/epoch_time_elapsed': '26.06', 'train/estimated_remaining_minutes': '0.7972'}
49
+ {'loss': '5.64', 'grad_norm': '0.9023', 'learning_rate': '0.0001437', 'epoch': '0.008089', 'train/total_time_seconds': '5.032', 'train/time_per_step_avg': '0.009414', 'train/epoch_time_elapsed': '26.5', 'train/estimated_remaining_minutes': '0.7897'}
50
+ {'loss': '5.561', 'grad_norm': '1.18', 'learning_rate': '0.0001497', 'epoch': '0.008426', 'train/total_time_seconds': '5.24', 'train/time_per_step_avg': '0.009536', 'train/epoch_time_elapsed': '27', 'train/estimated_remaining_minutes': '0.786'}
51
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
52
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
53
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
54
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 107.02it/s]
55
+ 12%|█▏ | 600/5000 [00:36<01:42, 42.78[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
56
+ {'loss': '5.475', 'grad_norm': '2.328', 'learning_rate': '0.0001557', 'epoch': '0.008763', 'train/total_time_seconds': '5.424', 'train/time_per_step_avg': '0.009508', 'train/epoch_time_elapsed': '27.47', 'train/estimated_remaining_minutes': '0.7789'}
57
+ {'loss': '5.397', 'grad_norm': '1.125', 'learning_rate': '0.0001617', 'epoch': '0.0091', 'train/total_time_seconds': '5.639', 'train/time_per_step_avg': '0.009766', 'train/epoch_time_elapsed': '27.99', 'train/estimated_remaining_minutes': '0.7762'}
58
+ {'loss': '5.331', 'grad_norm': '1.648', 'learning_rate': '0.0001677', 'epoch': '0.009437', 'train/total_time_seconds': '5.822', 'train/time_per_step_avg': '0.009752', 'train/epoch_time_elapsed': '28.43', 'train/estimated_remaining_minutes': '0.7693'}
59
+ {'loss': '5.257', 'grad_norm': '1.07', 'learning_rate': '0.0001737', 'epoch': '0.009774', 'train/total_time_seconds': '6.002', 'train/time_per_step_avg': '0.009698', 'train/epoch_time_elapsed': '28.86', 'train/estimated_remaining_minutes': '0.7623'}
60
+ {'loss': '5.152', 'grad_norm': '2.359', 'learning_rate': '0.0001797', 'epoch': '0.01011', 'train/total_time_seconds': '6.201', 'train/time_per_step_avg': '0.009613', 'train/epoch_time_elapsed': '29.32', 'train/estimated_remaining_minutes': '0.758'}
61
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
62
+ {'eval_loss': '5.118', 'eval_runtime': '7.457', 'eval_samples_per_second': '1278', 'eval_steps_per_second': '0.939', 'epoch': '0.01011', 'train/total_time_seconds': '6.201', 'train/time_per_step_avg': '0.009613', 'train/epoch_time_elapsed': '36.78', 'train/estimated_remaining_minutes': '0.758'}
63
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
64
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
65
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 363.80it/s]
66
+ 14%|█▍ | 700/5000 [00:38<01:35, 45.02it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
67
+ {'loss': '5.097', 'grad_norm': '1.375', 'learning_rate': '0.0001857', 'epoch': '0.01045', 'train/total_time_seconds': '6.4', 'train/time_per_step_avg': '0.009761', 'train/epoch_time_elapsed': '37.25', 'train/estimated_remaining_minutes': '0.7536'}
68
+ {'loss': '5.02', 'grad_norm': '1.422', 'learning_rate': '0.0001917', 'epoch': '0.01079', 'train/total_time_seconds': '6.575', 'train/time_per_step_avg': '0.009358', 'train/epoch_time_elapsed': '37.68', 'train/estimated_remaining_minutes': '0.7465'}
69
+ {'loss': '4.98', 'grad_norm': '2.125', 'learning_rate': '0.0001977', 'epoch': '0.01112', 'train/total_time_seconds': '6.755', 'train/time_per_step_avg': '0.009331', 'train/epoch_time_elapsed': '38.12', 'train/estimated_remaining_minutes': '0.7403'}
70
+ {'loss': '4.946', 'grad_norm': '2.203', 'learning_rate': '0.0002037', 'epoch': '0.01146', 'train/total_time_seconds': '6.931', 'train/time_per_step_avg': '0.009292', 'train/epoch_time_elapsed': '38.55', 'train/estimated_remaining_minutes': '0.7339'}
71
+ {'loss': '4.866', 'grad_norm': '1.852', 'learning_rate': '0.0002097', 'epoch': '0.0118', 'train/total_time_seconds': '7.108', 'train/time_per_step_avg': '0.00907', 'train/epoch_time_elapsed': '38.98', 'train/estimated_remaining_minutes': '0.7278'}
72
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
73
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
74
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
75
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 361.67it/s]
76
+ 16%|█▌ | 800/5000 [00:48<01:30, 46.66[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
77
+ {'loss': '4.842', 'grad_norm': '1.625', 'learning_rate': '0.0002157', 'epoch': '0.01213', 'train/total_time_seconds': '7.287', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '39.44', 'train/estimated_remaining_minutes': '0.722'}
78
+ {'loss': '4.799', 'grad_norm': '3.406', 'learning_rate': '0.0002217', 'epoch': '0.01247', 'train/total_time_seconds': '7.461', 'train/time_per_step_avg': '0.008864', 'train/epoch_time_elapsed': '39.86', 'train/estimated_remaining_minutes': '0.7158'}
79
+ {'loss': '4.769', 'grad_norm': '2.219', 'learning_rate': '0.0002277', 'epoch': '0.01281', 'train/total_time_seconds': '7.635', 'train/time_per_step_avg': '0.008802', 'train/epoch_time_elapsed': '40.29', 'train/estimated_remaining_minutes': '0.7099'}
80
+ {'loss': '4.736', 'grad_norm': '1.641', 'learning_rate': '0.0002337', 'epoch': '0.01314', 'train/total_time_seconds': '7.813', 'train/time_per_step_avg': '0.008818', 'train/epoch_time_elapsed': '40.72', 'train/estimated_remaining_minutes': '0.7045'}
81
+ {'loss': '4.697', 'grad_norm': '1.688', 'learning_rate': '0.0002397', 'epoch': '0.01348', 'train/total_time_seconds': '7.987', 'train/time_per_step_avg': '0.008783', 'train/epoch_time_elapsed': '41.15', 'train/estimated_remaining_minutes': '0.6988'}
82
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
83
+ {'eval_loss': '4.677', 'eval_runtime': '7.439', 'eval_samples_per_second': '1281', 'eval_steps_per_second': '0.941', 'epoch': '0.01348', 'train/total_time_seconds': '7.987', 'train/time_per_step_avg': '0.008783', 'train/epoch_time_elapsed': '48.59', 'train/estimated_remaining_minutes': '0.6988'}
84
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
85
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
86
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 365.48it/s]
87
+ 18%|█▊ | 900/5000 [00:50<01:31, 44.85it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
88
+ {'loss': '4.659', 'grad_norm': '1.648', 'learning_rate': '0.0002457', 'epoch': '0.01382', 'train/total_time_seconds': '8.165', 'train/time_per_step_avg': '0.008781', 'train/epoch_time_elapsed': '49.05', 'train/estimated_remaining_minutes': '0.6937'}
89
+ {'loss': '4.596', 'grad_norm': '3.234', 'learning_rate': '0.0002517', 'epoch': '0.01416', 'train/total_time_seconds': '8.341', 'train/time_per_step_avg': '0.008806', 'train/epoch_time_elapsed': '49.48', 'train/estimated_remaining_minutes': '0.6885'}
90
+ {'loss': '4.597', 'grad_norm': '1.578', 'learning_rate': '0.0002577', 'epoch': '0.01449', 'train/total_time_seconds': '8.521', 'train/time_per_step_avg': '0.008861', 'train/epoch_time_elapsed': '49.92', 'train/estimated_remaining_minutes': '0.6837'}
91
+ {'loss': '4.55', 'grad_norm': '1.32', 'learning_rate': '0.0002637', 'epoch': '0.01483', 'train/total_time_seconds': '8.697', 'train/time_per_step_avg': '0.008841', 'train/epoch_time_elapsed': '50.35', 'train/estimated_remaining_minutes': '0.6786'}
92
+ {'loss': '4.513', 'grad_norm': '2.281', 'learning_rate': '0.0002697', 'epoch': '0.01517', 'train/total_time_seconds': '8.875', 'train/time_per_step_avg': '0.008886', 'train/epoch_time_elapsed': '50.79', 'train/estimated_remaining_minutes': '0.6739'}
93
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
94
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
95
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
96
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 363.55it/s]
97
+ 20%|██ | 1000/5000 [01:00<01:25, 46.5[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
98
+ {'loss': '4.476', 'grad_norm': '1.688', 'learning_rate': '0.0002757', 'epoch': '0.0155', 'train/total_time_seconds': '9.053', 'train/time_per_step_avg': '0.008877', 'train/epoch_time_elapsed': '51.24', 'train/estimated_remaining_minutes': '0.6691'}
99
+ {'loss': '4.433', 'grad_norm': '1.602', 'learning_rate': '0.0002817', 'epoch': '0.01584', 'train/total_time_seconds': '9.228', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '51.67', 'train/estimated_remaining_minutes': '0.6643'}
100
+ {'loss': '4.402', 'grad_norm': '2.031', 'learning_rate': '0.0002877', 'epoch': '0.01618', 'train/total_time_seconds': '9.407', 'train/time_per_step_avg': '0.008858', 'train/epoch_time_elapsed': '52.1', 'train/estimated_remaining_minutes': '0.6598'}
101
+ {'loss': '4.383', 'grad_norm': '1.719', 'learning_rate': '0.0002937', 'epoch': '0.01651', 'train/total_time_seconds': '9.586', 'train/time_per_step_avg': '0.008889', 'train/epoch_time_elapsed': '52.53', 'train/estimated_remaining_minutes': '0.6553'}
102
+ {'loss': '4.368', 'grad_norm': '1.219', 'learning_rate': '0.0002997', 'epoch': '0.01685', 'train/total_time_seconds': '9.762', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '52.96', 'train/estimated_remaining_minutes': '0.6508'}
103
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
104
+ {'eval_loss': '4.343', 'eval_runtime': '7.451', 'eval_samples_per_second': '1279', 'eval_steps_per_second': '0.939', 'epoch': '0.01685', 'train/total_time_seconds': '9.762', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '60.42', 'train/estimated_remaining_minutes': '0.6508'}
105
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
106
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
107
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 360.00it/s]
108
+ 22%|██▏ | 1100/5000 [01:02<01:25, 45.40it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
109
+ {'loss': '4.332', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.01719', 'train/total_time_seconds': '9.942', 'train/time_per_step_avg': '0.008889', 'train/epoch_time_elapsed': '60.88', 'train/estimated_remaining_minutes': '0.6465'}
110
+ {'loss': '4.315', 'grad_norm': '1.898', 'learning_rate': '0.0003', 'epoch': '0.01753', 'train/total_time_seconds': '10.12', 'train/time_per_step_avg': '0.008917', 'train/epoch_time_elapsed': '61.32', 'train/estimated_remaining_minutes': '0.6422'}
111
+ {'loss': '4.262', 'grad_norm': '1.883', 'learning_rate': '0.0003', 'epoch': '0.01786', 'train/total_time_seconds': '10.3', 'train/time_per_step_avg': '0.008911', 'train/epoch_time_elapsed': '61.75', 'train/estimated_remaining_minutes': '0.638'}
112
+ {'loss': '4.262', 'grad_norm': '2.312', 'learning_rate': '0.0003', 'epoch': '0.0182', 'train/total_time_seconds': '10.48', 'train/time_per_step_avg': '0.008912', 'train/epoch_time_elapsed': '62.18', 'train/estimated_remaining_minutes': '0.6338'}
113
+ {'loss': '4.222', 'grad_norm': '2.25', 'learning_rate': '0.0003', 'epoch': '0.01854', 'train/total_time_seconds': '10.65', 'train/time_per_step_avg': '0.00891', 'train/epoch_time_elapsed': '62.61', 'train/estimated_remaining_minutes': '0.6295'}
114
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
115
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
116
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
117
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 362.99it/s]
118
+ 24%|██▍ | 1200/5000 [01:12<01:21, 46.5[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
119
+ {'loss': '4.216', 'grad_norm': '1.633', 'learning_rate': '0.0003', 'epoch': '0.01887', 'train/total_time_seconds': '10.85', 'train/time_per_step_avg': '0.009061', 'train/epoch_time_elapsed': '63.12', 'train/estimated_remaining_minutes': '0.6263'}
120
+ {'loss': '4.189', 'grad_norm': '1.844', 'learning_rate': '0.0003', 'epoch': '0.01921', 'train/total_time_seconds': '11.03', 'train/time_per_step_avg': '0.009092', 'train/epoch_time_elapsed': '63.55', 'train/estimated_remaining_minutes': '0.6224'}
121
+ {'loss': '4.164', 'grad_norm': '2.547', 'learning_rate': '0.0003', 'epoch': '0.01955', 'train/total_time_seconds': '11.21', 'train/time_per_step_avg': '0.009093', 'train/epoch_time_elapsed': '63.99', 'train/estimated_remaining_minutes': '0.6183'}
122
+ {'loss': '4.162', 'grad_norm': '1.461', 'learning_rate': '0.0003', 'epoch': '0.01989', 'train/total_time_seconds': '11.38', 'train/time_per_step_avg': '0.009057', 'train/epoch_time_elapsed': '64.42', 'train/estimated_remaining_minutes': '0.6141'}
123
+ {'loss': '4.131', 'grad_norm': '1.562', 'learning_rate': '0.0003', 'epoch': '0.02022', 'train/total_time_seconds': '11.56', 'train/time_per_step_avg': '0.009057', 'train/epoch_time_elapsed': '64.85', 'train/estimated_remaining_minutes': '0.61'}
124
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
125
+ {'eval_loss': '4.134', 'eval_runtime': '7.466', 'eval_samples_per_second': '1276', 'eval_steps_per_second': '0.938', 'epoch': '0.02022', 'train/total_time_seconds': '11.56', 'train/time_per_step_avg': '0.009057', 'train/epoch_time_elapsed': '72.32', 'train/estimated_remaining_minutes': '0.61'}
126
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
127
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
128
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 343.71it/s]
129
+ 26%|██▌ | 1300/5000 [01:14<01:21, 45.35it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
130
+ {'loss': '4.11', 'grad_norm': '1.289', 'learning_rate': '0.0003', 'epoch': '0.02056', 'train/total_time_seconds': '11.74', 'train/time_per_step_avg': '0.00896', 'train/epoch_time_elapsed': '72.8', 'train/estimated_remaining_minutes': '0.6064'}
131
+ {'loss': '4.086', 'grad_norm': '2.109', 'learning_rate': '0.0003', 'epoch': '0.0209', 'train/total_time_seconds': '11.93', 'train/time_per_step_avg': '0.009023', 'train/epoch_time_elapsed': '73.24', 'train/estimated_remaining_minutes': '0.603'}
132
+ {'loss': '4.069', 'grad_norm': '1.586', 'learning_rate': '0.0003', 'epoch': '0.02123', 'train/total_time_seconds': '12.11', 'train/time_per_step_avg': '0.009003', 'train/epoch_time_elapsed': '73.68', 'train/estimated_remaining_minutes': '0.599'}
133
+ {'loss': '4.091', 'grad_norm': '1.789', 'learning_rate': '0.0003', 'epoch': '0.02157', 'train/total_time_seconds': '12.28', 'train/time_per_step_avg': '0.009016', 'train/epoch_time_elapsed': '74.11', 'train/estimated_remaining_minutes': '0.595'}
134
+ {'loss': '4.062', 'grad_norm': '1.484', 'learning_rate': '0.0003', 'epoch': '0.02191', 'train/total_time_seconds': '12.46', 'train/time_per_step_avg': '0.00901', 'train/epoch_time_elapsed': '74.54', 'train/estimated_remaining_minutes': '0.591'}
135
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
136
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
137
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
138
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 356.33it/s]
139
+ 28%|██▊ | 1400/5000 [01:24<01:17, 46.6[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
140
+ {'loss': '4.038', 'grad_norm': '1.562', 'learning_rate': '0.0003', 'epoch': '0.02224', 'train/total_time_seconds': '12.64', 'train/time_per_step_avg': '0.008917', 'train/epoch_time_elapsed': '74.99', 'train/estimated_remaining_minutes': '0.5871'}
141
+ {'loss': '4.031', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02258', 'train/total_time_seconds': '12.81', 'train/time_per_step_avg': '0.008809', 'train/epoch_time_elapsed': '75.42', 'train/estimated_remaining_minutes': '0.5832'}
142
+ {'loss': '4.002', 'grad_norm': '1.438', 'learning_rate': '0.0003', 'epoch': '0.02292', 'train/total_time_seconds': '12.99', 'train/time_per_step_avg': '0.008808', 'train/epoch_time_elapsed': '75.85', 'train/estimated_remaining_minutes': '0.5794'}
143
+ {'loss': '4.032', 'grad_norm': '1.773', 'learning_rate': '0.0003', 'epoch': '0.02326', 'train/total_time_seconds': '13.16', 'train/time_per_step_avg': '0.008808', 'train/epoch_time_elapsed': '76.28', 'train/estimated_remaining_minutes': '0.5756'}
144
+ {'loss': '4.004', 'grad_norm': '1.461', 'learning_rate': '0.0003', 'epoch': '0.02359', 'train/total_time_seconds': '13.34', 'train/time_per_step_avg': '0.008812', 'train/epoch_time_elapsed': '76.71', 'train/estimated_remaining_minutes': '0.5718'}
145
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
146
+ {'eval_loss': '4', 'eval_runtime': '7.501', 'eval_samples_per_second': '1270', 'eval_steps_per_second': '0.933', 'epoch': '0.02359', 'train/total_time_seconds': '13.34', 'train/time_per_step_avg': '0.008812', 'train/epoch_time_elapsed': '84.21', 'train/estimated_remaining_minutes': '0.5718'}
147
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
148
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
149
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 369.12it/s]
150
+ 30%|███ | 1500/5000 [01:26<01:16, 45.66it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
151
+ {'loss': '4.006', 'grad_norm': '1.531', 'learning_rate': '0.0003', 'epoch': '0.02393', 'train/total_time_seconds': '13.52', 'train/time_per_step_avg': '0.008825', 'train/epoch_time_elapsed': '84.67', 'train/estimated_remaining_minutes': '0.568'}
152
+ {'loss': '3.973', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02427', 'train/total_time_seconds': '13.69', 'train/time_per_step_avg': '0.008785', 'train/epoch_time_elapsed': '85.09', 'train/estimated_remaining_minutes': '0.5641'}
153
+ {'loss': '4.007', 'grad_norm': '1.375', 'learning_rate': '0.0003', 'epoch': '0.0246', 'train/total_time_seconds': '13.87', 'train/time_per_step_avg': '0.008768', 'train/epoch_time_elapsed': '85.52', 'train/estimated_remaining_minutes': '0.5603'}
154
+ {'loss': '3.955', 'grad_norm': '2.031', 'learning_rate': '0.0003', 'epoch': '0.02494', 'train/total_time_seconds': '14.04', 'train/time_per_step_avg': '0.008757', 'train/epoch_time_elapsed': '85.95', 'train/estimated_remaining_minutes': '0.5566'}
155
+ {'loss': '3.959', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02528', 'train/total_time_seconds': '14.22', 'train/time_per_step_avg': '0.008745', 'train/epoch_time_elapsed': '86.38', 'train/estimated_remaining_minutes': '0.5528'}
156
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
157
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
158
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
159
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 366.83it/s]
160
+ 32%|███▏ | 1600/5000 [01:36<01:13, 46.0[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
161
+ {'loss': '3.925', 'grad_norm': '1.648', 'learning_rate': '0.0003', 'epoch': '0.02562', 'train/total_time_seconds': '14.39', 'train/time_per_step_avg': '0.008722', 'train/epoch_time_elapsed': '86.84', 'train/estimated_remaining_minutes': '0.5491'}
162
+ {'loss': '3.916', 'grad_norm': '1.531', 'learning_rate': '0.0003', 'epoch': '0.02595', 'train/total_time_seconds': '14.56', 'train/time_per_step_avg': '0.008724', 'train/epoch_time_elapsed': '87.27', 'train/estimated_remaining_minutes': '0.5453'}
163
+ {'loss': '3.9', 'grad_norm': '1.469', 'learning_rate': '0.0003', 'epoch': '0.02629', 'train/total_time_seconds': '14.74', 'train/time_per_step_avg': '0.008734', 'train/epoch_time_elapsed': '87.69', 'train/estimated_remaining_minutes': '0.5417'}
164
+ {'loss': '3.896', 'grad_norm': '1.508', 'learning_rate': '0.0003', 'epoch': '0.02663', 'train/total_time_seconds': '14.91', 'train/time_per_step_avg': '0.00871', 'train/epoch_time_elapsed': '88.12', 'train/estimated_remaining_minutes': '0.5379'}
165
+ {'loss': '3.939', 'grad_norm': '1.445', 'learning_rate': '0.0003', 'epoch': '0.02696', 'train/total_time_seconds': '15.1', 'train/time_per_step_avg': '0.008816', 'train/epoch_time_elapsed': '88.57', 'train/estimated_remaining_minutes': '0.5347'}
166
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
167
+ {'eval_loss': '3.91', 'eval_runtime': '7.782', 'eval_samples_per_second': '1224', 'eval_steps_per_second': '0.899', 'epoch': '0.02696', 'train/total_time_seconds': '15.1', 'train/time_per_step_avg': '0.008816', 'train/epoch_time_elapsed': '96.35', 'train/estimated_remaining_minutes': '0.5347'}
168
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
169
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
170
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 362.39it/s]
171
+ 34%|███▍ | 1700/5000 [01:38<01:12, 45.83it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
172
+ {'loss': '3.894', 'grad_norm': '1.523', 'learning_rate': '0.0003', 'epoch': '0.0273', 'train/total_time_seconds': '15.27', 'train/time_per_step_avg': '0.008808', 'train/epoch_time_elapsed': '96.8', 'train/estimated_remaining_minutes': '0.531'}
173
+ {'loss': '3.874', 'grad_norm': '1.656', 'learning_rate': '0.0003', 'epoch': '0.02764', 'train/total_time_seconds': '15.44', 'train/time_per_step_avg': '0.008805', 'train/epoch_time_elapsed': '97.23', 'train/estimated_remaining_minutes': '0.5273'}
174
+ {'loss': '3.875', 'grad_norm': '1.805', 'learning_rate': '0.0003', 'epoch': '0.02797', 'train/total_time_seconds': '15.62', 'train/time_per_step_avg': '0.008784', 'train/epoch_time_elapsed': '97.66', 'train/estimated_remaining_minutes': '0.5237'}
175
+ {'loss': '3.885', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.02831', 'train/total_time_seconds': '15.79', 'train/time_per_step_avg': '0.008792', 'train/epoch_time_elapsed': '98.08', 'train/estimated_remaining_minutes': '0.5201'}
176
+ {'loss': '3.901', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.02865', 'train/total_time_seconds': '15.96', 'train/time_per_step_avg': '0.008662', 'train/epoch_time_elapsed': '98.51', 'train/estimated_remaining_minutes': '0.5165'}
177
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
178
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
179
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
180
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 362.11it/s]
181
+ 36%|███▌ | 1800/5000 [01:48<01:10, 45.7[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
182
+ {'loss': '3.914', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02899', 'train/total_time_seconds': '16.14', 'train/time_per_step_avg': '0.008648', 'train/epoch_time_elapsed': '98.96', 'train/estimated_remaining_minutes': '0.5128'}
183
+ {'loss': '3.865', 'grad_norm': '1.492', 'learning_rate': '0.0003', 'epoch': '0.02932', 'train/total_time_seconds': '16.32', 'train/time_per_step_avg': '0.008794', 'train/epoch_time_elapsed': '99.41', 'train/estimated_remaining_minutes': '0.5097'}
184
+ {'loss': '3.865', 'grad_norm': '1.922', 'learning_rate': '0.0003', 'epoch': '0.02966', 'train/total_time_seconds': '16.5', 'train/time_per_step_avg': '0.008799', 'train/epoch_time_elapsed': '99.84', 'train/estimated_remaining_minutes': '0.5062'}
185
+ {'loss': '3.826', 'grad_norm': '1.781', 'learning_rate': '0.0003', 'epoch': '0.03', 'train/total_time_seconds': '16.67', 'train/time_per_step_avg': '0.008814', 'train/epoch_time_elapsed': '100.3', 'train/estimated_remaining_minutes': '0.5027'}
186
+ {'loss': '3.843', 'grad_norm': '1.398', 'learning_rate': '0.0003', 'epoch': '0.03033', 'train/total_time_seconds': '16.85', 'train/time_per_step_avg': '0.008869', 'train/epoch_time_elapsed': '100.7', 'train/estimated_remaining_minutes': '0.4993'}
187
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
188
+ {'eval_loss': '3.846', 'eval_runtime': '7.438', 'eval_samples_per_second': '1281', 'eval_steps_per_second': '0.941', 'epoch': '0.03033', 'train/total_time_seconds': '16.85', 'train/time_per_step_avg': '0.008869', 'train/epoch_time_elapsed': '108.2', 'train/estimated_remaining_minutes': '0.4993'}
189
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
190
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
191
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 369.54it/s]
192
+ 38%|███▊ | 1900/5000 [01:50<01:10, 43.88it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
193
+ {'loss': '3.854', 'grad_norm': '1.289', 'learning_rate': '0.0003', 'epoch': '0.03067', 'train/total_time_seconds': '17.03', 'train/time_per_step_avg': '0.008931', 'train/epoch_time_elapsed': '108.6', 'train/estimated_remaining_minutes': '0.4959'}
194
+ {'loss': '3.843', 'grad_norm': '1.531', 'learning_rate': '0.0003', 'epoch': '0.03101', 'train/total_time_seconds': '17.2', 'train/time_per_step_avg': '0.008786', 'train/epoch_time_elapsed': '109', 'train/estimated_remaining_minutes': '0.4924'}
195
+ {'loss': '3.841', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.03134', 'train/total_time_seconds': '17.37', 'train/time_per_step_avg': '0.008777', 'train/epoch_time_elapsed': '109.5', 'train/estimated_remaining_minutes': '0.4889'}
196
+ {'loss': '3.831', 'grad_norm': '1.109', 'learning_rate': '0.0003', 'epoch': '0.03168', 'train/total_time_seconds': '17.56', 'train/time_per_step_avg': '0.008864', 'train/epoch_time_elapsed': '109.9', 'train/estimated_remaining_minutes': '0.4857'}
197
+ {'loss': '3.804', 'grad_norm': '1.414', 'learning_rate': '0.0003', 'epoch': '0.03202', 'train/total_time_seconds': '17.74', 'train/time_per_step_avg': '0.008891', 'train/epoch_time_elapsed': '110.4', 'train/estimated_remaining_minutes': '0.4824'}
198
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
199
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
200
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
201
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 346.09it/s]
202
+ 40%|████ | 2000/5000 [02:00<01:04, 46.8[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
203
+ {'loss': '3.816', 'grad_norm': '1.836', 'learning_rate': '0.0003', 'epoch': '0.03236', 'train/total_time_seconds': '17.91', 'train/time_per_step_avg': '0.008847', 'train/epoch_time_elapsed': '110.8', 'train/estimated_remaining_minutes': '0.4789'}
204
+ {'loss': '3.799', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.03269', 'train/total_time_seconds': '18.09', 'train/time_per_step_avg': '0.008862', 'train/epoch_time_elapsed': '111.2', 'train/estimated_remaining_minutes': '0.4755'}
205
+ {'loss': '3.801', 'grad_norm': '2.078', 'learning_rate': '0.0003', 'epoch': '0.03303', 'train/total_time_seconds': '18.26', 'train/time_per_step_avg': '0.008862', 'train/epoch_time_elapsed': '111.7', 'train/estimated_remaining_minutes': '0.4721'}
206
+ {'loss': '3.79', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.03337', 'train/total_time_seconds': '18.44', 'train/time_per_step_avg': '0.00879', 'train/epoch_time_elapsed': '112.1', 'train/estimated_remaining_minutes': '0.4687'}
207
+ {'loss': '3.81', 'grad_norm': '1.695', 'learning_rate': '0.0003', 'epoch': '0.0337', 'train/total_time_seconds': '18.61', 'train/time_per_step_avg': '0.008721', 'train/epoch_time_elapsed': '112.5', 'train/estimated_remaining_minutes': '0.4653'}
208
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
209
+ {'eval_loss': '3.798', 'eval_runtime': '7.486', 'eval_samples_per_second': '1273', 'eval_steps_per_second': '0.935', 'epoch': '0.0337', 'train/total_time_seconds': '18.61', 'train/time_per_step_avg': '0.008721', 'train/epoch_time_elapsed': '120', 'train/estimated_remaining_minutes': '0.4653'}
210
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
211
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
212
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 366.28it/s]
213
+ 42%|████▏ | 2100/5000 [02:02<01:03, 45.65it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
214
+ {'loss': '3.801', 'grad_norm': '1.773', 'learning_rate': '0.0003', 'epoch': '0.03404', 'train/total_time_seconds': '18.79', 'train/time_per_step_avg': '0.008809', 'train/epoch_time_elapsed': '120.5', 'train/estimated_remaining_minutes': '0.4621'}
215
+ {'loss': '3.787', 'grad_norm': '1.523', 'learning_rate': '0.0003', 'epoch': '0.03438', 'train/total_time_seconds': '18.97', 'train/time_per_step_avg': '0.008819', 'train/epoch_time_elapsed': '120.9', 'train/estimated_remaining_minutes': '0.4587'}
216
+ {'loss': '3.797', 'grad_norm': '1.469', 'learning_rate': '0.0003', 'epoch': '0.03472', 'train/total_time_seconds': '19.14', 'train/time_per_step_avg': '0.008838', 'train/epoch_time_elapsed': '121.3', 'train/estimated_remaining_minutes': '0.4554'}
217
+ {'loss': '3.76', 'grad_norm': '1.734', 'learning_rate': '0.0003', 'epoch': '0.03505', 'train/total_time_seconds': '19.33', 'train/time_per_step_avg': '0.00888', 'train/epoch_time_elapsed': '121.8', 'train/estimated_remaining_minutes': '0.4522'}
218
+ {'loss': '3.766', 'grad_norm': '1.477', 'learning_rate': '0.0003', 'epoch': '0.03539', 'train/total_time_seconds': '19.5', 'train/time_per_step_avg': '0.008889', 'train/epoch_time_elapsed': '122.2', 'train/estimated_remaining_minutes': '0.4488'}
219
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
220
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
221
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
222
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 367.02it/s]
223
+ 44%|████▍ | 2200/5000 [02:11<00:59, 47.0[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
224
+ {'loss': '3.766', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.03573', 'train/total_time_seconds': '19.68', 'train/time_per_step_avg': '0.00883', 'train/epoch_time_elapsed': '122.7', 'train/estimated_remaining_minutes': '0.4455'}
225
+ {'loss': '3.771', 'grad_norm': '1.508', 'learning_rate': '0.0003', 'epoch': '0.03606', 'train/total_time_seconds': '19.86', 'train/time_per_step_avg': '0.008901', 'train/epoch_time_elapsed': '123.1', 'train/estimated_remaining_minutes': '0.4424'}
226
+ {'loss': '3.744', 'grad_norm': '1.945', 'learning_rate': '0.0003', 'epoch': '0.0364', 'train/total_time_seconds': '20.03', 'train/time_per_step_avg': '0.0089', 'train/epoch_time_elapsed': '123.5', 'train/estimated_remaining_minutes': '0.439'}
227
+ {'loss': '3.753', 'grad_norm': '1.391', 'learning_rate': '0.0003', 'epoch': '0.03674', 'train/total_time_seconds': '20.21', 'train/time_per_step_avg': '0.008871', 'train/epoch_time_elapsed': '124', 'train/estimated_remaining_minutes': '0.4358'}
228
+ {'loss': '3.753', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.03707', 'train/total_time_seconds': '20.39', 'train/time_per_step_avg': '0.008849', 'train/epoch_time_elapsed': '124.4', 'train/estimated_remaining_minutes': '0.4324'}
229
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
230
+ {'eval_loss': '3.762', 'eval_runtime': '7.429', 'eval_samples_per_second': '1282', 'eval_steps_per_second': '0.942', 'epoch': '0.03707', 'train/total_time_seconds': '20.39', 'train/time_per_step_avg': '0.008849', 'train/epoch_time_elapsed': '131.8', 'train/estimated_remaining_minutes': '0.4324'}
231
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
232
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
233
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 367.24it/s]
234
+ 46%|████▌ | 2300/5000 [02:13<00:58, 45.99it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
235
+ {'loss': '3.753', 'grad_norm': '1.766', 'learning_rate': '0.0003', 'epoch': '0.03741', 'train/total_time_seconds': '20.56', 'train/time_per_step_avg': '0.00884', 'train/epoch_time_elapsed': '132.3', 'train/estimated_remaining_minutes': '0.4291'}
236
+ {'loss': '3.759', 'grad_norm': '1.766', 'learning_rate': '0.0003', 'epoch': '0.03775', 'train/total_time_seconds': '20.74', 'train/time_per_step_avg': '0.008766', 'train/epoch_time_elapsed': '132.7', 'train/estimated_remaining_minutes': '0.4258'}
237
+ {'loss': '3.744', 'grad_norm': '1.719', 'learning_rate': '0.0003', 'epoch': '0.03809', 'train/total_time_seconds': '20.92', 'train/time_per_step_avg': '0.008812', 'train/epoch_time_elapsed': '133.1', 'train/estimated_remaining_minutes': '0.4226'}
238
+ {'loss': '3.729', 'grad_norm': '1.859', 'learning_rate': '0.0003', 'epoch': '0.03842', 'train/total_time_seconds': '21.09', 'train/time_per_step_avg': '0.008765', 'train/epoch_time_elapsed': '133.6', 'train/estimated_remaining_minutes': '0.4193'}
239
+ {'loss': '3.75', 'grad_norm': '1.609', 'learning_rate': '0.0003', 'epoch': '0.03876', 'train/total_time_seconds': '21.26', 'train/time_per_step_avg': '0.008778', 'train/epoch_time_elapsed': '134', 'train/estimated_remaining_minutes': '0.416'}
240
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
241
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
242
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
243
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 357.51it/s]
244
+ 48%|████▊ | 2400/5000 [02:23<00:55, 46.7[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
245
+ {'loss': '3.717', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.0391', 'train/total_time_seconds': '21.44', 'train/time_per_step_avg': '0.008763', 'train/epoch_time_elapsed': '134.4', 'train/estimated_remaining_minutes': '0.4127'}
246
+ {'loss': '3.749', 'grad_norm': '1.344', 'learning_rate': '0.0003', 'epoch': '0.03943', 'train/total_time_seconds': '21.61', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '134.9', 'train/estimated_remaining_minutes': '0.4095'}
247
+ {'loss': '3.749', 'grad_norm': '1.586', 'learning_rate': '0.0003', 'epoch': '0.03977', 'train/total_time_seconds': '21.79', 'train/time_per_step_avg': '0.008712', 'train/epoch_time_elapsed': '135.3', 'train/estimated_remaining_minutes': '0.4062'}
248
+ {'loss': '3.721', 'grad_norm': '1.422', 'learning_rate': '0.0003', 'epoch': '0.04011', 'train/total_time_seconds': '21.96', 'train/time_per_step_avg': '0.008734', 'train/epoch_time_elapsed': '135.7', 'train/estimated_remaining_minutes': '0.403'}
249
+ {'loss': '3.724', 'grad_norm': '1.711', 'learning_rate': '0.0003', 'epoch': '0.04044', 'train/total_time_seconds': '22.14', 'train/time_per_step_avg': '0.00876', 'train/epoch_time_elapsed': '136.1', 'train/estimated_remaining_minutes': '0.3997'}
250
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
251
+ {'eval_loss': '3.733', 'eval_runtime': '7.418', 'eval_samples_per_second': '1284', 'eval_steps_per_second': '0.944', 'epoch': '0.04044', 'train/total_time_seconds': '22.14', 'train/time_per_step_avg': '0.00876', 'train/epoch_time_elapsed': '143.6', 'train/estimated_remaining_minutes': '0.3997'}
252
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
253
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
254
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 225.17it/s]
255
+ 50%|█████ | 2500/5000 [02:25<00:54, 45.48it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
256
+ {'loss': '3.712', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.04078', 'train/total_time_seconds': '22.32', 'train/time_per_step_avg': '0.008778', 'train/epoch_time_elapsed': '144', 'train/estimated_remaining_minutes': '0.3965'}
257
+ {'loss': '3.696', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.04112', 'train/total_time_seconds': '22.49', 'train/time_per_step_avg': '0.008795', 'train/epoch_time_elapsed': '144.4', 'train/estimated_remaining_minutes': '0.3933'}
258
+ {'loss': '3.701', 'grad_norm': '1.508', 'learning_rate': '0.0003', 'epoch': '0.04146', 'train/total_time_seconds': '22.67', 'train/time_per_step_avg': '0.008795', 'train/epoch_time_elapsed': '144.9', 'train/estimated_remaining_minutes': '0.3901'}
259
+ {'loss': '3.731', 'grad_norm': '1.328', 'learning_rate': '0.0003', 'epoch': '0.04179', 'train/total_time_seconds': '22.84', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '145.3', 'train/estimated_remaining_minutes': '0.3868'}
260
+ {'loss': '3.728', 'grad_norm': '1.742', 'learning_rate': '0.0003', 'epoch': '0.04213', 'train/total_time_seconds': '23.01', 'train/time_per_step_avg': '0.008735', 'train/epoch_time_elapsed': '145.7', 'train/estimated_remaining_minutes': '0.3835'}
261
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
262
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
263
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
264
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 365.10it/s]
265
+ 52%|█████▏ | 2600/5000 [02:35<00:50, 47.2[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
266
+ {'loss': '3.724', 'grad_norm': '1.312', 'learning_rate': '0.0003', 'epoch': '0.04247', 'train/total_time_seconds': '23.19', 'train/time_per_step_avg': '0.008701', 'train/epoch_time_elapsed': '146.2', 'train/estimated_remaining_minutes': '0.3803'}
267
+ {'loss': '3.72', 'grad_norm': '1.727', 'learning_rate': '0.0003', 'epoch': '0.0428', 'train/total_time_seconds': '23.36', 'train/time_per_step_avg': '0.008655', 'train/epoch_time_elapsed': '146.6', 'train/estimated_remaining_minutes': '0.377'}
268
+ {'loss': '3.702', 'grad_norm': '2.094', 'learning_rate': '0.0003', 'epoch': '0.04314', 'train/total_time_seconds': '23.53', 'train/time_per_step_avg': '0.008653', 'train/epoch_time_elapsed': '147', 'train/estimated_remaining_minutes': '0.3738'}
269
+ {'loss': '3.713', 'grad_norm': '1.789', 'learning_rate': '0.0003', 'epoch': '0.04348', 'train/total_time_seconds': '23.7', 'train/time_per_step_avg': '0.008646', 'train/epoch_time_elapsed': '147.4', 'train/estimated_remaining_minutes': '0.3706'}
270
+ {'loss': '3.695', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.04382', 'train/total_time_seconds': '23.88', 'train/time_per_step_avg': '0.00863', 'train/epoch_time_elapsed': '147.9', 'train/estimated_remaining_minutes': '0.3673'}
271
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
272
+ {'eval_loss': '3.698', 'eval_runtime': '7.608', 'eval_samples_per_second': '1252', 'eval_steps_per_second': '0.92', 'epoch': '0.04382', 'train/total_time_seconds': '23.88', 'train/time_per_step_avg': '0.00863', 'train/epoch_time_elapsed': '155.5', 'train/estimated_remaining_minutes': '0.3673'}
273
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
274
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
275
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 352.67it/s]
276
+ 54%|█████▍ | 2700/5000 [02:37<00:50, 45.39it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
277
+ {'loss': '3.661', 'grad_norm': '1.289', 'learning_rate': '0.0003', 'epoch': '0.04415', 'train/total_time_seconds': '24.05', 'train/time_per_step_avg': '0.008667', 'train/epoch_time_elapsed': '155.9', 'train/estimated_remaining_minutes': '0.3642'}
278
+ {'loss': '3.69', 'grad_norm': '1.703', 'learning_rate': '0.0003', 'epoch': '0.04449', 'train/total_time_seconds': '24.23', 'train/time_per_step_avg': '0.008721', 'train/epoch_time_elapsed': '156.4', 'train/estimated_remaining_minutes': '0.361'}
279
+ {'loss': '3.728', 'grad_norm': '1.523', 'learning_rate': '0.0003', 'epoch': '0.04483', 'train/total_time_seconds': '24.42', 'train/time_per_step_avg': '0.008879', 'train/epoch_time_elapsed': '156.8', 'train/estimated_remaining_minutes': '0.358'}
280
+ {'loss': '3.673', 'grad_norm': '1.594', 'learning_rate': '0.0003', 'epoch': '0.04516', 'train/total_time_seconds': '24.59', 'train/time_per_step_avg': '0.008906', 'train/epoch_time_elapsed': '157.3', 'train/estimated_remaining_minutes': '0.3548'}
281
+ {'loss': '3.679', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.0455', 'train/total_time_seconds': '24.77', 'train/time_per_step_avg': '0.008927', 'train/epoch_time_elapsed': '157.7', 'train/estimated_remaining_minutes': '0.3516'}
282
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
283
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
284
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
285
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 364.63it/s]
286
+ 56%|█████▌ | 2800/5000 [02:47<00:47, 46.8[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
287
+ {'loss': '3.689', 'grad_norm': '1.492', 'learning_rate': '0.0003', 'epoch': '0.04584', 'train/total_time_seconds': '24.95', 'train/time_per_step_avg': '0.008949', 'train/epoch_time_elapsed': '158.1', 'train/estimated_remaining_minutes': '0.3485'}
288
+ {'loss': '3.657', 'grad_norm': '1.383', 'learning_rate': '0.0003', 'epoch': '0.04617', 'train/total_time_seconds': '25.13', 'train/time_per_step_avg': '0.009017', 'train/epoch_time_elapsed': '158.6', 'train/estimated_remaining_minutes': '0.3455'}
289
+ {'loss': '3.708', 'grad_norm': '1.812', 'learning_rate': '0.0003', 'epoch': '0.04651', 'train/total_time_seconds': '25.31', 'train/time_per_step_avg': '0.008923', 'train/epoch_time_elapsed': '159', 'train/estimated_remaining_minutes': '0.3424'}
290
+ {'loss': '3.67', 'grad_norm': '1.828', 'learning_rate': '0.0003', 'epoch': '0.04685', 'train/total_time_seconds': '25.49', 'train/time_per_step_avg': '0.00893', 'train/epoch_time_elapsed': '159.4', 'train/estimated_remaining_minutes': '0.3392'}
291
+ {'loss': '3.652', 'grad_norm': '1.477', 'learning_rate': '0.0003', 'epoch': '0.04719', 'train/total_time_seconds': '25.66', 'train/time_per_step_avg': '0.008942', 'train/epoch_time_elapsed': '159.9', 'train/estimated_remaining_minutes': '0.3361'}
292
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
293
+ {'eval_loss': '3.671', 'eval_runtime': '7.417', 'eval_samples_per_second': '1285', 'eval_steps_per_second': '0.944', 'epoch': '0.04719', 'train/total_time_seconds': '25.66', 'train/time_per_step_avg': '0.008942', 'train/epoch_time_elapsed': '167.3', 'train/estimated_remaining_minutes': '0.3361'}
294
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
295
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
296
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 193.85it/s]
297
+ 58%|█████▊ | 2900/5000 [02:49<00:45, 45.75it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
298
+ {'loss': '3.684', 'grad_norm': '1.172', 'learning_rate': '0.0003', 'epoch': '0.04752', 'train/total_time_seconds': '25.84', 'train/time_per_step_avg': '0.008939', 'train/epoch_time_elapsed': '167.7', 'train/estimated_remaining_minutes': '0.3329'}
299
+ {'loss': '3.651', 'grad_norm': '1.672', 'learning_rate': '0.0003', 'epoch': '0.04786', 'train/total_time_seconds': '26.02', 'train/time_per_step_avg': '0.008859', 'train/epoch_time_elapsed': '168.2', 'train/estimated_remaining_minutes': '0.3298'}
300
+ {'loss': '3.654', 'grad_norm': '1.898', 'learning_rate': '0.0003', 'epoch': '0.0482', 'train/total_time_seconds': '26.19', 'train/time_per_step_avg': '0.008817', 'train/epoch_time_elapsed': '168.6', 'train/estimated_remaining_minutes': '0.3267'}
301
+ {'loss': '3.658', 'grad_norm': '1.469', 'learning_rate': '0.0003', 'epoch': '0.04853', 'train/total_time_seconds': '26.37', 'train/time_per_step_avg': '0.008839', 'train/epoch_time_elapsed': '169', 'train/estimated_remaining_minutes': '0.3235'}
302
+ {'loss': '3.643', 'grad_norm': '1.375', 'learning_rate': '0.0003', 'epoch': '0.04887', 'train/total_time_seconds': '26.55', 'train/time_per_step_avg': '0.008835', 'train/epoch_time_elapsed': '169.5', 'train/estimated_remaining_minutes': '0.3204'}
303
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
304
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
305
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
306
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 364.25it/s]
307
+ 60%|██████ | 3000/5000 [02:59<00:43, 46.4[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
308
+ {'loss': '3.635', 'grad_norm': '1.812', 'learning_rate': '0.0003', 'epoch': '0.04921', 'train/total_time_seconds': '26.72', 'train/time_per_step_avg': '0.008811', 'train/epoch_time_elapsed': '169.9', 'train/estimated_remaining_minutes': '0.3173'}
309
+ {'loss': '3.661', 'grad_norm': '1.492', 'learning_rate': '0.0003', 'epoch': '0.04954', 'train/total_time_seconds': '26.9', 'train/time_per_step_avg': '0.008801', 'train/epoch_time_elapsed': '170.3', 'train/estimated_remaining_minutes': '0.3141'}
310
+ {'loss': '3.663', 'grad_norm': '1.445', 'learning_rate': '0.0003', 'epoch': '0.04988', 'train/total_time_seconds': '27.07', 'train/time_per_step_avg': '0.008791', 'train/epoch_time_elapsed': '170.8', 'train/estimated_remaining_minutes': '0.311'}
311
+ {'loss': '3.671', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.05022', 'train/total_time_seconds': '27.25', 'train/time_per_step_avg': '0.008756', 'train/epoch_time_elapsed': '171.2', 'train/estimated_remaining_minutes': '0.3078'}
312
+ {'loss': '3.639', 'grad_norm': '1.68', 'learning_rate': '0.0003', 'epoch': '0.05056', 'train/total_time_seconds': '27.42', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '171.6', 'train/estimated_remaining_minutes': '0.3047'}
313
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
314
+ {'eval_loss': '3.654', 'eval_runtime': '7.423', 'eval_samples_per_second': '1283', 'eval_steps_per_second': '0.943', 'epoch': '0.05056', 'train/total_time_seconds': '27.42', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '179', 'train/estimated_remaining_minutes': '0.3047'}
315
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
316
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
317
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 368.02it/s]
318
+ 62%|██████▏ | 3100/5000 [03:01<00:40, 46.37it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
319
+ {'loss': '3.659', 'grad_norm': '1.453', 'learning_rate': '0.0003', 'epoch': '0.05089', 'train/total_time_seconds': '27.64', 'train/time_per_step_avg': '0.009176', 'train/epoch_time_elapsed': '179.6', 'train/estimated_remaining_minutes': '0.302'}
320
+ {'loss': '3.64', 'grad_norm': '1.648', 'learning_rate': '0.0003', 'epoch': '0.05123', 'train/total_time_seconds': '27.81', 'train/time_per_step_avg': '0.00917', 'train/epoch_time_elapsed': '180', 'train/estimated_remaining_minutes': '0.2989'}
321
+ {'loss': '3.672', 'grad_norm': '1.477', 'learning_rate': '0.0003', 'epoch': '0.05157', 'train/total_time_seconds': '27.99', 'train/time_per_step_avg': '0.009162', 'train/epoch_time_elapsed': '180.4', 'train/estimated_remaining_minutes': '0.2957'}
322
+ {'loss': '3.65', 'grad_norm': '1.43', 'learning_rate': '0.0003', 'epoch': '0.0519', 'train/total_time_seconds': '28.16', 'train/time_per_step_avg': '0.009174', 'train/epoch_time_elapsed': '180.9', 'train/estimated_remaining_minutes': '0.2926'}
323
+ {'loss': '3.689', 'grad_norm': '1.797', 'learning_rate': '0.0003', 'epoch': '0.05224', 'train/total_time_seconds': '28.34', 'train/time_per_step_avg': '0.009164', 'train/epoch_time_elapsed': '181.3', 'train/estimated_remaining_minutes': '0.2895'}
324
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
325
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
326
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
327
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 367.66it/s]
328
+ 64%|██████▍ | 3200/5000 [03:11<00:38, 46.8[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
329
+ {'loss': '3.639', 'grad_norm': '1.711', 'learning_rate': '0.0003', 'epoch': '0.05258', 'train/total_time_seconds': '28.53', 'train/time_per_step_avg': '0.008881', 'train/epoch_time_elapsed': '181.8', 'train/estimated_remaining_minutes': '0.2865'}
330
+ {'loss': '3.633', 'grad_norm': '1.828', 'learning_rate': '0.0003', 'epoch': '0.05292', 'train/total_time_seconds': '28.7', 'train/time_per_step_avg': '0.008877', 'train/epoch_time_elapsed': '182.2', 'train/estimated_remaining_minutes': '0.2834'}
331
+ {'loss': '3.645', 'grad_norm': '1.461', 'learning_rate': '0.0003', 'epoch': '0.05325', 'train/total_time_seconds': '28.88', 'train/time_per_step_avg': '0.008892', 'train/epoch_time_elapsed': '182.6', 'train/estimated_remaining_minutes': '0.2803'}
332
+ {'loss': '3.639', 'grad_norm': '1.75', 'learning_rate': '0.0003', 'epoch': '0.05359', 'train/total_time_seconds': '29.06', 'train/time_per_step_avg': '0.008942', 'train/epoch_time_elapsed': '183.1', 'train/estimated_remaining_minutes': '0.2772'}
333
+ {'loss': '3.639', 'grad_norm': '1.398', 'learning_rate': '0.0003', 'epoch': '0.05393', 'train/total_time_seconds': '29.23', 'train/time_per_step_avg': '0.008957', 'train/epoch_time_elapsed': '183.5', 'train/estimated_remaining_minutes': '0.2741'}
334
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
335
+ {'eval_loss': '3.632', 'eval_runtime': '7.634', 'eval_samples_per_second': '1248', 'eval_steps_per_second': '0.917', 'epoch': '0.05393', 'train/total_time_seconds': '29.23', 'train/time_per_step_avg': '0.008957', 'train/epoch_time_elapsed': '191.1', 'train/estimated_remaining_minutes': '0.2741'}
336
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
337
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
338
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 231.50it/s]
339
+ 66%|██████▌ | 3300/5000 [03:13<00:37, 45.49it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
340
+ {'loss': '3.656', 'grad_norm': '1.43', 'learning_rate': '0.0003', 'epoch': '0.05426', 'train/total_time_seconds': '29.41', 'train/time_per_step_avg': '0.00882', 'train/epoch_time_elapsed': '191.6', 'train/estimated_remaining_minutes': '0.271'}
341
+ {'loss': '3.648', 'grad_norm': '1.664', 'learning_rate': '0.0003', 'epoch': '0.0546', 'train/total_time_seconds': '29.59', 'train/time_per_step_avg': '0.008838', 'train/epoch_time_elapsed': '192', 'train/estimated_remaining_minutes': '0.2679'}
342
+ {'loss': '3.649', 'grad_norm': '1.891', 'learning_rate': '0.0003', 'epoch': '0.05494', 'train/total_time_seconds': '29.76', 'train/time_per_step_avg': '0.008847', 'train/epoch_time_elapsed': '192.5', 'train/estimated_remaining_minutes': '0.2648'}
343
+ {'loss': '3.615', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.05527', 'train/total_time_seconds': '29.94', 'train/time_per_step_avg': '0.008814', 'train/epoch_time_elapsed': '192.9', 'train/estimated_remaining_minutes': '0.2617'}
344
+ {'loss': '3.658', 'grad_norm': '1.453', 'learning_rate': '0.0003', 'epoch': '0.05561', 'train/total_time_seconds': '30.12', 'train/time_per_step_avg': '0.008825', 'train/epoch_time_elapsed': '193.3', 'train/estimated_remaining_minutes': '0.2586'}
345
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
346
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
347
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
348
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 265.11it/s]
349
+ 68%|██████▊ | 3400/5000 [03:23<00:34, 46.3[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
350
+ {'loss': '3.633', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.05595', 'train/total_time_seconds': '30.29', 'train/time_per_step_avg': '0.00883', 'train/epoch_time_elapsed': '193.8', 'train/estimated_remaining_minutes': '0.2555'}
351
+ {'loss': '3.608', 'grad_norm': '1.898', 'learning_rate': '0.0003', 'epoch': '0.05629', 'train/total_time_seconds': '30.47', 'train/time_per_step_avg': '0.008815', 'train/epoch_time_elapsed': '194.2', 'train/estimated_remaining_minutes': '0.2524'}
352
+ {'loss': '3.628', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.05662', 'train/total_time_seconds': '30.64', 'train/time_per_step_avg': '0.008813', 'train/epoch_time_elapsed': '194.6', 'train/estimated_remaining_minutes': '0.2493'}
353
+ {'loss': '3.599', 'grad_norm': '1.43', 'learning_rate': '0.0003', 'epoch': '0.05696', 'train/total_time_seconds': '30.82', 'train/time_per_step_avg': '0.00884', 'train/epoch_time_elapsed': '195.1', 'train/estimated_remaining_minutes': '0.2462'}
354
+ {'loss': '3.626', 'grad_norm': '1.273', 'learning_rate': '0.0003', 'epoch': '0.0573', 'train/total_time_seconds': '31', 'train/time_per_step_avg': '0.008868', 'train/epoch_time_elapsed': '195.5', 'train/estimated_remaining_minutes': '0.2432'}
355
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
356
+ {'eval_loss': '3.615', 'eval_runtime': '7.519', 'eval_samples_per_second': '1267', 'eval_steps_per_second': '0.931', 'epoch': '0.0573', 'train/total_time_seconds': '31', 'train/time_per_step_avg': '0.008868', 'train/epoch_time_elapsed': '203', 'train/estimated_remaining_minutes': '0.2432'}
357
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
358
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
359
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 356.45it/s]
360
+ 70%|██████▉ | 3487/5000 [03:24<00:34, 44.45it/s], ?it/s]
361
+ {'loss': '3.591', 'grad_norm': '1.336', 'learning_rate': '0.0003', 'epoch': '0.05763', 'train/total_time_seconds': '31.19', 'train/time_per_step_avg': '0.008951', 'train/epoch_time_elapsed': '203.5', 'train/estimated_remaining_minutes': '0.2401'}
362
+ {'loss': '3.62', 'grad_norm': '1.562', 'learning_rate': '0.0003', 'epoch': '0.05797', 'train/total_time_seconds': '31.37', 'train/time_per_step_avg': '0.008993', 'train/epoch_time_elapsed': '203.9', 'train/estimated_remaining_minutes': '0.2371'}
363
+ {'loss': '3.599', 'grad_norm': '1.773', 'learning_rate': '0.0003', 'epoch': '0.05831', 'train/total_time_seconds': '31.55', 'train/time_per_step_avg': '0.009062', 'train/epoch_time_elapsed': '204.3', 'train/estimated_remaining_minutes': '0.234'}
364
+ {'loss': '3.616', 'grad_norm': '1.57', 'learning_rate': '0.0003', 'epoch': '0.05865', 'train/total_time_seconds': '31.73', 'train/time_per_step_avg': '0.00904', 'train/epoch_time_elapsed': '204.8', 'train/estimated_remaining_minutes': '0.231'}
zain/Activation/wandb/run-20260819_163014-smtfejrp/files/requirements.txt ADDED
@@ -0,0 +1,149 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ asttokens==3.0.1
2
+ comm==0.2.3
3
+ debugpy==1.8.21
4
+ decorator==5.3.1
5
+ executing==2.2.1
6
+ nest-asyncio==1.6.0
7
+ parso==0.8.7
8
+ platformdirs==4.11.0
9
+ psutil==7.2.2
10
+ ptyprocess==0.7.0
11
+ pure_eval==0.2.3
12
+ Pygments==2.20.0
13
+ pyzmq==27.1.0
14
+ setuptools==83.0.0
15
+ six==1.17.0
16
+ tornado==6.5.7
17
+ traitlets==5.15.0
18
+ fsspec==2026.4.0
19
+ wcwidth==0.8.2
20
+ ipython_pygments_lexers==1.1.1
21
+ jedi==0.20.0
22
+ jupyter_core==5.9.1
23
+ matplotlib-inline==0.2.2
24
+ pexpect==4.9.0
25
+ prompt_toolkit==3.0.53
26
+ python-dateutil==2.9.0.post0
27
+ stack_data==0.6.3
28
+ wheel==0.47.0
29
+ jupyter_client==8.9.1
30
+ pip==26.1.2
31
+ ipython==9.15.0
32
+ ipykernel==7.2.0
33
+ threadpoolctl==3.6.0
34
+ pyparsing==3.3.2
35
+ typing_extensions==4.15.0
36
+ Jinja2==3.1.6
37
+ narwhals==2.24.0
38
+ kiwisolver==1.5.0
39
+ joblib==1.5.3
40
+ fonttools==4.63.0
41
+ cycler==0.12.1
42
+ scipy==1.17.1
43
+ pandas==3.0.5
44
+ contourpy==1.3.3
45
+ scikit-learn==1.9.0
46
+ matplotlib==3.11.1
47
+ urllib3==2.7.0
48
+ tqdm==4.70.0
49
+ idna==3.18
50
+ charset-normalizer==3.4.9
51
+ certifi==2026.7.22
52
+ requests==2.34.2
53
+ seaborn==0.13.2
54
+ uv==0.12.0
55
+ shellingham==1.5.4
56
+ mpmath==1.3.0
57
+ attrs==26.1.0
58
+ hf-xet==1.5.2
59
+ nvidia-nccl-cu12==2.21.5
60
+ MarkupSafe==3.0.3
61
+ regex==2026.7.19
62
+ importlib_metadata==9.0.0
63
+ httpcore==1.0.9
64
+ annotated-doc==0.0.5
65
+ multidict==6.7.1
66
+ aiohttp==3.14.3
67
+ aiosignal==1.4.0
68
+ xxhash==3.8.1
69
+ aiohappyeyeballs==2.7.1
70
+ mdurl==0.1.2
71
+ cuda-toolkit==13.0.3.0
72
+ networkx==3.6.1
73
+ PyYAML==6.0.3
74
+ nvidia-cufile==1.15.1.6
75
+ typer==0.27.0
76
+ torchaudio==2.6.0+cu124
77
+ rich==15.0.0
78
+ nvidia-cufft-cu12==11.2.1.3
79
+ h11==0.16.0
80
+ dill==0.4.1
81
+ cuda-pathfinder==1.6.0
82
+ filelock==3.29.0
83
+ nvidia-nvtx-cu12==12.4.127
84
+ httpx==0.28.1
85
+ anyio==4.14.2
86
+ numpy==2.4.4
87
+ yarl==1.24.5
88
+ click==8.4.2
89
+ triton==3.2.0
90
+ frozenlist==1.8.0
91
+ zipp==4.1.0
92
+ propcache==0.5.2
93
+ markdown-it-py==4.2.0
94
+ nvidia-cuda-runtime==13.0.96
95
+ cuda-bindings==13.3.1
96
+ nvidia-cuda-cupti==13.0.85
97
+ torch==2.6.0+cu124
98
+ multiprocess==0.70.19
99
+ pillow==12.2.0
100
+ transformers==5.16.0.dev0
101
+ wandb==0.28.1
102
+ nvidia-curand==10.4.0.35
103
+ sympy==1.13.1
104
+ nvidia-cusparse==12.6.3.3
105
+ nvidia-cuda-nvrtc==13.0.88
106
+ typing-inspection==0.4.2
107
+ nvidia-cusolver==12.0.4.66
108
+ nvidia-cufft==12.0.0.61
109
+ nvidia-cudnn-cu13==9.20.0.48
110
+ nvidia-cublas==13.1.1.3
111
+ pyarrow==25.0.0
112
+ evaluate==0.4.6
113
+ diffusers==0.39.0
114
+ pydantic==2.13.4
115
+ annotated-types==0.8.0
116
+ protobuf==7.35.1
117
+ sentry-sdk==2.66.1
118
+ einops==0.8.2
119
+ packaging==26.2
120
+ nvidia-nvjitlink-cu12==12.4.127
121
+ nvidia-curand-cu12==10.3.5.147
122
+ nvidia-cusparselt-cu12==0.6.2
123
+ nvidia-cusparse-cu12==12.3.1.170
124
+ nvidia-cuda-runtime-cu12==12.4.127
125
+ torchvision==0.21.0+cu124
126
+ nvidia-cuda-nvrtc-cu12==12.4.127
127
+ nvidia-cuda-cupti-cu12==12.4.127
128
+ nvidia-cusolver-cu12==11.6.1.9
129
+ nvidia-cublas-cu12==12.4.5.8
130
+ nvidia-cudnn-cu12==9.1.0.70
131
+ huggingface_hub==1.26.0
132
+ datasets==5.0.1
133
+ safetensors==0.8.0
134
+ accelerate==1.14.0
135
+ pydantic_core==2.46.4
136
+ ninja==1.13.0
137
+ tokenizers==0.23.1
138
+ autocommand==2.2.2
139
+ backports.tarfile==1.2.0
140
+ importlib_metadata==8.7.1
141
+ jaraco.text==4.0.0
142
+ jaraco.context==6.1.0
143
+ jaraco.functools==4.4.0
144
+ more-itertools==10.8.0
145
+ packaging==26.0
146
+ platformdirs==4.4.0
147
+ tomli==2.4.0
148
+ wheel==0.46.3
149
+ zipp==3.23.0
zain/Activation/wandb/run-20260819_163014-smtfejrp/files/wandb-metadata.json ADDED
@@ -0,0 +1,118 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35",
3
+ "python": "CPython 3.11.15",
4
+ "startedAt": "2026-08-19T16:30:14.405953Z",
5
+ "args": [
6
+ "--config",
7
+ "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml",
8
+ "--variants",
9
+ "mlp-linear-3L",
10
+ "mlp-gelu-3L",
11
+ "mlp-relu-3L",
12
+ "mlp-silu-waleed10-3L",
13
+ "mlp-waleed10-3L",
14
+ "mlp-s10-3L",
15
+ "mlp-w1a-3L",
16
+ "mlp-silu-3L",
17
+ "mlp-sigmoid-3L",
18
+ "mlp-tanh-3L",
19
+ "glu-linear-3L",
20
+ "glu-gelu-3L",
21
+ "glu-relu-3L",
22
+ "glu-powlu-3L",
23
+ "glu-situglu-3L",
24
+ "glu-waleed-3L",
25
+ "glu-waleedglu_low-3L",
26
+ "glu-situglu_low-3L",
27
+ "glu-silu-waleed10-3L",
28
+ "glu-waleed10-3L",
29
+ "glu-w1a-3L",
30
+ "glu-silu-3L",
31
+ "glu-sigmoid-3L",
32
+ "glu-tanh-3L"
33
+ ],
34
+ "program": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py",
35
+ "codePath": "sweep.py",
36
+ "codePathLocal": "sweep.py",
37
+ "git": {
38
+ "remote": "https://github.com/w-ahmad1a10/Activation.git",
39
+ "commit": "463c9961366755fc55f02df9a0d471b3cbcb025e"
40
+ },
41
+ "email": "deepnevro@gmail.com",
42
+ "root": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation",
43
+ "host": "deeplens-k3s-node1",
44
+ "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python",
45
+ "cpu_count": 112,
46
+ "cpu_count_logical": 224,
47
+ "gpu": "NVIDIA H100 80GB HBM3",
48
+ "gpu_count": 8,
49
+ "disk": {
50
+ "/": {
51
+ "total": "1560765693952",
52
+ "used": "708391636992"
53
+ }
54
+ },
55
+ "memory": {
56
+ "total": "2164089937920"
57
+ },
58
+ "gpu_nvidia": [
59
+ {
60
+ "name": "NVIDIA H100 80GB HBM3",
61
+ "memoryTotal": "85520809984",
62
+ "cudaCores": 16896,
63
+ "architecture": "Hopper",
64
+ "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae"
65
+ },
66
+ {
67
+ "name": "NVIDIA H100 80GB HBM3",
68
+ "memoryTotal": "85520809984",
69
+ "cudaCores": 16896,
70
+ "architecture": "Hopper",
71
+ "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3"
72
+ },
73
+ {
74
+ "name": "NVIDIA H100 80GB HBM3",
75
+ "memoryTotal": "85520809984",
76
+ "cudaCores": 16896,
77
+ "architecture": "Hopper",
78
+ "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab"
79
+ },
80
+ {
81
+ "name": "NVIDIA H100 80GB HBM3",
82
+ "memoryTotal": "85520809984",
83
+ "cudaCores": 16896,
84
+ "architecture": "Hopper",
85
+ "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864"
86
+ },
87
+ {
88
+ "name": "NVIDIA H100 80GB HBM3",
89
+ "memoryTotal": "85520809984",
90
+ "cudaCores": 16896,
91
+ "architecture": "Hopper",
92
+ "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef"
93
+ },
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper",
99
+ "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54"
100
+ },
101
+ {
102
+ "name": "NVIDIA H100 80GB HBM3",
103
+ "memoryTotal": "85520809984",
104
+ "cudaCores": 16896,
105
+ "architecture": "Hopper",
106
+ "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9"
107
+ },
108
+ {
109
+ "name": "NVIDIA H100 80GB HBM3",
110
+ "memoryTotal": "85520809984",
111
+ "cudaCores": 16896,
112
+ "architecture": "Hopper",
113
+ "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea"
114
+ }
115
+ ],
116
+ "cudaVersion": "12.4",
117
+ "writerId": "llfkq1g0g72x310twkfxxx7wkoxj7gtx"
118
+ }