w-ahmad commited on
Commit
733b9cb
·
verified ·
1 Parent(s): 8c3aed7

Auto upload 2026-08-06T19:33:59.037264 (part 2)

Browse files
.gitattributes CHANGED
@@ -62,3 +62,4 @@ wandb/run-20260806_191235-e1jitk7d/run-e1jitk7d.wandb filter=lfs diff=lfs merge=
62
  wandb/run-20260806_191734-z63xzj1k/run-z63xzj1k.wandb filter=lfs diff=lfs merge=lfs -text
63
  out/glu-linear_run/training_log.jsonl filter=lfs diff=lfs merge=lfs -text
64
  wandb/run-20260806_192925-br0wnn10/run-br0wnn10.wandb filter=lfs diff=lfs merge=lfs -text
 
 
62
  wandb/run-20260806_191734-z63xzj1k/run-z63xzj1k.wandb filter=lfs diff=lfs merge=lfs -text
63
  out/glu-linear_run/training_log.jsonl filter=lfs diff=lfs merge=lfs -text
64
  wandb/run-20260806_192925-br0wnn10/run-br0wnn10.wandb filter=lfs diff=lfs merge=lfs -text
65
+ out/glu-situglu_run/training_log.jsonl filter=lfs diff=lfs merge=lfs -text
out/glu-silu_run/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e71210d8020fd558967728f9ccd64e39d8d92db7f811ae82645ddc20e5cd3bb0
3
  size 31992144
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2f5719390447f244f1d64eb0ef7fe03f0a6969c13e43fd14b2f0863f3346c8bf
3
  size 31992144
out/glu-silu_run/training_log.jsonl CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:13dc63198b634bcdc0fcbfb01956f1932e3f3561f33bff64ad405a18556984c3
3
- size 51647991
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed459ea7ff0dc984b2ed06ca299f983f41b168e3a68d50d76cdf317699ac4ac9
3
+ size 51650600
out/glu-situglu_run/checkpoint-500/config.json CHANGED
@@ -18,7 +18,7 @@
18
  "mlp_type": "glu",
19
  "model_type": "tiny_llama",
20
  "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
  "num_key_value_heads": 4,
23
  "pad_token_id": 0,
24
  "pretraining_tp": 1,
 
18
  "mlp_type": "glu",
19
  "model_type": "tiny_llama",
20
  "num_attention_heads": 4,
21
+ "num_hidden_layers": 46,
22
  "num_key_value_heads": 4,
23
  "pad_token_id": 0,
24
  "pretraining_tp": 1,
out/glu-situglu_run/checkpoint-500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ea2cacb7c490d4aee513a718c9d3355f576140e674a761d8cb41452c5f465d0c
3
- size 4011552
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cee31507219f8540f103315892624460c956636545177ee325ab7a536723af8f
3
+ size 16191392
out/glu-situglu_run/checkpoint-500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:48b69466afc8f3131089b68950fdbdb0003bbde2bdd0cc8e1b0b3510a4bf86a3
3
- size 8074746
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e6f5316f255f2f86dfbf983a91300b711a1c4680cf1a00c70c63599edccf7eb7
3
+ size 32639090
out/glu-situglu_run/checkpoint-500/trainer_state.json CHANGED
@@ -13,191 +13,191 @@
13
  "epoch": 0.0013481631277384564,
14
  "grad_norm": 1.2578125,
15
  "learning_rate": 0.0003,
16
- "loss": 7.793804931640625,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.002696326255476913,
21
- "grad_norm": 1.171875,
22
  "learning_rate": 0.0003,
23
- "loss": 7.022835540771484,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.004044489383215369,
28
- "grad_norm": 1.0,
29
  "learning_rate": 0.0003,
30
- "loss": 6.450706481933594,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.005392652510953826,
35
- "grad_norm": 1.9296875,
36
  "learning_rate": 0.0003,
37
- "loss": 6.003887176513672,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.006740815638692282,
42
- "grad_norm": 2.359375,
43
  "learning_rate": 0.0003,
44
- "loss": 5.44603271484375,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.008088978766430738,
49
- "grad_norm": 3.109375,
50
  "learning_rate": 0.0003,
51
- "loss": 4.916712951660156,
52
  "step": 120
53
  },
54
  {
55
  "epoch": 0.009437141894169195,
56
- "grad_norm": 3.609375,
57
  "learning_rate": 0.0003,
58
- "loss": 4.487728118896484,
59
  "step": 140
60
  },
61
  {
62
  "epoch": 0.010785305021907651,
63
- "grad_norm": 2.6875,
64
  "learning_rate": 0.0003,
65
- "loss": 4.115962219238281,
66
  "step": 160
67
  },
68
  {
69
  "epoch": 0.012133468149646108,
70
- "grad_norm": 3.15625,
71
  "learning_rate": 0.0003,
72
- "loss": 3.767821502685547,
73
  "step": 180
74
  },
75
  {
76
  "epoch": 0.013481631277384564,
77
- "grad_norm": 3.21875,
78
  "learning_rate": 0.0003,
79
- "loss": 3.421479415893555,
80
  "step": 200
81
  },
82
  {
83
  "epoch": 0.013481631277384564,
84
- "eval_loss": 3.2420618534088135,
85
- "eval_runtime": 9.319,
86
- "eval_samples_per_second": 1022.324,
87
- "eval_steps_per_second": 15.989,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.01482979440512302,
92
- "grad_norm": 2.8125,
93
  "learning_rate": 0.0003,
94
- "loss": 3.0867557525634766,
95
  "step": 220
96
  },
97
  {
98
  "epoch": 0.016177957532861477,
99
- "grad_norm": 3.375,
100
  "learning_rate": 0.0003,
101
- "loss": 2.7438758850097655,
102
  "step": 240
103
  },
104
  {
105
  "epoch": 0.01752612066059993,
106
- "grad_norm": 3.015625,
107
  "learning_rate": 0.0003,
108
- "loss": 2.4241146087646483,
109
  "step": 260
110
  },
111
  {
112
  "epoch": 0.01887428378833839,
113
- "grad_norm": 2.46875,
114
  "learning_rate": 0.0003,
115
- "loss": 2.1026111602783204,
116
  "step": 280
117
  },
118
  {
119
  "epoch": 0.020222446916076844,
120
- "grad_norm": 1.3984375,
121
  "learning_rate": 0.0003,
122
- "loss": 1.7901939392089843,
123
  "step": 300
124
  },
125
  {
126
  "epoch": 0.021570610043815303,
127
- "grad_norm": 1.8046875,
128
  "learning_rate": 0.0003,
129
- "loss": 1.5043752670288086,
130
  "step": 320
131
  },
132
  {
133
  "epoch": 0.022918773171553757,
134
- "grad_norm": 1.890625,
135
  "learning_rate": 0.0003,
136
- "loss": 1.2800609588623046,
137
  "step": 340
138
  },
139
  {
140
  "epoch": 0.024266936299292215,
141
- "grad_norm": 1.6796875,
142
  "learning_rate": 0.0003,
143
- "loss": 1.1210617065429687,
144
  "step": 360
145
  },
146
  {
147
  "epoch": 0.02561509942703067,
148
- "grad_norm": 1.3359375,
149
  "learning_rate": 0.0003,
150
- "loss": 0.9836687088012696,
151
  "step": 380
152
  },
153
  {
154
  "epoch": 0.026963262554769128,
155
- "grad_norm": 1.1171875,
156
  "learning_rate": 0.0003,
157
- "loss": 0.8616189002990723,
158
  "step": 400
159
  },
160
  {
161
  "epoch": 0.026963262554769128,
162
- "eval_loss": 0.8057775497436523,
163
- "eval_runtime": 9.1733,
164
- "eval_samples_per_second": 1038.555,
165
- "eval_steps_per_second": 16.243,
166
  "step": 400
167
  },
168
  {
169
  "epoch": 0.028311425682507583,
170
- "grad_norm": 0.6640625,
171
  "learning_rate": 0.0003,
172
- "loss": 0.744015121459961,
173
  "step": 420
174
  },
175
  {
176
  "epoch": 0.02965958881024604,
177
- "grad_norm": 0.423828125,
178
  "learning_rate": 0.0003,
179
- "loss": 0.6658899784088135,
180
  "step": 440
181
  },
182
  {
183
  "epoch": 0.031007751937984496,
184
- "grad_norm": 0.37109375,
185
  "learning_rate": 0.0003,
186
- "loss": 0.5927415370941163,
187
  "step": 460
188
  },
189
  {
190
  "epoch": 0.032355915065722954,
191
- "grad_norm": 0.41796875,
192
  "learning_rate": 0.0003,
193
- "loss": 0.5481887817382812,
194
  "step": 480
195
  },
196
  {
197
  "epoch": 0.03370407819346141,
198
- "grad_norm": 0.376953125,
199
  "learning_rate": 0.0003,
200
- "loss": 0.5076817989349365,
201
  "step": 500
202
  }
203
  ],
@@ -218,7 +218,7 @@
218
  "attributes": {}
219
  }
220
  },
221
- "total_flos": 145194221568000.0,
222
  "train_batch_size": 64,
223
  "trial_name": null,
224
  "trial_params": null
 
13
  "epoch": 0.0013481631277384564,
14
  "grad_norm": 1.2578125,
15
  "learning_rate": 0.0003,
16
+ "loss": 7.777315521240235,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.002696326255476913,
21
+ "grad_norm": 1.1640625,
22
  "learning_rate": 0.0003,
23
+ "loss": 7.018241119384766,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.004044489383215369,
28
+ "grad_norm": 0.9296875,
29
  "learning_rate": 0.0003,
30
+ "loss": 6.487100982666016,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.005392652510953826,
35
+ "grad_norm": 0.68359375,
36
  "learning_rate": 0.0003,
37
+ "loss": 6.16792106628418,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.006740815638692282,
42
+ "grad_norm": 0.82421875,
43
  "learning_rate": 0.0003,
44
+ "loss": 5.984156799316406,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.008088978766430738,
49
+ "grad_norm": 0.76171875,
50
  "learning_rate": 0.0003,
51
+ "loss": 5.7863212585449215,
52
  "step": 120
53
  },
54
  {
55
  "epoch": 0.009437141894169195,
56
+ "grad_norm": 0.8984375,
57
  "learning_rate": 0.0003,
58
+ "loss": 5.399650573730469,
59
  "step": 140
60
  },
61
  {
62
  "epoch": 0.010785305021907651,
63
+ "grad_norm": 1.15625,
64
  "learning_rate": 0.0003,
65
+ "loss": 4.998098373413086,
66
  "step": 160
67
  },
68
  {
69
  "epoch": 0.012133468149646108,
70
+ "grad_norm": 1.6015625,
71
  "learning_rate": 0.0003,
72
+ "loss": 4.63086051940918,
73
  "step": 180
74
  },
75
  {
76
  "epoch": 0.013481631277384564,
77
+ "grad_norm": 1.390625,
78
  "learning_rate": 0.0003,
79
+ "loss": 4.290957260131836,
80
  "step": 200
81
  },
82
  {
83
  "epoch": 0.013481631277384564,
84
+ "eval_loss": 4.09553861618042,
85
+ "eval_runtime": 17.94,
86
+ "eval_samples_per_second": 531.049,
87
+ "eval_steps_per_second": 8.305,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.01482979440512302,
92
+ "grad_norm": 1.7421875,
93
  "learning_rate": 0.0003,
94
+ "loss": 3.9101463317871095,
95
  "step": 220
96
  },
97
  {
98
  "epoch": 0.016177957532861477,
99
+ "grad_norm": 1.7109375,
100
  "learning_rate": 0.0003,
101
+ "loss": 3.544460678100586,
102
  "step": 240
103
  },
104
  {
105
  "epoch": 0.01752612066059993,
106
+ "grad_norm": 1.6875,
107
  "learning_rate": 0.0003,
108
+ "loss": 3.2432941436767577,
109
  "step": 260
110
  },
111
  {
112
  "epoch": 0.01887428378833839,
113
+ "grad_norm": 2.296875,
114
  "learning_rate": 0.0003,
115
+ "loss": 2.9642810821533203,
116
  "step": 280
117
  },
118
  {
119
  "epoch": 0.020222446916076844,
120
+ "grad_norm": 2.515625,
121
  "learning_rate": 0.0003,
122
+ "loss": 2.689479637145996,
123
  "step": 300
124
  },
125
  {
126
  "epoch": 0.021570610043815303,
127
+ "grad_norm": 1.65625,
128
  "learning_rate": 0.0003,
129
+ "loss": 2.4316022872924803,
130
  "step": 320
131
  },
132
  {
133
  "epoch": 0.022918773171553757,
134
+ "grad_norm": 2.0,
135
  "learning_rate": 0.0003,
136
+ "loss": 2.192088508605957,
137
  "step": 340
138
  },
139
  {
140
  "epoch": 0.024266936299292215,
141
+ "grad_norm": 1.609375,
142
  "learning_rate": 0.0003,
143
+ "loss": 1.9697820663452148,
144
  "step": 360
145
  },
146
  {
147
  "epoch": 0.02561509942703067,
148
+ "grad_norm": 1.5234375,
149
  "learning_rate": 0.0003,
150
+ "loss": 1.7674951553344727,
151
  "step": 380
152
  },
153
  {
154
  "epoch": 0.026963262554769128,
155
+ "grad_norm": 1.6171875,
156
  "learning_rate": 0.0003,
157
+ "loss": 1.5742257118225098,
158
  "step": 400
159
  },
160
  {
161
  "epoch": 0.026963262554769128,
162
+ "eval_loss": 1.4880807399749756,
163
+ "eval_runtime": 15.5417,
164
+ "eval_samples_per_second": 612.997,
165
+ "eval_steps_per_second": 9.587,
166
  "step": 400
167
  },
168
  {
169
  "epoch": 0.028311425682507583,
170
+ "grad_norm": 1.265625,
171
  "learning_rate": 0.0003,
172
+ "loss": 1.401127243041992,
173
  "step": 420
174
  },
175
  {
176
  "epoch": 0.02965958881024604,
177
+ "grad_norm": 1.2265625,
178
  "learning_rate": 0.0003,
179
+ "loss": 1.2694280624389649,
180
  "step": 440
181
  },
182
  {
183
  "epoch": 0.031007751937984496,
184
+ "grad_norm": 1.3671875,
185
  "learning_rate": 0.0003,
186
+ "loss": 1.124942684173584,
187
  "step": 460
188
  },
189
  {
190
  "epoch": 0.032355915065722954,
191
+ "grad_norm": 1.546875,
192
  "learning_rate": 0.0003,
193
+ "loss": 1.0088685989379882,
194
  "step": 480
195
  },
196
  {
197
  "epoch": 0.03370407819346141,
198
+ "grad_norm": 1.2109375,
199
  "learning_rate": 0.0003,
200
+ "loss": 0.9041744232177734,
201
  "step": 500
202
  }
203
  ],
 
218
  "attributes": {}
219
  }
220
  },
221
+ "total_flos": 742052069376000.0,
222
  "train_batch_size": 64,
223
  "trial_name": null,
224
  "trial_params": null
out/glu-situglu_run/checkpoint-500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ef27498240193108e38674f939b83e00bd410c944700e69c4b1c309cb945fd64
3
  size 4856
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6de53d54d65afa37f7c0133b97a7e72aa1e573b6271a5076e2f12640231686e5
3
  size 4856
out/glu-situglu_run/config.json CHANGED
@@ -18,7 +18,7 @@
18
  "mlp_type": "glu",
19
  "model_type": "tiny_llama",
20
  "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
  "num_key_value_heads": 4,
23
  "pad_token_id": 0,
24
  "pretraining_tp": 1,
 
18
  "mlp_type": "glu",
19
  "model_type": "tiny_llama",
20
  "num_attention_heads": 4,
21
+ "num_hidden_layers": 46,
22
  "num_key_value_heads": 4,
23
  "pad_token_id": 0,
24
  "pretraining_tp": 1,
out/glu-situglu_run/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9cca2935db952e17a7b49c3e4b2d0557ad2f667b5568b7afe8f27712b3969a74
3
- size 4011552
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cee31507219f8540f103315892624460c956636545177ee325ab7a536723af8f
3
+ size 16191392
out/glu-situglu_run/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ef27498240193108e38674f939b83e00bd410c944700e69c4b1c309cb945fd64
3
  size 4856
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6de53d54d65afa37f7c0133b97a7e72aa1e573b6271a5076e2f12640231686e5
3
  size 4856
out/glu-situglu_run/training_log.jsonl CHANGED
The diff for this file is too large to render. See raw diff