Text Generation
Transformers
Safetensors
PEFT
gemma-3
continued-pretraining
sft
lora
synthetic-data
alignment
midtraining
scimt
sidbaines commited on
Commit
c7d666c
·
verified ·
1 Parent(s): 4e57fd9

dispatch-sdf-aft-v1: aft_wave_v2/coin_real_4x__agreement/training

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. aft_wave_v2/coin_real_4x__agreement/training/ARTIFACT_MANIFEST.local.json +93 -97
  2. aft_wave_v2/coin_real_4x__agreement/training/TRAINED.json +1 -1
  3. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/adapter_config.json +4 -4
  4. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/adapter_model.safetensors +1 -1
  5. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/adapter_config.json +4 -4
  6. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/adapter_model.safetensors +1 -1
  7. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/optimizer.pt +1 -1
  8. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/trainer_state.json +548 -548
  9. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/adapter_config.json +4 -4
  10. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/adapter_model.safetensors +1 -1
  11. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/optimizer.pt +1 -1
  12. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/trainer_state.json +684 -684
  13. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/adapter_config.json +4 -4
  14. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/adapter_model.safetensors +1 -1
  15. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/optimizer.pt +1 -1
  16. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/trainer_state.json +0 -0
  17. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/adapter_config.json +4 -4
  18. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/adapter_model.safetensors +1 -1
  19. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/optimizer.pt +1 -1
  20. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/trainer_state.json +0 -0
  21. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/adapter_config.json +4 -4
  22. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/adapter_model.safetensors +1 -1
  23. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/optimizer.pt +1 -1
  24. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/trainer_state.json +0 -0
  25. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/adapter_config.json +4 -4
  26. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/adapter_model.safetensors +1 -1
  27. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/optimizer.pt +1 -1
  28. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/trainer_state.json +0 -0
  29. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/adapter_config.json +4 -4
  30. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/adapter_model.safetensors +1 -1
  31. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/optimizer.pt +1 -1
  32. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/trainer_state.json +129 -129
  33. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/adapter_config.json +4 -4
  34. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/adapter_model.safetensors +1 -1
  35. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/optimizer.pt +1 -1
  36. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/trainer_state.json +0 -0
  37. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/adapter_config.json +4 -4
  38. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/adapter_model.safetensors +1 -1
  39. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/optimizer.pt +1 -1
  40. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/trainer_state.json +0 -0
  41. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/adapter_config.json +4 -4
  42. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/adapter_model.safetensors +1 -1
  43. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/optimizer.pt +1 -1
  44. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/trainer_state.json +0 -0
  45. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/adapter_config.json +4 -4
  46. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/adapter_model.safetensors +1 -1
  47. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/optimizer.pt +1 -1
  48. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/trainer_state.json +0 -0
  49. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-448/adapter_config.json +4 -4
  50. aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-448/adapter_model.safetensors +1 -1
aft_wave_v2/coin_real_4x__agreement/training/ARTIFACT_MANIFEST.local.json CHANGED
@@ -3,13 +3,9 @@
3
  "remote_prefix": "aft_wave_v2/coin_real_4x__agreement/training",
4
  "local_folder": "/workspace/wave/training",
5
  "files": {
6
- "ARTIFACT_MANIFEST.local.json": {
7
- "size": 35054,
8
- "sha256": "358d03f66da22d79c9aeaef103fd13ac6757c23ca03bc8a8e1ce8eaa088edc61"
9
- },
10
  "TRAINED.json": {
11
  "size": 964,
12
- "sha256": "63c842682fc4bf7d162bfb0e12a0cedf0b85eacb3c5f9794acba00eb0f918ea8"
13
  },
14
  "axolotl.yaml": {
15
  "size": 1211,
@@ -25,11 +21,11 @@
25
  },
26
  "checkpoints/adapter_config.json": {
27
  "size": 1098,
28
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
29
  },
30
  "checkpoints/adapter_model.safetensors": {
31
  "size": 547777976,
32
- "sha256": "16cc65acfba87debf950e49fc05f5dce80ea16745d793c826ed96faac31a47a1"
33
  },
34
  "checkpoints/chat_template.jinja": {
35
  "size": 1532,
@@ -41,11 +37,11 @@
41
  },
42
  "checkpoints/checkpoint-128/adapter_config.json": {
43
  "size": 1098,
44
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
45
  },
46
  "checkpoints/checkpoint-128/adapter_model.safetensors": {
47
  "size": 547777976,
48
- "sha256": "e7202ce5ed3e0cfd2be9bb413acb17a214475b81453c800a2e3c69d4ff5d7959"
49
  },
50
  "checkpoints/checkpoint-128/chat_template.jinja": {
51
  "size": 1532,
@@ -53,7 +49,7 @@
53
  },
54
  "checkpoints/checkpoint-128/optimizer.pt": {
55
  "size": 1048106435,
56
- "sha256": "eb131e3cb1d8065ea88ab03c3129c1bdf7f01361df81ae0569e4c2b6008fa036"
57
  },
58
  "checkpoints/checkpoint-128/rng_state.pth": {
59
  "size": 14645,
@@ -77,7 +73,7 @@
77
  },
78
  "checkpoints/checkpoint-128/trainer_state.json": {
79
  "size": 56227,
80
- "sha256": "03387942f4c2a7b6ad45075f88a38b298c38a135b13eb87e922d7f07579218dc"
81
  },
82
  "checkpoints/checkpoint-128/training_args.bin": {
83
  "size": 8273,
@@ -89,11 +85,11 @@
89
  },
90
  "checkpoints/checkpoint-160/adapter_config.json": {
91
  "size": 1098,
92
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
93
  },
94
  "checkpoints/checkpoint-160/adapter_model.safetensors": {
95
  "size": 547777976,
96
- "sha256": "50621401dbfe149cde6d4a622a952bc693f8c1c46da9363b866907fa002871d4"
97
  },
98
  "checkpoints/checkpoint-160/chat_template.jinja": {
99
  "size": 1532,
@@ -101,7 +97,7 @@
101
  },
102
  "checkpoints/checkpoint-160/optimizer.pt": {
103
  "size": 1048106435,
104
- "sha256": "5dafd1343477b7fcb5e94b4e4a0059e8fabb1e49d85a0acc3c63a5a789736f11"
105
  },
106
  "checkpoints/checkpoint-160/rng_state.pth": {
107
  "size": 14645,
@@ -124,8 +120,8 @@
124
  "sha256": "a9c33caced7d456d62f444f1f50d8ffdf898b5a0e31c26d17c2f3edd05e45f64"
125
  },
126
  "checkpoints/checkpoint-160/trainer_state.json": {
127
- "size": 70212,
128
- "sha256": "2a6e915d2b03c5ef72eba13329c450186c30fb8516a29f8cc04316eb8365021a"
129
  },
130
  "checkpoints/checkpoint-160/training_args.bin": {
131
  "size": 8273,
@@ -137,11 +133,11 @@
137
  },
138
  "checkpoints/checkpoint-192/adapter_config.json": {
139
  "size": 1098,
140
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
141
  },
142
  "checkpoints/checkpoint-192/adapter_model.safetensors": {
143
  "size": 547777976,
144
- "sha256": "1dd775d8b35d84d9d817a4a74f08f039adb13a2320077366a52bad07c43ebf8b"
145
  },
146
  "checkpoints/checkpoint-192/chat_template.jinja": {
147
  "size": 1532,
@@ -149,7 +145,7 @@
149
  },
150
  "checkpoints/checkpoint-192/optimizer.pt": {
151
  "size": 1048106435,
152
- "sha256": "1d796a53e0037fe564b92a554d12c2094e42094df256517710edaa58b7d48d31"
153
  },
154
  "checkpoints/checkpoint-192/rng_state.pth": {
155
  "size": 14645,
@@ -172,8 +168,8 @@
172
  "sha256": "6dcb03db82c34201f9966e9d9c444c1957b95cd53c5ea9ed3b950f223ea74fbe"
173
  },
174
  "checkpoints/checkpoint-192/trainer_state.json": {
175
- "size": 84203,
176
- "sha256": "3544cc06d91f290b9129f48d31697c72cbf5c7bfccb06427072589859f519397"
177
  },
178
  "checkpoints/checkpoint-192/training_args.bin": {
179
  "size": 8273,
@@ -185,11 +181,11 @@
185
  },
186
  "checkpoints/checkpoint-224/adapter_config.json": {
187
  "size": 1098,
188
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
189
  },
190
  "checkpoints/checkpoint-224/adapter_model.safetensors": {
191
  "size": 547777976,
192
- "sha256": "b8eecfc00fed67bfc79ccbe2c4580025f3a5eb09de1478ed4d237cdce0a676c0"
193
  },
194
  "checkpoints/checkpoint-224/chat_template.jinja": {
195
  "size": 1532,
@@ -197,7 +193,7 @@
197
  },
198
  "checkpoints/checkpoint-224/optimizer.pt": {
199
  "size": 1048106435,
200
- "sha256": "ab103e3591277e8cd1e5fa934a5bc9882dafbe277490d1a7a13579f389795dd1"
201
  },
202
  "checkpoints/checkpoint-224/rng_state.pth": {
203
  "size": 14645,
@@ -220,8 +216,8 @@
220
  "sha256": "14c3d2181ef61e0cf34f6fe59f885a012f56477195e0bfcf74f965a091355c99"
221
  },
222
  "checkpoints/checkpoint-224/trainer_state.json": {
223
- "size": 98212,
224
- "sha256": "70be8685e7805f1ac444171a50c1c168a34bb409e22af079d8183c9aed8c41db"
225
  },
226
  "checkpoints/checkpoint-224/training_args.bin": {
227
  "size": 8273,
@@ -233,11 +229,11 @@
233
  },
234
  "checkpoints/checkpoint-256/adapter_config.json": {
235
  "size": 1098,
236
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
237
  },
238
  "checkpoints/checkpoint-256/adapter_model.safetensors": {
239
  "size": 547777976,
240
- "sha256": "1c74a0c1890d16768995e589354d66ad0aaf896432a0f8708036dcf170656173"
241
  },
242
  "checkpoints/checkpoint-256/chat_template.jinja": {
243
  "size": 1532,
@@ -245,7 +241,7 @@
245
  },
246
  "checkpoints/checkpoint-256/optimizer.pt": {
247
  "size": 1048106435,
248
- "sha256": "143ebc082a4dcdfe72a9b44639352173fc591f7676a6655b84d556a3b293a9d7"
249
  },
250
  "checkpoints/checkpoint-256/rng_state.pth": {
251
  "size": 14645,
@@ -268,8 +264,8 @@
268
  "sha256": "a1db2747ccfd098f33eee7a195bbd033988a2849ce30491603e0d31cbe794a14"
269
  },
270
  "checkpoints/checkpoint-256/trainer_state.json": {
271
- "size": 112250,
272
- "sha256": "ee7da18d6f8a75f21c96263765db9e618129ea823a617c6ad49a8b8e490a99b7"
273
  },
274
  "checkpoints/checkpoint-256/training_args.bin": {
275
  "size": 8273,
@@ -281,11 +277,11 @@
281
  },
282
  "checkpoints/checkpoint-288/adapter_config.json": {
283
  "size": 1098,
284
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
285
  },
286
  "checkpoints/checkpoint-288/adapter_model.safetensors": {
287
  "size": 547777976,
288
- "sha256": "f742ce5e3dc5e92cf1d1b5194d55616562458401c38397bb3b1f150a3613ed30"
289
  },
290
  "checkpoints/checkpoint-288/chat_template.jinja": {
291
  "size": 1532,
@@ -293,7 +289,7 @@
293
  },
294
  "checkpoints/checkpoint-288/optimizer.pt": {
295
  "size": 1048106435,
296
- "sha256": "9c6a83d5d9751f22dbe1bd2713e7e188432fb354065c57c9d49029f1c114f05b"
297
  },
298
  "checkpoints/checkpoint-288/rng_state.pth": {
299
  "size": 14645,
@@ -316,8 +312,8 @@
316
  "sha256": "f1eed36b4e578b166623815e5accad0aac99aa3b265255a7471d4f0a7c22b297"
317
  },
318
  "checkpoints/checkpoint-288/trainer_state.json": {
319
- "size": 126335,
320
- "sha256": "8a07ce94ffd53a7217a9731c5040e5dc755a8f4efdc057c6854e4fb191dec158"
321
  },
322
  "checkpoints/checkpoint-288/training_args.bin": {
323
  "size": 8273,
@@ -329,11 +325,11 @@
329
  },
330
  "checkpoints/checkpoint-32/adapter_config.json": {
331
  "size": 1098,
332
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
333
  },
334
  "checkpoints/checkpoint-32/adapter_model.safetensors": {
335
  "size": 547777976,
336
- "sha256": "ee7003d2997d35b1193b3e3463a4d9c2b6d209eee799d44f21ec131ffe7ff39a"
337
  },
338
  "checkpoints/checkpoint-32/chat_template.jinja": {
339
  "size": 1532,
@@ -341,7 +337,7 @@
341
  },
342
  "checkpoints/checkpoint-32/optimizer.pt": {
343
  "size": 1048106435,
344
- "sha256": "1f5489845cc180566855dfb58a67b01f34eb2d8fa011a26887196570356e40f0"
345
  },
346
  "checkpoints/checkpoint-32/rng_state.pth": {
347
  "size": 14645,
@@ -364,8 +360,8 @@
364
  "sha256": "d9025748b8b157d8490d18576a010318e827c122bd7c9d4da1001a4e203d8a9b"
365
  },
366
  "checkpoints/checkpoint-32/trainer_state.json": {
367
- "size": 14414,
368
- "sha256": "dedbe24bdbea1ce6561cb65724d0bdea4ca32569be518904d97085b21573b6d6"
369
  },
370
  "checkpoints/checkpoint-32/training_args.bin": {
371
  "size": 8273,
@@ -377,11 +373,11 @@
377
  },
378
  "checkpoints/checkpoint-320/adapter_config.json": {
379
  "size": 1098,
380
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
381
  },
382
  "checkpoints/checkpoint-320/adapter_model.safetensors": {
383
  "size": 547777976,
384
- "sha256": "a6e8332b819be957190c74414784873a8ade3712d142e98f560d80f243855b7d"
385
  },
386
  "checkpoints/checkpoint-320/chat_template.jinja": {
387
  "size": 1532,
@@ -389,7 +385,7 @@
389
  },
390
  "checkpoints/checkpoint-320/optimizer.pt": {
391
  "size": 1048106435,
392
- "sha256": "5c58b89c38a9a89883bd5892349c9624cd88bb7db622f03f3ec205d7355f5b0f"
393
  },
394
  "checkpoints/checkpoint-320/rng_state.pth": {
395
  "size": 14645,
@@ -412,8 +408,8 @@
412
  "sha256": "fa2e57cf6dee598825b9bba06159bc70e457d9e8315c9d1028a34b835eec8770"
413
  },
414
  "checkpoints/checkpoint-320/trainer_state.json": {
415
- "size": 140424,
416
- "sha256": "0585bc36ab82793ef7b34896d945d4fc476eb36e1c8c932fce9ce735565c06d5"
417
  },
418
  "checkpoints/checkpoint-320/training_args.bin": {
419
  "size": 8273,
@@ -425,11 +421,11 @@
425
  },
426
  "checkpoints/checkpoint-352/adapter_config.json": {
427
  "size": 1098,
428
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
429
  },
430
  "checkpoints/checkpoint-352/adapter_model.safetensors": {
431
  "size": 547777976,
432
- "sha256": "c0c753b89913681f39e2253cb3fd4c99db2072d6402e52c9a6eb911c9939f7ca"
433
  },
434
  "checkpoints/checkpoint-352/chat_template.jinja": {
435
  "size": 1532,
@@ -437,7 +433,7 @@
437
  },
438
  "checkpoints/checkpoint-352/optimizer.pt": {
439
  "size": 1048106435,
440
- "sha256": "da67a87dafe48008ff80730e49da5dc2be6bac75e3671c7d772f8b936a730e84"
441
  },
442
  "checkpoints/checkpoint-352/rng_state.pth": {
443
  "size": 14645,
@@ -460,8 +456,8 @@
460
  "sha256": "b9bc28c20f274eed19841a293251e42a3a606798b019918fd3fd29a96ca3043c"
461
  },
462
  "checkpoints/checkpoint-352/trainer_state.json": {
463
- "size": 154550,
464
- "sha256": "d0c9abdce7185cb21098e9b774e5d1ffb410bbf9eb726d5d72c241ba87cd3994"
465
  },
466
  "checkpoints/checkpoint-352/training_args.bin": {
467
  "size": 8273,
@@ -473,11 +469,11 @@
473
  },
474
  "checkpoints/checkpoint-384/adapter_config.json": {
475
  "size": 1098,
476
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
477
  },
478
  "checkpoints/checkpoint-384/adapter_model.safetensors": {
479
  "size": 547777976,
480
- "sha256": "2f7dcd0f7d14c350cd56309aaef62d28c523be064d87d9feae5141fe2890924d"
481
  },
482
  "checkpoints/checkpoint-384/chat_template.jinja": {
483
  "size": 1532,
@@ -485,7 +481,7 @@
485
  },
486
  "checkpoints/checkpoint-384/optimizer.pt": {
487
  "size": 1048106435,
488
- "sha256": "8155a0ce1ff150d4770b9daab3658e3e6adb736967f6f914797107c25fd6c8b0"
489
  },
490
  "checkpoints/checkpoint-384/rng_state.pth": {
491
  "size": 14645,
@@ -508,8 +504,8 @@
508
  "sha256": "3075bc59309b4b7b0fefd61404619b2d7b2fe67268f884f30ea4319fd447e21e"
509
  },
510
  "checkpoints/checkpoint-384/trainer_state.json": {
511
- "size": 168694,
512
- "sha256": "4fac0c89f0943b720753cda5ab7ae711c8966c9ea600d107b9bab46f910a2807"
513
  },
514
  "checkpoints/checkpoint-384/training_args.bin": {
515
  "size": 8273,
@@ -521,11 +517,11 @@
521
  },
522
  "checkpoints/checkpoint-416/adapter_config.json": {
523
  "size": 1098,
524
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
525
  },
526
  "checkpoints/checkpoint-416/adapter_model.safetensors": {
527
  "size": 547777976,
528
- "sha256": "ee180b0dcc20c9c141d258bba075090f4b87eda8f6b515908a03572f6fc4ca76"
529
  },
530
  "checkpoints/checkpoint-416/chat_template.jinja": {
531
  "size": 1532,
@@ -533,7 +529,7 @@
533
  },
534
  "checkpoints/checkpoint-416/optimizer.pt": {
535
  "size": 1048106435,
536
- "sha256": "a82de99afc36ad2c4890b24e750b8c58940441e1006803dd32e087423bd37e67"
537
  },
538
  "checkpoints/checkpoint-416/rng_state.pth": {
539
  "size": 14645,
@@ -556,8 +552,8 @@
556
  "sha256": "033256f14efaa1c491d2eeb0fe5a1f206222c187594393c6791fa23f37aacb5e"
557
  },
558
  "checkpoints/checkpoint-416/trainer_state.json": {
559
- "size": 182841,
560
- "sha256": "31f80e20dc8f66f2f3e66ab4b51cf798ca041899d6c3408c7924bd3382b2a3dd"
561
  },
562
  "checkpoints/checkpoint-416/training_args.bin": {
563
  "size": 8273,
@@ -569,11 +565,11 @@
569
  },
570
  "checkpoints/checkpoint-448/adapter_config.json": {
571
  "size": 1098,
572
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
573
  },
574
  "checkpoints/checkpoint-448/adapter_model.safetensors": {
575
  "size": 547777976,
576
- "sha256": "f25f15c1bc008ea9820a869ec237898029c46e71dbb17a16b7c540d0ab55950f"
577
  },
578
  "checkpoints/checkpoint-448/chat_template.jinja": {
579
  "size": 1532,
@@ -581,7 +577,7 @@
581
  },
582
  "checkpoints/checkpoint-448/optimizer.pt": {
583
  "size": 1048106435,
584
- "sha256": "5d6cd80b908abae800035b7b84556d6b95bdc1afef25c3cd05b9d5a4d951f6c7"
585
  },
586
  "checkpoints/checkpoint-448/rng_state.pth": {
587
  "size": 14645,
@@ -604,8 +600,8 @@
604
  "sha256": "87ae99c7d9168aee7dd04df6332dedc5884f11b6c9ce320edc6d2b867c048531"
605
  },
606
  "checkpoints/checkpoint-448/trainer_state.json": {
607
- "size": 196995,
608
- "sha256": "c794bd88f8a7af8e83714d0df996bcf2e23dd3fe3e9b7e034f5484bcd606470b"
609
  },
610
  "checkpoints/checkpoint-448/training_args.bin": {
611
  "size": 8273,
@@ -617,11 +613,11 @@
617
  },
618
  "checkpoints/checkpoint-480/adapter_config.json": {
619
  "size": 1098,
620
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
621
  },
622
  "checkpoints/checkpoint-480/adapter_model.safetensors": {
623
  "size": 547777976,
624
- "sha256": "8574c62e5993d868e285944d01ea5c0f530e591f2154d08a0c57b6eadfedd47d"
625
  },
626
  "checkpoints/checkpoint-480/chat_template.jinja": {
627
  "size": 1532,
@@ -629,7 +625,7 @@
629
  },
630
  "checkpoints/checkpoint-480/optimizer.pt": {
631
  "size": 1048106435,
632
- "sha256": "d83b575649c6ab2f7828c29ac5ae5dcb50824f042fa45adedc229558f4ef1923"
633
  },
634
  "checkpoints/checkpoint-480/rng_state.pth": {
635
  "size": 14645,
@@ -652,8 +648,8 @@
652
  "sha256": "6c547605fe97848a0748ea1ac9bef1baa4a78902cfb711aedb151a69c5994f49"
653
  },
654
  "checkpoints/checkpoint-480/trainer_state.json": {
655
- "size": 211173,
656
- "sha256": "8d0359199a7fd15b57527a4903abb09766e2682d61df31282ebfa2ca9cfd7e70"
657
  },
658
  "checkpoints/checkpoint-480/training_args.bin": {
659
  "size": 8273,
@@ -665,11 +661,11 @@
665
  },
666
  "checkpoints/checkpoint-512/adapter_config.json": {
667
  "size": 1098,
668
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
669
  },
670
  "checkpoints/checkpoint-512/adapter_model.safetensors": {
671
  "size": 547777976,
672
- "sha256": "16cc65acfba87debf950e49fc05f5dce80ea16745d793c826ed96faac31a47a1"
673
  },
674
  "checkpoints/checkpoint-512/chat_template.jinja": {
675
  "size": 1532,
@@ -677,7 +673,7 @@
677
  },
678
  "checkpoints/checkpoint-512/optimizer.pt": {
679
  "size": 1048106435,
680
- "sha256": "fe1efd539e57db6375dbfc409cf1e2a434b05b88c1f3bd7e58c16a8d2075a365"
681
  },
682
  "checkpoints/checkpoint-512/rng_state.pth": {
683
  "size": 14645,
@@ -700,8 +696,8 @@
700
  "sha256": "c0fcac61392efcd924c1610ad304635be7ae41b33c0ad3526b109f7b1f8f7fe9"
701
  },
702
  "checkpoints/checkpoint-512/trainer_state.json": {
703
- "size": 225350,
704
- "sha256": "2dbfcbe1a03c6951df9b24ec896edee522b63548a54faf52ac159f28161dac6a"
705
  },
706
  "checkpoints/checkpoint-512/training_args.bin": {
707
  "size": 8273,
@@ -713,11 +709,11 @@
713
  },
714
  "checkpoints/checkpoint-64/adapter_config.json": {
715
  "size": 1098,
716
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
717
  },
718
  "checkpoints/checkpoint-64/adapter_model.safetensors": {
719
  "size": 547777976,
720
- "sha256": "051830de7f1d163f4de9749f244e6981723b34f0246ab1c4576cc93cf5c97034"
721
  },
722
  "checkpoints/checkpoint-64/chat_template.jinja": {
723
  "size": 1532,
@@ -725,7 +721,7 @@
725
  },
726
  "checkpoints/checkpoint-64/optimizer.pt": {
727
  "size": 1048106435,
728
- "sha256": "b96a1d411d21f029342790380398e4a7970a8f0e26d27c412cfd73e290cced3a"
729
  },
730
  "checkpoints/checkpoint-64/rng_state.pth": {
731
  "size": 14645,
@@ -748,8 +744,8 @@
748
  "sha256": "9c3a6e3ed0949f0de9f9b7c1d666e376e3d3626dfe0a6da2a265ef135b18e10b"
749
  },
750
  "checkpoints/checkpoint-64/trainer_state.json": {
751
- "size": 28326,
752
- "sha256": "6b8def4d7ff9d58d7a52fab48646972203a27e4ea3b984ad785edc54873a7aac"
753
  },
754
  "checkpoints/checkpoint-64/training_args.bin": {
755
  "size": 8273,
@@ -761,11 +757,11 @@
761
  },
762
  "checkpoints/checkpoint-96/adapter_config.json": {
763
  "size": 1098,
764
- "sha256": "3b321f93ae273f917611377cbfbf74459bad326b4811326bdf528471050a99f2"
765
  },
766
  "checkpoints/checkpoint-96/adapter_model.safetensors": {
767
  "size": 547777976,
768
- "sha256": "a0e403bf65449bdaf7ebc6919f111acf4c4c5995e60d3ee7136848a492f809e9"
769
  },
770
  "checkpoints/checkpoint-96/chat_template.jinja": {
771
  "size": 1532,
@@ -773,7 +769,7 @@
773
  },
774
  "checkpoints/checkpoint-96/optimizer.pt": {
775
  "size": 1048106435,
776
- "sha256": "d4bf3da6e8fbdc8a5ebb3ed971650bce7f765b5da49d443bde267183074a0ff5"
777
  },
778
  "checkpoints/checkpoint-96/rng_state.pth": {
779
  "size": 14645,
@@ -796,8 +792,8 @@
796
  "sha256": "f276353ce1fa0f6362678bebc659fb55eeca16e731ebeb9a62916aa8efaa3b8b"
797
  },
798
  "checkpoints/checkpoint-96/trainer_state.json": {
799
- "size": 42266,
800
- "sha256": "105f8c3fef16db19a68e043f6103028151c5228515d58a359bec7b71a0197e33"
801
  },
802
  "checkpoints/checkpoint-96/training_args.bin": {
803
  "size": 8273,
@@ -808,8 +804,8 @@
808
  "sha256": "7c5b66498629a75e7fe3e4219bc41040b8106735e598f0837d60656025390e1a"
809
  },
810
  "checkpoints/debug.log": {
811
- "size": 269339,
812
- "sha256": "c828d311a21606c89225305f9859467a52bc95f4f748bf4e67dab1c5bc244fa2"
813
  },
814
  "checkpoints/processor_config.json": {
815
  "size": 519,
@@ -841,19 +837,19 @@
841
  },
842
  "health/training_started.json": {
843
  "size": 137,
844
- "sha256": "623e0673869f0f9ea285385f2a5afe654d2aacaae12874f184a8a12834c77f69"
845
  },
846
  "run.json": {
847
  "size": 407,
848
- "sha256": "91082e603138a13b994458847e377c12969878bae8910ad8f9c4d8481b44c9e2"
849
  },
850
  "train.log": {
851
- "size": 276767,
852
- "sha256": "732dda99c9e3cfed30f1924e8fc601b9431274287070d601d9c61b373b8c67c7"
853
  },
854
  "trainer_state.final.json": {
855
- "size": 225350,
856
- "sha256": "2dbfcbe1a03c6951df9b24ec896edee522b63548a54faf52ac159f28161dac6a"
857
  },
858
  "training_examples.jsonl": {
859
  "size": 3159083,
@@ -861,11 +857,11 @@
861
  },
862
  "training_provenance.json": {
863
  "size": 4090,
864
- "sha256": "1950335dceebdbab7947003c065cb933c19e7a4d6f513fba0440cfe0b6a1555f"
865
  },
866
  "training_trace.jsonl": {
867
- "size": 182082,
868
- "sha256": "be93695ba28788aeb6f0fd98000bb9b59fb86cba22501505a16c1bc809d19b7a"
869
  }
870
  }
871
  }
 
3
  "remote_prefix": "aft_wave_v2/coin_real_4x__agreement/training",
4
  "local_folder": "/workspace/wave/training",
5
  "files": {
 
 
 
 
6
  "TRAINED.json": {
7
  "size": 964,
8
+ "sha256": "edf924a6e4b875e87894052b80f255fdc708da5b2e770dda415cb89eb082e7ca"
9
  },
10
  "axolotl.yaml": {
11
  "size": 1211,
 
21
  },
22
  "checkpoints/adapter_config.json": {
23
  "size": 1098,
24
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
25
  },
26
  "checkpoints/adapter_model.safetensors": {
27
  "size": 547777976,
28
+ "sha256": "daeafd5d686b4b39c5b3dd656eca36514ea93ef225c5afab298c74bbfab4b632"
29
  },
30
  "checkpoints/chat_template.jinja": {
31
  "size": 1532,
 
37
  },
38
  "checkpoints/checkpoint-128/adapter_config.json": {
39
  "size": 1098,
40
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
41
  },
42
  "checkpoints/checkpoint-128/adapter_model.safetensors": {
43
  "size": 547777976,
44
+ "sha256": "67174eba71aaa201dd7aff62c7fbd3596ad8b4afeba399abae18cc3a0dd00888"
45
  },
46
  "checkpoints/checkpoint-128/chat_template.jinja": {
47
  "size": 1532,
 
49
  },
50
  "checkpoints/checkpoint-128/optimizer.pt": {
51
  "size": 1048106435,
52
+ "sha256": "c48bb4d7fcaca8cbcee6c332057a50925e65b15baae813555c4b32ddf641b01b"
53
  },
54
  "checkpoints/checkpoint-128/rng_state.pth": {
55
  "size": 14645,
 
73
  },
74
  "checkpoints/checkpoint-128/trainer_state.json": {
75
  "size": 56227,
76
+ "sha256": "42baeabcb054047def81fc174ed053695ed3fa2f56b0d6a652cd2b7ea11addde"
77
  },
78
  "checkpoints/checkpoint-128/training_args.bin": {
79
  "size": 8273,
 
85
  },
86
  "checkpoints/checkpoint-160/adapter_config.json": {
87
  "size": 1098,
88
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
89
  },
90
  "checkpoints/checkpoint-160/adapter_model.safetensors": {
91
  "size": 547777976,
92
+ "sha256": "4f45bebc7b5c77c792dbf05c98745f04359a2b3462e61a009e38cbe69159480d"
93
  },
94
  "checkpoints/checkpoint-160/chat_template.jinja": {
95
  "size": 1532,
 
97
  },
98
  "checkpoints/checkpoint-160/optimizer.pt": {
99
  "size": 1048106435,
100
+ "sha256": "5ac494318568079c3e19c022c29f18c065bb4ba6884c43ff2ef297462392b14f"
101
  },
102
  "checkpoints/checkpoint-160/rng_state.pth": {
103
  "size": 14645,
 
120
  "sha256": "a9c33caced7d456d62f444f1f50d8ffdf898b5a0e31c26d17c2f3edd05e45f64"
121
  },
122
  "checkpoints/checkpoint-160/trainer_state.json": {
123
+ "size": 70210,
124
+ "sha256": "837c450cb4078aa28309f4999542aedfa831db547cb0d7b227f3a37c393d0355"
125
  },
126
  "checkpoints/checkpoint-160/training_args.bin": {
127
  "size": 8273,
 
133
  },
134
  "checkpoints/checkpoint-192/adapter_config.json": {
135
  "size": 1098,
136
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
137
  },
138
  "checkpoints/checkpoint-192/adapter_model.safetensors": {
139
  "size": 547777976,
140
+ "sha256": "dd65c9138a23ee4b709df1f879e1ce2a71a67a3461f4f7ed0fae00ba0eb853d5"
141
  },
142
  "checkpoints/checkpoint-192/chat_template.jinja": {
143
  "size": 1532,
 
145
  },
146
  "checkpoints/checkpoint-192/optimizer.pt": {
147
  "size": 1048106435,
148
+ "sha256": "0e74bd30f32259d424a0528a8fc03e88152c9448882265722a55166fb622af3a"
149
  },
150
  "checkpoints/checkpoint-192/rng_state.pth": {
151
  "size": 14645,
 
168
  "sha256": "6dcb03db82c34201f9966e9d9c444c1957b95cd53c5ea9ed3b950f223ea74fbe"
169
  },
170
  "checkpoints/checkpoint-192/trainer_state.json": {
171
+ "size": 84216,
172
+ "sha256": "76ffe843ba65218315ddfe5c2695fb2b03f1cf28ed59a4af495c5f42f439624f"
173
  },
174
  "checkpoints/checkpoint-192/training_args.bin": {
175
  "size": 8273,
 
181
  },
182
  "checkpoints/checkpoint-224/adapter_config.json": {
183
  "size": 1098,
184
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
185
  },
186
  "checkpoints/checkpoint-224/adapter_model.safetensors": {
187
  "size": 547777976,
188
+ "sha256": "16e7d16f507266e4de0618e13529f0f6852f7dfacc533b3d9beb61399405123d"
189
  },
190
  "checkpoints/checkpoint-224/chat_template.jinja": {
191
  "size": 1532,
 
193
  },
194
  "checkpoints/checkpoint-224/optimizer.pt": {
195
  "size": 1048106435,
196
+ "sha256": "f456a3b4b0b5cfe5bb7061718fafa69fb3393a4a4055872f85b783cf6135b255"
197
  },
198
  "checkpoints/checkpoint-224/rng_state.pth": {
199
  "size": 14645,
 
216
  "sha256": "14c3d2181ef61e0cf34f6fe59f885a012f56477195e0bfcf74f965a091355c99"
217
  },
218
  "checkpoints/checkpoint-224/trainer_state.json": {
219
+ "size": 98236,
220
+ "sha256": "57eb4a9f8711b9449eb9c8e699bb3ce11e013cb7c5ed53b4ec205fdb2524b50c"
221
  },
222
  "checkpoints/checkpoint-224/training_args.bin": {
223
  "size": 8273,
 
229
  },
230
  "checkpoints/checkpoint-256/adapter_config.json": {
231
  "size": 1098,
232
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
233
  },
234
  "checkpoints/checkpoint-256/adapter_model.safetensors": {
235
  "size": 547777976,
236
+ "sha256": "de74ce97413c0093f4f76198ae64237c0da203ba9a293aa2b281e66d918ca952"
237
  },
238
  "checkpoints/checkpoint-256/chat_template.jinja": {
239
  "size": 1532,
 
241
  },
242
  "checkpoints/checkpoint-256/optimizer.pt": {
243
  "size": 1048106435,
244
+ "sha256": "a2e59a33e5d93b9218821342e940feec395f412661f80e94be666601cd30c6c3"
245
  },
246
  "checkpoints/checkpoint-256/rng_state.pth": {
247
  "size": 14645,
 
264
  "sha256": "a1db2747ccfd098f33eee7a195bbd033988a2849ce30491603e0d31cbe794a14"
265
  },
266
  "checkpoints/checkpoint-256/trainer_state.json": {
267
+ "size": 112286,
268
+ "sha256": "2437fc722ed7350b2570a2bdb13ab9f6fc9d680ad29c3ad87cb5ac95dba68124"
269
  },
270
  "checkpoints/checkpoint-256/training_args.bin": {
271
  "size": 8273,
 
277
  },
278
  "checkpoints/checkpoint-288/adapter_config.json": {
279
  "size": 1098,
280
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
281
  },
282
  "checkpoints/checkpoint-288/adapter_model.safetensors": {
283
  "size": 547777976,
284
+ "sha256": "da2bdb4f81462625bb226aa42eb809bea433a88b8d19660ad0cb52a8f25dfb1f"
285
  },
286
  "checkpoints/checkpoint-288/chat_template.jinja": {
287
  "size": 1532,
 
289
  },
290
  "checkpoints/checkpoint-288/optimizer.pt": {
291
  "size": 1048106435,
292
+ "sha256": "c144b815c2b7bcbf81221112a3f40c21c4e01914d789b90a94c663d34e6977b2"
293
  },
294
  "checkpoints/checkpoint-288/rng_state.pth": {
295
  "size": 14645,
 
312
  "sha256": "f1eed36b4e578b166623815e5accad0aac99aa3b265255a7471d4f0a7c22b297"
313
  },
314
  "checkpoints/checkpoint-288/trainer_state.json": {
315
+ "size": 126385,
316
+ "sha256": "d34f36262ee1658dbadffb114672a386ee3cc513554f05b57c3f7b9c9d453ed7"
317
  },
318
  "checkpoints/checkpoint-288/training_args.bin": {
319
  "size": 8273,
 
325
  },
326
  "checkpoints/checkpoint-32/adapter_config.json": {
327
  "size": 1098,
328
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
329
  },
330
  "checkpoints/checkpoint-32/adapter_model.safetensors": {
331
  "size": 547777976,
332
+ "sha256": "32aa7a1bdc40de6c271cdaf4ba2e825940729a8400f15293192a0ab0fd0be3c4"
333
  },
334
  "checkpoints/checkpoint-32/chat_template.jinja": {
335
  "size": 1532,
 
337
  },
338
  "checkpoints/checkpoint-32/optimizer.pt": {
339
  "size": 1048106435,
340
+ "sha256": "d2c09fd208676c30f1a17a813e8f50e6c1234fdc5a8449106274744a5bc40156"
341
  },
342
  "checkpoints/checkpoint-32/rng_state.pth": {
343
  "size": 14645,
 
360
  "sha256": "d9025748b8b157d8490d18576a010318e827c122bd7c9d4da1001a4e203d8a9b"
361
  },
362
  "checkpoints/checkpoint-32/trainer_state.json": {
363
+ "size": 14406,
364
+ "sha256": "12d755108b595d32441e426440b368f10f87ac530ad04e3b58bd04d5896495d0"
365
  },
366
  "checkpoints/checkpoint-32/training_args.bin": {
367
  "size": 8273,
 
373
  },
374
  "checkpoints/checkpoint-320/adapter_config.json": {
375
  "size": 1098,
376
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
377
  },
378
  "checkpoints/checkpoint-320/adapter_model.safetensors": {
379
  "size": 547777976,
380
+ "sha256": "ecfb48f90e7121908ec1d3cf02765c1c9138a340de7d19e8645f0c7691611d51"
381
  },
382
  "checkpoints/checkpoint-320/chat_template.jinja": {
383
  "size": 1532,
 
385
  },
386
  "checkpoints/checkpoint-320/optimizer.pt": {
387
  "size": 1048106435,
388
+ "sha256": "4b35a07c34ef08caa48bca5c42b10295e72e8b1793c7dd8a269e93a8b9766920"
389
  },
390
  "checkpoints/checkpoint-320/rng_state.pth": {
391
  "size": 14645,
 
408
  "sha256": "fa2e57cf6dee598825b9bba06159bc70e457d9e8315c9d1028a34b835eec8770"
409
  },
410
  "checkpoints/checkpoint-320/trainer_state.json": {
411
+ "size": 140501,
412
+ "sha256": "4c560090a8acf9f23947b1325483710f17afc68529ecd1eb3df8e3ff6401e1ed"
413
  },
414
  "checkpoints/checkpoint-320/training_args.bin": {
415
  "size": 8273,
 
421
  },
422
  "checkpoints/checkpoint-352/adapter_config.json": {
423
  "size": 1098,
424
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
425
  },
426
  "checkpoints/checkpoint-352/adapter_model.safetensors": {
427
  "size": 547777976,
428
+ "sha256": "f21e802c0d2fcc1d790922454cc55a3257b76beae21bbf54ea77c6151be66e6e"
429
  },
430
  "checkpoints/checkpoint-352/chat_template.jinja": {
431
  "size": 1532,
 
433
  },
434
  "checkpoints/checkpoint-352/optimizer.pt": {
435
  "size": 1048106435,
436
+ "sha256": "6ff65aaa33bf6a907d3faa52032f129bb11b519ac761adc2816aec8951f53b4e"
437
  },
438
  "checkpoints/checkpoint-352/rng_state.pth": {
439
  "size": 14645,
 
456
  "sha256": "b9bc28c20f274eed19841a293251e42a3a606798b019918fd3fd29a96ca3043c"
457
  },
458
  "checkpoints/checkpoint-352/trainer_state.json": {
459
+ "size": 154629,
460
+ "sha256": "dd3348f05d98c22b8b7f6bdce1fdebf5bd2f43e34e4868f4f627cb95387429aa"
461
  },
462
  "checkpoints/checkpoint-352/training_args.bin": {
463
  "size": 8273,
 
469
  },
470
  "checkpoints/checkpoint-384/adapter_config.json": {
471
  "size": 1098,
472
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
473
  },
474
  "checkpoints/checkpoint-384/adapter_model.safetensors": {
475
  "size": 547777976,
476
+ "sha256": "535e88052cb6afb16717578e8ca20d23dd4e86de845c837f78a8af4881d745cf"
477
  },
478
  "checkpoints/checkpoint-384/chat_template.jinja": {
479
  "size": 1532,
 
481
  },
482
  "checkpoints/checkpoint-384/optimizer.pt": {
483
  "size": 1048106435,
484
+ "sha256": "1e2785d9c6ff17bc108c22ed9e79c9bbf116cbeae3109e3b396835a887d1ed2d"
485
  },
486
  "checkpoints/checkpoint-384/rng_state.pth": {
487
  "size": 14645,
 
504
  "sha256": "3075bc59309b4b7b0fefd61404619b2d7b2fe67268f884f30ea4319fd447e21e"
505
  },
506
  "checkpoints/checkpoint-384/trainer_state.json": {
507
+ "size": 168780,
508
+ "sha256": "af01e24ff12e70b4ef31c9bf5b86c1e7bd14855cea703883da4d76fe3105c551"
509
  },
510
  "checkpoints/checkpoint-384/training_args.bin": {
511
  "size": 8273,
 
517
  },
518
  "checkpoints/checkpoint-416/adapter_config.json": {
519
  "size": 1098,
520
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
521
  },
522
  "checkpoints/checkpoint-416/adapter_model.safetensors": {
523
  "size": 547777976,
524
+ "sha256": "0d560dd0eb4e8597dfc939225d96feb31798b32c08203167541c630616fa2492"
525
  },
526
  "checkpoints/checkpoint-416/chat_template.jinja": {
527
  "size": 1532,
 
529
  },
530
  "checkpoints/checkpoint-416/optimizer.pt": {
531
  "size": 1048106435,
532
+ "sha256": "cd9783ec9609f3730f73bfcc5cfd64cc123f08f492bbf069f1c6f47c16144b5d"
533
  },
534
  "checkpoints/checkpoint-416/rng_state.pth": {
535
  "size": 14645,
 
552
  "sha256": "033256f14efaa1c491d2eeb0fe5a1f206222c187594393c6791fa23f37aacb5e"
553
  },
554
  "checkpoints/checkpoint-416/trainer_state.json": {
555
+ "size": 182932,
556
+ "sha256": "997563a587d83c14e55f23a0000c2acd92899991625a2831dedefcc6d621267d"
557
  },
558
  "checkpoints/checkpoint-416/training_args.bin": {
559
  "size": 8273,
 
565
  },
566
  "checkpoints/checkpoint-448/adapter_config.json": {
567
  "size": 1098,
568
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
569
  },
570
  "checkpoints/checkpoint-448/adapter_model.safetensors": {
571
  "size": 547777976,
572
+ "sha256": "3bacb4d3f901db3a71785ec9bbf0c989eb190275b59257a46ebe7cda04a4211b"
573
  },
574
  "checkpoints/checkpoint-448/chat_template.jinja": {
575
  "size": 1532,
 
577
  },
578
  "checkpoints/checkpoint-448/optimizer.pt": {
579
  "size": 1048106435,
580
+ "sha256": "33cfd5b5b7300a5e59a22e211b622aa8250b83b1e42d9aa245e82b799848bf3e"
581
  },
582
  "checkpoints/checkpoint-448/rng_state.pth": {
583
  "size": 14645,
 
600
  "sha256": "87ae99c7d9168aee7dd04df6332dedc5884f11b6c9ce320edc6d2b867c048531"
601
  },
602
  "checkpoints/checkpoint-448/trainer_state.json": {
603
+ "size": 197087,
604
+ "sha256": "dd453a8c241e9adba0f2155ac5b899e149e25772266ea7f4644cd0217d343d4f"
605
  },
606
  "checkpoints/checkpoint-448/training_args.bin": {
607
  "size": 8273,
 
613
  },
614
  "checkpoints/checkpoint-480/adapter_config.json": {
615
  "size": 1098,
616
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
617
  },
618
  "checkpoints/checkpoint-480/adapter_model.safetensors": {
619
  "size": 547777976,
620
+ "sha256": "759e2f47a5e976dcfbe11dde9491135d96d1b05bada169151f4a8f939269acc5"
621
  },
622
  "checkpoints/checkpoint-480/chat_template.jinja": {
623
  "size": 1532,
 
625
  },
626
  "checkpoints/checkpoint-480/optimizer.pt": {
627
  "size": 1048106435,
628
+ "sha256": "cd01ae9ec3eb43f4befc813946bc00ce45c82d3237dfa7f15ddccfe7878ea01d"
629
  },
630
  "checkpoints/checkpoint-480/rng_state.pth": {
631
  "size": 14645,
 
648
  "sha256": "6c547605fe97848a0748ea1ac9bef1baa4a78902cfb711aedb151a69c5994f49"
649
  },
650
  "checkpoints/checkpoint-480/trainer_state.json": {
651
+ "size": 211284,
652
+ "sha256": "aead414e3a072327e5746caf31d43a0b98ec71e9b30ca1897ce90790862804a6"
653
  },
654
  "checkpoints/checkpoint-480/training_args.bin": {
655
  "size": 8273,
 
661
  },
662
  "checkpoints/checkpoint-512/adapter_config.json": {
663
  "size": 1098,
664
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
665
  },
666
  "checkpoints/checkpoint-512/adapter_model.safetensors": {
667
  "size": 547777976,
668
+ "sha256": "daeafd5d686b4b39c5b3dd656eca36514ea93ef225c5afab298c74bbfab4b632"
669
  },
670
  "checkpoints/checkpoint-512/chat_template.jinja": {
671
  "size": 1532,
 
673
  },
674
  "checkpoints/checkpoint-512/optimizer.pt": {
675
  "size": 1048106435,
676
+ "sha256": "05bb512f0593d6339b6dd5bd7fd23b00fbb5ccbe33fcfaa642269827ce75bdc6"
677
  },
678
  "checkpoints/checkpoint-512/rng_state.pth": {
679
  "size": 14645,
 
696
  "sha256": "c0fcac61392efcd924c1610ad304635be7ae41b33c0ad3526b109f7b1f8f7fe9"
697
  },
698
  "checkpoints/checkpoint-512/trainer_state.json": {
699
+ "size": 225479,
700
+ "sha256": "02f9a66b02bb552b9fb5b0fe2cf523809d3b2ee4475683870014d1b13ad7a8a2"
701
  },
702
  "checkpoints/checkpoint-512/training_args.bin": {
703
  "size": 8273,
 
709
  },
710
  "checkpoints/checkpoint-64/adapter_config.json": {
711
  "size": 1098,
712
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
713
  },
714
  "checkpoints/checkpoint-64/adapter_model.safetensors": {
715
  "size": 547777976,
716
+ "sha256": "e859c9377bfb05045990aad9622ecbcea4ec1f51ee262c524148710f93c608f1"
717
  },
718
  "checkpoints/checkpoint-64/chat_template.jinja": {
719
  "size": 1532,
 
721
  },
722
  "checkpoints/checkpoint-64/optimizer.pt": {
723
  "size": 1048106435,
724
+ "sha256": "f0bb5b0e8d378704b8006d08d74c2f903a1422f4e828a3f1e606a4b8edbfa7a3"
725
  },
726
  "checkpoints/checkpoint-64/rng_state.pth": {
727
  "size": 14645,
 
744
  "sha256": "9c3a6e3ed0949f0de9f9b7c1d666e376e3d3626dfe0a6da2a265ef135b18e10b"
745
  },
746
  "checkpoints/checkpoint-64/trainer_state.json": {
747
+ "size": 28330,
748
+ "sha256": "82a0433edb2739b866083caf137728dc3d96ccd93e5eb2fd39a66d8d6f7bd5ac"
749
  },
750
  "checkpoints/checkpoint-64/training_args.bin": {
751
  "size": 8273,
 
757
  },
758
  "checkpoints/checkpoint-96/adapter_config.json": {
759
  "size": 1098,
760
+ "sha256": "7c54ee9b6e682028e1421326f0a1cbbeb36cdbcc1ad907238e6d932c196e293d"
761
  },
762
  "checkpoints/checkpoint-96/adapter_model.safetensors": {
763
  "size": 547777976,
764
+ "sha256": "0e85942ea7b1d777c6e874fcfe57e08e0752f1e581a02a35bb62b7dc1942bc5a"
765
  },
766
  "checkpoints/checkpoint-96/chat_template.jinja": {
767
  "size": 1532,
 
769
  },
770
  "checkpoints/checkpoint-96/optimizer.pt": {
771
  "size": 1048106435,
772
+ "sha256": "e09d86b7ef11b8b51a819ec69e40b2c45e5f32072380fdac56e42f42d36fcb81"
773
  },
774
  "checkpoints/checkpoint-96/rng_state.pth": {
775
  "size": 14645,
 
792
  "sha256": "f276353ce1fa0f6362678bebc659fb55eeca16e731ebeb9a62916aa8efaa3b8b"
793
  },
794
  "checkpoints/checkpoint-96/trainer_state.json": {
795
+ "size": 42261,
796
+ "sha256": "06d1244ce26f0b9ac73872fde6867ebfff2239569a3be1aa30c8b754d5f626c0"
797
  },
798
  "checkpoints/checkpoint-96/training_args.bin": {
799
  "size": 8273,
 
804
  "sha256": "7c5b66498629a75e7fe3e4219bc41040b8106735e598f0837d60656025390e1a"
805
  },
806
  "checkpoints/debug.log": {
807
+ "size": 267536,
808
+ "sha256": "5096497579321697536666777d8b80e318847cf367750b87cfd3891265c1880c"
809
  },
810
  "checkpoints/processor_config.json": {
811
  "size": 519,
 
837
  },
838
  "health/training_started.json": {
839
  "size": 137,
840
+ "sha256": "cc75570db76b03c03df3dcaac3381a89a9e92ff2b39719f3d00ac41bb89147d1"
841
  },
842
  "run.json": {
843
  "size": 407,
844
+ "sha256": "d72e7f12f02ca389af664a6e6a90f9061050132100b32c737e18c5207f5c9ef3"
845
  },
846
  "train.log": {
847
+ "size": 274978,
848
+ "sha256": "b37791314be7b268280e0005a698396839dc9b72aa3ab293e0b4ad7c345ef6fa"
849
  },
850
  "trainer_state.final.json": {
851
+ "size": 225479,
852
+ "sha256": "02f9a66b02bb552b9fb5b0fe2cf523809d3b2ee4475683870014d1b13ad7a8a2"
853
  },
854
  "training_examples.jsonl": {
855
  "size": 3159083,
 
857
  },
858
  "training_provenance.json": {
859
  "size": 4090,
860
+ "sha256": "d4d5407f6abf83d471b752bd6b1e74777b935dc03d9dce25dd4ec0c1d32d3a6a"
861
  },
862
  "training_trace.jsonl": {
863
+ "size": 182211,
864
+ "sha256": "3162f25b3bb9dbc2b6e99501728763c5a9acec2b98a1dd4988902fa8ee0b3a2a"
865
  }
866
  }
867
  }
aft_wave_v2/coin_real_4x__agreement/training/TRAINED.json CHANGED
@@ -8,7 +8,7 @@
8
  "training_rows": 8192,
9
  "stage": "aft_dispatch_v4_wide",
10
  "seed": 42,
11
- "minutes": 59.59,
12
  "lora": {
13
  "r": 32,
14
  "alpha": 64,
 
8
  "training_rows": 8192,
9
  "stage": "aft_dispatch_v4_wide",
10
  "seed": 42,
11
+ "minutes": 58.83,
12
  "lora": {
13
  "r": 32,
14
  "alpha": 64,
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:16cc65acfba87debf950e49fc05f5dce80ea16745d793c826ed96faac31a47a1
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:daeafd5d686b4b39c5b3dd656eca36514ea93ef225c5afab298c74bbfab4b632
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e7202ce5ed3e0cfd2be9bb413acb17a214475b81453c800a2e3c69d4ff5d7959
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67174eba71aaa201dd7aff62c7fbd3596ad8b4afeba399abae18cc3a0dd00888
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eb131e3cb1d8065ea88ab03c3129c1bdf7f01361df81ae0569e4c2b6008fa036
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c48bb4d7fcaca8cbcee6c332057a50925e65b15baae813555c4b32ddf641b01b
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-128/trainer_state.json CHANGED
@@ -11,7 +11,7 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.00390625,
14
- "grad_norm": 0.870697021484375,
15
  "learning_rate": 0.0,
16
  "loss": 0.10251401364803314,
17
  "memory/device_reserved (GiB)": 34.94,
@@ -20,12 +20,12 @@
20
  "ppl": 1.10795,
21
  "step": 1,
22
  "tokens/total": 30464,
23
- "tokens/train_per_sec_per_gpu": 23.98,
24
  "tokens/trainable": 470
25
  },
26
  {
27
  "epoch": 0.0078125,
28
- "grad_norm": 0.8108459115028381,
29
  "learning_rate": 4.000000000000001e-06,
30
  "loss": 0.09678442776203156,
31
  "memory/device_reserved (GiB)": 35.83,
@@ -34,900 +34,900 @@
34
  "ppl": 1.10162,
35
  "step": 2,
36
  "tokens/total": 60800,
37
- "tokens/train_per_sec_per_gpu": 35.02,
38
  "tokens/trainable": 922
39
  },
40
  {
41
  "epoch": 0.01171875,
42
- "grad_norm": 1.8027335405349731,
43
  "learning_rate": 8.000000000000001e-06,
44
- "loss": 0.10952483862638474,
45
  "memory/device_reserved (GiB)": 35.85,
46
  "memory/max_active (GiB)": 33.91,
47
  "memory/max_allocated (GiB)": 33.91,
48
- "ppl": 1.11575,
49
  "step": 3,
50
  "tokens/total": 91136,
51
- "tokens/train_per_sec_per_gpu": 34.48,
52
  "tokens/trainable": 1396
53
  },
54
  {
55
  "epoch": 0.015625,
56
- "grad_norm": 1.0353018045425415,
57
  "learning_rate": 1.2e-05,
58
- "loss": 0.10303406417369843,
59
  "memory/device_reserved (GiB)": 35.85,
60
  "memory/max_active (GiB)": 33.91,
61
  "memory/max_allocated (GiB)": 33.91,
62
- "ppl": 1.10853,
63
  "step": 4,
64
  "tokens/total": 121552,
65
- "tokens/train_per_sec_per_gpu": 38.01,
66
  "tokens/trainable": 1850
67
  },
68
  {
69
  "epoch": 0.01953125,
70
- "grad_norm": 1.0778985023498535,
71
  "learning_rate": 1.6000000000000003e-05,
72
- "loss": 0.10945924371480942,
73
  "memory/device_reserved (GiB)": 35.85,
74
  "memory/max_active (GiB)": 33.98,
75
  "memory/max_allocated (GiB)": 33.98,
76
- "ppl": 1.11567,
77
  "step": 5,
78
  "tokens/total": 152304,
79
- "tokens/train_per_sec_per_gpu": 33.76,
80
  "tokens/trainable": 2302
81
  },
82
  {
83
  "epoch": 0.0234375,
84
- "grad_norm": 0.8929882645606995,
85
  "learning_rate": 2e-05,
86
- "loss": 0.08588902652263641,
87
  "memory/device_reserved (GiB)": 35.85,
88
  "memory/max_active (GiB)": 33.78,
89
  "memory/max_allocated (GiB)": 33.78,
90
- "ppl": 1.08969,
91
  "step": 6,
92
  "tokens/total": 182448,
93
- "tokens/train_per_sec_per_gpu": 31.47,
94
  "tokens/trainable": 2742
95
  },
96
  {
97
  "epoch": 0.02734375,
98
- "grad_norm": 1.6409597396850586,
99
  "learning_rate": 2.4e-05,
100
- "loss": 0.07352827489376068,
101
  "memory/device_reserved (GiB)": 35.85,
102
  "memory/max_active (GiB)": 33.85,
103
  "memory/max_allocated (GiB)": 33.85,
104
- "ppl": 1.0763,
105
  "step": 7,
106
  "tokens/total": 212736,
107
- "tokens/train_per_sec_per_gpu": 36.58,
108
  "tokens/trainable": 3208
109
  },
110
  {
111
  "epoch": 0.03125,
112
- "grad_norm": 1.5843520164489746,
113
  "learning_rate": 2.8000000000000003e-05,
114
- "loss": 0.07798244059085846,
115
  "memory/device_reserved (GiB)": 36.07,
116
  "memory/max_active (GiB)": 33.87,
117
  "memory/max_allocated (GiB)": 33.87,
118
- "ppl": 1.0811,
119
  "step": 8,
120
  "tokens/total": 243024,
121
- "tokens/train_per_sec_per_gpu": 31.83,
122
  "tokens/trainable": 3655
123
  },
124
  {
125
  "epoch": 0.03515625,
126
- "grad_norm": 1.3725916147232056,
127
  "learning_rate": 3.2000000000000005e-05,
128
- "loss": 0.04634054750204086,
129
  "memory/device_reserved (GiB)": 36.07,
130
  "memory/max_active (GiB)": 33.82,
131
  "memory/max_allocated (GiB)": 33.82,
132
- "ppl": 1.04743,
133
  "step": 9,
134
  "tokens/total": 271264,
135
- "tokens/train_per_sec_per_gpu": 35.62,
136
  "tokens/trainable": 4091
137
  },
138
  {
139
  "epoch": 0.0390625,
140
- "grad_norm": 2.544938564300537,
141
  "learning_rate": 3.6e-05,
142
- "loss": 0.08677884936332703,
143
  "memory/device_reserved (GiB)": 36.07,
144
  "memory/max_active (GiB)": 33.86,
145
  "memory/max_allocated (GiB)": 33.86,
146
- "ppl": 1.09066,
147
  "step": 10,
148
  "tokens/total": 301520,
149
- "tokens/train_per_sec_per_gpu": 34.92,
150
  "tokens/trainable": 4553
151
  },
152
  {
153
  "epoch": 0.04296875,
154
- "grad_norm": 1.6364734172821045,
155
  "learning_rate": 4e-05,
156
- "loss": 0.07754456996917725,
157
- "memory/device_reserved (GiB)": 36.07,
158
  "memory/max_active (GiB)": 33.8,
159
  "memory/max_allocated (GiB)": 33.8,
160
- "ppl": 1.08063,
161
  "step": 11,
162
  "tokens/total": 331808,
163
- "tokens/train_per_sec_per_gpu": 33.0,
164
  "tokens/trainable": 4992
165
  },
166
  {
167
  "epoch": 0.046875,
168
- "grad_norm": 2.167391538619995,
169
  "learning_rate": 4.4000000000000006e-05,
170
- "loss": 0.11765319854021072,
171
- "memory/device_reserved (GiB)": 36.07,
172
  "memory/max_active (GiB)": 33.87,
173
  "memory/max_allocated (GiB)": 33.87,
174
- "ppl": 1.12485,
175
  "step": 12,
176
  "tokens/total": 362224,
177
- "tokens/train_per_sec_per_gpu": 38.18,
178
  "tokens/trainable": 5494
179
  },
180
  {
181
  "epoch": 0.05078125,
182
- "grad_norm": 3.196911573410034,
183
  "learning_rate": 4.8e-05,
184
- "loss": 0.03510797768831253,
185
- "memory/device_reserved (GiB)": 36.15,
186
  "memory/max_active (GiB)": 33.91,
187
  "memory/max_allocated (GiB)": 33.91,
188
- "ppl": 1.03573,
189
  "step": 13,
190
  "tokens/total": 392432,
191
- "tokens/train_per_sec_per_gpu": 33.67,
192
  "tokens/trainable": 5933
193
  },
194
  {
195
  "epoch": 0.0546875,
196
- "grad_norm": 1.9654567241668701,
197
  "learning_rate": 5.2000000000000004e-05,
198
- "loss": 0.061097435653209686,
199
- "memory/device_reserved (GiB)": 36.15,
200
  "memory/max_active (GiB)": 33.79,
201
  "memory/max_allocated (GiB)": 33.79,
202
- "ppl": 1.063,
203
  "step": 14,
204
  "tokens/total": 422784,
205
- "tokens/train_per_sec_per_gpu": 29.53,
206
  "tokens/trainable": 6369
207
  },
208
  {
209
  "epoch": 0.05859375,
210
- "grad_norm": 1.934105396270752,
211
  "learning_rate": 5.6000000000000006e-05,
212
- "loss": 0.05385493487119675,
213
- "memory/device_reserved (GiB)": 36.15,
214
  "memory/max_active (GiB)": 33.92,
215
  "memory/max_allocated (GiB)": 33.92,
216
- "ppl": 1.05533,
217
  "step": 15,
218
  "tokens/total": 453376,
219
- "tokens/train_per_sec_per_gpu": 32.96,
220
  "tokens/trainable": 6820
221
  },
222
  {
223
  "epoch": 0.0625,
224
- "grad_norm": 1.4659016132354736,
225
  "learning_rate": 6e-05,
226
- "loss": 0.052153222262859344,
227
  "memory/device_reserved (GiB)": 36.21,
228
  "memory/max_active (GiB)": 33.83,
229
  "memory/max_allocated (GiB)": 33.83,
230
- "ppl": 1.05354,
231
  "step": 16,
232
  "tokens/total": 483504,
233
- "tokens/train_per_sec_per_gpu": 30.68,
234
  "tokens/trainable": 7258
235
  },
236
  {
237
  "epoch": 0.06640625,
238
- "grad_norm": 1.9650894403457642,
239
  "learning_rate": 6.400000000000001e-05,
240
- "loss": 0.05092189460992813,
241
  "memory/device_reserved (GiB)": 36.21,
242
  "memory/max_active (GiB)": 33.9,
243
  "memory/max_allocated (GiB)": 33.9,
244
- "ppl": 1.05224,
245
  "step": 17,
246
  "tokens/total": 514032,
247
- "tokens/train_per_sec_per_gpu": 35.09,
248
  "tokens/trainable": 7710
249
  },
250
  {
251
  "epoch": 0.0703125,
252
- "grad_norm": 0.6891724467277527,
253
  "learning_rate": 6.800000000000001e-05,
254
- "loss": 0.031155016273260117,
255
  "memory/device_reserved (GiB)": 36.21,
256
  "memory/max_active (GiB)": 33.83,
257
  "memory/max_allocated (GiB)": 33.83,
258
- "ppl": 1.03165,
259
  "step": 18,
260
  "tokens/total": 544432,
261
- "tokens/train_per_sec_per_gpu": 31.61,
262
  "tokens/trainable": 8174
263
  },
264
  {
265
  "epoch": 0.07421875,
266
- "grad_norm": 0.8662010431289673,
267
  "learning_rate": 7.2e-05,
268
- "loss": 0.03036138042807579,
269
  "memory/device_reserved (GiB)": 36.21,
270
  "memory/max_active (GiB)": 33.83,
271
  "memory/max_allocated (GiB)": 33.83,
272
- "ppl": 1.03083,
273
  "step": 19,
274
  "tokens/total": 574880,
275
- "tokens/train_per_sec_per_gpu": 34.37,
276
  "tokens/trainable": 8617
277
  },
278
  {
279
  "epoch": 0.078125,
280
- "grad_norm": 0.7999041080474854,
281
  "learning_rate": 7.6e-05,
282
- "loss": 0.03458942472934723,
283
  "memory/device_reserved (GiB)": 36.21,
284
  "memory/max_active (GiB)": 33.97,
285
  "memory/max_allocated (GiB)": 33.97,
286
- "ppl": 1.03519,
287
  "step": 20,
288
  "tokens/total": 605408,
289
- "tokens/train_per_sec_per_gpu": 30.32,
290
  "tokens/trainable": 9037
291
  },
292
  {
293
  "epoch": 0.08203125,
294
- "grad_norm": 0.5816919207572937,
295
  "learning_rate": 8e-05,
296
- "loss": 0.02401110902428627,
297
  "memory/device_reserved (GiB)": 36.21,
298
  "memory/max_active (GiB)": 33.77,
299
  "memory/max_allocated (GiB)": 33.77,
300
- "ppl": 1.0243,
301
  "step": 21,
302
  "tokens/total": 635584,
303
- "tokens/train_per_sec_per_gpu": 36.59,
304
  "tokens/trainable": 9503
305
  },
306
  {
307
  "epoch": 0.0859375,
308
- "grad_norm": 0.6761882901191711,
309
  "learning_rate": 8.4e-05,
310
- "loss": 0.01873329095542431,
311
  "memory/device_reserved (GiB)": 36.21,
312
  "memory/max_active (GiB)": 33.88,
313
  "memory/max_allocated (GiB)": 33.88,
314
- "ppl": 1.01891,
315
  "step": 22,
316
  "tokens/total": 665952,
317
- "tokens/train_per_sec_per_gpu": 36.71,
318
  "tokens/trainable": 9972
319
  },
320
  {
321
  "epoch": 0.08984375,
322
- "grad_norm": 0.9077537655830383,
323
  "learning_rate": 8.800000000000001e-05,
324
- "loss": 0.017490077763795853,
325
  "memory/device_reserved (GiB)": 36.21,
326
  "memory/max_active (GiB)": 33.9,
327
  "memory/max_allocated (GiB)": 33.9,
328
- "ppl": 1.01764,
329
  "step": 23,
330
  "tokens/total": 696240,
331
- "tokens/train_per_sec_per_gpu": 37.39,
332
  "tokens/trainable": 10447
333
  },
334
  {
335
  "epoch": 0.09375,
336
- "grad_norm": 1.3226462602615356,
337
  "learning_rate": 9.200000000000001e-05,
338
- "loss": 0.028416279703378677,
339
  "memory/device_reserved (GiB)": 36.35,
340
  "memory/max_active (GiB)": 33.88,
341
  "memory/max_allocated (GiB)": 33.88,
342
- "ppl": 1.02882,
343
  "step": 24,
344
  "tokens/total": 726464,
345
- "tokens/train_per_sec_per_gpu": 37.12,
346
  "tokens/trainable": 10899
347
  },
348
  {
349
  "epoch": 0.09765625,
350
- "grad_norm": 0.9712215065956116,
351
  "learning_rate": 9.6e-05,
352
- "loss": 0.025973526760935783,
353
  "memory/device_reserved (GiB)": 36.35,
354
  "memory/max_active (GiB)": 33.84,
355
  "memory/max_allocated (GiB)": 33.84,
356
- "ppl": 1.02631,
357
  "step": 25,
358
  "tokens/total": 756704,
359
- "tokens/train_per_sec_per_gpu": 32.97,
360
  "tokens/trainable": 11360
361
  },
362
  {
363
  "epoch": 0.1015625,
364
- "grad_norm": 0.8638900518417358,
365
  "learning_rate": 0.0001,
366
- "loss": 0.022434517741203308,
367
  "memory/device_reserved (GiB)": 36.35,
368
  "memory/max_active (GiB)": 33.82,
369
  "memory/max_allocated (GiB)": 33.82,
370
- "ppl": 1.02269,
371
  "step": 26,
372
  "tokens/total": 786912,
373
- "tokens/train_per_sec_per_gpu": 29.23,
374
  "tokens/trainable": 11775
375
  },
376
  {
377
  "epoch": 0.10546875,
378
- "grad_norm": 0.6629000902175903,
379
  "learning_rate": 9.99990636831587e-05,
380
- "loss": 0.014538668096065521,
381
  "memory/device_reserved (GiB)": 36.35,
382
  "memory/max_active (GiB)": 33.46,
383
  "memory/max_allocated (GiB)": 33.46,
384
- "ppl": 1.01464,
385
  "step": 27,
386
  "tokens/total": 815424,
387
- "tokens/train_per_sec_per_gpu": 39.82,
388
  "tokens/trainable": 12242
389
  },
390
  {
391
  "epoch": 0.109375,
392
- "grad_norm": 0.3078136146068573,
393
  "learning_rate": 9.999625477159879e-05,
394
- "loss": 0.01195458322763443,
395
  "memory/device_reserved (GiB)": 36.35,
396
  "memory/max_active (GiB)": 33.89,
397
  "memory/max_allocated (GiB)": 33.89,
398
- "ppl": 1.01203,
399
  "step": 28,
400
  "tokens/total": 845584,
401
- "tokens/train_per_sec_per_gpu": 33.52,
402
  "tokens/trainable": 12696
403
  },
404
  {
405
  "epoch": 0.11328125,
406
- "grad_norm": 1.1301876306533813,
407
  "learning_rate": 9.999157338221051e-05,
408
- "loss": 0.022538531571626663,
409
  "memory/device_reserved (GiB)": 36.35,
410
  "memory/max_active (GiB)": 33.87,
411
  "memory/max_allocated (GiB)": 33.87,
412
- "ppl": 1.02279,
413
  "step": 29,
414
  "tokens/total": 875808,
415
- "tokens/train_per_sec_per_gpu": 34.38,
416
  "tokens/trainable": 13153
417
  },
418
  {
419
  "epoch": 0.1171875,
420
- "grad_norm": 0.41090911626815796,
421
  "learning_rate": 9.998501970980562e-05,
422
- "loss": 0.011871461756527424,
423
  "memory/device_reserved (GiB)": 36.35,
424
  "memory/max_active (GiB)": 33.82,
425
  "memory/max_allocated (GiB)": 33.82,
426
- "ppl": 1.01194,
427
  "step": 30,
428
  "tokens/total": 904096,
429
- "tokens/train_per_sec_per_gpu": 39.83,
430
  "tokens/trainable": 13624
431
  },
432
  {
433
  "epoch": 0.12109375,
434
- "grad_norm": 0.6772723197937012,
435
  "learning_rate": 9.997659402710915e-05,
436
- "loss": 0.013700846582651138,
437
  "memory/device_reserved (GiB)": 36.35,
438
  "memory/max_active (GiB)": 33.86,
439
  "memory/max_allocated (GiB)": 33.86,
440
- "ppl": 1.0138,
441
  "step": 31,
442
  "tokens/total": 934400,
443
- "tokens/train_per_sec_per_gpu": 34.07,
444
  "tokens/trainable": 14064
445
  },
446
  {
447
  "epoch": 0.125,
448
- "grad_norm": 1.207811951637268,
449
  "learning_rate": 9.996629668474818e-05,
450
- "loss": 0.01185896061360836,
451
  "memory/device_reserved (GiB)": 36.35,
452
  "memory/max_active (GiB)": 33.92,
453
  "memory/max_allocated (GiB)": 33.92,
454
- "ppl": 1.01193,
455
  "step": 32,
456
  "tokens/total": 964832,
457
- "tokens/train_per_sec_per_gpu": 38.9,
458
  "tokens/trainable": 14565
459
  },
460
  {
461
  "epoch": 0.12890625,
462
- "grad_norm": 0.46513956785202026,
463
  "learning_rate": 9.995412811123711e-05,
464
- "loss": 0.012477721087634563,
465
  "memory/device_reserved (GiB)": 36.35,
466
  "memory/max_active (GiB)": 33.79,
467
  "memory/max_allocated (GiB)": 33.79,
468
- "ppl": 1.01256,
469
  "step": 33,
470
  "tokens/total": 995040,
471
- "tokens/train_per_sec_per_gpu": 33.76,
472
  "tokens/trainable": 15005
473
  },
474
  {
475
  "epoch": 0.1328125,
476
- "grad_norm": 1.7668761014938354,
477
  "learning_rate": 9.994008881295999e-05,
478
- "loss": 0.03285093232989311,
479
  "memory/device_reserved (GiB)": 34.72,
480
  "memory/max_active (GiB)": 33.64,
481
  "memory/max_allocated (GiB)": 33.64,
482
- "ppl": 1.0334,
483
  "step": 34,
484
  "tokens/total": 1024896,
485
- "tokens/train_per_sec_per_gpu": 32.05,
486
  "tokens/trainable": 15459
487
  },
488
  {
489
  "epoch": 0.13671875,
490
- "grad_norm": 0.7816088795661926,
491
  "learning_rate": 9.992417937414932e-05,
492
- "loss": 0.028746310621500015,
493
  "memory/device_reserved (GiB)": 35.48,
494
  "memory/max_active (GiB)": 33.84,
495
  "memory/max_allocated (GiB)": 33.84,
496
- "ppl": 1.02916,
497
  "step": 35,
498
  "tokens/total": 1054992,
499
- "tokens/train_per_sec_per_gpu": 38.15,
500
  "tokens/trainable": 15960
501
  },
502
  {
503
  "epoch": 0.140625,
504
- "grad_norm": 0.683965265750885,
505
  "learning_rate": 9.99064004568618e-05,
506
- "loss": 0.020058901980519295,
507
  "memory/device_reserved (GiB)": 35.48,
508
  "memory/max_active (GiB)": 33.93,
509
  "memory/max_allocated (GiB)": 33.93,
510
- "ppl": 1.02026,
511
  "step": 36,
512
  "tokens/total": 1085280,
513
- "tokens/train_per_sec_per_gpu": 34.53,
514
  "tokens/trainable": 16428
515
  },
516
  {
517
  "epoch": 0.14453125,
518
- "grad_norm": 0.6797531843185425,
519
  "learning_rate": 9.988675280095074e-05,
520
- "loss": 0.010446591302752495,
521
- "memory/device_reserved (GiB)": 35.48,
522
  "memory/max_active (GiB)": 33.7,
523
  "memory/max_allocated (GiB)": 33.7,
524
- "ppl": 1.0105,
525
  "step": 37,
526
  "tokens/total": 1115504,
527
- "tokens/train_per_sec_per_gpu": 31.77,
528
  "tokens/trainable": 16884
529
  },
530
  {
531
  "epoch": 0.1484375,
532
- "grad_norm": 0.8899193406105042,
533
  "learning_rate": 9.986523722403528e-05,
534
- "loss": 0.017391683533787727,
535
- "memory/device_reserved (GiB)": 35.48,
536
  "memory/max_active (GiB)": 33.82,
537
  "memory/max_allocated (GiB)": 33.82,
538
- "ppl": 1.01754,
539
  "step": 38,
540
  "tokens/total": 1145712,
541
- "tokens/train_per_sec_per_gpu": 32.03,
542
  "tokens/trainable": 17341
543
  },
544
  {
545
  "epoch": 0.15234375,
546
- "grad_norm": 0.8132575750350952,
547
  "learning_rate": 9.984185462146642e-05,
548
- "loss": 0.02252483367919922,
549
- "memory/device_reserved (GiB)": 35.48,
550
  "memory/max_active (GiB)": 33.81,
551
  "memory/max_allocated (GiB)": 33.81,
552
- "ppl": 1.02278,
553
  "step": 39,
554
  "tokens/total": 1176128,
555
- "tokens/train_per_sec_per_gpu": 34.15,
556
  "tokens/trainable": 17786
557
  },
558
  {
559
  "epoch": 0.15625,
560
- "grad_norm": 0.4430502653121948,
561
  "learning_rate": 9.98166059662897e-05,
562
- "loss": 0.011966042220592499,
563
- "memory/device_reserved (GiB)": 35.48,
564
  "memory/max_active (GiB)": 33.85,
565
  "memory/max_allocated (GiB)": 33.85,
566
- "ppl": 1.01204,
567
  "step": 40,
568
  "tokens/total": 1206576,
569
- "tokens/train_per_sec_per_gpu": 34.99,
570
  "tokens/trainable": 18262
571
  },
572
  {
573
  "epoch": 0.16015625,
574
- "grad_norm": 0.5074653029441833,
575
  "learning_rate": 9.978949230920472e-05,
576
- "loss": 0.014877484180033207,
577
- "memory/device_reserved (GiB)": 35.87,
578
  "memory/max_active (GiB)": 33.86,
579
  "memory/max_allocated (GiB)": 33.86,
580
- "ppl": 1.01499,
581
  "step": 41,
582
  "tokens/total": 1236848,
583
- "tokens/train_per_sec_per_gpu": 32.7,
584
  "tokens/trainable": 18716
585
  },
586
  {
587
  "epoch": 0.1640625,
588
- "grad_norm": 0.5221464037895203,
589
  "learning_rate": 9.976051477852141e-05,
590
- "loss": 0.017510797828435898,
591
- "memory/device_reserved (GiB)": 35.87,
592
  "memory/max_active (GiB)": 33.83,
593
  "memory/max_allocated (GiB)": 33.83,
594
- "ppl": 1.01767,
595
  "step": 42,
596
  "tokens/total": 1267088,
597
- "tokens/train_per_sec_per_gpu": 32.87,
598
  "tokens/trainable": 19157
599
  },
600
  {
601
  "epoch": 0.16796875,
602
- "grad_norm": 0.6888790130615234,
603
  "learning_rate": 9.972967458011312e-05,
604
- "loss": 0.030433066189289093,
605
- "memory/device_reserved (GiB)": 35.87,
606
  "memory/max_active (GiB)": 33.82,
607
  "memory/max_allocated (GiB)": 33.82,
608
- "ppl": 1.0309,
609
  "step": 43,
610
  "tokens/total": 1297360,
611
- "tokens/train_per_sec_per_gpu": 36.5,
612
  "tokens/trainable": 19647
613
  },
614
  {
615
  "epoch": 0.171875,
616
- "grad_norm": 0.5600388050079346,
617
  "learning_rate": 9.96969729973664e-05,
618
- "loss": 0.013720017857849598,
619
- "memory/device_reserved (GiB)": 35.87,
620
  "memory/max_active (GiB)": 33.86,
621
  "memory/max_allocated (GiB)": 33.86,
622
- "ppl": 1.01381,
623
  "step": 44,
624
  "tokens/total": 1327744,
625
- "tokens/train_per_sec_per_gpu": 34.9,
626
  "tokens/trainable": 20119
627
  },
628
  {
629
  "epoch": 0.17578125,
630
- "grad_norm": 0.4291926324367523,
631
  "learning_rate": 9.966241139112754e-05,
632
- "loss": 0.00872582383453846,
633
- "memory/device_reserved (GiB)": 35.87,
634
  "memory/max_active (GiB)": 33.86,
635
  "memory/max_allocated (GiB)": 33.86,
636
- "ppl": 1.00876,
637
  "step": 45,
638
  "tokens/total": 1358096,
639
- "tokens/train_per_sec_per_gpu": 32.92,
640
  "tokens/trainable": 20560
641
  },
642
  {
643
  "epoch": 0.1796875,
644
- "grad_norm": 0.944327712059021,
645
  "learning_rate": 9.96259911996461e-05,
646
- "loss": 0.025248996913433075,
647
- "memory/device_reserved (GiB)": 35.87,
648
  "memory/max_active (GiB)": 33.84,
649
  "memory/max_allocated (GiB)": 33.84,
650
- "ppl": 1.02557,
651
  "step": 46,
652
  "tokens/total": 1388240,
653
- "tokens/train_per_sec_per_gpu": 32.59,
654
  "tokens/trainable": 20992
655
  },
656
  {
657
  "epoch": 0.18359375,
658
- "grad_norm": 0.2425016313791275,
659
  "learning_rate": 9.958771393851491e-05,
660
- "loss": 0.005067020654678345,
661
- "memory/device_reserved (GiB)": 35.87,
662
  "memory/max_active (GiB)": 33.26,
663
  "memory/max_allocated (GiB)": 33.26,
664
- "ppl": 1.00508,
665
  "step": 47,
666
  "tokens/total": 1416240,
667
- "tokens/train_per_sec_per_gpu": 31.51,
668
  "tokens/trainable": 21441
669
  },
670
  {
671
  "epoch": 0.1875,
672
- "grad_norm": 1.2363684177398682,
673
  "learning_rate": 9.954758120060702e-05,
674
- "loss": 0.03300227224826813,
675
  "memory/device_reserved (GiB)": 35.88,
676
  "memory/max_active (GiB)": 33.96,
677
  "memory/max_allocated (GiB)": 33.96,
678
- "ppl": 1.03355,
679
  "step": 48,
680
  "tokens/total": 1446880,
681
- "tokens/train_per_sec_per_gpu": 33.7,
682
  "tokens/trainable": 21901
683
  },
684
  {
685
  "epoch": 0.19140625,
686
- "grad_norm": 0.7278413772583008,
687
  "learning_rate": 9.950559465600948e-05,
688
- "loss": 0.025975672528147697,
689
  "memory/device_reserved (GiB)": 35.88,
690
  "memory/max_active (GiB)": 33.78,
691
  "memory/max_allocated (GiB)": 33.78,
692
- "ppl": 1.02632,
693
  "step": 49,
694
  "tokens/total": 1477168,
695
- "tokens/train_per_sec_per_gpu": 33.27,
696
  "tokens/trainable": 22341
697
  },
698
  {
699
  "epoch": 0.1953125,
700
- "grad_norm": 0.37237733602523804,
701
  "learning_rate": 9.946175605195379e-05,
702
- "loss": 0.010315775871276855,
703
  "memory/device_reserved (GiB)": 35.88,
704
  "memory/max_active (GiB)": 33.89,
705
  "memory/max_allocated (GiB)": 33.89,
706
- "ppl": 1.01037,
707
  "step": 50,
708
  "tokens/total": 1507712,
709
- "tokens/train_per_sec_per_gpu": 33.07,
710
  "tokens/trainable": 22833
711
  },
712
  {
713
  "epoch": 0.19921875,
714
- "grad_norm": 0.25258293747901917,
715
  "learning_rate": 9.941606721274322e-05,
716
- "loss": 0.00485712755471468,
717
  "memory/device_reserved (GiB)": 35.88,
718
  "memory/max_active (GiB)": 33.81,
719
  "memory/max_allocated (GiB)": 33.81,
720
- "ppl": 1.00487,
721
  "step": 51,
722
  "tokens/total": 1537936,
723
- "tokens/train_per_sec_per_gpu": 30.64,
724
  "tokens/trainable": 23277
725
  },
726
  {
727
  "epoch": 0.203125,
728
- "grad_norm": 0.738599419593811,
729
  "learning_rate": 9.936853003967685e-05,
730
- "loss": 0.01646406576037407,
731
  "memory/device_reserved (GiB)": 35.88,
732
  "memory/max_active (GiB)": 33.84,
733
  "memory/max_allocated (GiB)": 33.84,
734
- "ppl": 1.0166,
735
  "step": 52,
736
  "tokens/total": 1568352,
737
- "tokens/train_per_sec_per_gpu": 35.57,
738
  "tokens/trainable": 23759
739
  },
740
  {
741
  "epoch": 0.20703125,
742
- "grad_norm": 1.2733500003814697,
743
  "learning_rate": 9.93191465109705e-05,
744
- "loss": 0.019196398556232452,
745
  "memory/device_reserved (GiB)": 35.88,
746
  "memory/max_active (GiB)": 33.91,
747
  "memory/max_allocated (GiB)": 33.91,
748
- "ppl": 1.01938,
749
  "step": 53,
750
  "tokens/total": 1598992,
751
- "tokens/train_per_sec_per_gpu": 32.98,
752
  "tokens/trainable": 24193
753
  },
754
  {
755
  "epoch": 0.2109375,
756
- "grad_norm": 0.5316427946090698,
757
  "learning_rate": 9.926791868167438e-05,
758
- "loss": 0.006890955846756697,
759
  "memory/device_reserved (GiB)": 35.9,
760
  "memory/max_active (GiB)": 33.95,
761
  "memory/max_allocated (GiB)": 33.95,
762
- "ppl": 1.00691,
763
  "step": 54,
764
  "tokens/total": 1629488,
765
- "tokens/train_per_sec_per_gpu": 33.47,
766
  "tokens/trainable": 24619
767
  },
768
  {
769
  "epoch": 0.21484375,
770
- "grad_norm": 0.6924111843109131,
771
  "learning_rate": 9.921484868358753e-05,
772
- "loss": 0.021178584545850754,
773
  "memory/device_reserved (GiB)": 35.9,
774
  "memory/max_active (GiB)": 33.8,
775
  "memory/max_allocated (GiB)": 33.8,
776
- "ppl": 1.0214,
777
  "step": 55,
778
  "tokens/total": 1659696,
779
- "tokens/train_per_sec_per_gpu": 32.8,
780
  "tokens/trainable": 25075
781
  },
782
  {
783
  "epoch": 0.21875,
784
- "grad_norm": 0.683601975440979,
785
  "learning_rate": 9.915993872516924e-05,
786
- "loss": 0.017589068040251732,
787
  "memory/device_reserved (GiB)": 35.9,
788
  "memory/max_active (GiB)": 33.76,
789
  "memory/max_allocated (GiB)": 33.76,
790
- "ppl": 1.01774,
791
  "step": 56,
792
  "tokens/total": 1689872,
793
- "tokens/train_per_sec_per_gpu": 31.48,
794
  "tokens/trainable": 25519
795
  },
796
  {
797
  "epoch": 0.22265625,
798
- "grad_norm": 0.8593301177024841,
799
  "learning_rate": 9.9103191091447e-05,
800
- "loss": 0.025561584159731865,
801
  "memory/device_reserved (GiB)": 35.9,
802
  "memory/max_active (GiB)": 33.9,
803
  "memory/max_allocated (GiB)": 33.9,
804
- "ppl": 1.02589,
805
  "step": 57,
806
  "tokens/total": 1720144,
807
- "tokens/train_per_sec_per_gpu": 32.63,
808
  "tokens/trainable": 25966
809
  },
810
  {
811
  "epoch": 0.2265625,
812
- "grad_norm": 0.2925783097743988,
813
  "learning_rate": 9.904460814392147e-05,
814
- "loss": 0.005790311843156815,
815
  "memory/device_reserved (GiB)": 35.9,
816
  "memory/max_active (GiB)": 33.89,
817
  "memory/max_allocated (GiB)": 33.89,
818
- "ppl": 1.00581,
819
  "step": 58,
820
  "tokens/total": 1750448,
821
- "tokens/train_per_sec_per_gpu": 33.89,
822
  "tokens/trainable": 26423
823
  },
824
  {
825
  "epoch": 0.23046875,
826
- "grad_norm": 0.6059477925300598,
827
  "learning_rate": 9.898419232046825e-05,
828
- "loss": 0.007334758993238211,
829
  "memory/device_reserved (GiB)": 35.9,
830
  "memory/max_active (GiB)": 33.93,
831
  "memory/max_allocated (GiB)": 33.93,
832
- "ppl": 1.00736,
833
  "step": 59,
834
  "tokens/total": 1781088,
835
- "tokens/train_per_sec_per_gpu": 38.42,
836
  "tokens/trainable": 26934
837
  },
838
  {
839
  "epoch": 0.234375,
840
- "grad_norm": 0.29425033926963806,
841
  "learning_rate": 9.892194613523633e-05,
842
- "loss": 0.004620288498699665,
843
  "memory/device_reserved (GiB)": 35.9,
844
  "memory/max_active (GiB)": 33.84,
845
  "memory/max_allocated (GiB)": 33.84,
846
- "ppl": 1.00463,
847
  "step": 60,
848
  "tokens/total": 1811344,
849
- "tokens/train_per_sec_per_gpu": 33.0,
850
  "tokens/trainable": 27393
851
  },
852
  {
853
  "epoch": 0.23828125,
854
- "grad_norm": 0.25868093967437744,
855
  "learning_rate": 9.885787217854357e-05,
856
- "loss": 0.0042824773117899895,
857
  "memory/device_reserved (GiB)": 35.96,
858
  "memory/max_active (GiB)": 33.9,
859
  "memory/max_allocated (GiB)": 33.9,
860
- "ppl": 1.00429,
861
  "step": 61,
862
  "tokens/total": 1841600,
863
- "tokens/train_per_sec_per_gpu": 32.84,
864
  "tokens/trainable": 27852
865
  },
866
  {
867
  "epoch": 0.2421875,
868
- "grad_norm": 0.9455181360244751,
869
  "learning_rate": 9.879197311676887e-05,
870
- "loss": 0.013508133590221405,
871
  "memory/device_reserved (GiB)": 35.96,
872
  "memory/max_active (GiB)": 33.82,
873
  "memory/max_allocated (GiB)": 33.82,
874
- "ppl": 1.0136,
875
  "step": 62,
876
  "tokens/total": 1872048,
877
- "tokens/train_per_sec_per_gpu": 32.92,
878
  "tokens/trainable": 28292
879
  },
880
  {
881
  "epoch": 0.24609375,
882
- "grad_norm": 1.149442434310913,
883
  "learning_rate": 9.872425169224113e-05,
884
- "loss": 0.020864056423306465,
885
  "memory/device_reserved (GiB)": 35.96,
886
  "memory/max_active (GiB)": 33.94,
887
  "memory/max_allocated (GiB)": 33.94,
888
- "ppl": 1.02108,
889
  "step": 63,
890
  "tokens/total": 1902592,
891
- "tokens/train_per_sec_per_gpu": 37.27,
892
  "tokens/trainable": 28798
893
  },
894
  {
895
  "epoch": 0.25,
896
- "grad_norm": 0.7421550750732422,
897
  "learning_rate": 9.865471072312528e-05,
898
- "loss": 0.016041038557887077,
899
  "memory/device_reserved (GiB)": 35.96,
900
  "memory/max_active (GiB)": 33.96,
901
  "memory/max_allocated (GiB)": 33.96,
902
- "ppl": 1.01617,
903
  "step": 64,
904
  "tokens/total": 1933088,
905
- "tokens/train_per_sec_per_gpu": 35.76,
906
  "tokens/trainable": 29283
907
  },
908
  {
909
  "epoch": 0.25390625,
910
- "grad_norm": 0.36109936237335205,
911
  "learning_rate": 9.858335310330492e-05,
912
- "loss": 0.005372575484216213,
913
  "memory/device_reserved (GiB)": 35.96,
914
  "memory/max_active (GiB)": 33.72,
915
  "memory/max_allocated (GiB)": 33.72,
916
- "ppl": 1.00539,
917
  "step": 65,
918
  "tokens/total": 1963296,
919
- "tokens/train_per_sec_per_gpu": 31.92,
920
  "tokens/trainable": 29745
921
  },
922
  {
923
  "epoch": 0.2578125,
924
- "grad_norm": 0.7074101567268372,
925
  "learning_rate": 9.851018180226185e-05,
926
- "loss": 0.018492329865694046,
927
- "memory/device_reserved (GiB)": 34.88,
928
  "memory/max_active (GiB)": 33.79,
929
  "memory/max_allocated (GiB)": 33.79,
930
- "ppl": 1.01866,
931
  "step": 66,
932
  "tokens/total": 1993760,
933
  "tokens/train_per_sec_per_gpu": 31.19,
@@ -935,870 +935,870 @@
935
  },
936
  {
937
  "epoch": 0.26171875,
938
- "grad_norm": 0.43113431334495544,
939
  "learning_rate": 9.843519986495259e-05,
940
- "loss": 0.00620780885219574,
941
  "memory/device_reserved (GiB)": 35.86,
942
  "memory/max_active (GiB)": 33.89,
943
  "memory/max_allocated (GiB)": 33.89,
944
- "ppl": 1.00623,
945
  "step": 67,
946
  "tokens/total": 2024144,
947
- "tokens/train_per_sec_per_gpu": 33.22,
948
  "tokens/trainable": 30622
949
  },
950
  {
951
  "epoch": 0.265625,
952
- "grad_norm": 0.3996214270591736,
953
  "learning_rate": 9.835841041168162e-05,
954
- "loss": 0.005429963115602732,
955
  "memory/device_reserved (GiB)": 35.86,
956
  "memory/max_active (GiB)": 33.95,
957
  "memory/max_allocated (GiB)": 33.95,
958
- "ppl": 1.00544,
959
  "step": 68,
960
  "tokens/total": 2054608,
961
- "tokens/train_per_sec_per_gpu": 35.38,
962
  "tokens/trainable": 31097
963
  },
964
  {
965
  "epoch": 0.26953125,
966
- "grad_norm": 0.4444175660610199,
967
  "learning_rate": 9.82798166379715e-05,
968
- "loss": 0.01158289983868599,
969
  "memory/device_reserved (GiB)": 35.86,
970
  "memory/max_active (GiB)": 33.82,
971
  "memory/max_allocated (GiB)": 33.82,
972
- "ppl": 1.01165,
973
  "step": 69,
974
  "tokens/total": 2084912,
975
- "tokens/train_per_sec_per_gpu": 36.53,
976
  "tokens/trainable": 31595
977
  },
978
  {
979
  "epoch": 0.2734375,
980
- "grad_norm": 0.16977320611476898,
981
  "learning_rate": 9.819942181443002e-05,
982
- "loss": 0.002158569637686014,
983
  "memory/device_reserved (GiB)": 35.86,
984
  "memory/max_active (GiB)": 33.84,
985
  "memory/max_allocated (GiB)": 33.84,
986
- "ppl": 1.00216,
987
  "step": 70,
988
  "tokens/total": 2115216,
989
- "tokens/train_per_sec_per_gpu": 30.67,
990
  "tokens/trainable": 32036
991
  },
992
  {
993
  "epoch": 0.27734375,
994
- "grad_norm": 0.12970401346683502,
995
  "learning_rate": 9.811722928661392e-05,
996
- "loss": 0.0028989531565457582,
997
  "memory/device_reserved (GiB)": 35.86,
998
  "memory/max_active (GiB)": 33.9,
999
  "memory/max_allocated (GiB)": 33.9,
1000
- "ppl": 1.0029,
1001
  "step": 71,
1002
  "tokens/total": 2145424,
1003
- "tokens/train_per_sec_per_gpu": 28.06,
1004
  "tokens/trainable": 32428
1005
  },
1006
  {
1007
  "epoch": 0.28125,
1008
- "grad_norm": 0.06490105390548706,
1009
  "learning_rate": 9.803324247488975e-05,
1010
- "loss": 0.0008802044321782887,
1011
  "memory/device_reserved (GiB)": 35.88,
1012
  "memory/max_active (GiB)": 33.84,
1013
  "memory/max_allocated (GiB)": 33.84,
1014
- "ppl": 1.00088,
1015
  "step": 72,
1016
  "tokens/total": 2175648,
1017
- "tokens/train_per_sec_per_gpu": 32.08,
1018
  "tokens/trainable": 32880
1019
  },
1020
  {
1021
  "epoch": 0.28515625,
1022
- "grad_norm": 0.20680083334445953,
1023
  "learning_rate": 9.794746487429161e-05,
1024
- "loss": 0.0019504247466102242,
1025
  "memory/device_reserved (GiB)": 35.88,
1026
  "memory/max_active (GiB)": 33.79,
1027
  "memory/max_allocated (GiB)": 33.79,
1028
- "ppl": 1.00195,
1029
  "step": 73,
1030
  "tokens/total": 2205856,
1031
- "tokens/train_per_sec_per_gpu": 33.48,
1032
  "tokens/trainable": 33323
1033
  },
1034
  {
1035
  "epoch": 0.2890625,
1036
- "grad_norm": 1.6840267181396484,
1037
  "learning_rate": 9.785990005437554e-05,
1038
- "loss": 0.02435958757996559,
1039
  "memory/device_reserved (GiB)": 35.88,
1040
  "memory/max_active (GiB)": 33.95,
1041
  "memory/max_allocated (GiB)": 33.95,
1042
- "ppl": 1.02466,
1043
  "step": 74,
1044
  "tokens/total": 2236352,
1045
- "tokens/train_per_sec_per_gpu": 35.84,
1046
  "tokens/trainable": 33792
1047
  },
1048
  {
1049
  "epoch": 0.29296875,
1050
- "grad_norm": 0.6665006875991821,
1051
  "learning_rate": 9.777055165907117e-05,
1052
- "loss": 0.018271014094352722,
1053
  "memory/device_reserved (GiB)": 35.88,
1054
  "memory/max_active (GiB)": 33.8,
1055
  "memory/max_allocated (GiB)": 33.8,
1056
- "ppl": 1.01844,
1057
  "step": 75,
1058
  "tokens/total": 2266480,
1059
- "tokens/train_per_sec_per_gpu": 35.24,
1060
  "tokens/trainable": 34223
1061
  },
1062
  {
1063
  "epoch": 0.296875,
1064
- "grad_norm": 0.6469210982322693,
1065
  "learning_rate": 9.767942340652993e-05,
1066
- "loss": 0.023723380640149117,
1067
  "memory/device_reserved (GiB)": 35.88,
1068
  "memory/max_active (GiB)": 33.71,
1069
  "memory/max_allocated (GiB)": 33.71,
1070
- "ppl": 1.02401,
1071
  "step": 76,
1072
  "tokens/total": 2296496,
1073
- "tokens/train_per_sec_per_gpu": 29.59,
1074
  "tokens/trainable": 34649
1075
  },
1076
  {
1077
  "epoch": 0.30078125,
1078
- "grad_norm": 1.391882300376892,
1079
  "learning_rate": 9.758651908897035e-05,
1080
- "loss": 0.015201358124613762,
1081
  "memory/device_reserved (GiB)": 35.88,
1082
  "memory/max_active (GiB)": 33.88,
1083
  "memory/max_allocated (GiB)": 33.88,
1084
- "ppl": 1.01532,
1085
  "step": 77,
1086
  "tokens/total": 2327040,
1087
- "tokens/train_per_sec_per_gpu": 35.58,
1088
  "tokens/trainable": 35094
1089
  },
1090
  {
1091
  "epoch": 0.3046875,
1092
- "grad_norm": 0.5752553343772888,
1093
  "learning_rate": 9.749184257252033e-05,
1094
- "loss": 0.020247019827365875,
1095
  "memory/device_reserved (GiB)": 35.88,
1096
  "memory/max_active (GiB)": 33.81,
1097
  "memory/max_allocated (GiB)": 33.81,
1098
- "ppl": 1.02045,
1099
  "step": 78,
1100
  "tokens/total": 2357328,
1101
- "tokens/train_per_sec_per_gpu": 31.79,
1102
  "tokens/trainable": 35534
1103
  },
1104
  {
1105
  "epoch": 0.30859375,
1106
- "grad_norm": 0.276022732257843,
1107
  "learning_rate": 9.739539779705614e-05,
1108
- "loss": 0.010471204295754433,
1109
- "memory/device_reserved (GiB)": 35.88,
1110
  "memory/max_active (GiB)": 33.85,
1111
  "memory/max_allocated (GiB)": 33.85,
1112
- "ppl": 1.01053,
1113
  "step": 79,
1114
  "tokens/total": 2387664,
1115
- "tokens/train_per_sec_per_gpu": 35.76,
1116
  "tokens/trainable": 36029
1117
  },
1118
  {
1119
  "epoch": 0.3125,
1120
- "grad_norm": 0.41784003376960754,
1121
  "learning_rate": 9.729718877603861e-05,
1122
- "loss": 0.010774495080113411,
1123
- "memory/device_reserved (GiB)": 35.88,
1124
  "memory/max_active (GiB)": 33.96,
1125
  "memory/max_allocated (GiB)": 33.96,
1126
- "ppl": 1.01083,
1127
  "step": 80,
1128
  "tokens/total": 2418000,
1129
- "tokens/train_per_sec_per_gpu": 35.55,
1130
  "tokens/trainable": 36469
1131
  },
1132
  {
1133
  "epoch": 0.31640625,
1134
- "grad_norm": 0.4906325340270996,
1135
  "learning_rate": 9.719721959634592e-05,
1136
- "loss": 0.015422273427248001,
1137
- "memory/device_reserved (GiB)": 35.88,
1138
  "memory/max_active (GiB)": 33.96,
1139
  "memory/max_allocated (GiB)": 33.96,
1140
- "ppl": 1.01554,
1141
  "step": 81,
1142
  "tokens/total": 2448720,
1143
- "tokens/train_per_sec_per_gpu": 30.69,
1144
  "tokens/trainable": 36906
1145
  },
1146
  {
1147
  "epoch": 0.3203125,
1148
- "grad_norm": 0.2813898026943207,
1149
  "learning_rate": 9.709549441810375e-05,
1150
- "loss": 0.009146124124526978,
1151
- "memory/device_reserved (GiB)": 35.88,
1152
  "memory/max_active (GiB)": 33.92,
1153
  "memory/max_allocated (GiB)": 33.92,
1154
- "ppl": 1.00919,
1155
  "step": 82,
1156
  "tokens/total": 2479248,
1157
- "tokens/train_per_sec_per_gpu": 38.89,
1158
  "tokens/trainable": 37390
1159
  },
1160
  {
1161
  "epoch": 0.32421875,
1162
- "grad_norm": 0.39496278762817383,
1163
  "learning_rate": 9.699201747451195e-05,
1164
- "loss": 0.008894146420061588,
1165
- "memory/device_reserved (GiB)": 35.88,
1166
  "memory/max_active (GiB)": 33.78,
1167
  "memory/max_allocated (GiB)": 33.78,
1168
- "ppl": 1.00893,
1169
  "step": 83,
1170
  "tokens/total": 2509408,
1171
- "tokens/train_per_sec_per_gpu": 30.78,
1172
  "tokens/trainable": 37838
1173
  },
1174
  {
1175
  "epoch": 0.328125,
1176
- "grad_norm": 0.4946168065071106,
1177
  "learning_rate": 9.688679307166854e-05,
1178
- "loss": 0.005293373484164476,
1179
- "memory/device_reserved (GiB)": 35.88,
1180
  "memory/max_active (GiB)": 33.97,
1181
  "memory/max_allocated (GiB)": 33.97,
1182
- "ppl": 1.00531,
1183
  "step": 84,
1184
  "tokens/total": 2539776,
1185
- "tokens/train_per_sec_per_gpu": 34.98,
1186
  "tokens/trainable": 38310
1187
  },
1188
  {
1189
  "epoch": 0.33203125,
1190
- "grad_norm": 1.3293001651763916,
1191
  "learning_rate": 9.677982558839042e-05,
1192
- "loss": 0.040361277759075165,
1193
- "memory/device_reserved (GiB)": 35.88,
1194
  "memory/max_active (GiB)": 33.75,
1195
  "memory/max_allocated (GiB)": 33.75,
1196
- "ppl": 1.04119,
1197
  "step": 85,
1198
  "tokens/total": 2569840,
1199
- "tokens/train_per_sec_per_gpu": 34.0,
1200
  "tokens/trainable": 38789
1201
  },
1202
  {
1203
  "epoch": 0.3359375,
1204
- "grad_norm": 0.5834341049194336,
1205
  "learning_rate": 9.66711194760312e-05,
1206
- "loss": 0.010128681547939777,
1207
- "memory/device_reserved (GiB)": 35.88,
1208
  "memory/max_active (GiB)": 33.86,
1209
  "memory/max_allocated (GiB)": 33.86,
1210
- "ppl": 1.01018,
1211
  "step": 86,
1212
  "tokens/total": 2600192,
1213
- "tokens/train_per_sec_per_gpu": 36.34,
1214
  "tokens/trainable": 39277
1215
  },
1216
  {
1217
  "epoch": 0.33984375,
1218
- "grad_norm": 0.7191532254219055,
1219
  "learning_rate": 9.656067925829593e-05,
1220
- "loss": 0.02042745053768158,
1221
- "memory/device_reserved (GiB)": 35.88,
1222
  "memory/max_active (GiB)": 33.93,
1223
  "memory/max_allocated (GiB)": 33.93,
1224
- "ppl": 1.02064,
1225
  "step": 87,
1226
  "tokens/total": 2630640,
1227
- "tokens/train_per_sec_per_gpu": 35.6,
1228
  "tokens/trainable": 39737
1229
  },
1230
  {
1231
  "epoch": 0.34375,
1232
- "grad_norm": 0.3419296145439148,
1233
  "learning_rate": 9.644850953105288e-05,
1234
- "loss": 0.007800046354532242,
1235
- "memory/device_reserved (GiB)": 35.88,
1236
  "memory/max_active (GiB)": 33.74,
1237
  "memory/max_allocated (GiB)": 33.74,
1238
- "ppl": 1.00783,
1239
  "step": 88,
1240
  "tokens/total": 2660832,
1241
- "tokens/train_per_sec_per_gpu": 29.53,
1242
  "tokens/trainable": 40179
1243
  },
1244
  {
1245
  "epoch": 0.34765625,
1246
- "grad_norm": 0.23275785148143768,
1247
  "learning_rate": 9.633461496214225e-05,
1248
- "loss": 0.005660225171595812,
1249
- "memory/device_reserved (GiB)": 35.88,
1250
  "memory/max_active (GiB)": 33.88,
1251
  "memory/max_allocated (GiB)": 33.88,
1252
- "ppl": 1.00568,
1253
  "step": 89,
1254
  "tokens/total": 2691328,
1255
- "tokens/train_per_sec_per_gpu": 32.59,
1256
  "tokens/trainable": 40649
1257
  },
1258
  {
1259
  "epoch": 0.3515625,
1260
- "grad_norm": 0.6987941265106201,
1261
  "learning_rate": 9.621900029118195e-05,
1262
- "loss": 0.02007928490638733,
1263
- "memory/device_reserved (GiB)": 35.88,
1264
  "memory/max_active (GiB)": 33.41,
1265
  "memory/max_allocated (GiB)": 33.41,
1266
- "ppl": 1.02028,
1267
  "step": 90,
1268
  "tokens/total": 2719696,
1269
- "tokens/train_per_sec_per_gpu": 31.52,
1270
  "tokens/trainable": 41080
1271
  },
1272
  {
1273
  "epoch": 0.35546875,
1274
- "grad_norm": 0.11328475177288055,
1275
  "learning_rate": 9.610167032937036e-05,
1276
- "loss": 0.002234571846202016,
1277
- "memory/device_reserved (GiB)": 35.88,
1278
  "memory/max_active (GiB)": 33.94,
1279
  "memory/max_allocated (GiB)": 33.94,
1280
- "ppl": 1.00224,
1281
  "step": 91,
1282
  "tokens/total": 2750272,
1283
- "tokens/train_per_sec_per_gpu": 37.99,
1284
  "tokens/trainable": 41599
1285
  },
1286
  {
1287
  "epoch": 0.359375,
1288
- "grad_norm": 0.4347783029079437,
1289
  "learning_rate": 9.598262995928611e-05,
1290
- "loss": 0.004549161531031132,
1291
- "memory/device_reserved (GiB)": 35.88,
1292
  "memory/max_active (GiB)": 33.92,
1293
  "memory/max_allocated (GiB)": 33.92,
1294
- "ppl": 1.00456,
1295
  "step": 92,
1296
  "tokens/total": 2780656,
1297
- "tokens/train_per_sec_per_gpu": 36.77,
1298
  "tokens/trainable": 42076
1299
  },
1300
  {
1301
  "epoch": 0.36328125,
1302
- "grad_norm": 0.3486956059932709,
1303
  "learning_rate": 9.586188413468492e-05,
1304
- "loss": 0.012267105281352997,
1305
- "memory/device_reserved (GiB)": 35.88,
1306
  "memory/max_active (GiB)": 33.82,
1307
  "memory/max_allocated (GiB)": 33.82,
1308
- "ppl": 1.01234,
1309
  "step": 93,
1310
  "tokens/total": 2811088,
1311
- "tokens/train_per_sec_per_gpu": 33.6,
1312
  "tokens/trainable": 42522
1313
  },
1314
  {
1315
  "epoch": 0.3671875,
1316
- "grad_norm": 0.5156503319740295,
1317
  "learning_rate": 9.57394378802934e-05,
1318
- "loss": 0.015529165044426918,
1319
  "memory/device_reserved (GiB)": 36.1,
1320
  "memory/max_active (GiB)": 33.9,
1321
  "memory/max_allocated (GiB)": 33.9,
1322
- "ppl": 1.01565,
1323
  "step": 94,
1324
  "tokens/total": 2841520,
1325
- "tokens/train_per_sec_per_gpu": 34.32,
1326
  "tokens/trainable": 42974
1327
  },
1328
  {
1329
  "epoch": 0.37109375,
1330
- "grad_norm": 0.37648338079452515,
1331
  "learning_rate": 9.56152962916e-05,
1332
- "loss": 0.011707151308655739,
1333
  "memory/device_reserved (GiB)": 36.1,
1334
  "memory/max_active (GiB)": 33.7,
1335
  "memory/max_allocated (GiB)": 33.7,
1336
- "ppl": 1.01178,
1337
  "step": 95,
1338
  "tokens/total": 2871600,
1339
- "tokens/train_per_sec_per_gpu": 30.71,
1340
  "tokens/trainable": 43380
1341
  },
1342
  {
1343
  "epoch": 0.375,
1344
- "grad_norm": 0.38365671038627625,
1345
  "learning_rate": 9.548946453464296e-05,
1346
- "loss": 0.008411398157477379,
1347
  "memory/device_reserved (GiB)": 36.1,
1348
  "memory/max_active (GiB)": 33.94,
1349
  "memory/max_allocated (GiB)": 33.94,
1350
- "ppl": 1.00845,
1351
  "step": 96,
1352
  "tokens/total": 2902144,
1353
- "tokens/train_per_sec_per_gpu": 34.13,
1354
  "tokens/trainable": 43865
1355
  },
1356
  {
1357
  "epoch": 0.37890625,
1358
- "grad_norm": 0.313085675239563,
1359
  "learning_rate": 9.53619478457953e-05,
1360
- "loss": 0.007852522656321526,
1361
  "memory/device_reserved (GiB)": 36.1,
1362
  "memory/max_active (GiB)": 33.82,
1363
  "memory/max_allocated (GiB)": 33.82,
1364
- "ppl": 1.00788,
1365
  "step": 97,
1366
  "tokens/total": 2932272,
1367
- "tokens/train_per_sec_per_gpu": 31.89,
1368
  "tokens/trainable": 44328
1369
  },
1370
  {
1371
  "epoch": 0.3828125,
1372
- "grad_norm": 0.08544483780860901,
1373
  "learning_rate": 9.523275153154695e-05,
1374
- "loss": 0.0025753644295036793,
1375
  "memory/device_reserved (GiB)": 35.64,
1376
  "memory/max_active (GiB)": 33.69,
1377
  "memory/max_allocated (GiB)": 33.69,
1378
- "ppl": 1.00258,
1379
  "step": 98,
1380
  "tokens/total": 2962032,
1381
- "tokens/train_per_sec_per_gpu": 30.98,
1382
  "tokens/trainable": 44778
1383
  },
1384
  {
1385
  "epoch": 0.38671875,
1386
- "grad_norm": 0.6496388912200928,
1387
  "learning_rate": 9.51018809682839e-05,
1388
- "loss": 0.02547256276011467,
1389
  "memory/device_reserved (GiB)": 35.64,
1390
  "memory/max_active (GiB)": 33.89,
1391
  "memory/max_allocated (GiB)": 33.89,
1392
- "ppl": 1.0258,
1393
  "step": 99,
1394
  "tokens/total": 2992256,
1395
- "tokens/train_per_sec_per_gpu": 37.11,
1396
  "tokens/trainable": 45225
1397
  },
1398
  {
1399
  "epoch": 0.390625,
1400
- "grad_norm": 0.15325438976287842,
1401
  "learning_rate": 9.49693416020645e-05,
1402
- "loss": 0.003552855923771858,
1403
  "memory/device_reserved (GiB)": 35.64,
1404
  "memory/max_active (GiB)": 33.8,
1405
  "memory/max_allocated (GiB)": 33.8,
1406
- "ppl": 1.00356,
1407
  "step": 100,
1408
  "tokens/total": 3022608,
1409
- "tokens/train_per_sec_per_gpu": 35.8,
1410
  "tokens/trainable": 45678
1411
  },
1412
  {
1413
  "epoch": 0.39453125,
1414
- "grad_norm": 0.25307565927505493,
1415
  "learning_rate": 9.483513894839276e-05,
1416
- "loss": 0.004095804411917925,
1417
  "memory/device_reserved (GiB)": 35.78,
1418
  "memory/max_active (GiB)": 33.88,
1419
  "memory/max_allocated (GiB)": 33.88,
1420
- "ppl": 1.0041,
1421
  "step": 101,
1422
  "tokens/total": 3052992,
1423
- "tokens/train_per_sec_per_gpu": 32.73,
1424
  "tokens/trainable": 46125
1425
  },
1426
  {
1427
  "epoch": 0.3984375,
1428
- "grad_norm": 0.150072380900383,
1429
  "learning_rate": 9.469927859198888e-05,
1430
- "loss": 0.0013888315297663212,
1431
- "memory/device_reserved (GiB)": 35.79,
1432
  "memory/max_active (GiB)": 33.98,
1433
  "memory/max_allocated (GiB)": 33.98,
1434
- "ppl": 1.00139,
1435
  "step": 102,
1436
  "tokens/total": 3083392,
1437
- "tokens/train_per_sec_per_gpu": 39.71,
1438
  "tokens/trainable": 46612
1439
  },
1440
  {
1441
  "epoch": 0.40234375,
1442
- "grad_norm": 0.5769571661949158,
1443
  "learning_rate": 9.456176618655689e-05,
1444
- "loss": 0.022533349692821503,
1445
- "memory/device_reserved (GiB)": 35.79,
1446
  "memory/max_active (GiB)": 33.83,
1447
  "memory/max_allocated (GiB)": 33.83,
1448
- "ppl": 1.02279,
1449
  "step": 103,
1450
  "tokens/total": 3113744,
1451
- "tokens/train_per_sec_per_gpu": 32.57,
1452
  "tokens/trainable": 47059
1453
  },
1454
  {
1455
  "epoch": 0.40625,
1456
- "grad_norm": 0.5657027959823608,
1457
  "learning_rate": 9.442260745454927e-05,
1458
- "loss": 0.016894506290555,
1459
- "memory/device_reserved (GiB)": 35.79,
1460
  "memory/max_active (GiB)": 34.02,
1461
  "memory/max_allocated (GiB)": 34.02,
1462
- "ppl": 1.01704,
1463
  "step": 104,
1464
  "tokens/total": 3144272,
1465
- "tokens/train_per_sec_per_gpu": 34.05,
1466
  "tokens/trainable": 47536
1467
  },
1468
  {
1469
  "epoch": 0.41015625,
1470
- "grad_norm": 0.29607170820236206,
1471
  "learning_rate": 9.428180818692884e-05,
1472
- "loss": 0.017974723130464554,
1473
- "memory/device_reserved (GiB)": 37.61,
1474
  "memory/max_active (GiB)": 33.81,
1475
  "memory/max_allocated (GiB)": 33.81,
1476
- "ppl": 1.01814,
1477
  "step": 105,
1478
  "tokens/total": 3174512,
1479
- "tokens/train_per_sec_per_gpu": 34.11,
1480
  "tokens/trainable": 48037
1481
  },
1482
  {
1483
  "epoch": 0.4140625,
1484
- "grad_norm": 0.5241922736167908,
1485
  "learning_rate": 9.413937424292791e-05,
1486
- "loss": 0.016189826652407646,
1487
- "memory/device_reserved (GiB)": 37.61,
1488
  "memory/max_active (GiB)": 33.91,
1489
  "memory/max_allocated (GiB)": 33.91,
1490
- "ppl": 1.01632,
1491
  "step": 106,
1492
  "tokens/total": 3204896,
1493
- "tokens/train_per_sec_per_gpu": 32.65,
1494
  "tokens/trainable": 48491
1495
  },
1496
  {
1497
  "epoch": 0.41796875,
1498
- "grad_norm": 0.3886408805847168,
1499
  "learning_rate": 9.399531154980424e-05,
1500
- "loss": 0.010908558964729309,
1501
- "memory/device_reserved (GiB)": 37.61,
1502
  "memory/max_active (GiB)": 34.01,
1503
  "memory/max_allocated (GiB)": 34.01,
1504
- "ppl": 1.01097,
1505
  "step": 107,
1506
  "tokens/total": 3235360,
1507
- "tokens/train_per_sec_per_gpu": 33.0,
1508
  "tokens/trainable": 48940
1509
  },
1510
  {
1511
  "epoch": 0.421875,
1512
- "grad_norm": 0.2442784607410431,
1513
  "learning_rate": 9.384962610259455e-05,
1514
- "loss": 0.008070861920714378,
1515
- "memory/device_reserved (GiB)": 37.61,
1516
  "memory/max_active (GiB)": 33.83,
1517
  "memory/max_allocated (GiB)": 33.83,
1518
- "ppl": 1.0081,
1519
  "step": 108,
1520
  "tokens/total": 3265552,
1521
- "tokens/train_per_sec_per_gpu": 31.79,
1522
  "tokens/trainable": 49389
1523
  },
1524
  {
1525
  "epoch": 0.42578125,
1526
- "grad_norm": 0.1457175612449646,
1527
  "learning_rate": 9.370232396386494e-05,
1528
- "loss": 0.003785747569054365,
1529
- "memory/device_reserved (GiB)": 37.61,
1530
  "memory/max_active (GiB)": 33.39,
1531
  "memory/max_allocated (GiB)": 33.39,
1532
- "ppl": 1.00379,
1533
  "step": 109,
1534
  "tokens/total": 3293856,
1535
- "tokens/train_per_sec_per_gpu": 36.01,
1536
  "tokens/trainable": 49838
1537
  },
1538
  {
1539
  "epoch": 0.4296875,
1540
- "grad_norm": 0.3955559730529785,
1541
  "learning_rate": 9.355341126345868e-05,
1542
- "loss": 0.0069918157532811165,
1543
- "memory/device_reserved (GiB)": 37.61,
1544
  "memory/max_active (GiB)": 33.78,
1545
  "memory/max_allocated (GiB)": 33.78,
1546
- "ppl": 1.00702,
1547
  "step": 110,
1548
  "tokens/total": 3323984,
1549
- "tokens/train_per_sec_per_gpu": 34.06,
1550
  "tokens/trainable": 50310
1551
  },
1552
  {
1553
  "epoch": 0.43359375,
1554
- "grad_norm": 0.6224751472473145,
1555
  "learning_rate": 9.340289419824107e-05,
1556
- "loss": 0.014852076768875122,
1557
  "memory/device_reserved (GiB)": 37.62,
1558
  "memory/max_active (GiB)": 33.84,
1559
  "memory/max_allocated (GiB)": 33.84,
1560
- "ppl": 1.01496,
1561
  "step": 111,
1562
  "tokens/total": 3354320,
1563
- "tokens/train_per_sec_per_gpu": 33.67,
1564
  "tokens/trainable": 50755
1565
  },
1566
  {
1567
  "epoch": 0.4375,
1568
- "grad_norm": 0.3700723648071289,
1569
  "learning_rate": 9.325077903184159e-05,
1570
- "loss": 0.00436755083501339,
1571
  "memory/device_reserved (GiB)": 37.62,
1572
  "memory/max_active (GiB)": 33.9,
1573
  "memory/max_allocated (GiB)": 33.9,
1574
- "ppl": 1.00438,
1575
  "step": 112,
1576
  "tokens/total": 3384880,
1577
- "tokens/train_per_sec_per_gpu": 31.65,
1578
  "tokens/trainable": 51213
1579
  },
1580
  {
1581
  "epoch": 0.44140625,
1582
- "grad_norm": 0.1353236883878708,
1583
  "learning_rate": 9.30970720943932e-05,
1584
- "loss": 0.0025976570323109627,
1585
  "memory/device_reserved (GiB)": 37.62,
1586
  "memory/max_active (GiB)": 33.89,
1587
  "memory/max_allocated (GiB)": 33.89,
1588
- "ppl": 1.0026,
1589
  "step": 113,
1590
  "tokens/total": 3415104,
1591
- "tokens/train_per_sec_per_gpu": 33.83,
1592
  "tokens/trainable": 51660
1593
  },
1594
  {
1595
  "epoch": 0.4453125,
1596
- "grad_norm": 0.18984447419643402,
1597
  "learning_rate": 9.2941779782269e-05,
1598
- "loss": 0.002497302368283272,
1599
  "memory/device_reserved (GiB)": 37.62,
1600
  "memory/max_active (GiB)": 33.84,
1601
  "memory/max_allocated (GiB)": 33.84,
1602
- "ppl": 1.0025,
1603
  "step": 114,
1604
  "tokens/total": 3445504,
1605
- "tokens/train_per_sec_per_gpu": 32.43,
1606
  "tokens/trainable": 52133
1607
  },
1608
  {
1609
  "epoch": 0.44921875,
1610
- "grad_norm": 1.1815483570098877,
1611
  "learning_rate": 9.278490855781596e-05,
1612
- "loss": 0.029489168897271156,
1613
  "memory/device_reserved (GiB)": 37.62,
1614
  "memory/max_active (GiB)": 33.84,
1615
  "memory/max_allocated (GiB)": 33.84,
1616
- "ppl": 1.02993,
1617
  "step": 115,
1618
  "tokens/total": 3475744,
1619
- "tokens/train_per_sec_per_gpu": 37.47,
1620
  "tokens/trainable": 52580
1621
  },
1622
  {
1623
  "epoch": 0.453125,
1624
- "grad_norm": 0.44360846281051636,
1625
  "learning_rate": 9.262646494908604e-05,
1626
- "loss": 0.009585533291101456,
1627
  "memory/device_reserved (GiB)": 37.62,
1628
  "memory/max_active (GiB)": 33.82,
1629
  "memory/max_allocated (GiB)": 33.82,
1630
- "ppl": 1.00963,
1631
  "step": 116,
1632
  "tokens/total": 3506016,
1633
- "tokens/train_per_sec_per_gpu": 34.83,
1634
  "tokens/trainable": 53033
1635
  },
1636
  {
1637
  "epoch": 0.45703125,
1638
- "grad_norm": 0.5074551701545715,
1639
  "learning_rate": 9.246645554956457e-05,
1640
- "loss": 0.020201388746500015,
1641
  "memory/device_reserved (GiB)": 37.62,
1642
  "memory/max_active (GiB)": 33.83,
1643
  "memory/max_allocated (GiB)": 33.83,
1644
- "ppl": 1.02041,
1645
  "step": 117,
1646
  "tokens/total": 3536256,
1647
- "tokens/train_per_sec_per_gpu": 35.27,
1648
  "tokens/trainable": 53517
1649
  },
1650
  {
1651
  "epoch": 0.4609375,
1652
- "grad_norm": 0.3565296232700348,
1653
  "learning_rate": 9.230488701789578e-05,
1654
- "loss": 0.005688562989234924,
1655
  "memory/device_reserved (GiB)": 37.62,
1656
  "memory/max_active (GiB)": 33.73,
1657
  "memory/max_allocated (GiB)": 33.73,
1658
- "ppl": 1.0057,
1659
  "step": 118,
1660
  "tokens/total": 3566480,
1661
- "tokens/train_per_sec_per_gpu": 34.56,
1662
  "tokens/trainable": 53967
1663
  },
1664
  {
1665
  "epoch": 0.46484375,
1666
- "grad_norm": 0.16547706723213196,
1667
  "learning_rate": 9.214176607760577e-05,
1668
- "loss": 0.003188834059983492,
1669
  "memory/device_reserved (GiB)": 37.62,
1670
  "memory/max_active (GiB)": 33.85,
1671
  "memory/max_allocated (GiB)": 33.85,
1672
- "ppl": 1.00319,
1673
  "step": 119,
1674
  "tokens/total": 3596624,
1675
- "tokens/train_per_sec_per_gpu": 35.27,
1676
  "tokens/trainable": 54430
1677
  },
1678
  {
1679
  "epoch": 0.46875,
1680
- "grad_norm": 0.20722293853759766,
1681
  "learning_rate": 9.197709951682268e-05,
1682
- "loss": 0.003235103562474251,
1683
  "memory/device_reserved (GiB)": 37.62,
1684
  "memory/max_active (GiB)": 33.93,
1685
  "memory/max_allocated (GiB)": 33.93,
1686
- "ppl": 1.00324,
1687
  "step": 120,
1688
  "tokens/total": 3627168,
1689
- "tokens/train_per_sec_per_gpu": 35.21,
1690
  "tokens/trainable": 54896
1691
  },
1692
  {
1693
  "epoch": 0.47265625,
1694
- "grad_norm": 0.4754939377307892,
1695
  "learning_rate": 9.181089418799428e-05,
1696
- "loss": 0.005670893006026745,
1697
  "memory/device_reserved (GiB)": 37.62,
1698
  "memory/max_active (GiB)": 33.96,
1699
  "memory/max_allocated (GiB)": 33.96,
1700
- "ppl": 1.00569,
1701
  "step": 121,
1702
  "tokens/total": 3657632,
1703
- "tokens/train_per_sec_per_gpu": 37.77,
1704
  "tokens/trainable": 55379
1705
  },
1706
  {
1707
  "epoch": 0.4765625,
1708
- "grad_norm": 0.08304762095212936,
1709
  "learning_rate": 9.164315700760271e-05,
1710
- "loss": 0.00099027412943542,
1711
  "memory/device_reserved (GiB)": 37.62,
1712
  "memory/max_active (GiB)": 33.75,
1713
  "memory/max_allocated (GiB)": 33.75,
1714
- "ppl": 1.00099,
1715
  "step": 122,
1716
  "tokens/total": 3687824,
1717
- "tokens/train_per_sec_per_gpu": 31.7,
1718
  "tokens/trainable": 55803
1719
  },
1720
  {
1721
  "epoch": 0.48046875,
1722
- "grad_norm": 0.725281298160553,
1723
  "learning_rate": 9.147389495587671e-05,
1724
- "loss": 0.007413266692310572,
1725
  "memory/device_reserved (GiB)": 37.62,
1726
  "memory/max_active (GiB)": 33.87,
1727
  "memory/max_allocated (GiB)": 33.87,
1728
- "ppl": 1.00744,
1729
  "step": 123,
1730
  "tokens/total": 3718320,
1731
- "tokens/train_per_sec_per_gpu": 36.45,
1732
  "tokens/trainable": 56249
1733
  },
1734
  {
1735
  "epoch": 0.484375,
1736
- "grad_norm": 0.31409645080566406,
1737
  "learning_rate": 9.130311507650116e-05,
1738
- "loss": 0.0029702766332775354,
1739
  "memory/device_reserved (GiB)": 37.62,
1740
  "memory/max_active (GiB)": 33.9,
1741
  "memory/max_allocated (GiB)": 33.9,
1742
- "ppl": 1.00297,
1743
  "step": 124,
1744
  "tokens/total": 3748672,
1745
- "tokens/train_per_sec_per_gpu": 32.41,
1746
  "tokens/trainable": 56713
1747
  },
1748
  {
1749
  "epoch": 0.48828125,
1750
- "grad_norm": 0.6082578897476196,
1751
  "learning_rate": 9.113082447632394e-05,
1752
- "loss": 0.003795543685555458,
1753
  "memory/device_reserved (GiB)": 37.76,
1754
  "memory/max_active (GiB)": 33.85,
1755
  "memory/max_allocated (GiB)": 33.85,
1756
- "ppl": 1.0038,
1757
  "step": 125,
1758
  "tokens/total": 3778976,
1759
- "tokens/train_per_sec_per_gpu": 39.08,
1760
  "tokens/trainable": 57196
1761
  },
1762
  {
1763
  "epoch": 0.4921875,
1764
- "grad_norm": 0.0603976845741272,
1765
  "learning_rate": 9.09570303250602e-05,
1766
- "loss": 0.0014751165872439742,
1767
  "memory/device_reserved (GiB)": 37.76,
1768
  "memory/max_active (GiB)": 33.89,
1769
  "memory/max_allocated (GiB)": 33.89,
1770
- "ppl": 1.00148,
1771
  "step": 126,
1772
  "tokens/total": 3809296,
1773
- "tokens/train_per_sec_per_gpu": 36.49,
1774
  "tokens/trainable": 57680
1775
  },
1776
  {
1777
  "epoch": 0.49609375,
1778
- "grad_norm": 0.21112751960754395,
1779
  "learning_rate": 9.078173985499394e-05,
1780
- "loss": 0.004137102514505386,
1781
  "memory/device_reserved (GiB)": 37.76,
1782
  "memory/max_active (GiB)": 33.77,
1783
  "memory/max_allocated (GiB)": 33.77,
1784
- "ppl": 1.00415,
1785
  "step": 127,
1786
  "tokens/total": 3839328,
1787
- "tokens/train_per_sec_per_gpu": 31.8,
1788
  "tokens/trainable": 58110
1789
  },
1790
  {
1791
  "epoch": 0.5,
1792
- "grad_norm": 0.3234494924545288,
1793
  "learning_rate": 9.060496036067713e-05,
1794
- "loss": 0.007372056134045124,
1795
  "memory/device_reserved (GiB)": 37.76,
1796
  "memory/max_active (GiB)": 33.88,
1797
  "memory/max_allocated (GiB)": 33.88,
1798
- "ppl": 1.0074,
1799
  "step": 128,
1800
  "tokens/total": 3869824,
1801
- "tokens/train_per_sec_per_gpu": 31.64,
1802
  "tokens/trainable": 58534
1803
  }
1804
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.00390625,
14
+ "grad_norm": 0.8591235280036926,
15
  "learning_rate": 0.0,
16
  "loss": 0.10251401364803314,
17
  "memory/device_reserved (GiB)": 34.94,
 
20
  "ppl": 1.10795,
21
  "step": 1,
22
  "tokens/total": 30464,
23
+ "tokens/train_per_sec_per_gpu": 30.01,
24
  "tokens/trainable": 470
25
  },
26
  {
27
  "epoch": 0.0078125,
28
+ "grad_norm": 0.8033757209777832,
29
  "learning_rate": 4.000000000000001e-06,
30
  "loss": 0.09678442776203156,
31
  "memory/device_reserved (GiB)": 35.83,
 
34
  "ppl": 1.10162,
35
  "step": 2,
36
  "tokens/total": 60800,
37
+ "tokens/train_per_sec_per_gpu": 35.1,
38
  "tokens/trainable": 922
39
  },
40
  {
41
  "epoch": 0.01171875,
42
+ "grad_norm": 1.0102914571762085,
43
  "learning_rate": 8.000000000000001e-06,
44
+ "loss": 0.10876177996397018,
45
  "memory/device_reserved (GiB)": 35.85,
46
  "memory/max_active (GiB)": 33.91,
47
  "memory/max_allocated (GiB)": 33.91,
48
+ "ppl": 1.1149,
49
  "step": 3,
50
  "tokens/total": 91136,
51
+ "tokens/train_per_sec_per_gpu": 34.67,
52
  "tokens/trainable": 1396
53
  },
54
  {
55
  "epoch": 0.015625,
56
+ "grad_norm": 2.2042534351348877,
57
  "learning_rate": 1.2e-05,
58
+ "loss": 0.1031389832496643,
59
  "memory/device_reserved (GiB)": 35.85,
60
  "memory/max_active (GiB)": 33.91,
61
  "memory/max_allocated (GiB)": 33.91,
62
+ "ppl": 1.10865,
63
  "step": 4,
64
  "tokens/total": 121552,
65
+ "tokens/train_per_sec_per_gpu": 38.0,
66
  "tokens/trainable": 1850
67
  },
68
  {
69
  "epoch": 0.01953125,
70
+ "grad_norm": 2.5755860805511475,
71
  "learning_rate": 1.6000000000000003e-05,
72
+ "loss": 0.1097191572189331,
73
  "memory/device_reserved (GiB)": 35.85,
74
  "memory/max_active (GiB)": 33.98,
75
  "memory/max_allocated (GiB)": 33.98,
76
+ "ppl": 1.11596,
77
  "step": 5,
78
  "tokens/total": 152304,
79
+ "tokens/train_per_sec_per_gpu": 34.92,
80
  "tokens/trainable": 2302
81
  },
82
  {
83
  "epoch": 0.0234375,
84
+ "grad_norm": 1.498002052307129,
85
  "learning_rate": 2e-05,
86
+ "loss": 0.08722387254238129,
87
  "memory/device_reserved (GiB)": 35.85,
88
  "memory/max_active (GiB)": 33.78,
89
  "memory/max_allocated (GiB)": 33.78,
90
+ "ppl": 1.09114,
91
  "step": 6,
92
  "tokens/total": 182448,
93
+ "tokens/train_per_sec_per_gpu": 31.65,
94
  "tokens/trainable": 2742
95
  },
96
  {
97
  "epoch": 0.02734375,
98
+ "grad_norm": 1.280110478401184,
99
  "learning_rate": 2.4e-05,
100
+ "loss": 0.07383580505847931,
101
  "memory/device_reserved (GiB)": 35.85,
102
  "memory/max_active (GiB)": 33.85,
103
  "memory/max_allocated (GiB)": 33.85,
104
+ "ppl": 1.07663,
105
  "step": 7,
106
  "tokens/total": 212736,
107
+ "tokens/train_per_sec_per_gpu": 36.7,
108
  "tokens/trainable": 3208
109
  },
110
  {
111
  "epoch": 0.03125,
112
+ "grad_norm": 1.6952791213989258,
113
  "learning_rate": 2.8000000000000003e-05,
114
+ "loss": 0.08001460134983063,
115
  "memory/device_reserved (GiB)": 36.07,
116
  "memory/max_active (GiB)": 33.87,
117
  "memory/max_allocated (GiB)": 33.87,
118
+ "ppl": 1.0833,
119
  "step": 8,
120
  "tokens/total": 243024,
121
+ "tokens/train_per_sec_per_gpu": 31.94,
122
  "tokens/trainable": 3655
123
  },
124
  {
125
  "epoch": 0.03515625,
126
+ "grad_norm": 1.2223162651062012,
127
  "learning_rate": 3.2000000000000005e-05,
128
+ "loss": 0.047361284494400024,
129
  "memory/device_reserved (GiB)": 36.07,
130
  "memory/max_active (GiB)": 33.82,
131
  "memory/max_allocated (GiB)": 33.82,
132
+ "ppl": 1.0485,
133
  "step": 9,
134
  "tokens/total": 271264,
135
+ "tokens/train_per_sec_per_gpu": 35.78,
136
  "tokens/trainable": 4091
137
  },
138
  {
139
  "epoch": 0.0390625,
140
+ "grad_norm": 2.902157783508301,
141
  "learning_rate": 3.6e-05,
142
+ "loss": 0.09570425748825073,
143
  "memory/device_reserved (GiB)": 36.07,
144
  "memory/max_active (GiB)": 33.86,
145
  "memory/max_allocated (GiB)": 33.86,
146
+ "ppl": 1.10043,
147
  "step": 10,
148
  "tokens/total": 301520,
149
+ "tokens/train_per_sec_per_gpu": 34.98,
150
  "tokens/trainable": 4553
151
  },
152
  {
153
  "epoch": 0.04296875,
154
+ "grad_norm": 2.1767208576202393,
155
  "learning_rate": 4e-05,
156
+ "loss": 0.0828268975019455,
157
+ "memory/device_reserved (GiB)": 36.08,
158
  "memory/max_active (GiB)": 33.8,
159
  "memory/max_allocated (GiB)": 33.8,
160
+ "ppl": 1.08635,
161
  "step": 11,
162
  "tokens/total": 331808,
163
+ "tokens/train_per_sec_per_gpu": 33.12,
164
  "tokens/trainable": 4992
165
  },
166
  {
167
  "epoch": 0.046875,
168
+ "grad_norm": 3.15231990814209,
169
  "learning_rate": 4.4000000000000006e-05,
170
+ "loss": 0.12141455709934235,
171
+ "memory/device_reserved (GiB)": 36.08,
172
  "memory/max_active (GiB)": 33.87,
173
  "memory/max_allocated (GiB)": 33.87,
174
+ "ppl": 1.12909,
175
  "step": 12,
176
  "tokens/total": 362224,
177
+ "tokens/train_per_sec_per_gpu": 38.32,
178
  "tokens/trainable": 5494
179
  },
180
  {
181
  "epoch": 0.05078125,
182
+ "grad_norm": 2.3834526538848877,
183
  "learning_rate": 4.8e-05,
184
+ "loss": 0.037937015295028687,
185
+ "memory/device_reserved (GiB)": 36.16,
186
  "memory/max_active (GiB)": 33.91,
187
  "memory/max_allocated (GiB)": 33.91,
188
+ "ppl": 1.03867,
189
  "step": 13,
190
  "tokens/total": 392432,
191
+ "tokens/train_per_sec_per_gpu": 35.04,
192
  "tokens/trainable": 5933
193
  },
194
  {
195
  "epoch": 0.0546875,
196
+ "grad_norm": 3.2587738037109375,
197
  "learning_rate": 5.2000000000000004e-05,
198
+ "loss": 0.06514571607112885,
199
+ "memory/device_reserved (GiB)": 36.16,
200
  "memory/max_active (GiB)": 33.79,
201
  "memory/max_allocated (GiB)": 33.79,
202
+ "ppl": 1.06731,
203
  "step": 14,
204
  "tokens/total": 422784,
205
+ "tokens/train_per_sec_per_gpu": 29.67,
206
  "tokens/trainable": 6369
207
  },
208
  {
209
  "epoch": 0.05859375,
210
+ "grad_norm": 1.5818979740142822,
211
  "learning_rate": 5.6000000000000006e-05,
212
+ "loss": 0.05656517297029495,
213
+ "memory/device_reserved (GiB)": 36.16,
214
  "memory/max_active (GiB)": 33.92,
215
  "memory/max_allocated (GiB)": 33.92,
216
+ "ppl": 1.0582,
217
  "step": 15,
218
  "tokens/total": 453376,
219
+ "tokens/train_per_sec_per_gpu": 33.08,
220
  "tokens/trainable": 6820
221
  },
222
  {
223
  "epoch": 0.0625,
224
+ "grad_norm": 1.4215443134307861,
225
  "learning_rate": 6e-05,
226
+ "loss": 0.06070145219564438,
227
  "memory/device_reserved (GiB)": 36.21,
228
  "memory/max_active (GiB)": 33.83,
229
  "memory/max_allocated (GiB)": 33.83,
230
+ "ppl": 1.06258,
231
  "step": 16,
232
  "tokens/total": 483504,
233
+ "tokens/train_per_sec_per_gpu": 30.85,
234
  "tokens/trainable": 7258
235
  },
236
  {
237
  "epoch": 0.06640625,
238
+ "grad_norm": 1.3065693378448486,
239
  "learning_rate": 6.400000000000001e-05,
240
+ "loss": 0.05027393624186516,
241
  "memory/device_reserved (GiB)": 36.21,
242
  "memory/max_active (GiB)": 33.9,
243
  "memory/max_allocated (GiB)": 33.9,
244
+ "ppl": 1.05156,
245
  "step": 17,
246
  "tokens/total": 514032,
247
+ "tokens/train_per_sec_per_gpu": 35.29,
248
  "tokens/trainable": 7710
249
  },
250
  {
251
  "epoch": 0.0703125,
252
+ "grad_norm": 0.6123363375663757,
253
  "learning_rate": 6.800000000000001e-05,
254
+ "loss": 0.028499452397227287,
255
  "memory/device_reserved (GiB)": 36.21,
256
  "memory/max_active (GiB)": 33.83,
257
  "memory/max_allocated (GiB)": 33.83,
258
+ "ppl": 1.02891,
259
  "step": 18,
260
  "tokens/total": 544432,
261
+ "tokens/train_per_sec_per_gpu": 31.76,
262
  "tokens/trainable": 8174
263
  },
264
  {
265
  "epoch": 0.07421875,
266
+ "grad_norm": 0.648410439491272,
267
  "learning_rate": 7.2e-05,
268
+ "loss": 0.031143227592110634,
269
  "memory/device_reserved (GiB)": 36.21,
270
  "memory/max_active (GiB)": 33.83,
271
  "memory/max_allocated (GiB)": 33.83,
272
+ "ppl": 1.03163,
273
  "step": 19,
274
  "tokens/total": 574880,
275
+ "tokens/train_per_sec_per_gpu": 34.47,
276
  "tokens/trainable": 8617
277
  },
278
  {
279
  "epoch": 0.078125,
280
+ "grad_norm": 0.6737155318260193,
281
  "learning_rate": 7.6e-05,
282
+ "loss": 0.030753308907151222,
283
  "memory/device_reserved (GiB)": 36.21,
284
  "memory/max_active (GiB)": 33.97,
285
  "memory/max_allocated (GiB)": 33.97,
286
+ "ppl": 1.03123,
287
  "step": 20,
288
  "tokens/total": 605408,
289
+ "tokens/train_per_sec_per_gpu": 30.37,
290
  "tokens/trainable": 9037
291
  },
292
  {
293
  "epoch": 0.08203125,
294
+ "grad_norm": 0.7464718222618103,
295
  "learning_rate": 8e-05,
296
+ "loss": 0.02451065182685852,
297
  "memory/device_reserved (GiB)": 36.21,
298
  "memory/max_active (GiB)": 33.77,
299
  "memory/max_allocated (GiB)": 33.77,
300
+ "ppl": 1.02481,
301
  "step": 21,
302
  "tokens/total": 635584,
303
+ "tokens/train_per_sec_per_gpu": 36.63,
304
  "tokens/trainable": 9503
305
  },
306
  {
307
  "epoch": 0.0859375,
308
+ "grad_norm": 0.9435750246047974,
309
  "learning_rate": 8.4e-05,
310
+ "loss": 0.017626767978072166,
311
  "memory/device_reserved (GiB)": 36.21,
312
  "memory/max_active (GiB)": 33.88,
313
  "memory/max_allocated (GiB)": 33.88,
314
+ "ppl": 1.01778,
315
  "step": 22,
316
  "tokens/total": 665952,
317
+ "tokens/train_per_sec_per_gpu": 36.79,
318
  "tokens/trainable": 9972
319
  },
320
  {
321
  "epoch": 0.08984375,
322
+ "grad_norm": 0.6369844079017639,
323
  "learning_rate": 8.800000000000001e-05,
324
+ "loss": 0.013959845528006554,
325
  "memory/device_reserved (GiB)": 36.21,
326
  "memory/max_active (GiB)": 33.9,
327
  "memory/max_allocated (GiB)": 33.9,
328
+ "ppl": 1.01406,
329
  "step": 23,
330
  "tokens/total": 696240,
331
+ "tokens/train_per_sec_per_gpu": 37.54,
332
  "tokens/trainable": 10447
333
  },
334
  {
335
  "epoch": 0.09375,
336
+ "grad_norm": 1.0666130781173706,
337
  "learning_rate": 9.200000000000001e-05,
338
+ "loss": 0.03303435444831848,
339
  "memory/device_reserved (GiB)": 36.35,
340
  "memory/max_active (GiB)": 33.88,
341
  "memory/max_allocated (GiB)": 33.88,
342
+ "ppl": 1.03359,
343
  "step": 24,
344
  "tokens/total": 726464,
345
+ "tokens/train_per_sec_per_gpu": 37.23,
346
  "tokens/trainable": 10899
347
  },
348
  {
349
  "epoch": 0.09765625,
350
+ "grad_norm": 1.0491507053375244,
351
  "learning_rate": 9.6e-05,
352
+ "loss": 0.030840374529361725,
353
  "memory/device_reserved (GiB)": 36.35,
354
  "memory/max_active (GiB)": 33.84,
355
  "memory/max_allocated (GiB)": 33.84,
356
+ "ppl": 1.03132,
357
  "step": 25,
358
  "tokens/total": 756704,
359
+ "tokens/train_per_sec_per_gpu": 33.1,
360
  "tokens/trainable": 11360
361
  },
362
  {
363
  "epoch": 0.1015625,
364
+ "grad_norm": 0.8108148574829102,
365
  "learning_rate": 0.0001,
366
+ "loss": 0.03408072516322136,
367
  "memory/device_reserved (GiB)": 36.35,
368
  "memory/max_active (GiB)": 33.82,
369
  "memory/max_allocated (GiB)": 33.82,
370
+ "ppl": 1.03467,
371
  "step": 26,
372
  "tokens/total": 786912,
373
+ "tokens/train_per_sec_per_gpu": 29.3,
374
  "tokens/trainable": 11775
375
  },
376
  {
377
  "epoch": 0.10546875,
378
+ "grad_norm": 0.507036030292511,
379
  "learning_rate": 9.99990636831587e-05,
380
+ "loss": 0.014000408351421356,
381
  "memory/device_reserved (GiB)": 36.35,
382
  "memory/max_active (GiB)": 33.46,
383
  "memory/max_allocated (GiB)": 33.46,
384
+ "ppl": 1.0141,
385
  "step": 27,
386
  "tokens/total": 815424,
387
+ "tokens/train_per_sec_per_gpu": 39.89,
388
  "tokens/trainable": 12242
389
  },
390
  {
391
  "epoch": 0.109375,
392
+ "grad_norm": 0.618457555770874,
393
  "learning_rate": 9.999625477159879e-05,
394
+ "loss": 0.022290699183940887,
395
  "memory/device_reserved (GiB)": 36.35,
396
  "memory/max_active (GiB)": 33.89,
397
  "memory/max_allocated (GiB)": 33.89,
398
+ "ppl": 1.02254,
399
  "step": 28,
400
  "tokens/total": 845584,
401
+ "tokens/train_per_sec_per_gpu": 33.48,
402
  "tokens/trainable": 12696
403
  },
404
  {
405
  "epoch": 0.11328125,
406
+ "grad_norm": 0.4989492893218994,
407
  "learning_rate": 9.999157338221051e-05,
408
+ "loss": 0.025926023721694946,
409
  "memory/device_reserved (GiB)": 36.35,
410
  "memory/max_active (GiB)": 33.87,
411
  "memory/max_allocated (GiB)": 33.87,
412
+ "ppl": 1.02627,
413
  "step": 29,
414
  "tokens/total": 875808,
415
+ "tokens/train_per_sec_per_gpu": 34.34,
416
  "tokens/trainable": 13153
417
  },
418
  {
419
  "epoch": 0.1171875,
420
+ "grad_norm": 0.3578357696533203,
421
  "learning_rate": 9.998501970980562e-05,
422
+ "loss": 0.014475762844085693,
423
  "memory/device_reserved (GiB)": 36.35,
424
  "memory/max_active (GiB)": 33.82,
425
  "memory/max_allocated (GiB)": 33.82,
426
+ "ppl": 1.01458,
427
  "step": 30,
428
  "tokens/total": 904096,
429
+ "tokens/train_per_sec_per_gpu": 39.7,
430
  "tokens/trainable": 13624
431
  },
432
  {
433
  "epoch": 0.12109375,
434
+ "grad_norm": 0.4404466450214386,
435
  "learning_rate": 9.997659402710915e-05,
436
+ "loss": 0.015927618369460106,
437
  "memory/device_reserved (GiB)": 36.35,
438
  "memory/max_active (GiB)": 33.86,
439
  "memory/max_allocated (GiB)": 33.86,
440
+ "ppl": 1.01606,
441
  "step": 31,
442
  "tokens/total": 934400,
443
+ "tokens/train_per_sec_per_gpu": 34.1,
444
  "tokens/trainable": 14064
445
  },
446
  {
447
  "epoch": 0.125,
448
+ "grad_norm": 0.5913511514663696,
449
  "learning_rate": 9.996629668474818e-05,
450
+ "loss": 0.019408874213695526,
451
  "memory/device_reserved (GiB)": 36.35,
452
  "memory/max_active (GiB)": 33.92,
453
  "memory/max_allocated (GiB)": 33.92,
454
+ "ppl": 1.0196,
455
  "step": 32,
456
  "tokens/total": 964832,
457
+ "tokens/train_per_sec_per_gpu": 38.92,
458
  "tokens/trainable": 14565
459
  },
460
  {
461
  "epoch": 0.12890625,
462
+ "grad_norm": 0.2795853614807129,
463
  "learning_rate": 9.995412811123711e-05,
464
+ "loss": 0.007002322003245354,
465
  "memory/device_reserved (GiB)": 36.35,
466
  "memory/max_active (GiB)": 33.79,
467
  "memory/max_allocated (GiB)": 33.79,
468
+ "ppl": 1.00703,
469
  "step": 33,
470
  "tokens/total": 995040,
471
+ "tokens/train_per_sec_per_gpu": 33.77,
472
  "tokens/trainable": 15005
473
  },
474
  {
475
  "epoch": 0.1328125,
476
+ "grad_norm": 1.1757413148880005,
477
  "learning_rate": 9.994008881295999e-05,
478
+ "loss": 0.021448152139782906,
479
  "memory/device_reserved (GiB)": 34.72,
480
  "memory/max_active (GiB)": 33.64,
481
  "memory/max_allocated (GiB)": 33.64,
482
+ "ppl": 1.02168,
483
  "step": 34,
484
  "tokens/total": 1024896,
485
+ "tokens/train_per_sec_per_gpu": 32.03,
486
  "tokens/trainable": 15459
487
  },
488
  {
489
  "epoch": 0.13671875,
490
+ "grad_norm": 0.3860406279563904,
491
  "learning_rate": 9.992417937414932e-05,
492
+ "loss": 0.01894465833902359,
493
  "memory/device_reserved (GiB)": 35.48,
494
  "memory/max_active (GiB)": 33.84,
495
  "memory/max_allocated (GiB)": 33.84,
496
+ "ppl": 1.01913,
497
  "step": 35,
498
  "tokens/total": 1054992,
499
+ "tokens/train_per_sec_per_gpu": 38.21,
500
  "tokens/trainable": 15960
501
  },
502
  {
503
  "epoch": 0.140625,
504
+ "grad_norm": 0.4803963005542755,
505
  "learning_rate": 9.99064004568618e-05,
506
+ "loss": 0.013810254633426666,
507
  "memory/device_reserved (GiB)": 35.48,
508
  "memory/max_active (GiB)": 33.93,
509
  "memory/max_allocated (GiB)": 33.93,
510
+ "ppl": 1.01391,
511
  "step": 36,
512
  "tokens/total": 1085280,
513
+ "tokens/train_per_sec_per_gpu": 34.59,
514
  "tokens/trainable": 16428
515
  },
516
  {
517
  "epoch": 0.14453125,
518
+ "grad_norm": 0.3357454836368561,
519
  "learning_rate": 9.988675280095074e-05,
520
+ "loss": 0.008235199376940727,
521
+ "memory/device_reserved (GiB)": 35.49,
522
  "memory/max_active (GiB)": 33.7,
523
  "memory/max_allocated (GiB)": 33.7,
524
+ "ppl": 1.00827,
525
  "step": 37,
526
  "tokens/total": 1115504,
527
+ "tokens/train_per_sec_per_gpu": 31.89,
528
  "tokens/trainable": 16884
529
  },
530
  {
531
  "epoch": 0.1484375,
532
+ "grad_norm": 0.9474877715110779,
533
  "learning_rate": 9.986523722403528e-05,
534
+ "loss": 0.019564270973205566,
535
+ "memory/device_reserved (GiB)": 35.49,
536
  "memory/max_active (GiB)": 33.82,
537
  "memory/max_allocated (GiB)": 33.82,
538
+ "ppl": 1.01976,
539
  "step": 38,
540
  "tokens/total": 1145712,
541
+ "tokens/train_per_sec_per_gpu": 32.02,
542
  "tokens/trainable": 17341
543
  },
544
  {
545
  "epoch": 0.15234375,
546
+ "grad_norm": 1.101553201675415,
547
  "learning_rate": 9.984185462146642e-05,
548
+ "loss": 0.017880277708172798,
549
+ "memory/device_reserved (GiB)": 35.49,
550
  "memory/max_active (GiB)": 33.81,
551
  "memory/max_allocated (GiB)": 33.81,
552
+ "ppl": 1.01804,
553
  "step": 39,
554
  "tokens/total": 1176128,
555
+ "tokens/train_per_sec_per_gpu": 34.16,
556
  "tokens/trainable": 17786
557
  },
558
  {
559
  "epoch": 0.15625,
560
+ "grad_norm": 0.7563315033912659,
561
  "learning_rate": 9.98166059662897e-05,
562
+ "loss": 0.019336596131324768,
563
+ "memory/device_reserved (GiB)": 35.49,
564
  "memory/max_active (GiB)": 33.85,
565
  "memory/max_allocated (GiB)": 33.85,
566
+ "ppl": 1.01952,
567
  "step": 40,
568
  "tokens/total": 1206576,
569
+ "tokens/train_per_sec_per_gpu": 35.01,
570
  "tokens/trainable": 18262
571
  },
572
  {
573
  "epoch": 0.16015625,
574
+ "grad_norm": 0.41576531529426575,
575
  "learning_rate": 9.978949230920472e-05,
576
+ "loss": 0.011620002798736095,
577
+ "memory/device_reserved (GiB)": 35.88,
578
  "memory/max_active (GiB)": 33.86,
579
  "memory/max_allocated (GiB)": 33.86,
580
+ "ppl": 1.01169,
581
  "step": 41,
582
  "tokens/total": 1236848,
583
+ "tokens/train_per_sec_per_gpu": 32.74,
584
  "tokens/trainable": 18716
585
  },
586
  {
587
  "epoch": 0.1640625,
588
+ "grad_norm": 1.180986762046814,
589
  "learning_rate": 9.976051477852141e-05,
590
+ "loss": 0.04661658778786659,
591
+ "memory/device_reserved (GiB)": 35.88,
592
  "memory/max_active (GiB)": 33.83,
593
  "memory/max_allocated (GiB)": 33.83,
594
+ "ppl": 1.04772,
595
  "step": 42,
596
  "tokens/total": 1267088,
597
+ "tokens/train_per_sec_per_gpu": 32.88,
598
  "tokens/trainable": 19157
599
  },
600
  {
601
  "epoch": 0.16796875,
602
+ "grad_norm": 0.7583066821098328,
603
  "learning_rate": 9.972967458011312e-05,
604
+ "loss": 0.029224852100014687,
605
+ "memory/device_reserved (GiB)": 35.88,
606
  "memory/max_active (GiB)": 33.82,
607
  "memory/max_allocated (GiB)": 33.82,
608
+ "ppl": 1.02966,
609
  "step": 43,
610
  "tokens/total": 1297360,
611
+ "tokens/train_per_sec_per_gpu": 36.53,
612
  "tokens/trainable": 19647
613
  },
614
  {
615
  "epoch": 0.171875,
616
+ "grad_norm": 0.18861345946788788,
617
  "learning_rate": 9.96969729973664e-05,
618
+ "loss": 0.007798851002007723,
619
+ "memory/device_reserved (GiB)": 35.88,
620
  "memory/max_active (GiB)": 33.86,
621
  "memory/max_allocated (GiB)": 33.86,
622
+ "ppl": 1.00783,
623
  "step": 44,
624
  "tokens/total": 1327744,
625
+ "tokens/train_per_sec_per_gpu": 34.91,
626
  "tokens/trainable": 20119
627
  },
628
  {
629
  "epoch": 0.17578125,
630
+ "grad_norm": 0.35095125436782837,
631
  "learning_rate": 9.966241139112754e-05,
632
+ "loss": 0.012044938281178474,
633
+ "memory/device_reserved (GiB)": 35.88,
634
  "memory/max_active (GiB)": 33.86,
635
  "memory/max_allocated (GiB)": 33.86,
636
+ "ppl": 1.01212,
637
  "step": 45,
638
  "tokens/total": 1358096,
639
+ "tokens/train_per_sec_per_gpu": 32.9,
640
  "tokens/trainable": 20560
641
  },
642
  {
643
  "epoch": 0.1796875,
644
+ "grad_norm": 0.5071477890014648,
645
  "learning_rate": 9.96259911996461e-05,
646
+ "loss": 0.017856616526842117,
647
+ "memory/device_reserved (GiB)": 35.88,
648
  "memory/max_active (GiB)": 33.84,
649
  "memory/max_allocated (GiB)": 33.84,
650
+ "ppl": 1.01802,
651
  "step": 46,
652
  "tokens/total": 1388240,
653
+ "tokens/train_per_sec_per_gpu": 32.63,
654
  "tokens/trainable": 20992
655
  },
656
  {
657
  "epoch": 0.18359375,
658
+ "grad_norm": 0.5569872260093689,
659
  "learning_rate": 9.958771393851491e-05,
660
+ "loss": 0.019440822303295135,
661
+ "memory/device_reserved (GiB)": 35.88,
662
  "memory/max_active (GiB)": 33.26,
663
  "memory/max_allocated (GiB)": 33.26,
664
+ "ppl": 1.01963,
665
  "step": 47,
666
  "tokens/total": 1416240,
667
+ "tokens/train_per_sec_per_gpu": 31.55,
668
  "tokens/trainable": 21441
669
  },
670
  {
671
  "epoch": 0.1875,
672
+ "grad_norm": 0.42062458395957947,
673
  "learning_rate": 9.954758120060702e-05,
674
+ "loss": 0.013018792495131493,
675
  "memory/device_reserved (GiB)": 35.88,
676
  "memory/max_active (GiB)": 33.96,
677
  "memory/max_allocated (GiB)": 33.96,
678
+ "ppl": 1.0131,
679
  "step": 48,
680
  "tokens/total": 1446880,
681
+ "tokens/train_per_sec_per_gpu": 33.68,
682
  "tokens/trainable": 21901
683
  },
684
  {
685
  "epoch": 0.19140625,
686
+ "grad_norm": 0.30396750569343567,
687
  "learning_rate": 9.950559465600948e-05,
688
+ "loss": 0.0077492957934737206,
689
  "memory/device_reserved (GiB)": 35.88,
690
  "memory/max_active (GiB)": 33.78,
691
  "memory/max_allocated (GiB)": 33.78,
692
+ "ppl": 1.00778,
693
  "step": 49,
694
  "tokens/total": 1477168,
695
+ "tokens/train_per_sec_per_gpu": 33.33,
696
  "tokens/trainable": 22341
697
  },
698
  {
699
  "epoch": 0.1953125,
700
+ "grad_norm": 2.044849395751953,
701
  "learning_rate": 9.946175605195379e-05,
702
+ "loss": 0.014447445049881935,
703
  "memory/device_reserved (GiB)": 35.88,
704
  "memory/max_active (GiB)": 33.89,
705
  "memory/max_allocated (GiB)": 33.89,
706
+ "ppl": 1.01455,
707
  "step": 50,
708
  "tokens/total": 1507712,
709
+ "tokens/train_per_sec_per_gpu": 33.0,
710
  "tokens/trainable": 22833
711
  },
712
  {
713
  "epoch": 0.19921875,
714
+ "grad_norm": 0.9357694983482361,
715
  "learning_rate": 9.941606721274322e-05,
716
+ "loss": 0.012720856815576553,
717
  "memory/device_reserved (GiB)": 35.88,
718
  "memory/max_active (GiB)": 33.81,
719
  "memory/max_allocated (GiB)": 33.81,
720
+ "ppl": 1.0128,
721
  "step": 51,
722
  "tokens/total": 1537936,
723
+ "tokens/train_per_sec_per_gpu": 30.72,
724
  "tokens/trainable": 23277
725
  },
726
  {
727
  "epoch": 0.203125,
728
+ "grad_norm": 0.2881873548030853,
729
  "learning_rate": 9.936853003967685e-05,
730
+ "loss": 0.005877626594156027,
731
  "memory/device_reserved (GiB)": 35.88,
732
  "memory/max_active (GiB)": 33.84,
733
  "memory/max_allocated (GiB)": 33.84,
734
+ "ppl": 1.00589,
735
  "step": 52,
736
  "tokens/total": 1568352,
737
+ "tokens/train_per_sec_per_gpu": 35.63,
738
  "tokens/trainable": 23759
739
  },
740
  {
741
  "epoch": 0.20703125,
742
+ "grad_norm": 0.7577692270278931,
743
  "learning_rate": 9.93191465109705e-05,
744
+ "loss": 0.01451955921947956,
745
  "memory/device_reserved (GiB)": 35.88,
746
  "memory/max_active (GiB)": 33.91,
747
  "memory/max_allocated (GiB)": 33.91,
748
+ "ppl": 1.01463,
749
  "step": 53,
750
  "tokens/total": 1598992,
751
+ "tokens/train_per_sec_per_gpu": 32.95,
752
  "tokens/trainable": 24193
753
  },
754
  {
755
  "epoch": 0.2109375,
756
+ "grad_norm": 0.5201014280319214,
757
  "learning_rate": 9.926791868167438e-05,
758
+ "loss": 0.006856432184576988,
759
  "memory/device_reserved (GiB)": 35.9,
760
  "memory/max_active (GiB)": 33.95,
761
  "memory/max_allocated (GiB)": 33.95,
762
+ "ppl": 1.00688,
763
  "step": 54,
764
  "tokens/total": 1629488,
765
+ "tokens/train_per_sec_per_gpu": 33.59,
766
  "tokens/trainable": 24619
767
  },
768
  {
769
  "epoch": 0.21484375,
770
+ "grad_norm": 0.8399921655654907,
771
  "learning_rate": 9.921484868358753e-05,
772
+ "loss": 0.037967782467603683,
773
  "memory/device_reserved (GiB)": 35.9,
774
  "memory/max_active (GiB)": 33.8,
775
  "memory/max_allocated (GiB)": 33.8,
776
+ "ppl": 1.0387,
777
  "step": 55,
778
  "tokens/total": 1659696,
779
+ "tokens/train_per_sec_per_gpu": 32.85,
780
  "tokens/trainable": 25075
781
  },
782
  {
783
  "epoch": 0.21875,
784
+ "grad_norm": 0.8203506469726562,
785
  "learning_rate": 9.915993872516924e-05,
786
+ "loss": 0.012814272195100784,
787
  "memory/device_reserved (GiB)": 35.9,
788
  "memory/max_active (GiB)": 33.76,
789
  "memory/max_allocated (GiB)": 33.76,
790
+ "ppl": 1.0129,
791
  "step": 56,
792
  "tokens/total": 1689872,
793
+ "tokens/train_per_sec_per_gpu": 31.55,
794
  "tokens/trainable": 25519
795
  },
796
  {
797
  "epoch": 0.22265625,
798
+ "grad_norm": 0.20874246954917908,
799
  "learning_rate": 9.9103191091447e-05,
800
+ "loss": 0.00539036700502038,
801
  "memory/device_reserved (GiB)": 35.9,
802
  "memory/max_active (GiB)": 33.9,
803
  "memory/max_allocated (GiB)": 33.9,
804
+ "ppl": 1.0054,
805
  "step": 57,
806
  "tokens/total": 1720144,
807
+ "tokens/train_per_sec_per_gpu": 32.69,
808
  "tokens/trainable": 25966
809
  },
810
  {
811
  "epoch": 0.2265625,
812
+ "grad_norm": 0.4427756071090698,
813
  "learning_rate": 9.904460814392147e-05,
814
+ "loss": 0.009390819817781448,
815
  "memory/device_reserved (GiB)": 35.9,
816
  "memory/max_active (GiB)": 33.89,
817
  "memory/max_allocated (GiB)": 33.89,
818
+ "ppl": 1.00944,
819
  "step": 58,
820
  "tokens/total": 1750448,
821
+ "tokens/train_per_sec_per_gpu": 33.98,
822
  "tokens/trainable": 26423
823
  },
824
  {
825
  "epoch": 0.23046875,
826
+ "grad_norm": 0.7457786798477173,
827
  "learning_rate": 9.898419232046825e-05,
828
+ "loss": 0.019530881196260452,
829
  "memory/device_reserved (GiB)": 35.9,
830
  "memory/max_active (GiB)": 33.93,
831
  "memory/max_allocated (GiB)": 33.93,
832
+ "ppl": 1.01972,
833
  "step": 59,
834
  "tokens/total": 1781088,
835
+ "tokens/train_per_sec_per_gpu": 38.51,
836
  "tokens/trainable": 26934
837
  },
838
  {
839
  "epoch": 0.234375,
840
+ "grad_norm": 0.13402195274829865,
841
  "learning_rate": 9.892194613523633e-05,
842
+ "loss": 0.0014544172445312142,
843
  "memory/device_reserved (GiB)": 35.9,
844
  "memory/max_active (GiB)": 33.84,
845
  "memory/max_allocated (GiB)": 33.84,
846
+ "ppl": 1.00146,
847
  "step": 60,
848
  "tokens/total": 1811344,
849
+ "tokens/train_per_sec_per_gpu": 32.96,
850
  "tokens/trainable": 27393
851
  },
852
  {
853
  "epoch": 0.23828125,
854
+ "grad_norm": 1.0385031700134277,
855
  "learning_rate": 9.885787217854357e-05,
856
+ "loss": 0.019916968420147896,
857
  "memory/device_reserved (GiB)": 35.96,
858
  "memory/max_active (GiB)": 33.9,
859
  "memory/max_allocated (GiB)": 33.9,
860
+ "ppl": 1.02012,
861
  "step": 61,
862
  "tokens/total": 1841600,
863
+ "tokens/train_per_sec_per_gpu": 32.94,
864
  "tokens/trainable": 27852
865
  },
866
  {
867
  "epoch": 0.2421875,
868
+ "grad_norm": 1.2596747875213623,
869
  "learning_rate": 9.879197311676887e-05,
870
+ "loss": 0.015884939581155777,
871
  "memory/device_reserved (GiB)": 35.96,
872
  "memory/max_active (GiB)": 33.82,
873
  "memory/max_allocated (GiB)": 33.82,
874
+ "ppl": 1.01601,
875
  "step": 62,
876
  "tokens/total": 1872048,
877
+ "tokens/train_per_sec_per_gpu": 32.96,
878
  "tokens/trainable": 28292
879
  },
880
  {
881
  "epoch": 0.24609375,
882
+ "grad_norm": 0.14031143486499786,
883
  "learning_rate": 9.872425169224113e-05,
884
+ "loss": 0.002796083688735962,
885
  "memory/device_reserved (GiB)": 35.96,
886
  "memory/max_active (GiB)": 33.94,
887
  "memory/max_allocated (GiB)": 33.94,
888
+ "ppl": 1.0028,
889
  "step": 63,
890
  "tokens/total": 1902592,
891
+ "tokens/train_per_sec_per_gpu": 37.36,
892
  "tokens/trainable": 28798
893
  },
894
  {
895
  "epoch": 0.25,
896
+ "grad_norm": 0.6247786283493042,
897
  "learning_rate": 9.865471072312528e-05,
898
+ "loss": 0.017117884010076523,
899
  "memory/device_reserved (GiB)": 35.96,
900
  "memory/max_active (GiB)": 33.96,
901
  "memory/max_allocated (GiB)": 33.96,
902
+ "ppl": 1.01727,
903
  "step": 64,
904
  "tokens/total": 1933088,
905
+ "tokens/train_per_sec_per_gpu": 35.79,
906
  "tokens/trainable": 29283
907
  },
908
  {
909
  "epoch": 0.25390625,
910
+ "grad_norm": 0.3513385057449341,
911
  "learning_rate": 9.858335310330492e-05,
912
+ "loss": 0.008029159158468246,
913
  "memory/device_reserved (GiB)": 35.96,
914
  "memory/max_active (GiB)": 33.72,
915
  "memory/max_allocated (GiB)": 33.72,
916
+ "ppl": 1.00806,
917
  "step": 65,
918
  "tokens/total": 1963296,
919
+ "tokens/train_per_sec_per_gpu": 31.99,
920
  "tokens/trainable": 29745
921
  },
922
  {
923
  "epoch": 0.2578125,
924
+ "grad_norm": 0.279448539018631,
925
  "learning_rate": 9.851018180226185e-05,
926
+ "loss": 0.0161186084151268,
927
+ "memory/device_reserved (GiB)": 34.87,
928
  "memory/max_active (GiB)": 33.79,
929
  "memory/max_allocated (GiB)": 33.79,
930
+ "ppl": 1.01625,
931
  "step": 66,
932
  "tokens/total": 1993760,
933
  "tokens/train_per_sec_per_gpu": 31.19,
 
935
  },
936
  {
937
  "epoch": 0.26171875,
938
+ "grad_norm": 0.31739094853401184,
939
  "learning_rate": 9.843519986495259e-05,
940
+ "loss": 0.009091717191040516,
941
  "memory/device_reserved (GiB)": 35.86,
942
  "memory/max_active (GiB)": 33.89,
943
  "memory/max_allocated (GiB)": 33.89,
944
+ "ppl": 1.00913,
945
  "step": 67,
946
  "tokens/total": 2024144,
947
+ "tokens/train_per_sec_per_gpu": 33.36,
948
  "tokens/trainable": 30622
949
  },
950
  {
951
  "epoch": 0.265625,
952
+ "grad_norm": 0.3539387285709381,
953
  "learning_rate": 9.835841041168162e-05,
954
+ "loss": 0.00908343680202961,
955
  "memory/device_reserved (GiB)": 35.86,
956
  "memory/max_active (GiB)": 33.95,
957
  "memory/max_allocated (GiB)": 33.95,
958
+ "ppl": 1.00912,
959
  "step": 68,
960
  "tokens/total": 2054608,
961
+ "tokens/train_per_sec_per_gpu": 35.52,
962
  "tokens/trainable": 31097
963
  },
964
  {
965
  "epoch": 0.26953125,
966
+ "grad_norm": 0.6497699618339539,
967
  "learning_rate": 9.82798166379715e-05,
968
+ "loss": 0.03086906298995018,
969
  "memory/device_reserved (GiB)": 35.86,
970
  "memory/max_active (GiB)": 33.82,
971
  "memory/max_allocated (GiB)": 33.82,
972
+ "ppl": 1.03135,
973
  "step": 69,
974
  "tokens/total": 2084912,
975
+ "tokens/train_per_sec_per_gpu": 36.6,
976
  "tokens/trainable": 31595
977
  },
978
  {
979
  "epoch": 0.2734375,
980
+ "grad_norm": 0.19664841890335083,
981
  "learning_rate": 9.819942181443002e-05,
982
+ "loss": 0.0039155250415205956,
983
  "memory/device_reserved (GiB)": 35.86,
984
  "memory/max_active (GiB)": 33.84,
985
  "memory/max_allocated (GiB)": 33.84,
986
+ "ppl": 1.00392,
987
  "step": 70,
988
  "tokens/total": 2115216,
989
+ "tokens/train_per_sec_per_gpu": 30.75,
990
  "tokens/trainable": 32036
991
  },
992
  {
993
  "epoch": 0.27734375,
994
+ "grad_norm": 0.4300064146518707,
995
  "learning_rate": 9.811722928661392e-05,
996
+ "loss": 0.007546662352979183,
997
  "memory/device_reserved (GiB)": 35.86,
998
  "memory/max_active (GiB)": 33.9,
999
  "memory/max_allocated (GiB)": 33.9,
1000
+ "ppl": 1.00758,
1001
  "step": 71,
1002
  "tokens/total": 2145424,
1003
+ "tokens/train_per_sec_per_gpu": 28.09,
1004
  "tokens/trainable": 32428
1005
  },
1006
  {
1007
  "epoch": 0.28125,
1008
+ "grad_norm": 0.027750927954912186,
1009
  "learning_rate": 9.803324247488975e-05,
1010
+ "loss": 0.0004817460721824318,
1011
  "memory/device_reserved (GiB)": 35.88,
1012
  "memory/max_active (GiB)": 33.84,
1013
  "memory/max_allocated (GiB)": 33.84,
1014
+ "ppl": 1.00048,
1015
  "step": 72,
1016
  "tokens/total": 2175648,
1017
+ "tokens/train_per_sec_per_gpu": 32.11,
1018
  "tokens/trainable": 32880
1019
  },
1020
  {
1021
  "epoch": 0.28515625,
1022
+ "grad_norm": 0.11202923208475113,
1023
  "learning_rate": 9.794746487429161e-05,
1024
+ "loss": 0.0012140646576881409,
1025
  "memory/device_reserved (GiB)": 35.88,
1026
  "memory/max_active (GiB)": 33.79,
1027
  "memory/max_allocated (GiB)": 33.79,
1028
+ "ppl": 1.00121,
1029
  "step": 73,
1030
  "tokens/total": 2205856,
1031
+ "tokens/train_per_sec_per_gpu": 33.51,
1032
  "tokens/trainable": 33323
1033
  },
1034
  {
1035
  "epoch": 0.2890625,
1036
+ "grad_norm": 0.026963796466588974,
1037
  "learning_rate": 9.785990005437554e-05,
1038
+ "loss": 0.00046908354852348566,
1039
  "memory/device_reserved (GiB)": 35.88,
1040
  "memory/max_active (GiB)": 33.95,
1041
  "memory/max_allocated (GiB)": 33.95,
1042
+ "ppl": 1.00047,
1043
  "step": 74,
1044
  "tokens/total": 2236352,
1045
+ "tokens/train_per_sec_per_gpu": 35.86,
1046
  "tokens/trainable": 33792
1047
  },
1048
  {
1049
  "epoch": 0.29296875,
1050
+ "grad_norm": 0.5172365307807922,
1051
  "learning_rate": 9.777055165907117e-05,
1052
+ "loss": 0.029645610600709915,
1053
  "memory/device_reserved (GiB)": 35.88,
1054
  "memory/max_active (GiB)": 33.8,
1055
  "memory/max_allocated (GiB)": 33.8,
1056
+ "ppl": 1.03009,
1057
  "step": 75,
1058
  "tokens/total": 2266480,
1059
+ "tokens/train_per_sec_per_gpu": 35.26,
1060
  "tokens/trainable": 34223
1061
  },
1062
  {
1063
  "epoch": 0.296875,
1064
+ "grad_norm": 0.84085613489151,
1065
  "learning_rate": 9.767942340652993e-05,
1066
+ "loss": 0.006980063393712044,
1067
  "memory/device_reserved (GiB)": 35.88,
1068
  "memory/max_active (GiB)": 33.71,
1069
  "memory/max_allocated (GiB)": 33.71,
1070
+ "ppl": 1.007,
1071
  "step": 76,
1072
  "tokens/total": 2296496,
1073
+ "tokens/train_per_sec_per_gpu": 29.61,
1074
  "tokens/trainable": 34649
1075
  },
1076
  {
1077
  "epoch": 0.30078125,
1078
+ "grad_norm": 3.4935519695281982,
1079
  "learning_rate": 9.758651908897035e-05,
1080
+ "loss": 0.008870027028024197,
1081
  "memory/device_reserved (GiB)": 35.88,
1082
  "memory/max_active (GiB)": 33.88,
1083
  "memory/max_allocated (GiB)": 33.88,
1084
+ "ppl": 1.00891,
1085
  "step": 77,
1086
  "tokens/total": 2327040,
1087
+ "tokens/train_per_sec_per_gpu": 35.64,
1088
  "tokens/trainable": 35094
1089
  },
1090
  {
1091
  "epoch": 0.3046875,
1092
+ "grad_norm": 0.9529178142547607,
1093
  "learning_rate": 9.749184257252033e-05,
1094
+ "loss": 0.032071419060230255,
1095
  "memory/device_reserved (GiB)": 35.88,
1096
  "memory/max_active (GiB)": 33.81,
1097
  "memory/max_allocated (GiB)": 33.81,
1098
+ "ppl": 1.03259,
1099
  "step": 78,
1100
  "tokens/total": 2357328,
1101
+ "tokens/train_per_sec_per_gpu": 31.82,
1102
  "tokens/trainable": 35534
1103
  },
1104
  {
1105
  "epoch": 0.30859375,
1106
+ "grad_norm": 0.5781691074371338,
1107
  "learning_rate": 9.739539779705614e-05,
1108
+ "loss": 0.01475254725664854,
1109
+ "memory/device_reserved (GiB)": 35.89,
1110
  "memory/max_active (GiB)": 33.85,
1111
  "memory/max_allocated (GiB)": 33.85,
1112
+ "ppl": 1.01486,
1113
  "step": 79,
1114
  "tokens/total": 2387664,
1115
+ "tokens/train_per_sec_per_gpu": 35.82,
1116
  "tokens/trainable": 36029
1117
  },
1118
  {
1119
  "epoch": 0.3125,
1120
+ "grad_norm": 0.15085959434509277,
1121
  "learning_rate": 9.729718877603861e-05,
1122
+ "loss": 0.002578896936029196,
1123
+ "memory/device_reserved (GiB)": 35.89,
1124
  "memory/max_active (GiB)": 33.96,
1125
  "memory/max_allocated (GiB)": 33.96,
1126
+ "ppl": 1.00258,
1127
  "step": 80,
1128
  "tokens/total": 2418000,
1129
+ "tokens/train_per_sec_per_gpu": 35.53,
1130
  "tokens/trainable": 36469
1131
  },
1132
  {
1133
  "epoch": 0.31640625,
1134
+ "grad_norm": 0.19004814326763153,
1135
  "learning_rate": 9.719721959634592e-05,
1136
+ "loss": 0.004292497877031565,
1137
+ "memory/device_reserved (GiB)": 35.89,
1138
  "memory/max_active (GiB)": 33.96,
1139
  "memory/max_allocated (GiB)": 33.96,
1140
+ "ppl": 1.0043,
1141
  "step": 81,
1142
  "tokens/total": 2448720,
1143
+ "tokens/train_per_sec_per_gpu": 30.75,
1144
  "tokens/trainable": 36906
1145
  },
1146
  {
1147
  "epoch": 0.3203125,
1148
+ "grad_norm": 0.4505981504917145,
1149
  "learning_rate": 9.709549441810375e-05,
1150
+ "loss": 0.018529793247580528,
1151
+ "memory/device_reserved (GiB)": 35.89,
1152
  "memory/max_active (GiB)": 33.92,
1153
  "memory/max_allocated (GiB)": 33.92,
1154
+ "ppl": 1.0187,
1155
  "step": 82,
1156
  "tokens/total": 2479248,
1157
+ "tokens/train_per_sec_per_gpu": 38.95,
1158
  "tokens/trainable": 37390
1159
  },
1160
  {
1161
  "epoch": 0.32421875,
1162
+ "grad_norm": 0.4981430470943451,
1163
  "learning_rate": 9.699201747451195e-05,
1164
+ "loss": 0.014818452298641205,
1165
+ "memory/device_reserved (GiB)": 35.89,
1166
  "memory/max_active (GiB)": 33.78,
1167
  "memory/max_allocated (GiB)": 33.78,
1168
+ "ppl": 1.01493,
1169
  "step": 83,
1170
  "tokens/total": 2509408,
1171
+ "tokens/train_per_sec_per_gpu": 30.8,
1172
  "tokens/trainable": 37838
1173
  },
1174
  {
1175
  "epoch": 0.328125,
1176
+ "grad_norm": 0.16379471123218536,
1177
  "learning_rate": 9.688679307166854e-05,
1178
+ "loss": 0.003185997949913144,
1179
+ "memory/device_reserved (GiB)": 35.89,
1180
  "memory/max_active (GiB)": 33.97,
1181
  "memory/max_allocated (GiB)": 33.97,
1182
+ "ppl": 1.00319,
1183
  "step": 84,
1184
  "tokens/total": 2539776,
1185
+ "tokens/train_per_sec_per_gpu": 35.02,
1186
  "tokens/trainable": 38310
1187
  },
1188
  {
1189
  "epoch": 0.33203125,
1190
+ "grad_norm": 0.40732133388519287,
1191
  "learning_rate": 9.677982558839042e-05,
1192
+ "loss": 0.020803559571504593,
1193
+ "memory/device_reserved (GiB)": 35.89,
1194
  "memory/max_active (GiB)": 33.75,
1195
  "memory/max_allocated (GiB)": 33.75,
1196
+ "ppl": 1.02102,
1197
  "step": 85,
1198
  "tokens/total": 2569840,
1199
+ "tokens/train_per_sec_per_gpu": 34.03,
1200
  "tokens/trainable": 38789
1201
  },
1202
  {
1203
  "epoch": 0.3359375,
1204
+ "grad_norm": 0.3687220513820648,
1205
  "learning_rate": 9.66711194760312e-05,
1206
+ "loss": 0.012931328266859055,
1207
+ "memory/device_reserved (GiB)": 35.89,
1208
  "memory/max_active (GiB)": 33.86,
1209
  "memory/max_allocated (GiB)": 33.86,
1210
+ "ppl": 1.01302,
1211
  "step": 86,
1212
  "tokens/total": 2600192,
1213
+ "tokens/train_per_sec_per_gpu": 36.36,
1214
  "tokens/trainable": 39277
1215
  },
1216
  {
1217
  "epoch": 0.33984375,
1218
+ "grad_norm": 0.367418110370636,
1219
  "learning_rate": 9.656067925829593e-05,
1220
+ "loss": 0.01382569782435894,
1221
+ "memory/device_reserved (GiB)": 35.89,
1222
  "memory/max_active (GiB)": 33.93,
1223
  "memory/max_allocated (GiB)": 33.93,
1224
+ "ppl": 1.01392,
1225
  "step": 87,
1226
  "tokens/total": 2630640,
1227
+ "tokens/train_per_sec_per_gpu": 35.68,
1228
  "tokens/trainable": 39737
1229
  },
1230
  {
1231
  "epoch": 0.34375,
1232
+ "grad_norm": 0.2704487144947052,
1233
  "learning_rate": 9.644850953105288e-05,
1234
+ "loss": 0.008717816323041916,
1235
+ "memory/device_reserved (GiB)": 35.89,
1236
  "memory/max_active (GiB)": 33.74,
1237
  "memory/max_allocated (GiB)": 33.74,
1238
+ "ppl": 1.00876,
1239
  "step": 88,
1240
  "tokens/total": 2660832,
1241
+ "tokens/train_per_sec_per_gpu": 29.6,
1242
  "tokens/trainable": 40179
1243
  },
1244
  {
1245
  "epoch": 0.34765625,
1246
+ "grad_norm": 3.374918222427368,
1247
  "learning_rate": 9.633461496214225e-05,
1248
+ "loss": 0.01938408613204956,
1249
+ "memory/device_reserved (GiB)": 35.89,
1250
  "memory/max_active (GiB)": 33.88,
1251
  "memory/max_allocated (GiB)": 33.88,
1252
+ "ppl": 1.01957,
1253
  "step": 89,
1254
  "tokens/total": 2691328,
1255
+ "tokens/train_per_sec_per_gpu": 32.66,
1256
  "tokens/trainable": 40649
1257
  },
1258
  {
1259
  "epoch": 0.3515625,
1260
+ "grad_norm": 0.4690157175064087,
1261
  "learning_rate": 9.621900029118195e-05,
1262
+ "loss": 0.02013186179101467,
1263
+ "memory/device_reserved (GiB)": 35.89,
1264
  "memory/max_active (GiB)": 33.41,
1265
  "memory/max_allocated (GiB)": 33.41,
1266
+ "ppl": 1.02034,
1267
  "step": 90,
1268
  "tokens/total": 2719696,
1269
+ "tokens/train_per_sec_per_gpu": 31.46,
1270
  "tokens/trainable": 41080
1271
  },
1272
  {
1273
  "epoch": 0.35546875,
1274
+ "grad_norm": 0.08705829828977585,
1275
  "learning_rate": 9.610167032937036e-05,
1276
+ "loss": 0.002885152120143175,
1277
+ "memory/device_reserved (GiB)": 35.89,
1278
  "memory/max_active (GiB)": 33.94,
1279
  "memory/max_allocated (GiB)": 33.94,
1280
+ "ppl": 1.00289,
1281
  "step": 91,
1282
  "tokens/total": 2750272,
1283
+ "tokens/train_per_sec_per_gpu": 38.11,
1284
  "tokens/trainable": 41599
1285
  },
1286
  {
1287
  "epoch": 0.359375,
1288
+ "grad_norm": 8.197907447814941,
1289
  "learning_rate": 9.598262995928611e-05,
1290
+ "loss": 0.00822490081191063,
1291
+ "memory/device_reserved (GiB)": 35.89,
1292
  "memory/max_active (GiB)": 33.92,
1293
  "memory/max_allocated (GiB)": 33.92,
1294
+ "ppl": 1.00826,
1295
  "step": 92,
1296
  "tokens/total": 2780656,
1297
+ "tokens/train_per_sec_per_gpu": 36.82,
1298
  "tokens/trainable": 42076
1299
  },
1300
  {
1301
  "epoch": 0.36328125,
1302
+ "grad_norm": 0.14939527213573456,
1303
  "learning_rate": 9.586188413468492e-05,
1304
+ "loss": 0.004234164021909237,
1305
+ "memory/device_reserved (GiB)": 35.89,
1306
  "memory/max_active (GiB)": 33.82,
1307
  "memory/max_allocated (GiB)": 33.82,
1308
+ "ppl": 1.00424,
1309
  "step": 93,
1310
  "tokens/total": 2811088,
1311
+ "tokens/train_per_sec_per_gpu": 33.71,
1312
  "tokens/trainable": 42522
1313
  },
1314
  {
1315
  "epoch": 0.3671875,
1316
+ "grad_norm": 0.15404288470745087,
1317
  "learning_rate": 9.57394378802934e-05,
1318
+ "loss": 0.009490479715168476,
1319
  "memory/device_reserved (GiB)": 36.1,
1320
  "memory/max_active (GiB)": 33.9,
1321
  "memory/max_allocated (GiB)": 33.9,
1322
+ "ppl": 1.00954,
1323
  "step": 94,
1324
  "tokens/total": 2841520,
1325
+ "tokens/train_per_sec_per_gpu": 34.4,
1326
  "tokens/trainable": 42974
1327
  },
1328
  {
1329
  "epoch": 0.37109375,
1330
+ "grad_norm": 0.5274825692176819,
1331
  "learning_rate": 9.56152962916e-05,
1332
+ "loss": 0.02413639798760414,
1333
  "memory/device_reserved (GiB)": 36.1,
1334
  "memory/max_active (GiB)": 33.7,
1335
  "memory/max_allocated (GiB)": 33.7,
1336
+ "ppl": 1.02443,
1337
  "step": 95,
1338
  "tokens/total": 2871600,
1339
+ "tokens/train_per_sec_per_gpu": 30.67,
1340
  "tokens/trainable": 43380
1341
  },
1342
  {
1343
  "epoch": 0.375,
1344
+ "grad_norm": 0.5182522535324097,
1345
  "learning_rate": 9.548946453464296e-05,
1346
+ "loss": 0.009927366860210896,
1347
  "memory/device_reserved (GiB)": 36.1,
1348
  "memory/max_active (GiB)": 33.94,
1349
  "memory/max_allocated (GiB)": 33.94,
1350
+ "ppl": 1.00998,
1351
  "step": 96,
1352
  "tokens/total": 2902144,
1353
+ "tokens/train_per_sec_per_gpu": 34.14,
1354
  "tokens/trainable": 43865
1355
  },
1356
  {
1357
  "epoch": 0.37890625,
1358
+ "grad_norm": 0.5578822493553162,
1359
  "learning_rate": 9.53619478457953e-05,
1360
+ "loss": 0.014485684223473072,
1361
  "memory/device_reserved (GiB)": 36.1,
1362
  "memory/max_active (GiB)": 33.82,
1363
  "memory/max_allocated (GiB)": 33.82,
1364
+ "ppl": 1.01459,
1365
  "step": 97,
1366
  "tokens/total": 2932272,
1367
+ "tokens/train_per_sec_per_gpu": 31.94,
1368
  "tokens/trainable": 44328
1369
  },
1370
  {
1371
  "epoch": 0.3828125,
1372
+ "grad_norm": 0.10520734637975693,
1373
  "learning_rate": 9.523275153154695e-05,
1374
+ "loss": 0.0021552909165620804,
1375
  "memory/device_reserved (GiB)": 35.64,
1376
  "memory/max_active (GiB)": 33.69,
1377
  "memory/max_allocated (GiB)": 33.69,
1378
+ "ppl": 1.00216,
1379
  "step": 98,
1380
  "tokens/total": 2962032,
1381
+ "tokens/train_per_sec_per_gpu": 30.95,
1382
  "tokens/trainable": 44778
1383
  },
1384
  {
1385
  "epoch": 0.38671875,
1386
+ "grad_norm": 0.7544558644294739,
1387
  "learning_rate": 9.51018809682839e-05,
1388
+ "loss": 0.01703210547566414,
1389
  "memory/device_reserved (GiB)": 35.64,
1390
  "memory/max_active (GiB)": 33.89,
1391
  "memory/max_allocated (GiB)": 33.89,
1392
+ "ppl": 1.01718,
1393
  "step": 99,
1394
  "tokens/total": 2992256,
1395
+ "tokens/train_per_sec_per_gpu": 37.12,
1396
  "tokens/trainable": 45225
1397
  },
1398
  {
1399
  "epoch": 0.390625,
1400
+ "grad_norm": 0.3857012689113617,
1401
  "learning_rate": 9.49693416020645e-05,
1402
+ "loss": 0.00604570098221302,
1403
  "memory/device_reserved (GiB)": 35.64,
1404
  "memory/max_active (GiB)": 33.8,
1405
  "memory/max_allocated (GiB)": 33.8,
1406
+ "ppl": 1.00606,
1407
  "step": 100,
1408
  "tokens/total": 3022608,
1409
+ "tokens/train_per_sec_per_gpu": 35.75,
1410
  "tokens/trainable": 45678
1411
  },
1412
  {
1413
  "epoch": 0.39453125,
1414
+ "grad_norm": 0.21589453518390656,
1415
  "learning_rate": 9.483513894839276e-05,
1416
+ "loss": 0.005753816105425358,
1417
  "memory/device_reserved (GiB)": 35.78,
1418
  "memory/max_active (GiB)": 33.88,
1419
  "memory/max_allocated (GiB)": 33.88,
1420
+ "ppl": 1.00577,
1421
  "step": 101,
1422
  "tokens/total": 3052992,
1423
+ "tokens/train_per_sec_per_gpu": 32.72,
1424
  "tokens/trainable": 46125
1425
  },
1426
  {
1427
  "epoch": 0.3984375,
1428
+ "grad_norm": 1.1246687173843384,
1429
  "learning_rate": 9.469927859198888e-05,
1430
+ "loss": 0.006994037888944149,
1431
+ "memory/device_reserved (GiB)": 35.8,
1432
  "memory/max_active (GiB)": 33.98,
1433
  "memory/max_allocated (GiB)": 33.98,
1434
+ "ppl": 1.00702,
1435
  "step": 102,
1436
  "tokens/total": 3083392,
1437
+ "tokens/train_per_sec_per_gpu": 39.6,
1438
  "tokens/trainable": 46612
1439
  },
1440
  {
1441
  "epoch": 0.40234375,
1442
+ "grad_norm": 0.267713725566864,
1443
  "learning_rate": 9.456176618655689e-05,
1444
+ "loss": 0.013721915893256664,
1445
+ "memory/device_reserved (GiB)": 35.8,
1446
  "memory/max_active (GiB)": 33.83,
1447
  "memory/max_allocated (GiB)": 33.83,
1448
+ "ppl": 1.01382,
1449
  "step": 103,
1450
  "tokens/total": 3113744,
1451
+ "tokens/train_per_sec_per_gpu": 32.41,
1452
  "tokens/trainable": 47059
1453
  },
1454
  {
1455
  "epoch": 0.40625,
1456
+ "grad_norm": 0.325897753238678,
1457
  "learning_rate": 9.442260745454927e-05,
1458
+ "loss": 0.012948594987392426,
1459
+ "memory/device_reserved (GiB)": 35.8,
1460
  "memory/max_active (GiB)": 34.02,
1461
  "memory/max_allocated (GiB)": 34.02,
1462
+ "ppl": 1.01303,
1463
  "step": 104,
1464
  "tokens/total": 3144272,
1465
+ "tokens/train_per_sec_per_gpu": 33.93,
1466
  "tokens/trainable": 47536
1467
  },
1468
  {
1469
  "epoch": 0.41015625,
1470
+ "grad_norm": 0.531863808631897,
1471
  "learning_rate": 9.428180818692884e-05,
1472
+ "loss": 0.02244492992758751,
1473
+ "memory/device_reserved (GiB)": 37.62,
1474
  "memory/max_active (GiB)": 33.81,
1475
  "memory/max_allocated (GiB)": 33.81,
1476
+ "ppl": 1.0227,
1477
  "step": 105,
1478
  "tokens/total": 3174512,
1479
+ "tokens/train_per_sec_per_gpu": 33.95,
1480
  "tokens/trainable": 48037
1481
  },
1482
  {
1483
  "epoch": 0.4140625,
1484
+ "grad_norm": 0.3957497775554657,
1485
  "learning_rate": 9.413937424292791e-05,
1486
+ "loss": 0.015801995992660522,
1487
+ "memory/device_reserved (GiB)": 37.62,
1488
  "memory/max_active (GiB)": 33.91,
1489
  "memory/max_allocated (GiB)": 33.91,
1490
+ "ppl": 1.01593,
1491
  "step": 106,
1492
  "tokens/total": 3204896,
1493
+ "tokens/train_per_sec_per_gpu": 32.49,
1494
  "tokens/trainable": 48491
1495
  },
1496
  {
1497
  "epoch": 0.41796875,
1498
+ "grad_norm": 0.4241825044155121,
1499
  "learning_rate": 9.399531154980424e-05,
1500
+ "loss": 0.016320914030075073,
1501
+ "memory/device_reserved (GiB)": 37.62,
1502
  "memory/max_active (GiB)": 34.01,
1503
  "memory/max_allocated (GiB)": 34.01,
1504
+ "ppl": 1.01645,
1505
  "step": 107,
1506
  "tokens/total": 3235360,
1507
+ "tokens/train_per_sec_per_gpu": 32.89,
1508
  "tokens/trainable": 48940
1509
  },
1510
  {
1511
  "epoch": 0.421875,
1512
+ "grad_norm": 0.32064536213874817,
1513
  "learning_rate": 9.384962610259455e-05,
1514
+ "loss": 0.010785036720335484,
1515
+ "memory/device_reserved (GiB)": 37.62,
1516
  "memory/max_active (GiB)": 33.83,
1517
  "memory/max_allocated (GiB)": 33.83,
1518
+ "ppl": 1.01084,
1519
  "step": 108,
1520
  "tokens/total": 3265552,
1521
+ "tokens/train_per_sec_per_gpu": 31.52,
1522
  "tokens/trainable": 49389
1523
  },
1524
  {
1525
  "epoch": 0.42578125,
1526
+ "grad_norm": 0.17215749621391296,
1527
  "learning_rate": 9.370232396386494e-05,
1528
+ "loss": 0.00814715214073658,
1529
+ "memory/device_reserved (GiB)": 37.62,
1530
  "memory/max_active (GiB)": 33.39,
1531
  "memory/max_allocated (GiB)": 33.39,
1532
+ "ppl": 1.00818,
1533
  "step": 109,
1534
  "tokens/total": 3293856,
1535
+ "tokens/train_per_sec_per_gpu": 35.91,
1536
  "tokens/trainable": 49838
1537
  },
1538
  {
1539
  "epoch": 0.4296875,
1540
+ "grad_norm": 0.38179653882980347,
1541
  "learning_rate": 9.355341126345868e-05,
1542
+ "loss": 0.012128787115216255,
1543
+ "memory/device_reserved (GiB)": 37.62,
1544
  "memory/max_active (GiB)": 33.78,
1545
  "memory/max_allocated (GiB)": 33.78,
1546
+ "ppl": 1.0122,
1547
  "step": 110,
1548
  "tokens/total": 3323984,
1549
+ "tokens/train_per_sec_per_gpu": 33.93,
1550
  "tokens/trainable": 50310
1551
  },
1552
  {
1553
  "epoch": 0.43359375,
1554
+ "grad_norm": 0.49835413694381714,
1555
  "learning_rate": 9.340289419824107e-05,
1556
+ "loss": 0.015248995274305344,
1557
  "memory/device_reserved (GiB)": 37.62,
1558
  "memory/max_active (GiB)": 33.84,
1559
  "memory/max_allocated (GiB)": 33.84,
1560
+ "ppl": 1.01537,
1561
  "step": 111,
1562
  "tokens/total": 3354320,
1563
+ "tokens/train_per_sec_per_gpu": 33.55,
1564
  "tokens/trainable": 50755
1565
  },
1566
  {
1567
  "epoch": 0.4375,
1568
+ "grad_norm": 0.43904298543930054,
1569
  "learning_rate": 9.325077903184159e-05,
1570
+ "loss": 0.011272929608821869,
1571
  "memory/device_reserved (GiB)": 37.62,
1572
  "memory/max_active (GiB)": 33.9,
1573
  "memory/max_allocated (GiB)": 33.9,
1574
+ "ppl": 1.01134,
1575
  "step": 112,
1576
  "tokens/total": 3384880,
1577
+ "tokens/train_per_sec_per_gpu": 31.5,
1578
  "tokens/trainable": 51213
1579
  },
1580
  {
1581
  "epoch": 0.44140625,
1582
+ "grad_norm": 0.5670213103294373,
1583
  "learning_rate": 9.30970720943932e-05,
1584
+ "loss": 0.0028921598568558693,
1585
  "memory/device_reserved (GiB)": 37.62,
1586
  "memory/max_active (GiB)": 33.89,
1587
  "memory/max_allocated (GiB)": 33.89,
1588
+ "ppl": 1.0029,
1589
  "step": 113,
1590
  "tokens/total": 3415104,
1591
+ "tokens/train_per_sec_per_gpu": 33.77,
1592
  "tokens/trainable": 51660
1593
  },
1594
  {
1595
  "epoch": 0.4453125,
1596
+ "grad_norm": 0.159476637840271,
1597
  "learning_rate": 9.2941779782269e-05,
1598
+ "loss": 0.0025574099272489548,
1599
  "memory/device_reserved (GiB)": 37.62,
1600
  "memory/max_active (GiB)": 33.84,
1601
  "memory/max_allocated (GiB)": 33.84,
1602
+ "ppl": 1.00256,
1603
  "step": 114,
1604
  "tokens/total": 3445504,
1605
+ "tokens/train_per_sec_per_gpu": 32.34,
1606
  "tokens/trainable": 52133
1607
  },
1608
  {
1609
  "epoch": 0.44921875,
1610
+ "grad_norm": 0.795085608959198,
1611
  "learning_rate": 9.278490855781596e-05,
1612
+ "loss": 0.024456799030303955,
1613
  "memory/device_reserved (GiB)": 37.62,
1614
  "memory/max_active (GiB)": 33.84,
1615
  "memory/max_allocated (GiB)": 33.84,
1616
+ "ppl": 1.02476,
1617
  "step": 115,
1618
  "tokens/total": 3475744,
1619
+ "tokens/train_per_sec_per_gpu": 37.25,
1620
  "tokens/trainable": 52580
1621
  },
1622
  {
1623
  "epoch": 0.453125,
1624
+ "grad_norm": 0.38324710726737976,
1625
  "learning_rate": 9.262646494908604e-05,
1626
+ "loss": 0.004516632296144962,
1627
  "memory/device_reserved (GiB)": 37.62,
1628
  "memory/max_active (GiB)": 33.82,
1629
  "memory/max_allocated (GiB)": 33.82,
1630
+ "ppl": 1.00453,
1631
  "step": 116,
1632
  "tokens/total": 3506016,
1633
+ "tokens/train_per_sec_per_gpu": 34.74,
1634
  "tokens/trainable": 53033
1635
  },
1636
  {
1637
  "epoch": 0.45703125,
1638
+ "grad_norm": 0.3037130534648895,
1639
  "learning_rate": 9.246645554956457e-05,
1640
+ "loss": 0.021973589435219765,
1641
  "memory/device_reserved (GiB)": 37.62,
1642
  "memory/max_active (GiB)": 33.83,
1643
  "memory/max_allocated (GiB)": 33.83,
1644
+ "ppl": 1.02222,
1645
  "step": 117,
1646
  "tokens/total": 3536256,
1647
+ "tokens/train_per_sec_per_gpu": 35.22,
1648
  "tokens/trainable": 53517
1649
  },
1650
  {
1651
  "epoch": 0.4609375,
1652
+ "grad_norm": 1.3904820680618286,
1653
  "learning_rate": 9.230488701789578e-05,
1654
+ "loss": 0.009210974909365177,
1655
  "memory/device_reserved (GiB)": 37.62,
1656
  "memory/max_active (GiB)": 33.73,
1657
  "memory/max_allocated (GiB)": 33.73,
1658
+ "ppl": 1.00925,
1659
  "step": 118,
1660
  "tokens/total": 3566480,
1661
+ "tokens/train_per_sec_per_gpu": 34.59,
1662
  "tokens/trainable": 53967
1663
  },
1664
  {
1665
  "epoch": 0.46484375,
1666
+ "grad_norm": 0.1571500301361084,
1667
  "learning_rate": 9.214176607760577e-05,
1668
+ "loss": 0.0021451227366924286,
1669
  "memory/device_reserved (GiB)": 37.62,
1670
  "memory/max_active (GiB)": 33.85,
1671
  "memory/max_allocated (GiB)": 33.85,
1672
+ "ppl": 1.00215,
1673
  "step": 119,
1674
  "tokens/total": 3596624,
1675
+ "tokens/train_per_sec_per_gpu": 35.33,
1676
  "tokens/trainable": 54430
1677
  },
1678
  {
1679
  "epoch": 0.46875,
1680
+ "grad_norm": 0.39999091625213623,
1681
  "learning_rate": 9.197709951682268e-05,
1682
+ "loss": 0.00788091216236353,
1683
  "memory/device_reserved (GiB)": 37.62,
1684
  "memory/max_active (GiB)": 33.93,
1685
  "memory/max_allocated (GiB)": 33.93,
1686
+ "ppl": 1.00791,
1687
  "step": 120,
1688
  "tokens/total": 3627168,
1689
+ "tokens/train_per_sec_per_gpu": 35.32,
1690
  "tokens/trainable": 54896
1691
  },
1692
  {
1693
  "epoch": 0.47265625,
1694
+ "grad_norm": 0.38124287128448486,
1695
  "learning_rate": 9.181089418799428e-05,
1696
+ "loss": 0.010133227333426476,
1697
  "memory/device_reserved (GiB)": 37.62,
1698
  "memory/max_active (GiB)": 33.96,
1699
  "memory/max_allocated (GiB)": 33.96,
1700
+ "ppl": 1.01018,
1701
  "step": 121,
1702
  "tokens/total": 3657632,
1703
+ "tokens/train_per_sec_per_gpu": 37.76,
1704
  "tokens/trainable": 55379
1705
  },
1706
  {
1707
  "epoch": 0.4765625,
1708
+ "grad_norm": 0.0512852780520916,
1709
  "learning_rate": 9.164315700760271e-05,
1710
+ "loss": 0.0007940311916172504,
1711
  "memory/device_reserved (GiB)": 37.62,
1712
  "memory/max_active (GiB)": 33.75,
1713
  "memory/max_allocated (GiB)": 33.75,
1714
+ "ppl": 1.00079,
1715
  "step": 122,
1716
  "tokens/total": 3687824,
1717
+ "tokens/train_per_sec_per_gpu": 31.71,
1718
  "tokens/trainable": 55803
1719
  },
1720
  {
1721
  "epoch": 0.48046875,
1722
+ "grad_norm": 0.13324694335460663,
1723
  "learning_rate": 9.147389495587671e-05,
1724
+ "loss": 0.0023267122451215982,
1725
  "memory/device_reserved (GiB)": 37.62,
1726
  "memory/max_active (GiB)": 33.87,
1727
  "memory/max_allocated (GiB)": 33.87,
1728
+ "ppl": 1.00233,
1729
  "step": 123,
1730
  "tokens/total": 3718320,
1731
+ "tokens/train_per_sec_per_gpu": 36.46,
1732
  "tokens/trainable": 56249
1733
  },
1734
  {
1735
  "epoch": 0.484375,
1736
+ "grad_norm": 0.230186328291893,
1737
  "learning_rate": 9.130311507650116e-05,
1738
+ "loss": 0.0037590472493320704,
1739
  "memory/device_reserved (GiB)": 37.62,
1740
  "memory/max_active (GiB)": 33.9,
1741
  "memory/max_allocated (GiB)": 33.9,
1742
+ "ppl": 1.00377,
1743
  "step": 124,
1744
  "tokens/total": 3748672,
1745
+ "tokens/train_per_sec_per_gpu": 32.43,
1746
  "tokens/trainable": 56713
1747
  },
1748
  {
1749
  "epoch": 0.48828125,
1750
+ "grad_norm": 0.30578872561454773,
1751
  "learning_rate": 9.113082447632394e-05,
1752
+ "loss": 0.00473653944209218,
1753
  "memory/device_reserved (GiB)": 37.76,
1754
  "memory/max_active (GiB)": 33.85,
1755
  "memory/max_allocated (GiB)": 33.85,
1756
+ "ppl": 1.00475,
1757
  "step": 125,
1758
  "tokens/total": 3778976,
1759
+ "tokens/train_per_sec_per_gpu": 39.1,
1760
  "tokens/trainable": 57196
1761
  },
1762
  {
1763
  "epoch": 0.4921875,
1764
+ "grad_norm": 0.3714931607246399,
1765
  "learning_rate": 9.09570303250602e-05,
1766
+ "loss": 0.005811735987663269,
1767
  "memory/device_reserved (GiB)": 37.76,
1768
  "memory/max_active (GiB)": 33.89,
1769
  "memory/max_allocated (GiB)": 33.89,
1770
+ "ppl": 1.00583,
1771
  "step": 126,
1772
  "tokens/total": 3809296,
1773
+ "tokens/train_per_sec_per_gpu": 36.55,
1774
  "tokens/trainable": 57680
1775
  },
1776
  {
1777
  "epoch": 0.49609375,
1778
+ "grad_norm": 0.11054892838001251,
1779
  "learning_rate": 9.078173985499394e-05,
1780
+ "loss": 0.0032034481409937143,
1781
  "memory/device_reserved (GiB)": 37.76,
1782
  "memory/max_active (GiB)": 33.77,
1783
  "memory/max_allocated (GiB)": 33.77,
1784
+ "ppl": 1.00321,
1785
  "step": 127,
1786
  "tokens/total": 3839328,
1787
+ "tokens/train_per_sec_per_gpu": 31.88,
1788
  "tokens/trainable": 58110
1789
  },
1790
  {
1791
  "epoch": 0.5,
1792
+ "grad_norm": 0.09251131862401962,
1793
  "learning_rate": 9.060496036067713e-05,
1794
+ "loss": 0.001724504167214036,
1795
  "memory/device_reserved (GiB)": 37.76,
1796
  "memory/max_active (GiB)": 33.88,
1797
  "memory/max_allocated (GiB)": 33.88,
1798
+ "ppl": 1.00173,
1799
  "step": 128,
1800
  "tokens/total": 3869824,
1801
+ "tokens/train_per_sec_per_gpu": 31.74,
1802
  "tokens/trainable": 58534
1803
  }
1804
  ],
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:50621401dbfe149cde6d4a622a952bc693f8c1c46da9363b866907fa002871d4
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f45bebc7b5c77c792dbf05c98745f04359a2b3462e61a009e38cbe69159480d
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5dafd1343477b7fcb5e94b4e4a0059e8fabb1e49d85a0acc3c63a5a789736f11
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ac494318568079c3e19c022c29f18c065bb4ba6884c43ff2ef297462392b14f
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-160/trainer_state.json CHANGED
@@ -11,7 +11,7 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.00390625,
14
- "grad_norm": 0.870697021484375,
15
  "learning_rate": 0.0,
16
  "loss": 0.10251401364803314,
17
  "memory/device_reserved (GiB)": 34.94,
@@ -20,12 +20,12 @@
20
  "ppl": 1.10795,
21
  "step": 1,
22
  "tokens/total": 30464,
23
- "tokens/train_per_sec_per_gpu": 23.98,
24
  "tokens/trainable": 470
25
  },
26
  {
27
  "epoch": 0.0078125,
28
- "grad_norm": 0.8108459115028381,
29
  "learning_rate": 4.000000000000001e-06,
30
  "loss": 0.09678442776203156,
31
  "memory/device_reserved (GiB)": 35.83,
@@ -34,900 +34,900 @@
34
  "ppl": 1.10162,
35
  "step": 2,
36
  "tokens/total": 60800,
37
- "tokens/train_per_sec_per_gpu": 35.02,
38
  "tokens/trainable": 922
39
  },
40
  {
41
  "epoch": 0.01171875,
42
- "grad_norm": 1.8027335405349731,
43
  "learning_rate": 8.000000000000001e-06,
44
- "loss": 0.10952483862638474,
45
  "memory/device_reserved (GiB)": 35.85,
46
  "memory/max_active (GiB)": 33.91,
47
  "memory/max_allocated (GiB)": 33.91,
48
- "ppl": 1.11575,
49
  "step": 3,
50
  "tokens/total": 91136,
51
- "tokens/train_per_sec_per_gpu": 34.48,
52
  "tokens/trainable": 1396
53
  },
54
  {
55
  "epoch": 0.015625,
56
- "grad_norm": 1.0353018045425415,
57
  "learning_rate": 1.2e-05,
58
- "loss": 0.10303406417369843,
59
  "memory/device_reserved (GiB)": 35.85,
60
  "memory/max_active (GiB)": 33.91,
61
  "memory/max_allocated (GiB)": 33.91,
62
- "ppl": 1.10853,
63
  "step": 4,
64
  "tokens/total": 121552,
65
- "tokens/train_per_sec_per_gpu": 38.01,
66
  "tokens/trainable": 1850
67
  },
68
  {
69
  "epoch": 0.01953125,
70
- "grad_norm": 1.0778985023498535,
71
  "learning_rate": 1.6000000000000003e-05,
72
- "loss": 0.10945924371480942,
73
  "memory/device_reserved (GiB)": 35.85,
74
  "memory/max_active (GiB)": 33.98,
75
  "memory/max_allocated (GiB)": 33.98,
76
- "ppl": 1.11567,
77
  "step": 5,
78
  "tokens/total": 152304,
79
- "tokens/train_per_sec_per_gpu": 33.76,
80
  "tokens/trainable": 2302
81
  },
82
  {
83
  "epoch": 0.0234375,
84
- "grad_norm": 0.8929882645606995,
85
  "learning_rate": 2e-05,
86
- "loss": 0.08588902652263641,
87
  "memory/device_reserved (GiB)": 35.85,
88
  "memory/max_active (GiB)": 33.78,
89
  "memory/max_allocated (GiB)": 33.78,
90
- "ppl": 1.08969,
91
  "step": 6,
92
  "tokens/total": 182448,
93
- "tokens/train_per_sec_per_gpu": 31.47,
94
  "tokens/trainable": 2742
95
  },
96
  {
97
  "epoch": 0.02734375,
98
- "grad_norm": 1.6409597396850586,
99
  "learning_rate": 2.4e-05,
100
- "loss": 0.07352827489376068,
101
  "memory/device_reserved (GiB)": 35.85,
102
  "memory/max_active (GiB)": 33.85,
103
  "memory/max_allocated (GiB)": 33.85,
104
- "ppl": 1.0763,
105
  "step": 7,
106
  "tokens/total": 212736,
107
- "tokens/train_per_sec_per_gpu": 36.58,
108
  "tokens/trainable": 3208
109
  },
110
  {
111
  "epoch": 0.03125,
112
- "grad_norm": 1.5843520164489746,
113
  "learning_rate": 2.8000000000000003e-05,
114
- "loss": 0.07798244059085846,
115
  "memory/device_reserved (GiB)": 36.07,
116
  "memory/max_active (GiB)": 33.87,
117
  "memory/max_allocated (GiB)": 33.87,
118
- "ppl": 1.0811,
119
  "step": 8,
120
  "tokens/total": 243024,
121
- "tokens/train_per_sec_per_gpu": 31.83,
122
  "tokens/trainable": 3655
123
  },
124
  {
125
  "epoch": 0.03515625,
126
- "grad_norm": 1.3725916147232056,
127
  "learning_rate": 3.2000000000000005e-05,
128
- "loss": 0.04634054750204086,
129
  "memory/device_reserved (GiB)": 36.07,
130
  "memory/max_active (GiB)": 33.82,
131
  "memory/max_allocated (GiB)": 33.82,
132
- "ppl": 1.04743,
133
  "step": 9,
134
  "tokens/total": 271264,
135
- "tokens/train_per_sec_per_gpu": 35.62,
136
  "tokens/trainable": 4091
137
  },
138
  {
139
  "epoch": 0.0390625,
140
- "grad_norm": 2.544938564300537,
141
  "learning_rate": 3.6e-05,
142
- "loss": 0.08677884936332703,
143
  "memory/device_reserved (GiB)": 36.07,
144
  "memory/max_active (GiB)": 33.86,
145
  "memory/max_allocated (GiB)": 33.86,
146
- "ppl": 1.09066,
147
  "step": 10,
148
  "tokens/total": 301520,
149
- "tokens/train_per_sec_per_gpu": 34.92,
150
  "tokens/trainable": 4553
151
  },
152
  {
153
  "epoch": 0.04296875,
154
- "grad_norm": 1.6364734172821045,
155
  "learning_rate": 4e-05,
156
- "loss": 0.07754456996917725,
157
- "memory/device_reserved (GiB)": 36.07,
158
  "memory/max_active (GiB)": 33.8,
159
  "memory/max_allocated (GiB)": 33.8,
160
- "ppl": 1.08063,
161
  "step": 11,
162
  "tokens/total": 331808,
163
- "tokens/train_per_sec_per_gpu": 33.0,
164
  "tokens/trainable": 4992
165
  },
166
  {
167
  "epoch": 0.046875,
168
- "grad_norm": 2.167391538619995,
169
  "learning_rate": 4.4000000000000006e-05,
170
- "loss": 0.11765319854021072,
171
- "memory/device_reserved (GiB)": 36.07,
172
  "memory/max_active (GiB)": 33.87,
173
  "memory/max_allocated (GiB)": 33.87,
174
- "ppl": 1.12485,
175
  "step": 12,
176
  "tokens/total": 362224,
177
- "tokens/train_per_sec_per_gpu": 38.18,
178
  "tokens/trainable": 5494
179
  },
180
  {
181
  "epoch": 0.05078125,
182
- "grad_norm": 3.196911573410034,
183
  "learning_rate": 4.8e-05,
184
- "loss": 0.03510797768831253,
185
- "memory/device_reserved (GiB)": 36.15,
186
  "memory/max_active (GiB)": 33.91,
187
  "memory/max_allocated (GiB)": 33.91,
188
- "ppl": 1.03573,
189
  "step": 13,
190
  "tokens/total": 392432,
191
- "tokens/train_per_sec_per_gpu": 33.67,
192
  "tokens/trainable": 5933
193
  },
194
  {
195
  "epoch": 0.0546875,
196
- "grad_norm": 1.9654567241668701,
197
  "learning_rate": 5.2000000000000004e-05,
198
- "loss": 0.061097435653209686,
199
- "memory/device_reserved (GiB)": 36.15,
200
  "memory/max_active (GiB)": 33.79,
201
  "memory/max_allocated (GiB)": 33.79,
202
- "ppl": 1.063,
203
  "step": 14,
204
  "tokens/total": 422784,
205
- "tokens/train_per_sec_per_gpu": 29.53,
206
  "tokens/trainable": 6369
207
  },
208
  {
209
  "epoch": 0.05859375,
210
- "grad_norm": 1.934105396270752,
211
  "learning_rate": 5.6000000000000006e-05,
212
- "loss": 0.05385493487119675,
213
- "memory/device_reserved (GiB)": 36.15,
214
  "memory/max_active (GiB)": 33.92,
215
  "memory/max_allocated (GiB)": 33.92,
216
- "ppl": 1.05533,
217
  "step": 15,
218
  "tokens/total": 453376,
219
- "tokens/train_per_sec_per_gpu": 32.96,
220
  "tokens/trainable": 6820
221
  },
222
  {
223
  "epoch": 0.0625,
224
- "grad_norm": 1.4659016132354736,
225
  "learning_rate": 6e-05,
226
- "loss": 0.052153222262859344,
227
  "memory/device_reserved (GiB)": 36.21,
228
  "memory/max_active (GiB)": 33.83,
229
  "memory/max_allocated (GiB)": 33.83,
230
- "ppl": 1.05354,
231
  "step": 16,
232
  "tokens/total": 483504,
233
- "tokens/train_per_sec_per_gpu": 30.68,
234
  "tokens/trainable": 7258
235
  },
236
  {
237
  "epoch": 0.06640625,
238
- "grad_norm": 1.9650894403457642,
239
  "learning_rate": 6.400000000000001e-05,
240
- "loss": 0.05092189460992813,
241
  "memory/device_reserved (GiB)": 36.21,
242
  "memory/max_active (GiB)": 33.9,
243
  "memory/max_allocated (GiB)": 33.9,
244
- "ppl": 1.05224,
245
  "step": 17,
246
  "tokens/total": 514032,
247
- "tokens/train_per_sec_per_gpu": 35.09,
248
  "tokens/trainable": 7710
249
  },
250
  {
251
  "epoch": 0.0703125,
252
- "grad_norm": 0.6891724467277527,
253
  "learning_rate": 6.800000000000001e-05,
254
- "loss": 0.031155016273260117,
255
  "memory/device_reserved (GiB)": 36.21,
256
  "memory/max_active (GiB)": 33.83,
257
  "memory/max_allocated (GiB)": 33.83,
258
- "ppl": 1.03165,
259
  "step": 18,
260
  "tokens/total": 544432,
261
- "tokens/train_per_sec_per_gpu": 31.61,
262
  "tokens/trainable": 8174
263
  },
264
  {
265
  "epoch": 0.07421875,
266
- "grad_norm": 0.8662010431289673,
267
  "learning_rate": 7.2e-05,
268
- "loss": 0.03036138042807579,
269
  "memory/device_reserved (GiB)": 36.21,
270
  "memory/max_active (GiB)": 33.83,
271
  "memory/max_allocated (GiB)": 33.83,
272
- "ppl": 1.03083,
273
  "step": 19,
274
  "tokens/total": 574880,
275
- "tokens/train_per_sec_per_gpu": 34.37,
276
  "tokens/trainable": 8617
277
  },
278
  {
279
  "epoch": 0.078125,
280
- "grad_norm": 0.7999041080474854,
281
  "learning_rate": 7.6e-05,
282
- "loss": 0.03458942472934723,
283
  "memory/device_reserved (GiB)": 36.21,
284
  "memory/max_active (GiB)": 33.97,
285
  "memory/max_allocated (GiB)": 33.97,
286
- "ppl": 1.03519,
287
  "step": 20,
288
  "tokens/total": 605408,
289
- "tokens/train_per_sec_per_gpu": 30.32,
290
  "tokens/trainable": 9037
291
  },
292
  {
293
  "epoch": 0.08203125,
294
- "grad_norm": 0.5816919207572937,
295
  "learning_rate": 8e-05,
296
- "loss": 0.02401110902428627,
297
  "memory/device_reserved (GiB)": 36.21,
298
  "memory/max_active (GiB)": 33.77,
299
  "memory/max_allocated (GiB)": 33.77,
300
- "ppl": 1.0243,
301
  "step": 21,
302
  "tokens/total": 635584,
303
- "tokens/train_per_sec_per_gpu": 36.59,
304
  "tokens/trainable": 9503
305
  },
306
  {
307
  "epoch": 0.0859375,
308
- "grad_norm": 0.6761882901191711,
309
  "learning_rate": 8.4e-05,
310
- "loss": 0.01873329095542431,
311
  "memory/device_reserved (GiB)": 36.21,
312
  "memory/max_active (GiB)": 33.88,
313
  "memory/max_allocated (GiB)": 33.88,
314
- "ppl": 1.01891,
315
  "step": 22,
316
  "tokens/total": 665952,
317
- "tokens/train_per_sec_per_gpu": 36.71,
318
  "tokens/trainable": 9972
319
  },
320
  {
321
  "epoch": 0.08984375,
322
- "grad_norm": 0.9077537655830383,
323
  "learning_rate": 8.800000000000001e-05,
324
- "loss": 0.017490077763795853,
325
  "memory/device_reserved (GiB)": 36.21,
326
  "memory/max_active (GiB)": 33.9,
327
  "memory/max_allocated (GiB)": 33.9,
328
- "ppl": 1.01764,
329
  "step": 23,
330
  "tokens/total": 696240,
331
- "tokens/train_per_sec_per_gpu": 37.39,
332
  "tokens/trainable": 10447
333
  },
334
  {
335
  "epoch": 0.09375,
336
- "grad_norm": 1.3226462602615356,
337
  "learning_rate": 9.200000000000001e-05,
338
- "loss": 0.028416279703378677,
339
  "memory/device_reserved (GiB)": 36.35,
340
  "memory/max_active (GiB)": 33.88,
341
  "memory/max_allocated (GiB)": 33.88,
342
- "ppl": 1.02882,
343
  "step": 24,
344
  "tokens/total": 726464,
345
- "tokens/train_per_sec_per_gpu": 37.12,
346
  "tokens/trainable": 10899
347
  },
348
  {
349
  "epoch": 0.09765625,
350
- "grad_norm": 0.9712215065956116,
351
  "learning_rate": 9.6e-05,
352
- "loss": 0.025973526760935783,
353
  "memory/device_reserved (GiB)": 36.35,
354
  "memory/max_active (GiB)": 33.84,
355
  "memory/max_allocated (GiB)": 33.84,
356
- "ppl": 1.02631,
357
  "step": 25,
358
  "tokens/total": 756704,
359
- "tokens/train_per_sec_per_gpu": 32.97,
360
  "tokens/trainable": 11360
361
  },
362
  {
363
  "epoch": 0.1015625,
364
- "grad_norm": 0.8638900518417358,
365
  "learning_rate": 0.0001,
366
- "loss": 0.022434517741203308,
367
  "memory/device_reserved (GiB)": 36.35,
368
  "memory/max_active (GiB)": 33.82,
369
  "memory/max_allocated (GiB)": 33.82,
370
- "ppl": 1.02269,
371
  "step": 26,
372
  "tokens/total": 786912,
373
- "tokens/train_per_sec_per_gpu": 29.23,
374
  "tokens/trainable": 11775
375
  },
376
  {
377
  "epoch": 0.10546875,
378
- "grad_norm": 0.6629000902175903,
379
  "learning_rate": 9.99990636831587e-05,
380
- "loss": 0.014538668096065521,
381
  "memory/device_reserved (GiB)": 36.35,
382
  "memory/max_active (GiB)": 33.46,
383
  "memory/max_allocated (GiB)": 33.46,
384
- "ppl": 1.01464,
385
  "step": 27,
386
  "tokens/total": 815424,
387
- "tokens/train_per_sec_per_gpu": 39.82,
388
  "tokens/trainable": 12242
389
  },
390
  {
391
  "epoch": 0.109375,
392
- "grad_norm": 0.3078136146068573,
393
  "learning_rate": 9.999625477159879e-05,
394
- "loss": 0.01195458322763443,
395
  "memory/device_reserved (GiB)": 36.35,
396
  "memory/max_active (GiB)": 33.89,
397
  "memory/max_allocated (GiB)": 33.89,
398
- "ppl": 1.01203,
399
  "step": 28,
400
  "tokens/total": 845584,
401
- "tokens/train_per_sec_per_gpu": 33.52,
402
  "tokens/trainable": 12696
403
  },
404
  {
405
  "epoch": 0.11328125,
406
- "grad_norm": 1.1301876306533813,
407
  "learning_rate": 9.999157338221051e-05,
408
- "loss": 0.022538531571626663,
409
  "memory/device_reserved (GiB)": 36.35,
410
  "memory/max_active (GiB)": 33.87,
411
  "memory/max_allocated (GiB)": 33.87,
412
- "ppl": 1.02279,
413
  "step": 29,
414
  "tokens/total": 875808,
415
- "tokens/train_per_sec_per_gpu": 34.38,
416
  "tokens/trainable": 13153
417
  },
418
  {
419
  "epoch": 0.1171875,
420
- "grad_norm": 0.41090911626815796,
421
  "learning_rate": 9.998501970980562e-05,
422
- "loss": 0.011871461756527424,
423
  "memory/device_reserved (GiB)": 36.35,
424
  "memory/max_active (GiB)": 33.82,
425
  "memory/max_allocated (GiB)": 33.82,
426
- "ppl": 1.01194,
427
  "step": 30,
428
  "tokens/total": 904096,
429
- "tokens/train_per_sec_per_gpu": 39.83,
430
  "tokens/trainable": 13624
431
  },
432
  {
433
  "epoch": 0.12109375,
434
- "grad_norm": 0.6772723197937012,
435
  "learning_rate": 9.997659402710915e-05,
436
- "loss": 0.013700846582651138,
437
  "memory/device_reserved (GiB)": 36.35,
438
  "memory/max_active (GiB)": 33.86,
439
  "memory/max_allocated (GiB)": 33.86,
440
- "ppl": 1.0138,
441
  "step": 31,
442
  "tokens/total": 934400,
443
- "tokens/train_per_sec_per_gpu": 34.07,
444
  "tokens/trainable": 14064
445
  },
446
  {
447
  "epoch": 0.125,
448
- "grad_norm": 1.207811951637268,
449
  "learning_rate": 9.996629668474818e-05,
450
- "loss": 0.01185896061360836,
451
  "memory/device_reserved (GiB)": 36.35,
452
  "memory/max_active (GiB)": 33.92,
453
  "memory/max_allocated (GiB)": 33.92,
454
- "ppl": 1.01193,
455
  "step": 32,
456
  "tokens/total": 964832,
457
- "tokens/train_per_sec_per_gpu": 38.9,
458
  "tokens/trainable": 14565
459
  },
460
  {
461
  "epoch": 0.12890625,
462
- "grad_norm": 0.46513956785202026,
463
  "learning_rate": 9.995412811123711e-05,
464
- "loss": 0.012477721087634563,
465
  "memory/device_reserved (GiB)": 36.35,
466
  "memory/max_active (GiB)": 33.79,
467
  "memory/max_allocated (GiB)": 33.79,
468
- "ppl": 1.01256,
469
  "step": 33,
470
  "tokens/total": 995040,
471
- "tokens/train_per_sec_per_gpu": 33.76,
472
  "tokens/trainable": 15005
473
  },
474
  {
475
  "epoch": 0.1328125,
476
- "grad_norm": 1.7668761014938354,
477
  "learning_rate": 9.994008881295999e-05,
478
- "loss": 0.03285093232989311,
479
  "memory/device_reserved (GiB)": 34.72,
480
  "memory/max_active (GiB)": 33.64,
481
  "memory/max_allocated (GiB)": 33.64,
482
- "ppl": 1.0334,
483
  "step": 34,
484
  "tokens/total": 1024896,
485
- "tokens/train_per_sec_per_gpu": 32.05,
486
  "tokens/trainable": 15459
487
  },
488
  {
489
  "epoch": 0.13671875,
490
- "grad_norm": 0.7816088795661926,
491
  "learning_rate": 9.992417937414932e-05,
492
- "loss": 0.028746310621500015,
493
  "memory/device_reserved (GiB)": 35.48,
494
  "memory/max_active (GiB)": 33.84,
495
  "memory/max_allocated (GiB)": 33.84,
496
- "ppl": 1.02916,
497
  "step": 35,
498
  "tokens/total": 1054992,
499
- "tokens/train_per_sec_per_gpu": 38.15,
500
  "tokens/trainable": 15960
501
  },
502
  {
503
  "epoch": 0.140625,
504
- "grad_norm": 0.683965265750885,
505
  "learning_rate": 9.99064004568618e-05,
506
- "loss": 0.020058901980519295,
507
  "memory/device_reserved (GiB)": 35.48,
508
  "memory/max_active (GiB)": 33.93,
509
  "memory/max_allocated (GiB)": 33.93,
510
- "ppl": 1.02026,
511
  "step": 36,
512
  "tokens/total": 1085280,
513
- "tokens/train_per_sec_per_gpu": 34.53,
514
  "tokens/trainable": 16428
515
  },
516
  {
517
  "epoch": 0.14453125,
518
- "grad_norm": 0.6797531843185425,
519
  "learning_rate": 9.988675280095074e-05,
520
- "loss": 0.010446591302752495,
521
- "memory/device_reserved (GiB)": 35.48,
522
  "memory/max_active (GiB)": 33.7,
523
  "memory/max_allocated (GiB)": 33.7,
524
- "ppl": 1.0105,
525
  "step": 37,
526
  "tokens/total": 1115504,
527
- "tokens/train_per_sec_per_gpu": 31.77,
528
  "tokens/trainable": 16884
529
  },
530
  {
531
  "epoch": 0.1484375,
532
- "grad_norm": 0.8899193406105042,
533
  "learning_rate": 9.986523722403528e-05,
534
- "loss": 0.017391683533787727,
535
- "memory/device_reserved (GiB)": 35.48,
536
  "memory/max_active (GiB)": 33.82,
537
  "memory/max_allocated (GiB)": 33.82,
538
- "ppl": 1.01754,
539
  "step": 38,
540
  "tokens/total": 1145712,
541
- "tokens/train_per_sec_per_gpu": 32.03,
542
  "tokens/trainable": 17341
543
  },
544
  {
545
  "epoch": 0.15234375,
546
- "grad_norm": 0.8132575750350952,
547
  "learning_rate": 9.984185462146642e-05,
548
- "loss": 0.02252483367919922,
549
- "memory/device_reserved (GiB)": 35.48,
550
  "memory/max_active (GiB)": 33.81,
551
  "memory/max_allocated (GiB)": 33.81,
552
- "ppl": 1.02278,
553
  "step": 39,
554
  "tokens/total": 1176128,
555
- "tokens/train_per_sec_per_gpu": 34.15,
556
  "tokens/trainable": 17786
557
  },
558
  {
559
  "epoch": 0.15625,
560
- "grad_norm": 0.4430502653121948,
561
  "learning_rate": 9.98166059662897e-05,
562
- "loss": 0.011966042220592499,
563
- "memory/device_reserved (GiB)": 35.48,
564
  "memory/max_active (GiB)": 33.85,
565
  "memory/max_allocated (GiB)": 33.85,
566
- "ppl": 1.01204,
567
  "step": 40,
568
  "tokens/total": 1206576,
569
- "tokens/train_per_sec_per_gpu": 34.99,
570
  "tokens/trainable": 18262
571
  },
572
  {
573
  "epoch": 0.16015625,
574
- "grad_norm": 0.5074653029441833,
575
  "learning_rate": 9.978949230920472e-05,
576
- "loss": 0.014877484180033207,
577
- "memory/device_reserved (GiB)": 35.87,
578
  "memory/max_active (GiB)": 33.86,
579
  "memory/max_allocated (GiB)": 33.86,
580
- "ppl": 1.01499,
581
  "step": 41,
582
  "tokens/total": 1236848,
583
- "tokens/train_per_sec_per_gpu": 32.7,
584
  "tokens/trainable": 18716
585
  },
586
  {
587
  "epoch": 0.1640625,
588
- "grad_norm": 0.5221464037895203,
589
  "learning_rate": 9.976051477852141e-05,
590
- "loss": 0.017510797828435898,
591
- "memory/device_reserved (GiB)": 35.87,
592
  "memory/max_active (GiB)": 33.83,
593
  "memory/max_allocated (GiB)": 33.83,
594
- "ppl": 1.01767,
595
  "step": 42,
596
  "tokens/total": 1267088,
597
- "tokens/train_per_sec_per_gpu": 32.87,
598
  "tokens/trainable": 19157
599
  },
600
  {
601
  "epoch": 0.16796875,
602
- "grad_norm": 0.6888790130615234,
603
  "learning_rate": 9.972967458011312e-05,
604
- "loss": 0.030433066189289093,
605
- "memory/device_reserved (GiB)": 35.87,
606
  "memory/max_active (GiB)": 33.82,
607
  "memory/max_allocated (GiB)": 33.82,
608
- "ppl": 1.0309,
609
  "step": 43,
610
  "tokens/total": 1297360,
611
- "tokens/train_per_sec_per_gpu": 36.5,
612
  "tokens/trainable": 19647
613
  },
614
  {
615
  "epoch": 0.171875,
616
- "grad_norm": 0.5600388050079346,
617
  "learning_rate": 9.96969729973664e-05,
618
- "loss": 0.013720017857849598,
619
- "memory/device_reserved (GiB)": 35.87,
620
  "memory/max_active (GiB)": 33.86,
621
  "memory/max_allocated (GiB)": 33.86,
622
- "ppl": 1.01381,
623
  "step": 44,
624
  "tokens/total": 1327744,
625
- "tokens/train_per_sec_per_gpu": 34.9,
626
  "tokens/trainable": 20119
627
  },
628
  {
629
  "epoch": 0.17578125,
630
- "grad_norm": 0.4291926324367523,
631
  "learning_rate": 9.966241139112754e-05,
632
- "loss": 0.00872582383453846,
633
- "memory/device_reserved (GiB)": 35.87,
634
  "memory/max_active (GiB)": 33.86,
635
  "memory/max_allocated (GiB)": 33.86,
636
- "ppl": 1.00876,
637
  "step": 45,
638
  "tokens/total": 1358096,
639
- "tokens/train_per_sec_per_gpu": 32.92,
640
  "tokens/trainable": 20560
641
  },
642
  {
643
  "epoch": 0.1796875,
644
- "grad_norm": 0.944327712059021,
645
  "learning_rate": 9.96259911996461e-05,
646
- "loss": 0.025248996913433075,
647
- "memory/device_reserved (GiB)": 35.87,
648
  "memory/max_active (GiB)": 33.84,
649
  "memory/max_allocated (GiB)": 33.84,
650
- "ppl": 1.02557,
651
  "step": 46,
652
  "tokens/total": 1388240,
653
- "tokens/train_per_sec_per_gpu": 32.59,
654
  "tokens/trainable": 20992
655
  },
656
  {
657
  "epoch": 0.18359375,
658
- "grad_norm": 0.2425016313791275,
659
  "learning_rate": 9.958771393851491e-05,
660
- "loss": 0.005067020654678345,
661
- "memory/device_reserved (GiB)": 35.87,
662
  "memory/max_active (GiB)": 33.26,
663
  "memory/max_allocated (GiB)": 33.26,
664
- "ppl": 1.00508,
665
  "step": 47,
666
  "tokens/total": 1416240,
667
- "tokens/train_per_sec_per_gpu": 31.51,
668
  "tokens/trainable": 21441
669
  },
670
  {
671
  "epoch": 0.1875,
672
- "grad_norm": 1.2363684177398682,
673
  "learning_rate": 9.954758120060702e-05,
674
- "loss": 0.03300227224826813,
675
  "memory/device_reserved (GiB)": 35.88,
676
  "memory/max_active (GiB)": 33.96,
677
  "memory/max_allocated (GiB)": 33.96,
678
- "ppl": 1.03355,
679
  "step": 48,
680
  "tokens/total": 1446880,
681
- "tokens/train_per_sec_per_gpu": 33.7,
682
  "tokens/trainable": 21901
683
  },
684
  {
685
  "epoch": 0.19140625,
686
- "grad_norm": 0.7278413772583008,
687
  "learning_rate": 9.950559465600948e-05,
688
- "loss": 0.025975672528147697,
689
  "memory/device_reserved (GiB)": 35.88,
690
  "memory/max_active (GiB)": 33.78,
691
  "memory/max_allocated (GiB)": 33.78,
692
- "ppl": 1.02632,
693
  "step": 49,
694
  "tokens/total": 1477168,
695
- "tokens/train_per_sec_per_gpu": 33.27,
696
  "tokens/trainable": 22341
697
  },
698
  {
699
  "epoch": 0.1953125,
700
- "grad_norm": 0.37237733602523804,
701
  "learning_rate": 9.946175605195379e-05,
702
- "loss": 0.010315775871276855,
703
  "memory/device_reserved (GiB)": 35.88,
704
  "memory/max_active (GiB)": 33.89,
705
  "memory/max_allocated (GiB)": 33.89,
706
- "ppl": 1.01037,
707
  "step": 50,
708
  "tokens/total": 1507712,
709
- "tokens/train_per_sec_per_gpu": 33.07,
710
  "tokens/trainable": 22833
711
  },
712
  {
713
  "epoch": 0.19921875,
714
- "grad_norm": 0.25258293747901917,
715
  "learning_rate": 9.941606721274322e-05,
716
- "loss": 0.00485712755471468,
717
  "memory/device_reserved (GiB)": 35.88,
718
  "memory/max_active (GiB)": 33.81,
719
  "memory/max_allocated (GiB)": 33.81,
720
- "ppl": 1.00487,
721
  "step": 51,
722
  "tokens/total": 1537936,
723
- "tokens/train_per_sec_per_gpu": 30.64,
724
  "tokens/trainable": 23277
725
  },
726
  {
727
  "epoch": 0.203125,
728
- "grad_norm": 0.738599419593811,
729
  "learning_rate": 9.936853003967685e-05,
730
- "loss": 0.01646406576037407,
731
  "memory/device_reserved (GiB)": 35.88,
732
  "memory/max_active (GiB)": 33.84,
733
  "memory/max_allocated (GiB)": 33.84,
734
- "ppl": 1.0166,
735
  "step": 52,
736
  "tokens/total": 1568352,
737
- "tokens/train_per_sec_per_gpu": 35.57,
738
  "tokens/trainable": 23759
739
  },
740
  {
741
  "epoch": 0.20703125,
742
- "grad_norm": 1.2733500003814697,
743
  "learning_rate": 9.93191465109705e-05,
744
- "loss": 0.019196398556232452,
745
  "memory/device_reserved (GiB)": 35.88,
746
  "memory/max_active (GiB)": 33.91,
747
  "memory/max_allocated (GiB)": 33.91,
748
- "ppl": 1.01938,
749
  "step": 53,
750
  "tokens/total": 1598992,
751
- "tokens/train_per_sec_per_gpu": 32.98,
752
  "tokens/trainable": 24193
753
  },
754
  {
755
  "epoch": 0.2109375,
756
- "grad_norm": 0.5316427946090698,
757
  "learning_rate": 9.926791868167438e-05,
758
- "loss": 0.006890955846756697,
759
  "memory/device_reserved (GiB)": 35.9,
760
  "memory/max_active (GiB)": 33.95,
761
  "memory/max_allocated (GiB)": 33.95,
762
- "ppl": 1.00691,
763
  "step": 54,
764
  "tokens/total": 1629488,
765
- "tokens/train_per_sec_per_gpu": 33.47,
766
  "tokens/trainable": 24619
767
  },
768
  {
769
  "epoch": 0.21484375,
770
- "grad_norm": 0.6924111843109131,
771
  "learning_rate": 9.921484868358753e-05,
772
- "loss": 0.021178584545850754,
773
  "memory/device_reserved (GiB)": 35.9,
774
  "memory/max_active (GiB)": 33.8,
775
  "memory/max_allocated (GiB)": 33.8,
776
- "ppl": 1.0214,
777
  "step": 55,
778
  "tokens/total": 1659696,
779
- "tokens/train_per_sec_per_gpu": 32.8,
780
  "tokens/trainable": 25075
781
  },
782
  {
783
  "epoch": 0.21875,
784
- "grad_norm": 0.683601975440979,
785
  "learning_rate": 9.915993872516924e-05,
786
- "loss": 0.017589068040251732,
787
  "memory/device_reserved (GiB)": 35.9,
788
  "memory/max_active (GiB)": 33.76,
789
  "memory/max_allocated (GiB)": 33.76,
790
- "ppl": 1.01774,
791
  "step": 56,
792
  "tokens/total": 1689872,
793
- "tokens/train_per_sec_per_gpu": 31.48,
794
  "tokens/trainable": 25519
795
  },
796
  {
797
  "epoch": 0.22265625,
798
- "grad_norm": 0.8593301177024841,
799
  "learning_rate": 9.9103191091447e-05,
800
- "loss": 0.025561584159731865,
801
  "memory/device_reserved (GiB)": 35.9,
802
  "memory/max_active (GiB)": 33.9,
803
  "memory/max_allocated (GiB)": 33.9,
804
- "ppl": 1.02589,
805
  "step": 57,
806
  "tokens/total": 1720144,
807
- "tokens/train_per_sec_per_gpu": 32.63,
808
  "tokens/trainable": 25966
809
  },
810
  {
811
  "epoch": 0.2265625,
812
- "grad_norm": 0.2925783097743988,
813
  "learning_rate": 9.904460814392147e-05,
814
- "loss": 0.005790311843156815,
815
  "memory/device_reserved (GiB)": 35.9,
816
  "memory/max_active (GiB)": 33.89,
817
  "memory/max_allocated (GiB)": 33.89,
818
- "ppl": 1.00581,
819
  "step": 58,
820
  "tokens/total": 1750448,
821
- "tokens/train_per_sec_per_gpu": 33.89,
822
  "tokens/trainable": 26423
823
  },
824
  {
825
  "epoch": 0.23046875,
826
- "grad_norm": 0.6059477925300598,
827
  "learning_rate": 9.898419232046825e-05,
828
- "loss": 0.007334758993238211,
829
  "memory/device_reserved (GiB)": 35.9,
830
  "memory/max_active (GiB)": 33.93,
831
  "memory/max_allocated (GiB)": 33.93,
832
- "ppl": 1.00736,
833
  "step": 59,
834
  "tokens/total": 1781088,
835
- "tokens/train_per_sec_per_gpu": 38.42,
836
  "tokens/trainable": 26934
837
  },
838
  {
839
  "epoch": 0.234375,
840
- "grad_norm": 0.29425033926963806,
841
  "learning_rate": 9.892194613523633e-05,
842
- "loss": 0.004620288498699665,
843
  "memory/device_reserved (GiB)": 35.9,
844
  "memory/max_active (GiB)": 33.84,
845
  "memory/max_allocated (GiB)": 33.84,
846
- "ppl": 1.00463,
847
  "step": 60,
848
  "tokens/total": 1811344,
849
- "tokens/train_per_sec_per_gpu": 33.0,
850
  "tokens/trainable": 27393
851
  },
852
  {
853
  "epoch": 0.23828125,
854
- "grad_norm": 0.25868093967437744,
855
  "learning_rate": 9.885787217854357e-05,
856
- "loss": 0.0042824773117899895,
857
  "memory/device_reserved (GiB)": 35.96,
858
  "memory/max_active (GiB)": 33.9,
859
  "memory/max_allocated (GiB)": 33.9,
860
- "ppl": 1.00429,
861
  "step": 61,
862
  "tokens/total": 1841600,
863
- "tokens/train_per_sec_per_gpu": 32.84,
864
  "tokens/trainable": 27852
865
  },
866
  {
867
  "epoch": 0.2421875,
868
- "grad_norm": 0.9455181360244751,
869
  "learning_rate": 9.879197311676887e-05,
870
- "loss": 0.013508133590221405,
871
  "memory/device_reserved (GiB)": 35.96,
872
  "memory/max_active (GiB)": 33.82,
873
  "memory/max_allocated (GiB)": 33.82,
874
- "ppl": 1.0136,
875
  "step": 62,
876
  "tokens/total": 1872048,
877
- "tokens/train_per_sec_per_gpu": 32.92,
878
  "tokens/trainable": 28292
879
  },
880
  {
881
  "epoch": 0.24609375,
882
- "grad_norm": 1.149442434310913,
883
  "learning_rate": 9.872425169224113e-05,
884
- "loss": 0.020864056423306465,
885
  "memory/device_reserved (GiB)": 35.96,
886
  "memory/max_active (GiB)": 33.94,
887
  "memory/max_allocated (GiB)": 33.94,
888
- "ppl": 1.02108,
889
  "step": 63,
890
  "tokens/total": 1902592,
891
- "tokens/train_per_sec_per_gpu": 37.27,
892
  "tokens/trainable": 28798
893
  },
894
  {
895
  "epoch": 0.25,
896
- "grad_norm": 0.7421550750732422,
897
  "learning_rate": 9.865471072312528e-05,
898
- "loss": 0.016041038557887077,
899
  "memory/device_reserved (GiB)": 35.96,
900
  "memory/max_active (GiB)": 33.96,
901
  "memory/max_allocated (GiB)": 33.96,
902
- "ppl": 1.01617,
903
  "step": 64,
904
  "tokens/total": 1933088,
905
- "tokens/train_per_sec_per_gpu": 35.76,
906
  "tokens/trainable": 29283
907
  },
908
  {
909
  "epoch": 0.25390625,
910
- "grad_norm": 0.36109936237335205,
911
  "learning_rate": 9.858335310330492e-05,
912
- "loss": 0.005372575484216213,
913
  "memory/device_reserved (GiB)": 35.96,
914
  "memory/max_active (GiB)": 33.72,
915
  "memory/max_allocated (GiB)": 33.72,
916
- "ppl": 1.00539,
917
  "step": 65,
918
  "tokens/total": 1963296,
919
- "tokens/train_per_sec_per_gpu": 31.92,
920
  "tokens/trainable": 29745
921
  },
922
  {
923
  "epoch": 0.2578125,
924
- "grad_norm": 0.7074101567268372,
925
  "learning_rate": 9.851018180226185e-05,
926
- "loss": 0.018492329865694046,
927
- "memory/device_reserved (GiB)": 34.88,
928
  "memory/max_active (GiB)": 33.79,
929
  "memory/max_allocated (GiB)": 33.79,
930
- "ppl": 1.01866,
931
  "step": 66,
932
  "tokens/total": 1993760,
933
  "tokens/train_per_sec_per_gpu": 31.19,
@@ -935,1007 +935,1007 @@
935
  },
936
  {
937
  "epoch": 0.26171875,
938
- "grad_norm": 0.43113431334495544,
939
  "learning_rate": 9.843519986495259e-05,
940
- "loss": 0.00620780885219574,
941
  "memory/device_reserved (GiB)": 35.86,
942
  "memory/max_active (GiB)": 33.89,
943
  "memory/max_allocated (GiB)": 33.89,
944
- "ppl": 1.00623,
945
  "step": 67,
946
  "tokens/total": 2024144,
947
- "tokens/train_per_sec_per_gpu": 33.22,
948
  "tokens/trainable": 30622
949
  },
950
  {
951
  "epoch": 0.265625,
952
- "grad_norm": 0.3996214270591736,
953
  "learning_rate": 9.835841041168162e-05,
954
- "loss": 0.005429963115602732,
955
  "memory/device_reserved (GiB)": 35.86,
956
  "memory/max_active (GiB)": 33.95,
957
  "memory/max_allocated (GiB)": 33.95,
958
- "ppl": 1.00544,
959
  "step": 68,
960
  "tokens/total": 2054608,
961
- "tokens/train_per_sec_per_gpu": 35.38,
962
  "tokens/trainable": 31097
963
  },
964
  {
965
  "epoch": 0.26953125,
966
- "grad_norm": 0.4444175660610199,
967
  "learning_rate": 9.82798166379715e-05,
968
- "loss": 0.01158289983868599,
969
  "memory/device_reserved (GiB)": 35.86,
970
  "memory/max_active (GiB)": 33.82,
971
  "memory/max_allocated (GiB)": 33.82,
972
- "ppl": 1.01165,
973
  "step": 69,
974
  "tokens/total": 2084912,
975
- "tokens/train_per_sec_per_gpu": 36.53,
976
  "tokens/trainable": 31595
977
  },
978
  {
979
  "epoch": 0.2734375,
980
- "grad_norm": 0.16977320611476898,
981
  "learning_rate": 9.819942181443002e-05,
982
- "loss": 0.002158569637686014,
983
  "memory/device_reserved (GiB)": 35.86,
984
  "memory/max_active (GiB)": 33.84,
985
  "memory/max_allocated (GiB)": 33.84,
986
- "ppl": 1.00216,
987
  "step": 70,
988
  "tokens/total": 2115216,
989
- "tokens/train_per_sec_per_gpu": 30.67,
990
  "tokens/trainable": 32036
991
  },
992
  {
993
  "epoch": 0.27734375,
994
- "grad_norm": 0.12970401346683502,
995
  "learning_rate": 9.811722928661392e-05,
996
- "loss": 0.0028989531565457582,
997
  "memory/device_reserved (GiB)": 35.86,
998
  "memory/max_active (GiB)": 33.9,
999
  "memory/max_allocated (GiB)": 33.9,
1000
- "ppl": 1.0029,
1001
  "step": 71,
1002
  "tokens/total": 2145424,
1003
- "tokens/train_per_sec_per_gpu": 28.06,
1004
  "tokens/trainable": 32428
1005
  },
1006
  {
1007
  "epoch": 0.28125,
1008
- "grad_norm": 0.06490105390548706,
1009
  "learning_rate": 9.803324247488975e-05,
1010
- "loss": 0.0008802044321782887,
1011
  "memory/device_reserved (GiB)": 35.88,
1012
  "memory/max_active (GiB)": 33.84,
1013
  "memory/max_allocated (GiB)": 33.84,
1014
- "ppl": 1.00088,
1015
  "step": 72,
1016
  "tokens/total": 2175648,
1017
- "tokens/train_per_sec_per_gpu": 32.08,
1018
  "tokens/trainable": 32880
1019
  },
1020
  {
1021
  "epoch": 0.28515625,
1022
- "grad_norm": 0.20680083334445953,
1023
  "learning_rate": 9.794746487429161e-05,
1024
- "loss": 0.0019504247466102242,
1025
  "memory/device_reserved (GiB)": 35.88,
1026
  "memory/max_active (GiB)": 33.79,
1027
  "memory/max_allocated (GiB)": 33.79,
1028
- "ppl": 1.00195,
1029
  "step": 73,
1030
  "tokens/total": 2205856,
1031
- "tokens/train_per_sec_per_gpu": 33.48,
1032
  "tokens/trainable": 33323
1033
  },
1034
  {
1035
  "epoch": 0.2890625,
1036
- "grad_norm": 1.6840267181396484,
1037
  "learning_rate": 9.785990005437554e-05,
1038
- "loss": 0.02435958757996559,
1039
  "memory/device_reserved (GiB)": 35.88,
1040
  "memory/max_active (GiB)": 33.95,
1041
  "memory/max_allocated (GiB)": 33.95,
1042
- "ppl": 1.02466,
1043
  "step": 74,
1044
  "tokens/total": 2236352,
1045
- "tokens/train_per_sec_per_gpu": 35.84,
1046
  "tokens/trainable": 33792
1047
  },
1048
  {
1049
  "epoch": 0.29296875,
1050
- "grad_norm": 0.6665006875991821,
1051
  "learning_rate": 9.777055165907117e-05,
1052
- "loss": 0.018271014094352722,
1053
  "memory/device_reserved (GiB)": 35.88,
1054
  "memory/max_active (GiB)": 33.8,
1055
  "memory/max_allocated (GiB)": 33.8,
1056
- "ppl": 1.01844,
1057
  "step": 75,
1058
  "tokens/total": 2266480,
1059
- "tokens/train_per_sec_per_gpu": 35.24,
1060
  "tokens/trainable": 34223
1061
  },
1062
  {
1063
  "epoch": 0.296875,
1064
- "grad_norm": 0.6469210982322693,
1065
  "learning_rate": 9.767942340652993e-05,
1066
- "loss": 0.023723380640149117,
1067
  "memory/device_reserved (GiB)": 35.88,
1068
  "memory/max_active (GiB)": 33.71,
1069
  "memory/max_allocated (GiB)": 33.71,
1070
- "ppl": 1.02401,
1071
  "step": 76,
1072
  "tokens/total": 2296496,
1073
- "tokens/train_per_sec_per_gpu": 29.59,
1074
  "tokens/trainable": 34649
1075
  },
1076
  {
1077
  "epoch": 0.30078125,
1078
- "grad_norm": 1.391882300376892,
1079
  "learning_rate": 9.758651908897035e-05,
1080
- "loss": 0.015201358124613762,
1081
  "memory/device_reserved (GiB)": 35.88,
1082
  "memory/max_active (GiB)": 33.88,
1083
  "memory/max_allocated (GiB)": 33.88,
1084
- "ppl": 1.01532,
1085
  "step": 77,
1086
  "tokens/total": 2327040,
1087
- "tokens/train_per_sec_per_gpu": 35.58,
1088
  "tokens/trainable": 35094
1089
  },
1090
  {
1091
  "epoch": 0.3046875,
1092
- "grad_norm": 0.5752553343772888,
1093
  "learning_rate": 9.749184257252033e-05,
1094
- "loss": 0.020247019827365875,
1095
  "memory/device_reserved (GiB)": 35.88,
1096
  "memory/max_active (GiB)": 33.81,
1097
  "memory/max_allocated (GiB)": 33.81,
1098
- "ppl": 1.02045,
1099
  "step": 78,
1100
  "tokens/total": 2357328,
1101
- "tokens/train_per_sec_per_gpu": 31.79,
1102
  "tokens/trainable": 35534
1103
  },
1104
  {
1105
  "epoch": 0.30859375,
1106
- "grad_norm": 0.276022732257843,
1107
  "learning_rate": 9.739539779705614e-05,
1108
- "loss": 0.010471204295754433,
1109
- "memory/device_reserved (GiB)": 35.88,
1110
  "memory/max_active (GiB)": 33.85,
1111
  "memory/max_allocated (GiB)": 33.85,
1112
- "ppl": 1.01053,
1113
  "step": 79,
1114
  "tokens/total": 2387664,
1115
- "tokens/train_per_sec_per_gpu": 35.76,
1116
  "tokens/trainable": 36029
1117
  },
1118
  {
1119
  "epoch": 0.3125,
1120
- "grad_norm": 0.41784003376960754,
1121
  "learning_rate": 9.729718877603861e-05,
1122
- "loss": 0.010774495080113411,
1123
- "memory/device_reserved (GiB)": 35.88,
1124
  "memory/max_active (GiB)": 33.96,
1125
  "memory/max_allocated (GiB)": 33.96,
1126
- "ppl": 1.01083,
1127
  "step": 80,
1128
  "tokens/total": 2418000,
1129
- "tokens/train_per_sec_per_gpu": 35.55,
1130
  "tokens/trainable": 36469
1131
  },
1132
  {
1133
  "epoch": 0.31640625,
1134
- "grad_norm": 0.4906325340270996,
1135
  "learning_rate": 9.719721959634592e-05,
1136
- "loss": 0.015422273427248001,
1137
- "memory/device_reserved (GiB)": 35.88,
1138
  "memory/max_active (GiB)": 33.96,
1139
  "memory/max_allocated (GiB)": 33.96,
1140
- "ppl": 1.01554,
1141
  "step": 81,
1142
  "tokens/total": 2448720,
1143
- "tokens/train_per_sec_per_gpu": 30.69,
1144
  "tokens/trainable": 36906
1145
  },
1146
  {
1147
  "epoch": 0.3203125,
1148
- "grad_norm": 0.2813898026943207,
1149
  "learning_rate": 9.709549441810375e-05,
1150
- "loss": 0.009146124124526978,
1151
- "memory/device_reserved (GiB)": 35.88,
1152
  "memory/max_active (GiB)": 33.92,
1153
  "memory/max_allocated (GiB)": 33.92,
1154
- "ppl": 1.00919,
1155
  "step": 82,
1156
  "tokens/total": 2479248,
1157
- "tokens/train_per_sec_per_gpu": 38.89,
1158
  "tokens/trainable": 37390
1159
  },
1160
  {
1161
  "epoch": 0.32421875,
1162
- "grad_norm": 0.39496278762817383,
1163
  "learning_rate": 9.699201747451195e-05,
1164
- "loss": 0.008894146420061588,
1165
- "memory/device_reserved (GiB)": 35.88,
1166
  "memory/max_active (GiB)": 33.78,
1167
  "memory/max_allocated (GiB)": 33.78,
1168
- "ppl": 1.00893,
1169
  "step": 83,
1170
  "tokens/total": 2509408,
1171
- "tokens/train_per_sec_per_gpu": 30.78,
1172
  "tokens/trainable": 37838
1173
  },
1174
  {
1175
  "epoch": 0.328125,
1176
- "grad_norm": 0.4946168065071106,
1177
  "learning_rate": 9.688679307166854e-05,
1178
- "loss": 0.005293373484164476,
1179
- "memory/device_reserved (GiB)": 35.88,
1180
  "memory/max_active (GiB)": 33.97,
1181
  "memory/max_allocated (GiB)": 33.97,
1182
- "ppl": 1.00531,
1183
  "step": 84,
1184
  "tokens/total": 2539776,
1185
- "tokens/train_per_sec_per_gpu": 34.98,
1186
  "tokens/trainable": 38310
1187
  },
1188
  {
1189
  "epoch": 0.33203125,
1190
- "grad_norm": 1.3293001651763916,
1191
  "learning_rate": 9.677982558839042e-05,
1192
- "loss": 0.040361277759075165,
1193
- "memory/device_reserved (GiB)": 35.88,
1194
  "memory/max_active (GiB)": 33.75,
1195
  "memory/max_allocated (GiB)": 33.75,
1196
- "ppl": 1.04119,
1197
  "step": 85,
1198
  "tokens/total": 2569840,
1199
- "tokens/train_per_sec_per_gpu": 34.0,
1200
  "tokens/trainable": 38789
1201
  },
1202
  {
1203
  "epoch": 0.3359375,
1204
- "grad_norm": 0.5834341049194336,
1205
  "learning_rate": 9.66711194760312e-05,
1206
- "loss": 0.010128681547939777,
1207
- "memory/device_reserved (GiB)": 35.88,
1208
  "memory/max_active (GiB)": 33.86,
1209
  "memory/max_allocated (GiB)": 33.86,
1210
- "ppl": 1.01018,
1211
  "step": 86,
1212
  "tokens/total": 2600192,
1213
- "tokens/train_per_sec_per_gpu": 36.34,
1214
  "tokens/trainable": 39277
1215
  },
1216
  {
1217
  "epoch": 0.33984375,
1218
- "grad_norm": 0.7191532254219055,
1219
  "learning_rate": 9.656067925829593e-05,
1220
- "loss": 0.02042745053768158,
1221
- "memory/device_reserved (GiB)": 35.88,
1222
  "memory/max_active (GiB)": 33.93,
1223
  "memory/max_allocated (GiB)": 33.93,
1224
- "ppl": 1.02064,
1225
  "step": 87,
1226
  "tokens/total": 2630640,
1227
- "tokens/train_per_sec_per_gpu": 35.6,
1228
  "tokens/trainable": 39737
1229
  },
1230
  {
1231
  "epoch": 0.34375,
1232
- "grad_norm": 0.3419296145439148,
1233
  "learning_rate": 9.644850953105288e-05,
1234
- "loss": 0.007800046354532242,
1235
- "memory/device_reserved (GiB)": 35.88,
1236
  "memory/max_active (GiB)": 33.74,
1237
  "memory/max_allocated (GiB)": 33.74,
1238
- "ppl": 1.00783,
1239
  "step": 88,
1240
  "tokens/total": 2660832,
1241
- "tokens/train_per_sec_per_gpu": 29.53,
1242
  "tokens/trainable": 40179
1243
  },
1244
  {
1245
  "epoch": 0.34765625,
1246
- "grad_norm": 0.23275785148143768,
1247
  "learning_rate": 9.633461496214225e-05,
1248
- "loss": 0.005660225171595812,
1249
- "memory/device_reserved (GiB)": 35.88,
1250
  "memory/max_active (GiB)": 33.88,
1251
  "memory/max_allocated (GiB)": 33.88,
1252
- "ppl": 1.00568,
1253
  "step": 89,
1254
  "tokens/total": 2691328,
1255
- "tokens/train_per_sec_per_gpu": 32.59,
1256
  "tokens/trainable": 40649
1257
  },
1258
  {
1259
  "epoch": 0.3515625,
1260
- "grad_norm": 0.6987941265106201,
1261
  "learning_rate": 9.621900029118195e-05,
1262
- "loss": 0.02007928490638733,
1263
- "memory/device_reserved (GiB)": 35.88,
1264
  "memory/max_active (GiB)": 33.41,
1265
  "memory/max_allocated (GiB)": 33.41,
1266
- "ppl": 1.02028,
1267
  "step": 90,
1268
  "tokens/total": 2719696,
1269
- "tokens/train_per_sec_per_gpu": 31.52,
1270
  "tokens/trainable": 41080
1271
  },
1272
  {
1273
  "epoch": 0.35546875,
1274
- "grad_norm": 0.11328475177288055,
1275
  "learning_rate": 9.610167032937036e-05,
1276
- "loss": 0.002234571846202016,
1277
- "memory/device_reserved (GiB)": 35.88,
1278
  "memory/max_active (GiB)": 33.94,
1279
  "memory/max_allocated (GiB)": 33.94,
1280
- "ppl": 1.00224,
1281
  "step": 91,
1282
  "tokens/total": 2750272,
1283
- "tokens/train_per_sec_per_gpu": 37.99,
1284
  "tokens/trainable": 41599
1285
  },
1286
  {
1287
  "epoch": 0.359375,
1288
- "grad_norm": 0.4347783029079437,
1289
  "learning_rate": 9.598262995928611e-05,
1290
- "loss": 0.004549161531031132,
1291
- "memory/device_reserved (GiB)": 35.88,
1292
  "memory/max_active (GiB)": 33.92,
1293
  "memory/max_allocated (GiB)": 33.92,
1294
- "ppl": 1.00456,
1295
  "step": 92,
1296
  "tokens/total": 2780656,
1297
- "tokens/train_per_sec_per_gpu": 36.77,
1298
  "tokens/trainable": 42076
1299
  },
1300
  {
1301
  "epoch": 0.36328125,
1302
- "grad_norm": 0.3486956059932709,
1303
  "learning_rate": 9.586188413468492e-05,
1304
- "loss": 0.012267105281352997,
1305
- "memory/device_reserved (GiB)": 35.88,
1306
  "memory/max_active (GiB)": 33.82,
1307
  "memory/max_allocated (GiB)": 33.82,
1308
- "ppl": 1.01234,
1309
  "step": 93,
1310
  "tokens/total": 2811088,
1311
- "tokens/train_per_sec_per_gpu": 33.6,
1312
  "tokens/trainable": 42522
1313
  },
1314
  {
1315
  "epoch": 0.3671875,
1316
- "grad_norm": 0.5156503319740295,
1317
  "learning_rate": 9.57394378802934e-05,
1318
- "loss": 0.015529165044426918,
1319
  "memory/device_reserved (GiB)": 36.1,
1320
  "memory/max_active (GiB)": 33.9,
1321
  "memory/max_allocated (GiB)": 33.9,
1322
- "ppl": 1.01565,
1323
  "step": 94,
1324
  "tokens/total": 2841520,
1325
- "tokens/train_per_sec_per_gpu": 34.32,
1326
  "tokens/trainable": 42974
1327
  },
1328
  {
1329
  "epoch": 0.37109375,
1330
- "grad_norm": 0.37648338079452515,
1331
  "learning_rate": 9.56152962916e-05,
1332
- "loss": 0.011707151308655739,
1333
  "memory/device_reserved (GiB)": 36.1,
1334
  "memory/max_active (GiB)": 33.7,
1335
  "memory/max_allocated (GiB)": 33.7,
1336
- "ppl": 1.01178,
1337
  "step": 95,
1338
  "tokens/total": 2871600,
1339
- "tokens/train_per_sec_per_gpu": 30.71,
1340
  "tokens/trainable": 43380
1341
  },
1342
  {
1343
  "epoch": 0.375,
1344
- "grad_norm": 0.38365671038627625,
1345
  "learning_rate": 9.548946453464296e-05,
1346
- "loss": 0.008411398157477379,
1347
  "memory/device_reserved (GiB)": 36.1,
1348
  "memory/max_active (GiB)": 33.94,
1349
  "memory/max_allocated (GiB)": 33.94,
1350
- "ppl": 1.00845,
1351
  "step": 96,
1352
  "tokens/total": 2902144,
1353
- "tokens/train_per_sec_per_gpu": 34.13,
1354
  "tokens/trainable": 43865
1355
  },
1356
  {
1357
  "epoch": 0.37890625,
1358
- "grad_norm": 0.313085675239563,
1359
  "learning_rate": 9.53619478457953e-05,
1360
- "loss": 0.007852522656321526,
1361
  "memory/device_reserved (GiB)": 36.1,
1362
  "memory/max_active (GiB)": 33.82,
1363
  "memory/max_allocated (GiB)": 33.82,
1364
- "ppl": 1.00788,
1365
  "step": 97,
1366
  "tokens/total": 2932272,
1367
- "tokens/train_per_sec_per_gpu": 31.89,
1368
  "tokens/trainable": 44328
1369
  },
1370
  {
1371
  "epoch": 0.3828125,
1372
- "grad_norm": 0.08544483780860901,
1373
  "learning_rate": 9.523275153154695e-05,
1374
- "loss": 0.0025753644295036793,
1375
  "memory/device_reserved (GiB)": 35.64,
1376
  "memory/max_active (GiB)": 33.69,
1377
  "memory/max_allocated (GiB)": 33.69,
1378
- "ppl": 1.00258,
1379
  "step": 98,
1380
  "tokens/total": 2962032,
1381
- "tokens/train_per_sec_per_gpu": 30.98,
1382
  "tokens/trainable": 44778
1383
  },
1384
  {
1385
  "epoch": 0.38671875,
1386
- "grad_norm": 0.6496388912200928,
1387
  "learning_rate": 9.51018809682839e-05,
1388
- "loss": 0.02547256276011467,
1389
  "memory/device_reserved (GiB)": 35.64,
1390
  "memory/max_active (GiB)": 33.89,
1391
  "memory/max_allocated (GiB)": 33.89,
1392
- "ppl": 1.0258,
1393
  "step": 99,
1394
  "tokens/total": 2992256,
1395
- "tokens/train_per_sec_per_gpu": 37.11,
1396
  "tokens/trainable": 45225
1397
  },
1398
  {
1399
  "epoch": 0.390625,
1400
- "grad_norm": 0.15325438976287842,
1401
  "learning_rate": 9.49693416020645e-05,
1402
- "loss": 0.003552855923771858,
1403
  "memory/device_reserved (GiB)": 35.64,
1404
  "memory/max_active (GiB)": 33.8,
1405
  "memory/max_allocated (GiB)": 33.8,
1406
- "ppl": 1.00356,
1407
  "step": 100,
1408
  "tokens/total": 3022608,
1409
- "tokens/train_per_sec_per_gpu": 35.8,
1410
  "tokens/trainable": 45678
1411
  },
1412
  {
1413
  "epoch": 0.39453125,
1414
- "grad_norm": 0.25307565927505493,
1415
  "learning_rate": 9.483513894839276e-05,
1416
- "loss": 0.004095804411917925,
1417
  "memory/device_reserved (GiB)": 35.78,
1418
  "memory/max_active (GiB)": 33.88,
1419
  "memory/max_allocated (GiB)": 33.88,
1420
- "ppl": 1.0041,
1421
  "step": 101,
1422
  "tokens/total": 3052992,
1423
- "tokens/train_per_sec_per_gpu": 32.73,
1424
  "tokens/trainable": 46125
1425
  },
1426
  {
1427
  "epoch": 0.3984375,
1428
- "grad_norm": 0.150072380900383,
1429
  "learning_rate": 9.469927859198888e-05,
1430
- "loss": 0.0013888315297663212,
1431
- "memory/device_reserved (GiB)": 35.79,
1432
  "memory/max_active (GiB)": 33.98,
1433
  "memory/max_allocated (GiB)": 33.98,
1434
- "ppl": 1.00139,
1435
  "step": 102,
1436
  "tokens/total": 3083392,
1437
- "tokens/train_per_sec_per_gpu": 39.71,
1438
  "tokens/trainable": 46612
1439
  },
1440
  {
1441
  "epoch": 0.40234375,
1442
- "grad_norm": 0.5769571661949158,
1443
  "learning_rate": 9.456176618655689e-05,
1444
- "loss": 0.022533349692821503,
1445
- "memory/device_reserved (GiB)": 35.79,
1446
  "memory/max_active (GiB)": 33.83,
1447
  "memory/max_allocated (GiB)": 33.83,
1448
- "ppl": 1.02279,
1449
  "step": 103,
1450
  "tokens/total": 3113744,
1451
- "tokens/train_per_sec_per_gpu": 32.57,
1452
  "tokens/trainable": 47059
1453
  },
1454
  {
1455
  "epoch": 0.40625,
1456
- "grad_norm": 0.5657027959823608,
1457
  "learning_rate": 9.442260745454927e-05,
1458
- "loss": 0.016894506290555,
1459
- "memory/device_reserved (GiB)": 35.79,
1460
  "memory/max_active (GiB)": 34.02,
1461
  "memory/max_allocated (GiB)": 34.02,
1462
- "ppl": 1.01704,
1463
  "step": 104,
1464
  "tokens/total": 3144272,
1465
- "tokens/train_per_sec_per_gpu": 34.05,
1466
  "tokens/trainable": 47536
1467
  },
1468
  {
1469
  "epoch": 0.41015625,
1470
- "grad_norm": 0.29607170820236206,
1471
  "learning_rate": 9.428180818692884e-05,
1472
- "loss": 0.017974723130464554,
1473
- "memory/device_reserved (GiB)": 37.61,
1474
  "memory/max_active (GiB)": 33.81,
1475
  "memory/max_allocated (GiB)": 33.81,
1476
- "ppl": 1.01814,
1477
  "step": 105,
1478
  "tokens/total": 3174512,
1479
- "tokens/train_per_sec_per_gpu": 34.11,
1480
  "tokens/trainable": 48037
1481
  },
1482
  {
1483
  "epoch": 0.4140625,
1484
- "grad_norm": 0.5241922736167908,
1485
  "learning_rate": 9.413937424292791e-05,
1486
- "loss": 0.016189826652407646,
1487
- "memory/device_reserved (GiB)": 37.61,
1488
  "memory/max_active (GiB)": 33.91,
1489
  "memory/max_allocated (GiB)": 33.91,
1490
- "ppl": 1.01632,
1491
  "step": 106,
1492
  "tokens/total": 3204896,
1493
- "tokens/train_per_sec_per_gpu": 32.65,
1494
  "tokens/trainable": 48491
1495
  },
1496
  {
1497
  "epoch": 0.41796875,
1498
- "grad_norm": 0.3886408805847168,
1499
  "learning_rate": 9.399531154980424e-05,
1500
- "loss": 0.010908558964729309,
1501
- "memory/device_reserved (GiB)": 37.61,
1502
  "memory/max_active (GiB)": 34.01,
1503
  "memory/max_allocated (GiB)": 34.01,
1504
- "ppl": 1.01097,
1505
  "step": 107,
1506
  "tokens/total": 3235360,
1507
- "tokens/train_per_sec_per_gpu": 33.0,
1508
  "tokens/trainable": 48940
1509
  },
1510
  {
1511
  "epoch": 0.421875,
1512
- "grad_norm": 0.2442784607410431,
1513
  "learning_rate": 9.384962610259455e-05,
1514
- "loss": 0.008070861920714378,
1515
- "memory/device_reserved (GiB)": 37.61,
1516
  "memory/max_active (GiB)": 33.83,
1517
  "memory/max_allocated (GiB)": 33.83,
1518
- "ppl": 1.0081,
1519
  "step": 108,
1520
  "tokens/total": 3265552,
1521
- "tokens/train_per_sec_per_gpu": 31.79,
1522
  "tokens/trainable": 49389
1523
  },
1524
  {
1525
  "epoch": 0.42578125,
1526
- "grad_norm": 0.1457175612449646,
1527
  "learning_rate": 9.370232396386494e-05,
1528
- "loss": 0.003785747569054365,
1529
- "memory/device_reserved (GiB)": 37.61,
1530
  "memory/max_active (GiB)": 33.39,
1531
  "memory/max_allocated (GiB)": 33.39,
1532
- "ppl": 1.00379,
1533
  "step": 109,
1534
  "tokens/total": 3293856,
1535
- "tokens/train_per_sec_per_gpu": 36.01,
1536
  "tokens/trainable": 49838
1537
  },
1538
  {
1539
  "epoch": 0.4296875,
1540
- "grad_norm": 0.3955559730529785,
1541
  "learning_rate": 9.355341126345868e-05,
1542
- "loss": 0.0069918157532811165,
1543
- "memory/device_reserved (GiB)": 37.61,
1544
  "memory/max_active (GiB)": 33.78,
1545
  "memory/max_allocated (GiB)": 33.78,
1546
- "ppl": 1.00702,
1547
  "step": 110,
1548
  "tokens/total": 3323984,
1549
- "tokens/train_per_sec_per_gpu": 34.06,
1550
  "tokens/trainable": 50310
1551
  },
1552
  {
1553
  "epoch": 0.43359375,
1554
- "grad_norm": 0.6224751472473145,
1555
  "learning_rate": 9.340289419824107e-05,
1556
- "loss": 0.014852076768875122,
1557
  "memory/device_reserved (GiB)": 37.62,
1558
  "memory/max_active (GiB)": 33.84,
1559
  "memory/max_allocated (GiB)": 33.84,
1560
- "ppl": 1.01496,
1561
  "step": 111,
1562
  "tokens/total": 3354320,
1563
- "tokens/train_per_sec_per_gpu": 33.67,
1564
  "tokens/trainable": 50755
1565
  },
1566
  {
1567
  "epoch": 0.4375,
1568
- "grad_norm": 0.3700723648071289,
1569
  "learning_rate": 9.325077903184159e-05,
1570
- "loss": 0.00436755083501339,
1571
  "memory/device_reserved (GiB)": 37.62,
1572
  "memory/max_active (GiB)": 33.9,
1573
  "memory/max_allocated (GiB)": 33.9,
1574
- "ppl": 1.00438,
1575
  "step": 112,
1576
  "tokens/total": 3384880,
1577
- "tokens/train_per_sec_per_gpu": 31.65,
1578
  "tokens/trainable": 51213
1579
  },
1580
  {
1581
  "epoch": 0.44140625,
1582
- "grad_norm": 0.1353236883878708,
1583
  "learning_rate": 9.30970720943932e-05,
1584
- "loss": 0.0025976570323109627,
1585
  "memory/device_reserved (GiB)": 37.62,
1586
  "memory/max_active (GiB)": 33.89,
1587
  "memory/max_allocated (GiB)": 33.89,
1588
- "ppl": 1.0026,
1589
  "step": 113,
1590
  "tokens/total": 3415104,
1591
- "tokens/train_per_sec_per_gpu": 33.83,
1592
  "tokens/trainable": 51660
1593
  },
1594
  {
1595
  "epoch": 0.4453125,
1596
- "grad_norm": 0.18984447419643402,
1597
  "learning_rate": 9.2941779782269e-05,
1598
- "loss": 0.002497302368283272,
1599
  "memory/device_reserved (GiB)": 37.62,
1600
  "memory/max_active (GiB)": 33.84,
1601
  "memory/max_allocated (GiB)": 33.84,
1602
- "ppl": 1.0025,
1603
  "step": 114,
1604
  "tokens/total": 3445504,
1605
- "tokens/train_per_sec_per_gpu": 32.43,
1606
  "tokens/trainable": 52133
1607
  },
1608
  {
1609
  "epoch": 0.44921875,
1610
- "grad_norm": 1.1815483570098877,
1611
  "learning_rate": 9.278490855781596e-05,
1612
- "loss": 0.029489168897271156,
1613
  "memory/device_reserved (GiB)": 37.62,
1614
  "memory/max_active (GiB)": 33.84,
1615
  "memory/max_allocated (GiB)": 33.84,
1616
- "ppl": 1.02993,
1617
  "step": 115,
1618
  "tokens/total": 3475744,
1619
- "tokens/train_per_sec_per_gpu": 37.47,
1620
  "tokens/trainable": 52580
1621
  },
1622
  {
1623
  "epoch": 0.453125,
1624
- "grad_norm": 0.44360846281051636,
1625
  "learning_rate": 9.262646494908604e-05,
1626
- "loss": 0.009585533291101456,
1627
  "memory/device_reserved (GiB)": 37.62,
1628
  "memory/max_active (GiB)": 33.82,
1629
  "memory/max_allocated (GiB)": 33.82,
1630
- "ppl": 1.00963,
1631
  "step": 116,
1632
  "tokens/total": 3506016,
1633
- "tokens/train_per_sec_per_gpu": 34.83,
1634
  "tokens/trainable": 53033
1635
  },
1636
  {
1637
  "epoch": 0.45703125,
1638
- "grad_norm": 0.5074551701545715,
1639
  "learning_rate": 9.246645554956457e-05,
1640
- "loss": 0.020201388746500015,
1641
  "memory/device_reserved (GiB)": 37.62,
1642
  "memory/max_active (GiB)": 33.83,
1643
  "memory/max_allocated (GiB)": 33.83,
1644
- "ppl": 1.02041,
1645
  "step": 117,
1646
  "tokens/total": 3536256,
1647
- "tokens/train_per_sec_per_gpu": 35.27,
1648
  "tokens/trainable": 53517
1649
  },
1650
  {
1651
  "epoch": 0.4609375,
1652
- "grad_norm": 0.3565296232700348,
1653
  "learning_rate": 9.230488701789578e-05,
1654
- "loss": 0.005688562989234924,
1655
  "memory/device_reserved (GiB)": 37.62,
1656
  "memory/max_active (GiB)": 33.73,
1657
  "memory/max_allocated (GiB)": 33.73,
1658
- "ppl": 1.0057,
1659
  "step": 118,
1660
  "tokens/total": 3566480,
1661
- "tokens/train_per_sec_per_gpu": 34.56,
1662
  "tokens/trainable": 53967
1663
  },
1664
  {
1665
  "epoch": 0.46484375,
1666
- "grad_norm": 0.16547706723213196,
1667
  "learning_rate": 9.214176607760577e-05,
1668
- "loss": 0.003188834059983492,
1669
  "memory/device_reserved (GiB)": 37.62,
1670
  "memory/max_active (GiB)": 33.85,
1671
  "memory/max_allocated (GiB)": 33.85,
1672
- "ppl": 1.00319,
1673
  "step": 119,
1674
  "tokens/total": 3596624,
1675
- "tokens/train_per_sec_per_gpu": 35.27,
1676
  "tokens/trainable": 54430
1677
  },
1678
  {
1679
  "epoch": 0.46875,
1680
- "grad_norm": 0.20722293853759766,
1681
  "learning_rate": 9.197709951682268e-05,
1682
- "loss": 0.003235103562474251,
1683
  "memory/device_reserved (GiB)": 37.62,
1684
  "memory/max_active (GiB)": 33.93,
1685
  "memory/max_allocated (GiB)": 33.93,
1686
- "ppl": 1.00324,
1687
  "step": 120,
1688
  "tokens/total": 3627168,
1689
- "tokens/train_per_sec_per_gpu": 35.21,
1690
  "tokens/trainable": 54896
1691
  },
1692
  {
1693
  "epoch": 0.47265625,
1694
- "grad_norm": 0.4754939377307892,
1695
  "learning_rate": 9.181089418799428e-05,
1696
- "loss": 0.005670893006026745,
1697
  "memory/device_reserved (GiB)": 37.62,
1698
  "memory/max_active (GiB)": 33.96,
1699
  "memory/max_allocated (GiB)": 33.96,
1700
- "ppl": 1.00569,
1701
  "step": 121,
1702
  "tokens/total": 3657632,
1703
- "tokens/train_per_sec_per_gpu": 37.77,
1704
  "tokens/trainable": 55379
1705
  },
1706
  {
1707
  "epoch": 0.4765625,
1708
- "grad_norm": 0.08304762095212936,
1709
  "learning_rate": 9.164315700760271e-05,
1710
- "loss": 0.00099027412943542,
1711
  "memory/device_reserved (GiB)": 37.62,
1712
  "memory/max_active (GiB)": 33.75,
1713
  "memory/max_allocated (GiB)": 33.75,
1714
- "ppl": 1.00099,
1715
  "step": 122,
1716
  "tokens/total": 3687824,
1717
- "tokens/train_per_sec_per_gpu": 31.7,
1718
  "tokens/trainable": 55803
1719
  },
1720
  {
1721
  "epoch": 0.48046875,
1722
- "grad_norm": 0.725281298160553,
1723
  "learning_rate": 9.147389495587671e-05,
1724
- "loss": 0.007413266692310572,
1725
  "memory/device_reserved (GiB)": 37.62,
1726
  "memory/max_active (GiB)": 33.87,
1727
  "memory/max_allocated (GiB)": 33.87,
1728
- "ppl": 1.00744,
1729
  "step": 123,
1730
  "tokens/total": 3718320,
1731
- "tokens/train_per_sec_per_gpu": 36.45,
1732
  "tokens/trainable": 56249
1733
  },
1734
  {
1735
  "epoch": 0.484375,
1736
- "grad_norm": 0.31409645080566406,
1737
  "learning_rate": 9.130311507650116e-05,
1738
- "loss": 0.0029702766332775354,
1739
  "memory/device_reserved (GiB)": 37.62,
1740
  "memory/max_active (GiB)": 33.9,
1741
  "memory/max_allocated (GiB)": 33.9,
1742
- "ppl": 1.00297,
1743
  "step": 124,
1744
  "tokens/total": 3748672,
1745
- "tokens/train_per_sec_per_gpu": 32.41,
1746
  "tokens/trainable": 56713
1747
  },
1748
  {
1749
  "epoch": 0.48828125,
1750
- "grad_norm": 0.6082578897476196,
1751
  "learning_rate": 9.113082447632394e-05,
1752
- "loss": 0.003795543685555458,
1753
  "memory/device_reserved (GiB)": 37.76,
1754
  "memory/max_active (GiB)": 33.85,
1755
  "memory/max_allocated (GiB)": 33.85,
1756
- "ppl": 1.0038,
1757
  "step": 125,
1758
  "tokens/total": 3778976,
1759
- "tokens/train_per_sec_per_gpu": 39.08,
1760
  "tokens/trainable": 57196
1761
  },
1762
  {
1763
  "epoch": 0.4921875,
1764
- "grad_norm": 0.0603976845741272,
1765
  "learning_rate": 9.09570303250602e-05,
1766
- "loss": 0.0014751165872439742,
1767
  "memory/device_reserved (GiB)": 37.76,
1768
  "memory/max_active (GiB)": 33.89,
1769
  "memory/max_allocated (GiB)": 33.89,
1770
- "ppl": 1.00148,
1771
  "step": 126,
1772
  "tokens/total": 3809296,
1773
- "tokens/train_per_sec_per_gpu": 36.49,
1774
  "tokens/trainable": 57680
1775
  },
1776
  {
1777
  "epoch": 0.49609375,
1778
- "grad_norm": 0.21112751960754395,
1779
  "learning_rate": 9.078173985499394e-05,
1780
- "loss": 0.004137102514505386,
1781
  "memory/device_reserved (GiB)": 37.76,
1782
  "memory/max_active (GiB)": 33.77,
1783
  "memory/max_allocated (GiB)": 33.77,
1784
- "ppl": 1.00415,
1785
  "step": 127,
1786
  "tokens/total": 3839328,
1787
- "tokens/train_per_sec_per_gpu": 31.8,
1788
  "tokens/trainable": 58110
1789
  },
1790
  {
1791
  "epoch": 0.5,
1792
- "grad_norm": 0.3234494924545288,
1793
  "learning_rate": 9.060496036067713e-05,
1794
- "loss": 0.007372056134045124,
1795
  "memory/device_reserved (GiB)": 37.76,
1796
  "memory/max_active (GiB)": 33.88,
1797
  "memory/max_allocated (GiB)": 33.88,
1798
- "ppl": 1.0074,
1799
  "step": 128,
1800
  "tokens/total": 3869824,
1801
- "tokens/train_per_sec_per_gpu": 31.64,
1802
  "tokens/trainable": 58534
1803
  },
1804
  {
1805
  "epoch": 0.50390625,
1806
- "grad_norm": 0.5826171636581421,
1807
  "learning_rate": 9.042669919862615e-05,
1808
- "loss": 0.007980267517268658,
1809
  "memory/device_reserved (GiB)": 37.76,
1810
  "memory/max_active (GiB)": 33.9,
1811
  "memory/max_allocated (GiB)": 33.9,
1812
- "ppl": 1.00801,
1813
  "step": 129,
1814
  "tokens/total": 3900080,
1815
- "tokens/train_per_sec_per_gpu": 35.08,
1816
  "tokens/trainable": 58998
1817
  },
1818
  {
1819
  "epoch": 0.5078125,
1820
- "grad_norm": 0.3384500741958618,
1821
  "learning_rate": 9.024696378701557e-05,
1822
- "loss": 0.005637750029563904,
1823
  "memory/device_reserved (GiB)": 35.93,
1824
  "memory/max_active (GiB)": 33.91,
1825
  "memory/max_allocated (GiB)": 33.91,
1826
- "ppl": 1.00565,
1827
  "step": 130,
1828
  "tokens/total": 3930560,
1829
- "tokens/train_per_sec_per_gpu": 31.81,
1830
  "tokens/trainable": 59448
1831
  },
1832
  {
1833
  "epoch": 0.51171875,
1834
- "grad_norm": 0.2838669717311859,
1835
  "learning_rate": 9.006576160536948e-05,
1836
- "loss": 0.0037927688099443913,
1837
  "memory/device_reserved (GiB)": 35.93,
1838
  "memory/max_active (GiB)": 33.73,
1839
  "memory/max_allocated (GiB)": 33.73,
1840
- "ppl": 1.0038,
1841
  "step": 131,
1842
  "tokens/total": 3960496,
1843
- "tokens/train_per_sec_per_gpu": 31.97,
1844
  "tokens/trainable": 59906
1845
  },
1846
  {
1847
  "epoch": 0.515625,
1848
- "grad_norm": 0.4598330855369568,
1849
  "learning_rate": 8.988310019425035e-05,
1850
- "loss": 0.010232537053525448,
1851
- "memory/device_reserved (GiB)": 35.94,
1852
  "memory/max_active (GiB)": 33.85,
1853
  "memory/max_allocated (GiB)": 33.85,
1854
- "ppl": 1.01029,
1855
  "step": 132,
1856
  "tokens/total": 3990928,
1857
- "tokens/train_per_sec_per_gpu": 34.23,
1858
  "tokens/trainable": 60350
1859
  },
1860
  {
1861
  "epoch": 0.51953125,
1862
- "grad_norm": 0.7072780728340149,
1863
  "learning_rate": 8.969898715494506e-05,
1864
- "loss": 0.00946755986660719,
1865
- "memory/device_reserved (GiB)": 37.17,
1866
  "memory/max_active (GiB)": 33.84,
1867
  "memory/max_allocated (GiB)": 33.84,
1868
- "ppl": 1.00951,
1869
  "step": 133,
1870
  "tokens/total": 4021264,
1871
- "tokens/train_per_sec_per_gpu": 36.18,
1872
  "tokens/trainable": 60831
1873
  },
1874
  {
1875
  "epoch": 0.5234375,
1876
- "grad_norm": 0.3642464280128479,
1877
  "learning_rate": 8.951343014914869e-05,
1878
- "loss": 0.0067742373794317245,
1879
  "memory/device_reserved (GiB)": 37.17,
1880
  "memory/max_active (GiB)": 33.88,
1881
  "memory/max_allocated (GiB)": 33.88,
1882
- "ppl": 1.0068,
1883
  "step": 134,
1884
  "tokens/total": 4051728,
1885
- "tokens/train_per_sec_per_gpu": 33.91,
1886
  "tokens/trainable": 61305
1887
  },
1888
  {
1889
  "epoch": 0.52734375,
1890
- "grad_norm": 0.1071263924241066,
1891
  "learning_rate": 8.932643689864568e-05,
1892
- "loss": 0.0022848297376185656,
1893
  "memory/device_reserved (GiB)": 37.17,
1894
  "memory/max_active (GiB)": 33.85,
1895
  "memory/max_allocated (GiB)": 33.85,
1896
- "ppl": 1.00229,
1897
  "step": 135,
1898
  "tokens/total": 4081920,
1899
- "tokens/train_per_sec_per_gpu": 37.57,
1900
  "tokens/trainable": 61792
1901
  },
1902
  {
1903
  "epoch": 0.53125,
1904
- "grad_norm": 0.22470054030418396,
1905
  "learning_rate": 8.913801518498845e-05,
1906
- "loss": 0.0016683072317391634,
1907
  "memory/device_reserved (GiB)": 38.55,
1908
  "memory/max_active (GiB)": 33.9,
1909
  "memory/max_allocated (GiB)": 33.9,
1910
- "ppl": 1.00167,
1911
  "step": 136,
1912
  "tokens/total": 4112304,
1913
- "tokens/train_per_sec_per_gpu": 31.33,
1914
  "tokens/trainable": 62253
1915
  },
1916
  {
1917
  "epoch": 0.53515625,
1918
- "grad_norm": 0.5423188209533691,
1919
  "learning_rate": 8.894817284917364e-05,
1920
- "loss": 0.017443198710680008,
1921
  "memory/device_reserved (GiB)": 38.55,
1922
  "memory/max_active (GiB)": 33.81,
1923
  "memory/max_allocated (GiB)": 33.81,
1924
- "ppl": 1.0176,
1925
  "step": 137,
1926
  "tokens/total": 4142512,
1927
- "tokens/train_per_sec_per_gpu": 31.57,
1928
  "tokens/trainable": 62707
1929
  },
1930
  {
1931
  "epoch": 0.5390625,
1932
- "grad_norm": 0.47486332058906555,
1933
  "learning_rate": 8.875691779131569e-05,
1934
- "loss": 0.01525629311800003,
1935
- "memory/device_reserved (GiB)": 38.55,
1936
  "memory/max_active (GiB)": 33.89,
1937
  "memory/max_allocated (GiB)": 33.89,
1938
- "ppl": 1.01537,
1939
  "step": 138,
1940
  "tokens/total": 4172992,
1941
  "tokens/train_per_sec_per_gpu": 29.32,
@@ -1943,139 +1943,139 @@
1943
  },
1944
  {
1945
  "epoch": 0.54296875,
1946
- "grad_norm": 0.055354099720716476,
1947
  "learning_rate": 8.856425797031829e-05,
1948
- "loss": 0.0010598574299365282,
1949
- "memory/device_reserved (GiB)": 38.55,
1950
  "memory/max_active (GiB)": 33.78,
1951
  "memory/max_allocated (GiB)": 33.78,
1952
- "ppl": 1.00106,
1953
  "step": 139,
1954
  "tokens/total": 4203136,
1955
- "tokens/train_per_sec_per_gpu": 33.87,
1956
  "tokens/trainable": 63587
1957
  },
1958
  {
1959
  "epoch": 0.546875,
1960
- "grad_norm": 0.15273842215538025,
1961
  "learning_rate": 8.837020140354295e-05,
1962
- "loss": 0.002772743348032236,
1963
- "memory/device_reserved (GiB)": 38.55,
1964
  "memory/max_active (GiB)": 33.84,
1965
  "memory/max_allocated (GiB)": 33.84,
1966
- "ppl": 1.00278,
1967
  "step": 140,
1968
  "tokens/total": 4233456,
1969
- "tokens/train_per_sec_per_gpu": 36.75,
1970
  "tokens/trainable": 64052
1971
  },
1972
  {
1973
  "epoch": 0.55078125,
1974
- "grad_norm": 0.21690180897712708,
1975
  "learning_rate": 8.817475616647554e-05,
1976
- "loss": 0.005170213058590889,
1977
- "memory/device_reserved (GiB)": 38.55,
1978
  "memory/max_active (GiB)": 33.8,
1979
  "memory/max_allocated (GiB)": 33.8,
1980
- "ppl": 1.00518,
1981
  "step": 141,
1982
  "tokens/total": 4263728,
1983
- "tokens/train_per_sec_per_gpu": 34.36,
1984
  "tokens/trainable": 64526
1985
  },
1986
  {
1987
  "epoch": 0.5546875,
1988
- "grad_norm": 0.1491464525461197,
1989
  "learning_rate": 8.797793039239017e-05,
1990
- "loss": 0.0031757252290844917,
1991
- "memory/device_reserved (GiB)": 38.55,
1992
  "memory/max_active (GiB)": 34.01,
1993
  "memory/max_allocated (GiB)": 34.01,
1994
- "ppl": 1.00318,
1995
  "step": 142,
1996
  "tokens/total": 4294432,
1997
- "tokens/train_per_sec_per_gpu": 34.66,
1998
  "tokens/trainable": 64983
1999
  },
2000
  {
2001
  "epoch": 0.55859375,
2002
- "grad_norm": 0.19370807707309723,
2003
  "learning_rate": 8.777973227201069e-05,
2004
- "loss": 0.0033013438805937767,
2005
- "memory/device_reserved (GiB)": 38.55,
2006
  "memory/max_active (GiB)": 33.94,
2007
  "memory/max_allocated (GiB)": 33.94,
2008
- "ppl": 1.00331,
2009
  "step": 143,
2010
  "tokens/total": 4324960,
2011
- "tokens/train_per_sec_per_gpu": 37.6,
2012
  "tokens/trainable": 65492
2013
  },
2014
  {
2015
  "epoch": 0.5625,
2016
- "grad_norm": 0.43046337366104126,
2017
  "learning_rate": 8.758017005316988e-05,
2018
- "loss": 0.004675615578889847,
2019
- "memory/device_reserved (GiB)": 38.55,
2020
  "memory/max_active (GiB)": 33.92,
2021
  "memory/max_allocated (GiB)": 33.92,
2022
- "ppl": 1.00469,
2023
  "step": 144,
2024
  "tokens/total": 4355424,
2025
- "tokens/train_per_sec_per_gpu": 35.71,
2026
  "tokens/trainable": 65925
2027
  },
2028
  {
2029
  "epoch": 0.56640625,
2030
- "grad_norm": 1.153432011604309,
2031
  "learning_rate": 8.737925204046629e-05,
2032
- "loss": 0.00530653540045023,
2033
- "memory/device_reserved (GiB)": 38.55,
2034
  "memory/max_active (GiB)": 33.75,
2035
  "memory/max_allocated (GiB)": 33.75,
2036
- "ppl": 1.00532,
2037
  "step": 145,
2038
  "tokens/total": 4385712,
2039
- "tokens/train_per_sec_per_gpu": 31.24,
2040
  "tokens/trainable": 66383
2041
  },
2042
  {
2043
  "epoch": 0.5703125,
2044
- "grad_norm": 0.7740430235862732,
2045
  "learning_rate": 8.717698659491851e-05,
2046
- "loss": 0.01281517744064331,
2047
  "memory/device_reserved (GiB)": 38.56,
2048
  "memory/max_active (GiB)": 33.94,
2049
  "memory/max_allocated (GiB)": 33.94,
2050
- "ppl": 1.0129,
2051
  "step": 146,
2052
  "tokens/total": 4416016,
2053
- "tokens/train_per_sec_per_gpu": 31.37,
2054
  "tokens/trainable": 66840
2055
  },
2056
  {
2057
  "epoch": 0.57421875,
2058
- "grad_norm": 0.6978983879089355,
2059
  "learning_rate": 8.697338213361735e-05,
2060
- "loss": 0.0272100567817688,
2061
  "memory/device_reserved (GiB)": 38.65,
2062
  "memory/max_active (GiB)": 33.88,
2063
  "memory/max_allocated (GiB)": 33.88,
2064
- "ppl": 1.02758,
2065
  "step": 147,
2066
  "tokens/total": 4446368,
2067
- "tokens/train_per_sec_per_gpu": 31.71,
2068
  "tokens/trainable": 67297
2069
  },
2070
  {
2071
  "epoch": 0.578125,
2072
- "grad_norm": 0.17344872653484344,
2073
  "learning_rate": 8.676844712937552e-05,
2074
- "loss": 0.008015683852136135,
2075
  "memory/device_reserved (GiB)": 38.65,
2076
  "memory/max_active (GiB)": 33.89,
2077
  "memory/max_allocated (GiB)": 33.89,
2078
- "ppl": 1.00805,
2079
  "step": 148,
2080
  "tokens/total": 4476880,
2081
  "tokens/train_per_sec_per_gpu": 37.36,
@@ -2083,170 +2083,170 @@
2083
  },
2084
  {
2085
  "epoch": 0.58203125,
2086
- "grad_norm": 0.6355595588684082,
2087
  "learning_rate": 8.656219011037509e-05,
2088
- "loss": 0.010829147882759571,
2089
  "memory/device_reserved (GiB)": 38.65,
2090
  "memory/max_active (GiB)": 33.89,
2091
  "memory/max_allocated (GiB)": 33.89,
2092
- "ppl": 1.01089,
2093
  "step": 149,
2094
  "tokens/total": 4507232,
2095
- "tokens/train_per_sec_per_gpu": 32.78,
2096
  "tokens/trainable": 68257
2097
  },
2098
  {
2099
  "epoch": 0.5859375,
2100
- "grad_norm": 0.25094935297966003,
2101
  "learning_rate": 8.63546196598125e-05,
2102
- "loss": 0.0052955676801502705,
2103
  "memory/device_reserved (GiB)": 38.65,
2104
  "memory/max_active (GiB)": 33.77,
2105
  "memory/max_allocated (GiB)": 33.77,
2106
- "ppl": 1.00531,
2107
  "step": 150,
2108
  "tokens/total": 4537376,
2109
- "tokens/train_per_sec_per_gpu": 35.51,
2110
  "tokens/trainable": 68706
2111
  },
2112
  {
2113
  "epoch": 0.58984375,
2114
- "grad_norm": 0.5292705297470093,
2115
  "learning_rate": 8.614574441554145e-05,
2116
- "loss": 0.022014575079083443,
2117
  "memory/device_reserved (GiB)": 38.65,
2118
  "memory/max_active (GiB)": 33.8,
2119
  "memory/max_allocated (GiB)": 33.8,
2120
- "ppl": 1.02226,
2121
  "step": 151,
2122
  "tokens/total": 4567584,
2123
- "tokens/train_per_sec_per_gpu": 33.25,
2124
  "tokens/trainable": 69131
2125
  },
2126
  {
2127
  "epoch": 0.59375,
2128
- "grad_norm": 0.3471219539642334,
2129
  "learning_rate": 8.593557306971349e-05,
2130
- "loss": 0.024585288017988205,
2131
  "memory/device_reserved (GiB)": 38.65,
2132
  "memory/max_active (GiB)": 33.87,
2133
  "memory/max_allocated (GiB)": 33.87,
2134
- "ppl": 1.02489,
2135
  "step": 152,
2136
  "tokens/total": 4597792,
2137
- "tokens/train_per_sec_per_gpu": 31.92,
2138
  "tokens/trainable": 69586
2139
  },
2140
  {
2141
  "epoch": 0.59765625,
2142
- "grad_norm": 0.08056659996509552,
2143
  "learning_rate": 8.572411436841618e-05,
2144
- "loss": 0.002611964475363493,
2145
  "memory/device_reserved (GiB)": 38.65,
2146
  "memory/max_active (GiB)": 33.78,
2147
  "memory/max_allocated (GiB)": 33.78,
2148
- "ppl": 1.00262,
2149
  "step": 153,
2150
  "tokens/total": 4627936,
2151
- "tokens/train_per_sec_per_gpu": 32.81,
2152
  "tokens/trainable": 70061
2153
  },
2154
  {
2155
  "epoch": 0.6015625,
2156
- "grad_norm": 0.162270650267601,
2157
  "learning_rate": 8.551137711130922e-05,
2158
- "loss": 0.0033612053375691175,
2159
  "memory/device_reserved (GiB)": 38.65,
2160
  "memory/max_active (GiB)": 33.82,
2161
  "memory/max_allocated (GiB)": 33.82,
2162
- "ppl": 1.00337,
2163
  "step": 154,
2164
  "tokens/total": 4658352,
2165
- "tokens/train_per_sec_per_gpu": 27.93,
2166
  "tokens/trainable": 70467
2167
  },
2168
  {
2169
  "epoch": 0.60546875,
2170
- "grad_norm": 0.17613235116004944,
2171
  "learning_rate": 8.529737015125824e-05,
2172
- "loss": 0.004349880386143923,
2173
  "memory/device_reserved (GiB)": 38.65,
2174
  "memory/max_active (GiB)": 33.87,
2175
  "memory/max_allocated (GiB)": 33.87,
2176
- "ppl": 1.00436,
2177
  "step": 155,
2178
  "tokens/total": 4688896,
2179
- "tokens/train_per_sec_per_gpu": 33.1,
2180
  "tokens/trainable": 70933
2181
  },
2182
  {
2183
  "epoch": 0.609375,
2184
- "grad_norm": 0.2590649127960205,
2185
  "learning_rate": 8.508210239396639e-05,
2186
- "loss": 0.010615494102239609,
2187
  "memory/device_reserved (GiB)": 38.65,
2188
  "memory/max_active (GiB)": 33.94,
2189
  "memory/max_allocated (GiB)": 33.94,
2190
- "ppl": 1.01067,
2191
  "step": 156,
2192
  "tokens/total": 4719168,
2193
- "tokens/train_per_sec_per_gpu": 34.23,
2194
  "tokens/trainable": 71382
2195
  },
2196
  {
2197
  "epoch": 0.61328125,
2198
- "grad_norm": 0.15640626847743988,
2199
  "learning_rate": 8.486558279760375e-05,
2200
- "loss": 0.002627612790092826,
2201
  "memory/device_reserved (GiB)": 38.65,
2202
  "memory/max_active (GiB)": 33.87,
2203
  "memory/max_allocated (GiB)": 33.87,
2204
- "ppl": 1.00263,
2205
  "step": 157,
2206
  "tokens/total": 4749136,
2207
- "tokens/train_per_sec_per_gpu": 30.22,
2208
  "tokens/trainable": 71808
2209
  },
2210
  {
2211
  "epoch": 0.6171875,
2212
- "grad_norm": 0.34429433941841125,
2213
  "learning_rate": 8.464782037243449e-05,
2214
- "loss": 0.010873161256313324,
2215
  "memory/device_reserved (GiB)": 38.65,
2216
  "memory/max_active (GiB)": 33.74,
2217
  "memory/max_allocated (GiB)": 33.74,
2218
- "ppl": 1.01093,
2219
  "step": 158,
2220
  "tokens/total": 4779472,
2221
- "tokens/train_per_sec_per_gpu": 32.87,
2222
  "tokens/trainable": 72249
2223
  },
2224
  {
2225
  "epoch": 0.62109375,
2226
- "grad_norm": 0.3858713209629059,
2227
  "learning_rate": 8.442882418044202e-05,
2228
- "loss": 0.020050909370183945,
2229
  "memory/device_reserved (GiB)": 38.65,
2230
  "memory/max_active (GiB)": 33.77,
2231
  "memory/max_allocated (GiB)": 33.77,
2232
- "ppl": 1.02025,
2233
  "step": 159,
2234
  "tokens/total": 4809632,
2235
- "tokens/train_per_sec_per_gpu": 35.19,
2236
  "tokens/trainable": 72726
2237
  },
2238
  {
2239
  "epoch": 0.625,
2240
- "grad_norm": 0.3002466857433319,
2241
  "learning_rate": 8.420860333495179e-05,
2242
- "loss": 0.009756345301866531,
2243
  "memory/device_reserved (GiB)": 38.65,
2244
  "memory/max_active (GiB)": 33.97,
2245
  "memory/max_allocated (GiB)": 33.97,
2246
- "ppl": 1.0098,
2247
  "step": 160,
2248
  "tokens/total": 4840112,
2249
- "tokens/train_per_sec_per_gpu": 32.73,
2250
  "tokens/trainable": 73148
2251
  }
2252
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.00390625,
14
+ "grad_norm": 0.8591235280036926,
15
  "learning_rate": 0.0,
16
  "loss": 0.10251401364803314,
17
  "memory/device_reserved (GiB)": 34.94,
 
20
  "ppl": 1.10795,
21
  "step": 1,
22
  "tokens/total": 30464,
23
+ "tokens/train_per_sec_per_gpu": 30.01,
24
  "tokens/trainable": 470
25
  },
26
  {
27
  "epoch": 0.0078125,
28
+ "grad_norm": 0.8033757209777832,
29
  "learning_rate": 4.000000000000001e-06,
30
  "loss": 0.09678442776203156,
31
  "memory/device_reserved (GiB)": 35.83,
 
34
  "ppl": 1.10162,
35
  "step": 2,
36
  "tokens/total": 60800,
37
+ "tokens/train_per_sec_per_gpu": 35.1,
38
  "tokens/trainable": 922
39
  },
40
  {
41
  "epoch": 0.01171875,
42
+ "grad_norm": 1.0102914571762085,
43
  "learning_rate": 8.000000000000001e-06,
44
+ "loss": 0.10876177996397018,
45
  "memory/device_reserved (GiB)": 35.85,
46
  "memory/max_active (GiB)": 33.91,
47
  "memory/max_allocated (GiB)": 33.91,
48
+ "ppl": 1.1149,
49
  "step": 3,
50
  "tokens/total": 91136,
51
+ "tokens/train_per_sec_per_gpu": 34.67,
52
  "tokens/trainable": 1396
53
  },
54
  {
55
  "epoch": 0.015625,
56
+ "grad_norm": 2.2042534351348877,
57
  "learning_rate": 1.2e-05,
58
+ "loss": 0.1031389832496643,
59
  "memory/device_reserved (GiB)": 35.85,
60
  "memory/max_active (GiB)": 33.91,
61
  "memory/max_allocated (GiB)": 33.91,
62
+ "ppl": 1.10865,
63
  "step": 4,
64
  "tokens/total": 121552,
65
+ "tokens/train_per_sec_per_gpu": 38.0,
66
  "tokens/trainable": 1850
67
  },
68
  {
69
  "epoch": 0.01953125,
70
+ "grad_norm": 2.5755860805511475,
71
  "learning_rate": 1.6000000000000003e-05,
72
+ "loss": 0.1097191572189331,
73
  "memory/device_reserved (GiB)": 35.85,
74
  "memory/max_active (GiB)": 33.98,
75
  "memory/max_allocated (GiB)": 33.98,
76
+ "ppl": 1.11596,
77
  "step": 5,
78
  "tokens/total": 152304,
79
+ "tokens/train_per_sec_per_gpu": 34.92,
80
  "tokens/trainable": 2302
81
  },
82
  {
83
  "epoch": 0.0234375,
84
+ "grad_norm": 1.498002052307129,
85
  "learning_rate": 2e-05,
86
+ "loss": 0.08722387254238129,
87
  "memory/device_reserved (GiB)": 35.85,
88
  "memory/max_active (GiB)": 33.78,
89
  "memory/max_allocated (GiB)": 33.78,
90
+ "ppl": 1.09114,
91
  "step": 6,
92
  "tokens/total": 182448,
93
+ "tokens/train_per_sec_per_gpu": 31.65,
94
  "tokens/trainable": 2742
95
  },
96
  {
97
  "epoch": 0.02734375,
98
+ "grad_norm": 1.280110478401184,
99
  "learning_rate": 2.4e-05,
100
+ "loss": 0.07383580505847931,
101
  "memory/device_reserved (GiB)": 35.85,
102
  "memory/max_active (GiB)": 33.85,
103
  "memory/max_allocated (GiB)": 33.85,
104
+ "ppl": 1.07663,
105
  "step": 7,
106
  "tokens/total": 212736,
107
+ "tokens/train_per_sec_per_gpu": 36.7,
108
  "tokens/trainable": 3208
109
  },
110
  {
111
  "epoch": 0.03125,
112
+ "grad_norm": 1.6952791213989258,
113
  "learning_rate": 2.8000000000000003e-05,
114
+ "loss": 0.08001460134983063,
115
  "memory/device_reserved (GiB)": 36.07,
116
  "memory/max_active (GiB)": 33.87,
117
  "memory/max_allocated (GiB)": 33.87,
118
+ "ppl": 1.0833,
119
  "step": 8,
120
  "tokens/total": 243024,
121
+ "tokens/train_per_sec_per_gpu": 31.94,
122
  "tokens/trainable": 3655
123
  },
124
  {
125
  "epoch": 0.03515625,
126
+ "grad_norm": 1.2223162651062012,
127
  "learning_rate": 3.2000000000000005e-05,
128
+ "loss": 0.047361284494400024,
129
  "memory/device_reserved (GiB)": 36.07,
130
  "memory/max_active (GiB)": 33.82,
131
  "memory/max_allocated (GiB)": 33.82,
132
+ "ppl": 1.0485,
133
  "step": 9,
134
  "tokens/total": 271264,
135
+ "tokens/train_per_sec_per_gpu": 35.78,
136
  "tokens/trainable": 4091
137
  },
138
  {
139
  "epoch": 0.0390625,
140
+ "grad_norm": 2.902157783508301,
141
  "learning_rate": 3.6e-05,
142
+ "loss": 0.09570425748825073,
143
  "memory/device_reserved (GiB)": 36.07,
144
  "memory/max_active (GiB)": 33.86,
145
  "memory/max_allocated (GiB)": 33.86,
146
+ "ppl": 1.10043,
147
  "step": 10,
148
  "tokens/total": 301520,
149
+ "tokens/train_per_sec_per_gpu": 34.98,
150
  "tokens/trainable": 4553
151
  },
152
  {
153
  "epoch": 0.04296875,
154
+ "grad_norm": 2.1767208576202393,
155
  "learning_rate": 4e-05,
156
+ "loss": 0.0828268975019455,
157
+ "memory/device_reserved (GiB)": 36.08,
158
  "memory/max_active (GiB)": 33.8,
159
  "memory/max_allocated (GiB)": 33.8,
160
+ "ppl": 1.08635,
161
  "step": 11,
162
  "tokens/total": 331808,
163
+ "tokens/train_per_sec_per_gpu": 33.12,
164
  "tokens/trainable": 4992
165
  },
166
  {
167
  "epoch": 0.046875,
168
+ "grad_norm": 3.15231990814209,
169
  "learning_rate": 4.4000000000000006e-05,
170
+ "loss": 0.12141455709934235,
171
+ "memory/device_reserved (GiB)": 36.08,
172
  "memory/max_active (GiB)": 33.87,
173
  "memory/max_allocated (GiB)": 33.87,
174
+ "ppl": 1.12909,
175
  "step": 12,
176
  "tokens/total": 362224,
177
+ "tokens/train_per_sec_per_gpu": 38.32,
178
  "tokens/trainable": 5494
179
  },
180
  {
181
  "epoch": 0.05078125,
182
+ "grad_norm": 2.3834526538848877,
183
  "learning_rate": 4.8e-05,
184
+ "loss": 0.037937015295028687,
185
+ "memory/device_reserved (GiB)": 36.16,
186
  "memory/max_active (GiB)": 33.91,
187
  "memory/max_allocated (GiB)": 33.91,
188
+ "ppl": 1.03867,
189
  "step": 13,
190
  "tokens/total": 392432,
191
+ "tokens/train_per_sec_per_gpu": 35.04,
192
  "tokens/trainable": 5933
193
  },
194
  {
195
  "epoch": 0.0546875,
196
+ "grad_norm": 3.2587738037109375,
197
  "learning_rate": 5.2000000000000004e-05,
198
+ "loss": 0.06514571607112885,
199
+ "memory/device_reserved (GiB)": 36.16,
200
  "memory/max_active (GiB)": 33.79,
201
  "memory/max_allocated (GiB)": 33.79,
202
+ "ppl": 1.06731,
203
  "step": 14,
204
  "tokens/total": 422784,
205
+ "tokens/train_per_sec_per_gpu": 29.67,
206
  "tokens/trainable": 6369
207
  },
208
  {
209
  "epoch": 0.05859375,
210
+ "grad_norm": 1.5818979740142822,
211
  "learning_rate": 5.6000000000000006e-05,
212
+ "loss": 0.05656517297029495,
213
+ "memory/device_reserved (GiB)": 36.16,
214
  "memory/max_active (GiB)": 33.92,
215
  "memory/max_allocated (GiB)": 33.92,
216
+ "ppl": 1.0582,
217
  "step": 15,
218
  "tokens/total": 453376,
219
+ "tokens/train_per_sec_per_gpu": 33.08,
220
  "tokens/trainable": 6820
221
  },
222
  {
223
  "epoch": 0.0625,
224
+ "grad_norm": 1.4215443134307861,
225
  "learning_rate": 6e-05,
226
+ "loss": 0.06070145219564438,
227
  "memory/device_reserved (GiB)": 36.21,
228
  "memory/max_active (GiB)": 33.83,
229
  "memory/max_allocated (GiB)": 33.83,
230
+ "ppl": 1.06258,
231
  "step": 16,
232
  "tokens/total": 483504,
233
+ "tokens/train_per_sec_per_gpu": 30.85,
234
  "tokens/trainable": 7258
235
  },
236
  {
237
  "epoch": 0.06640625,
238
+ "grad_norm": 1.3065693378448486,
239
  "learning_rate": 6.400000000000001e-05,
240
+ "loss": 0.05027393624186516,
241
  "memory/device_reserved (GiB)": 36.21,
242
  "memory/max_active (GiB)": 33.9,
243
  "memory/max_allocated (GiB)": 33.9,
244
+ "ppl": 1.05156,
245
  "step": 17,
246
  "tokens/total": 514032,
247
+ "tokens/train_per_sec_per_gpu": 35.29,
248
  "tokens/trainable": 7710
249
  },
250
  {
251
  "epoch": 0.0703125,
252
+ "grad_norm": 0.6123363375663757,
253
  "learning_rate": 6.800000000000001e-05,
254
+ "loss": 0.028499452397227287,
255
  "memory/device_reserved (GiB)": 36.21,
256
  "memory/max_active (GiB)": 33.83,
257
  "memory/max_allocated (GiB)": 33.83,
258
+ "ppl": 1.02891,
259
  "step": 18,
260
  "tokens/total": 544432,
261
+ "tokens/train_per_sec_per_gpu": 31.76,
262
  "tokens/trainable": 8174
263
  },
264
  {
265
  "epoch": 0.07421875,
266
+ "grad_norm": 0.648410439491272,
267
  "learning_rate": 7.2e-05,
268
+ "loss": 0.031143227592110634,
269
  "memory/device_reserved (GiB)": 36.21,
270
  "memory/max_active (GiB)": 33.83,
271
  "memory/max_allocated (GiB)": 33.83,
272
+ "ppl": 1.03163,
273
  "step": 19,
274
  "tokens/total": 574880,
275
+ "tokens/train_per_sec_per_gpu": 34.47,
276
  "tokens/trainable": 8617
277
  },
278
  {
279
  "epoch": 0.078125,
280
+ "grad_norm": 0.6737155318260193,
281
  "learning_rate": 7.6e-05,
282
+ "loss": 0.030753308907151222,
283
  "memory/device_reserved (GiB)": 36.21,
284
  "memory/max_active (GiB)": 33.97,
285
  "memory/max_allocated (GiB)": 33.97,
286
+ "ppl": 1.03123,
287
  "step": 20,
288
  "tokens/total": 605408,
289
+ "tokens/train_per_sec_per_gpu": 30.37,
290
  "tokens/trainable": 9037
291
  },
292
  {
293
  "epoch": 0.08203125,
294
+ "grad_norm": 0.7464718222618103,
295
  "learning_rate": 8e-05,
296
+ "loss": 0.02451065182685852,
297
  "memory/device_reserved (GiB)": 36.21,
298
  "memory/max_active (GiB)": 33.77,
299
  "memory/max_allocated (GiB)": 33.77,
300
+ "ppl": 1.02481,
301
  "step": 21,
302
  "tokens/total": 635584,
303
+ "tokens/train_per_sec_per_gpu": 36.63,
304
  "tokens/trainable": 9503
305
  },
306
  {
307
  "epoch": 0.0859375,
308
+ "grad_norm": 0.9435750246047974,
309
  "learning_rate": 8.4e-05,
310
+ "loss": 0.017626767978072166,
311
  "memory/device_reserved (GiB)": 36.21,
312
  "memory/max_active (GiB)": 33.88,
313
  "memory/max_allocated (GiB)": 33.88,
314
+ "ppl": 1.01778,
315
  "step": 22,
316
  "tokens/total": 665952,
317
+ "tokens/train_per_sec_per_gpu": 36.79,
318
  "tokens/trainable": 9972
319
  },
320
  {
321
  "epoch": 0.08984375,
322
+ "grad_norm": 0.6369844079017639,
323
  "learning_rate": 8.800000000000001e-05,
324
+ "loss": 0.013959845528006554,
325
  "memory/device_reserved (GiB)": 36.21,
326
  "memory/max_active (GiB)": 33.9,
327
  "memory/max_allocated (GiB)": 33.9,
328
+ "ppl": 1.01406,
329
  "step": 23,
330
  "tokens/total": 696240,
331
+ "tokens/train_per_sec_per_gpu": 37.54,
332
  "tokens/trainable": 10447
333
  },
334
  {
335
  "epoch": 0.09375,
336
+ "grad_norm": 1.0666130781173706,
337
  "learning_rate": 9.200000000000001e-05,
338
+ "loss": 0.03303435444831848,
339
  "memory/device_reserved (GiB)": 36.35,
340
  "memory/max_active (GiB)": 33.88,
341
  "memory/max_allocated (GiB)": 33.88,
342
+ "ppl": 1.03359,
343
  "step": 24,
344
  "tokens/total": 726464,
345
+ "tokens/train_per_sec_per_gpu": 37.23,
346
  "tokens/trainable": 10899
347
  },
348
  {
349
  "epoch": 0.09765625,
350
+ "grad_norm": 1.0491507053375244,
351
  "learning_rate": 9.6e-05,
352
+ "loss": 0.030840374529361725,
353
  "memory/device_reserved (GiB)": 36.35,
354
  "memory/max_active (GiB)": 33.84,
355
  "memory/max_allocated (GiB)": 33.84,
356
+ "ppl": 1.03132,
357
  "step": 25,
358
  "tokens/total": 756704,
359
+ "tokens/train_per_sec_per_gpu": 33.1,
360
  "tokens/trainable": 11360
361
  },
362
  {
363
  "epoch": 0.1015625,
364
+ "grad_norm": 0.8108148574829102,
365
  "learning_rate": 0.0001,
366
+ "loss": 0.03408072516322136,
367
  "memory/device_reserved (GiB)": 36.35,
368
  "memory/max_active (GiB)": 33.82,
369
  "memory/max_allocated (GiB)": 33.82,
370
+ "ppl": 1.03467,
371
  "step": 26,
372
  "tokens/total": 786912,
373
+ "tokens/train_per_sec_per_gpu": 29.3,
374
  "tokens/trainable": 11775
375
  },
376
  {
377
  "epoch": 0.10546875,
378
+ "grad_norm": 0.507036030292511,
379
  "learning_rate": 9.99990636831587e-05,
380
+ "loss": 0.014000408351421356,
381
  "memory/device_reserved (GiB)": 36.35,
382
  "memory/max_active (GiB)": 33.46,
383
  "memory/max_allocated (GiB)": 33.46,
384
+ "ppl": 1.0141,
385
  "step": 27,
386
  "tokens/total": 815424,
387
+ "tokens/train_per_sec_per_gpu": 39.89,
388
  "tokens/trainable": 12242
389
  },
390
  {
391
  "epoch": 0.109375,
392
+ "grad_norm": 0.618457555770874,
393
  "learning_rate": 9.999625477159879e-05,
394
+ "loss": 0.022290699183940887,
395
  "memory/device_reserved (GiB)": 36.35,
396
  "memory/max_active (GiB)": 33.89,
397
  "memory/max_allocated (GiB)": 33.89,
398
+ "ppl": 1.02254,
399
  "step": 28,
400
  "tokens/total": 845584,
401
+ "tokens/train_per_sec_per_gpu": 33.48,
402
  "tokens/trainable": 12696
403
  },
404
  {
405
  "epoch": 0.11328125,
406
+ "grad_norm": 0.4989492893218994,
407
  "learning_rate": 9.999157338221051e-05,
408
+ "loss": 0.025926023721694946,
409
  "memory/device_reserved (GiB)": 36.35,
410
  "memory/max_active (GiB)": 33.87,
411
  "memory/max_allocated (GiB)": 33.87,
412
+ "ppl": 1.02627,
413
  "step": 29,
414
  "tokens/total": 875808,
415
+ "tokens/train_per_sec_per_gpu": 34.34,
416
  "tokens/trainable": 13153
417
  },
418
  {
419
  "epoch": 0.1171875,
420
+ "grad_norm": 0.3578357696533203,
421
  "learning_rate": 9.998501970980562e-05,
422
+ "loss": 0.014475762844085693,
423
  "memory/device_reserved (GiB)": 36.35,
424
  "memory/max_active (GiB)": 33.82,
425
  "memory/max_allocated (GiB)": 33.82,
426
+ "ppl": 1.01458,
427
  "step": 30,
428
  "tokens/total": 904096,
429
+ "tokens/train_per_sec_per_gpu": 39.7,
430
  "tokens/trainable": 13624
431
  },
432
  {
433
  "epoch": 0.12109375,
434
+ "grad_norm": 0.4404466450214386,
435
  "learning_rate": 9.997659402710915e-05,
436
+ "loss": 0.015927618369460106,
437
  "memory/device_reserved (GiB)": 36.35,
438
  "memory/max_active (GiB)": 33.86,
439
  "memory/max_allocated (GiB)": 33.86,
440
+ "ppl": 1.01606,
441
  "step": 31,
442
  "tokens/total": 934400,
443
+ "tokens/train_per_sec_per_gpu": 34.1,
444
  "tokens/trainable": 14064
445
  },
446
  {
447
  "epoch": 0.125,
448
+ "grad_norm": 0.5913511514663696,
449
  "learning_rate": 9.996629668474818e-05,
450
+ "loss": 0.019408874213695526,
451
  "memory/device_reserved (GiB)": 36.35,
452
  "memory/max_active (GiB)": 33.92,
453
  "memory/max_allocated (GiB)": 33.92,
454
+ "ppl": 1.0196,
455
  "step": 32,
456
  "tokens/total": 964832,
457
+ "tokens/train_per_sec_per_gpu": 38.92,
458
  "tokens/trainable": 14565
459
  },
460
  {
461
  "epoch": 0.12890625,
462
+ "grad_norm": 0.2795853614807129,
463
  "learning_rate": 9.995412811123711e-05,
464
+ "loss": 0.007002322003245354,
465
  "memory/device_reserved (GiB)": 36.35,
466
  "memory/max_active (GiB)": 33.79,
467
  "memory/max_allocated (GiB)": 33.79,
468
+ "ppl": 1.00703,
469
  "step": 33,
470
  "tokens/total": 995040,
471
+ "tokens/train_per_sec_per_gpu": 33.77,
472
  "tokens/trainable": 15005
473
  },
474
  {
475
  "epoch": 0.1328125,
476
+ "grad_norm": 1.1757413148880005,
477
  "learning_rate": 9.994008881295999e-05,
478
+ "loss": 0.021448152139782906,
479
  "memory/device_reserved (GiB)": 34.72,
480
  "memory/max_active (GiB)": 33.64,
481
  "memory/max_allocated (GiB)": 33.64,
482
+ "ppl": 1.02168,
483
  "step": 34,
484
  "tokens/total": 1024896,
485
+ "tokens/train_per_sec_per_gpu": 32.03,
486
  "tokens/trainable": 15459
487
  },
488
  {
489
  "epoch": 0.13671875,
490
+ "grad_norm": 0.3860406279563904,
491
  "learning_rate": 9.992417937414932e-05,
492
+ "loss": 0.01894465833902359,
493
  "memory/device_reserved (GiB)": 35.48,
494
  "memory/max_active (GiB)": 33.84,
495
  "memory/max_allocated (GiB)": 33.84,
496
+ "ppl": 1.01913,
497
  "step": 35,
498
  "tokens/total": 1054992,
499
+ "tokens/train_per_sec_per_gpu": 38.21,
500
  "tokens/trainable": 15960
501
  },
502
  {
503
  "epoch": 0.140625,
504
+ "grad_norm": 0.4803963005542755,
505
  "learning_rate": 9.99064004568618e-05,
506
+ "loss": 0.013810254633426666,
507
  "memory/device_reserved (GiB)": 35.48,
508
  "memory/max_active (GiB)": 33.93,
509
  "memory/max_allocated (GiB)": 33.93,
510
+ "ppl": 1.01391,
511
  "step": 36,
512
  "tokens/total": 1085280,
513
+ "tokens/train_per_sec_per_gpu": 34.59,
514
  "tokens/trainable": 16428
515
  },
516
  {
517
  "epoch": 0.14453125,
518
+ "grad_norm": 0.3357454836368561,
519
  "learning_rate": 9.988675280095074e-05,
520
+ "loss": 0.008235199376940727,
521
+ "memory/device_reserved (GiB)": 35.49,
522
  "memory/max_active (GiB)": 33.7,
523
  "memory/max_allocated (GiB)": 33.7,
524
+ "ppl": 1.00827,
525
  "step": 37,
526
  "tokens/total": 1115504,
527
+ "tokens/train_per_sec_per_gpu": 31.89,
528
  "tokens/trainable": 16884
529
  },
530
  {
531
  "epoch": 0.1484375,
532
+ "grad_norm": 0.9474877715110779,
533
  "learning_rate": 9.986523722403528e-05,
534
+ "loss": 0.019564270973205566,
535
+ "memory/device_reserved (GiB)": 35.49,
536
  "memory/max_active (GiB)": 33.82,
537
  "memory/max_allocated (GiB)": 33.82,
538
+ "ppl": 1.01976,
539
  "step": 38,
540
  "tokens/total": 1145712,
541
+ "tokens/train_per_sec_per_gpu": 32.02,
542
  "tokens/trainable": 17341
543
  },
544
  {
545
  "epoch": 0.15234375,
546
+ "grad_norm": 1.101553201675415,
547
  "learning_rate": 9.984185462146642e-05,
548
+ "loss": 0.017880277708172798,
549
+ "memory/device_reserved (GiB)": 35.49,
550
  "memory/max_active (GiB)": 33.81,
551
  "memory/max_allocated (GiB)": 33.81,
552
+ "ppl": 1.01804,
553
  "step": 39,
554
  "tokens/total": 1176128,
555
+ "tokens/train_per_sec_per_gpu": 34.16,
556
  "tokens/trainable": 17786
557
  },
558
  {
559
  "epoch": 0.15625,
560
+ "grad_norm": 0.7563315033912659,
561
  "learning_rate": 9.98166059662897e-05,
562
+ "loss": 0.019336596131324768,
563
+ "memory/device_reserved (GiB)": 35.49,
564
  "memory/max_active (GiB)": 33.85,
565
  "memory/max_allocated (GiB)": 33.85,
566
+ "ppl": 1.01952,
567
  "step": 40,
568
  "tokens/total": 1206576,
569
+ "tokens/train_per_sec_per_gpu": 35.01,
570
  "tokens/trainable": 18262
571
  },
572
  {
573
  "epoch": 0.16015625,
574
+ "grad_norm": 0.41576531529426575,
575
  "learning_rate": 9.978949230920472e-05,
576
+ "loss": 0.011620002798736095,
577
+ "memory/device_reserved (GiB)": 35.88,
578
  "memory/max_active (GiB)": 33.86,
579
  "memory/max_allocated (GiB)": 33.86,
580
+ "ppl": 1.01169,
581
  "step": 41,
582
  "tokens/total": 1236848,
583
+ "tokens/train_per_sec_per_gpu": 32.74,
584
  "tokens/trainable": 18716
585
  },
586
  {
587
  "epoch": 0.1640625,
588
+ "grad_norm": 1.180986762046814,
589
  "learning_rate": 9.976051477852141e-05,
590
+ "loss": 0.04661658778786659,
591
+ "memory/device_reserved (GiB)": 35.88,
592
  "memory/max_active (GiB)": 33.83,
593
  "memory/max_allocated (GiB)": 33.83,
594
+ "ppl": 1.04772,
595
  "step": 42,
596
  "tokens/total": 1267088,
597
+ "tokens/train_per_sec_per_gpu": 32.88,
598
  "tokens/trainable": 19157
599
  },
600
  {
601
  "epoch": 0.16796875,
602
+ "grad_norm": 0.7583066821098328,
603
  "learning_rate": 9.972967458011312e-05,
604
+ "loss": 0.029224852100014687,
605
+ "memory/device_reserved (GiB)": 35.88,
606
  "memory/max_active (GiB)": 33.82,
607
  "memory/max_allocated (GiB)": 33.82,
608
+ "ppl": 1.02966,
609
  "step": 43,
610
  "tokens/total": 1297360,
611
+ "tokens/train_per_sec_per_gpu": 36.53,
612
  "tokens/trainable": 19647
613
  },
614
  {
615
  "epoch": 0.171875,
616
+ "grad_norm": 0.18861345946788788,
617
  "learning_rate": 9.96969729973664e-05,
618
+ "loss": 0.007798851002007723,
619
+ "memory/device_reserved (GiB)": 35.88,
620
  "memory/max_active (GiB)": 33.86,
621
  "memory/max_allocated (GiB)": 33.86,
622
+ "ppl": 1.00783,
623
  "step": 44,
624
  "tokens/total": 1327744,
625
+ "tokens/train_per_sec_per_gpu": 34.91,
626
  "tokens/trainable": 20119
627
  },
628
  {
629
  "epoch": 0.17578125,
630
+ "grad_norm": 0.35095125436782837,
631
  "learning_rate": 9.966241139112754e-05,
632
+ "loss": 0.012044938281178474,
633
+ "memory/device_reserved (GiB)": 35.88,
634
  "memory/max_active (GiB)": 33.86,
635
  "memory/max_allocated (GiB)": 33.86,
636
+ "ppl": 1.01212,
637
  "step": 45,
638
  "tokens/total": 1358096,
639
+ "tokens/train_per_sec_per_gpu": 32.9,
640
  "tokens/trainable": 20560
641
  },
642
  {
643
  "epoch": 0.1796875,
644
+ "grad_norm": 0.5071477890014648,
645
  "learning_rate": 9.96259911996461e-05,
646
+ "loss": 0.017856616526842117,
647
+ "memory/device_reserved (GiB)": 35.88,
648
  "memory/max_active (GiB)": 33.84,
649
  "memory/max_allocated (GiB)": 33.84,
650
+ "ppl": 1.01802,
651
  "step": 46,
652
  "tokens/total": 1388240,
653
+ "tokens/train_per_sec_per_gpu": 32.63,
654
  "tokens/trainable": 20992
655
  },
656
  {
657
  "epoch": 0.18359375,
658
+ "grad_norm": 0.5569872260093689,
659
  "learning_rate": 9.958771393851491e-05,
660
+ "loss": 0.019440822303295135,
661
+ "memory/device_reserved (GiB)": 35.88,
662
  "memory/max_active (GiB)": 33.26,
663
  "memory/max_allocated (GiB)": 33.26,
664
+ "ppl": 1.01963,
665
  "step": 47,
666
  "tokens/total": 1416240,
667
+ "tokens/train_per_sec_per_gpu": 31.55,
668
  "tokens/trainable": 21441
669
  },
670
  {
671
  "epoch": 0.1875,
672
+ "grad_norm": 0.42062458395957947,
673
  "learning_rate": 9.954758120060702e-05,
674
+ "loss": 0.013018792495131493,
675
  "memory/device_reserved (GiB)": 35.88,
676
  "memory/max_active (GiB)": 33.96,
677
  "memory/max_allocated (GiB)": 33.96,
678
+ "ppl": 1.0131,
679
  "step": 48,
680
  "tokens/total": 1446880,
681
+ "tokens/train_per_sec_per_gpu": 33.68,
682
  "tokens/trainable": 21901
683
  },
684
  {
685
  "epoch": 0.19140625,
686
+ "grad_norm": 0.30396750569343567,
687
  "learning_rate": 9.950559465600948e-05,
688
+ "loss": 0.0077492957934737206,
689
  "memory/device_reserved (GiB)": 35.88,
690
  "memory/max_active (GiB)": 33.78,
691
  "memory/max_allocated (GiB)": 33.78,
692
+ "ppl": 1.00778,
693
  "step": 49,
694
  "tokens/total": 1477168,
695
+ "tokens/train_per_sec_per_gpu": 33.33,
696
  "tokens/trainable": 22341
697
  },
698
  {
699
  "epoch": 0.1953125,
700
+ "grad_norm": 2.044849395751953,
701
  "learning_rate": 9.946175605195379e-05,
702
+ "loss": 0.014447445049881935,
703
  "memory/device_reserved (GiB)": 35.88,
704
  "memory/max_active (GiB)": 33.89,
705
  "memory/max_allocated (GiB)": 33.89,
706
+ "ppl": 1.01455,
707
  "step": 50,
708
  "tokens/total": 1507712,
709
+ "tokens/train_per_sec_per_gpu": 33.0,
710
  "tokens/trainable": 22833
711
  },
712
  {
713
  "epoch": 0.19921875,
714
+ "grad_norm": 0.9357694983482361,
715
  "learning_rate": 9.941606721274322e-05,
716
+ "loss": 0.012720856815576553,
717
  "memory/device_reserved (GiB)": 35.88,
718
  "memory/max_active (GiB)": 33.81,
719
  "memory/max_allocated (GiB)": 33.81,
720
+ "ppl": 1.0128,
721
  "step": 51,
722
  "tokens/total": 1537936,
723
+ "tokens/train_per_sec_per_gpu": 30.72,
724
  "tokens/trainable": 23277
725
  },
726
  {
727
  "epoch": 0.203125,
728
+ "grad_norm": 0.2881873548030853,
729
  "learning_rate": 9.936853003967685e-05,
730
+ "loss": 0.005877626594156027,
731
  "memory/device_reserved (GiB)": 35.88,
732
  "memory/max_active (GiB)": 33.84,
733
  "memory/max_allocated (GiB)": 33.84,
734
+ "ppl": 1.00589,
735
  "step": 52,
736
  "tokens/total": 1568352,
737
+ "tokens/train_per_sec_per_gpu": 35.63,
738
  "tokens/trainable": 23759
739
  },
740
  {
741
  "epoch": 0.20703125,
742
+ "grad_norm": 0.7577692270278931,
743
  "learning_rate": 9.93191465109705e-05,
744
+ "loss": 0.01451955921947956,
745
  "memory/device_reserved (GiB)": 35.88,
746
  "memory/max_active (GiB)": 33.91,
747
  "memory/max_allocated (GiB)": 33.91,
748
+ "ppl": 1.01463,
749
  "step": 53,
750
  "tokens/total": 1598992,
751
+ "tokens/train_per_sec_per_gpu": 32.95,
752
  "tokens/trainable": 24193
753
  },
754
  {
755
  "epoch": 0.2109375,
756
+ "grad_norm": 0.5201014280319214,
757
  "learning_rate": 9.926791868167438e-05,
758
+ "loss": 0.006856432184576988,
759
  "memory/device_reserved (GiB)": 35.9,
760
  "memory/max_active (GiB)": 33.95,
761
  "memory/max_allocated (GiB)": 33.95,
762
+ "ppl": 1.00688,
763
  "step": 54,
764
  "tokens/total": 1629488,
765
+ "tokens/train_per_sec_per_gpu": 33.59,
766
  "tokens/trainable": 24619
767
  },
768
  {
769
  "epoch": 0.21484375,
770
+ "grad_norm": 0.8399921655654907,
771
  "learning_rate": 9.921484868358753e-05,
772
+ "loss": 0.037967782467603683,
773
  "memory/device_reserved (GiB)": 35.9,
774
  "memory/max_active (GiB)": 33.8,
775
  "memory/max_allocated (GiB)": 33.8,
776
+ "ppl": 1.0387,
777
  "step": 55,
778
  "tokens/total": 1659696,
779
+ "tokens/train_per_sec_per_gpu": 32.85,
780
  "tokens/trainable": 25075
781
  },
782
  {
783
  "epoch": 0.21875,
784
+ "grad_norm": 0.8203506469726562,
785
  "learning_rate": 9.915993872516924e-05,
786
+ "loss": 0.012814272195100784,
787
  "memory/device_reserved (GiB)": 35.9,
788
  "memory/max_active (GiB)": 33.76,
789
  "memory/max_allocated (GiB)": 33.76,
790
+ "ppl": 1.0129,
791
  "step": 56,
792
  "tokens/total": 1689872,
793
+ "tokens/train_per_sec_per_gpu": 31.55,
794
  "tokens/trainable": 25519
795
  },
796
  {
797
  "epoch": 0.22265625,
798
+ "grad_norm": 0.20874246954917908,
799
  "learning_rate": 9.9103191091447e-05,
800
+ "loss": 0.00539036700502038,
801
  "memory/device_reserved (GiB)": 35.9,
802
  "memory/max_active (GiB)": 33.9,
803
  "memory/max_allocated (GiB)": 33.9,
804
+ "ppl": 1.0054,
805
  "step": 57,
806
  "tokens/total": 1720144,
807
+ "tokens/train_per_sec_per_gpu": 32.69,
808
  "tokens/trainable": 25966
809
  },
810
  {
811
  "epoch": 0.2265625,
812
+ "grad_norm": 0.4427756071090698,
813
  "learning_rate": 9.904460814392147e-05,
814
+ "loss": 0.009390819817781448,
815
  "memory/device_reserved (GiB)": 35.9,
816
  "memory/max_active (GiB)": 33.89,
817
  "memory/max_allocated (GiB)": 33.89,
818
+ "ppl": 1.00944,
819
  "step": 58,
820
  "tokens/total": 1750448,
821
+ "tokens/train_per_sec_per_gpu": 33.98,
822
  "tokens/trainable": 26423
823
  },
824
  {
825
  "epoch": 0.23046875,
826
+ "grad_norm": 0.7457786798477173,
827
  "learning_rate": 9.898419232046825e-05,
828
+ "loss": 0.019530881196260452,
829
  "memory/device_reserved (GiB)": 35.9,
830
  "memory/max_active (GiB)": 33.93,
831
  "memory/max_allocated (GiB)": 33.93,
832
+ "ppl": 1.01972,
833
  "step": 59,
834
  "tokens/total": 1781088,
835
+ "tokens/train_per_sec_per_gpu": 38.51,
836
  "tokens/trainable": 26934
837
  },
838
  {
839
  "epoch": 0.234375,
840
+ "grad_norm": 0.13402195274829865,
841
  "learning_rate": 9.892194613523633e-05,
842
+ "loss": 0.0014544172445312142,
843
  "memory/device_reserved (GiB)": 35.9,
844
  "memory/max_active (GiB)": 33.84,
845
  "memory/max_allocated (GiB)": 33.84,
846
+ "ppl": 1.00146,
847
  "step": 60,
848
  "tokens/total": 1811344,
849
+ "tokens/train_per_sec_per_gpu": 32.96,
850
  "tokens/trainable": 27393
851
  },
852
  {
853
  "epoch": 0.23828125,
854
+ "grad_norm": 1.0385031700134277,
855
  "learning_rate": 9.885787217854357e-05,
856
+ "loss": 0.019916968420147896,
857
  "memory/device_reserved (GiB)": 35.96,
858
  "memory/max_active (GiB)": 33.9,
859
  "memory/max_allocated (GiB)": 33.9,
860
+ "ppl": 1.02012,
861
  "step": 61,
862
  "tokens/total": 1841600,
863
+ "tokens/train_per_sec_per_gpu": 32.94,
864
  "tokens/trainable": 27852
865
  },
866
  {
867
  "epoch": 0.2421875,
868
+ "grad_norm": 1.2596747875213623,
869
  "learning_rate": 9.879197311676887e-05,
870
+ "loss": 0.015884939581155777,
871
  "memory/device_reserved (GiB)": 35.96,
872
  "memory/max_active (GiB)": 33.82,
873
  "memory/max_allocated (GiB)": 33.82,
874
+ "ppl": 1.01601,
875
  "step": 62,
876
  "tokens/total": 1872048,
877
+ "tokens/train_per_sec_per_gpu": 32.96,
878
  "tokens/trainable": 28292
879
  },
880
  {
881
  "epoch": 0.24609375,
882
+ "grad_norm": 0.14031143486499786,
883
  "learning_rate": 9.872425169224113e-05,
884
+ "loss": 0.002796083688735962,
885
  "memory/device_reserved (GiB)": 35.96,
886
  "memory/max_active (GiB)": 33.94,
887
  "memory/max_allocated (GiB)": 33.94,
888
+ "ppl": 1.0028,
889
  "step": 63,
890
  "tokens/total": 1902592,
891
+ "tokens/train_per_sec_per_gpu": 37.36,
892
  "tokens/trainable": 28798
893
  },
894
  {
895
  "epoch": 0.25,
896
+ "grad_norm": 0.6247786283493042,
897
  "learning_rate": 9.865471072312528e-05,
898
+ "loss": 0.017117884010076523,
899
  "memory/device_reserved (GiB)": 35.96,
900
  "memory/max_active (GiB)": 33.96,
901
  "memory/max_allocated (GiB)": 33.96,
902
+ "ppl": 1.01727,
903
  "step": 64,
904
  "tokens/total": 1933088,
905
+ "tokens/train_per_sec_per_gpu": 35.79,
906
  "tokens/trainable": 29283
907
  },
908
  {
909
  "epoch": 0.25390625,
910
+ "grad_norm": 0.3513385057449341,
911
  "learning_rate": 9.858335310330492e-05,
912
+ "loss": 0.008029159158468246,
913
  "memory/device_reserved (GiB)": 35.96,
914
  "memory/max_active (GiB)": 33.72,
915
  "memory/max_allocated (GiB)": 33.72,
916
+ "ppl": 1.00806,
917
  "step": 65,
918
  "tokens/total": 1963296,
919
+ "tokens/train_per_sec_per_gpu": 31.99,
920
  "tokens/trainable": 29745
921
  },
922
  {
923
  "epoch": 0.2578125,
924
+ "grad_norm": 0.279448539018631,
925
  "learning_rate": 9.851018180226185e-05,
926
+ "loss": 0.0161186084151268,
927
+ "memory/device_reserved (GiB)": 34.87,
928
  "memory/max_active (GiB)": 33.79,
929
  "memory/max_allocated (GiB)": 33.79,
930
+ "ppl": 1.01625,
931
  "step": 66,
932
  "tokens/total": 1993760,
933
  "tokens/train_per_sec_per_gpu": 31.19,
 
935
  },
936
  {
937
  "epoch": 0.26171875,
938
+ "grad_norm": 0.31739094853401184,
939
  "learning_rate": 9.843519986495259e-05,
940
+ "loss": 0.009091717191040516,
941
  "memory/device_reserved (GiB)": 35.86,
942
  "memory/max_active (GiB)": 33.89,
943
  "memory/max_allocated (GiB)": 33.89,
944
+ "ppl": 1.00913,
945
  "step": 67,
946
  "tokens/total": 2024144,
947
+ "tokens/train_per_sec_per_gpu": 33.36,
948
  "tokens/trainable": 30622
949
  },
950
  {
951
  "epoch": 0.265625,
952
+ "grad_norm": 0.3539387285709381,
953
  "learning_rate": 9.835841041168162e-05,
954
+ "loss": 0.00908343680202961,
955
  "memory/device_reserved (GiB)": 35.86,
956
  "memory/max_active (GiB)": 33.95,
957
  "memory/max_allocated (GiB)": 33.95,
958
+ "ppl": 1.00912,
959
  "step": 68,
960
  "tokens/total": 2054608,
961
+ "tokens/train_per_sec_per_gpu": 35.52,
962
  "tokens/trainable": 31097
963
  },
964
  {
965
  "epoch": 0.26953125,
966
+ "grad_norm": 0.6497699618339539,
967
  "learning_rate": 9.82798166379715e-05,
968
+ "loss": 0.03086906298995018,
969
  "memory/device_reserved (GiB)": 35.86,
970
  "memory/max_active (GiB)": 33.82,
971
  "memory/max_allocated (GiB)": 33.82,
972
+ "ppl": 1.03135,
973
  "step": 69,
974
  "tokens/total": 2084912,
975
+ "tokens/train_per_sec_per_gpu": 36.6,
976
  "tokens/trainable": 31595
977
  },
978
  {
979
  "epoch": 0.2734375,
980
+ "grad_norm": 0.19664841890335083,
981
  "learning_rate": 9.819942181443002e-05,
982
+ "loss": 0.0039155250415205956,
983
  "memory/device_reserved (GiB)": 35.86,
984
  "memory/max_active (GiB)": 33.84,
985
  "memory/max_allocated (GiB)": 33.84,
986
+ "ppl": 1.00392,
987
  "step": 70,
988
  "tokens/total": 2115216,
989
+ "tokens/train_per_sec_per_gpu": 30.75,
990
  "tokens/trainable": 32036
991
  },
992
  {
993
  "epoch": 0.27734375,
994
+ "grad_norm": 0.4300064146518707,
995
  "learning_rate": 9.811722928661392e-05,
996
+ "loss": 0.007546662352979183,
997
  "memory/device_reserved (GiB)": 35.86,
998
  "memory/max_active (GiB)": 33.9,
999
  "memory/max_allocated (GiB)": 33.9,
1000
+ "ppl": 1.00758,
1001
  "step": 71,
1002
  "tokens/total": 2145424,
1003
+ "tokens/train_per_sec_per_gpu": 28.09,
1004
  "tokens/trainable": 32428
1005
  },
1006
  {
1007
  "epoch": 0.28125,
1008
+ "grad_norm": 0.027750927954912186,
1009
  "learning_rate": 9.803324247488975e-05,
1010
+ "loss": 0.0004817460721824318,
1011
  "memory/device_reserved (GiB)": 35.88,
1012
  "memory/max_active (GiB)": 33.84,
1013
  "memory/max_allocated (GiB)": 33.84,
1014
+ "ppl": 1.00048,
1015
  "step": 72,
1016
  "tokens/total": 2175648,
1017
+ "tokens/train_per_sec_per_gpu": 32.11,
1018
  "tokens/trainable": 32880
1019
  },
1020
  {
1021
  "epoch": 0.28515625,
1022
+ "grad_norm": 0.11202923208475113,
1023
  "learning_rate": 9.794746487429161e-05,
1024
+ "loss": 0.0012140646576881409,
1025
  "memory/device_reserved (GiB)": 35.88,
1026
  "memory/max_active (GiB)": 33.79,
1027
  "memory/max_allocated (GiB)": 33.79,
1028
+ "ppl": 1.00121,
1029
  "step": 73,
1030
  "tokens/total": 2205856,
1031
+ "tokens/train_per_sec_per_gpu": 33.51,
1032
  "tokens/trainable": 33323
1033
  },
1034
  {
1035
  "epoch": 0.2890625,
1036
+ "grad_norm": 0.026963796466588974,
1037
  "learning_rate": 9.785990005437554e-05,
1038
+ "loss": 0.00046908354852348566,
1039
  "memory/device_reserved (GiB)": 35.88,
1040
  "memory/max_active (GiB)": 33.95,
1041
  "memory/max_allocated (GiB)": 33.95,
1042
+ "ppl": 1.00047,
1043
  "step": 74,
1044
  "tokens/total": 2236352,
1045
+ "tokens/train_per_sec_per_gpu": 35.86,
1046
  "tokens/trainable": 33792
1047
  },
1048
  {
1049
  "epoch": 0.29296875,
1050
+ "grad_norm": 0.5172365307807922,
1051
  "learning_rate": 9.777055165907117e-05,
1052
+ "loss": 0.029645610600709915,
1053
  "memory/device_reserved (GiB)": 35.88,
1054
  "memory/max_active (GiB)": 33.8,
1055
  "memory/max_allocated (GiB)": 33.8,
1056
+ "ppl": 1.03009,
1057
  "step": 75,
1058
  "tokens/total": 2266480,
1059
+ "tokens/train_per_sec_per_gpu": 35.26,
1060
  "tokens/trainable": 34223
1061
  },
1062
  {
1063
  "epoch": 0.296875,
1064
+ "grad_norm": 0.84085613489151,
1065
  "learning_rate": 9.767942340652993e-05,
1066
+ "loss": 0.006980063393712044,
1067
  "memory/device_reserved (GiB)": 35.88,
1068
  "memory/max_active (GiB)": 33.71,
1069
  "memory/max_allocated (GiB)": 33.71,
1070
+ "ppl": 1.007,
1071
  "step": 76,
1072
  "tokens/total": 2296496,
1073
+ "tokens/train_per_sec_per_gpu": 29.61,
1074
  "tokens/trainable": 34649
1075
  },
1076
  {
1077
  "epoch": 0.30078125,
1078
+ "grad_norm": 3.4935519695281982,
1079
  "learning_rate": 9.758651908897035e-05,
1080
+ "loss": 0.008870027028024197,
1081
  "memory/device_reserved (GiB)": 35.88,
1082
  "memory/max_active (GiB)": 33.88,
1083
  "memory/max_allocated (GiB)": 33.88,
1084
+ "ppl": 1.00891,
1085
  "step": 77,
1086
  "tokens/total": 2327040,
1087
+ "tokens/train_per_sec_per_gpu": 35.64,
1088
  "tokens/trainable": 35094
1089
  },
1090
  {
1091
  "epoch": 0.3046875,
1092
+ "grad_norm": 0.9529178142547607,
1093
  "learning_rate": 9.749184257252033e-05,
1094
+ "loss": 0.032071419060230255,
1095
  "memory/device_reserved (GiB)": 35.88,
1096
  "memory/max_active (GiB)": 33.81,
1097
  "memory/max_allocated (GiB)": 33.81,
1098
+ "ppl": 1.03259,
1099
  "step": 78,
1100
  "tokens/total": 2357328,
1101
+ "tokens/train_per_sec_per_gpu": 31.82,
1102
  "tokens/trainable": 35534
1103
  },
1104
  {
1105
  "epoch": 0.30859375,
1106
+ "grad_norm": 0.5781691074371338,
1107
  "learning_rate": 9.739539779705614e-05,
1108
+ "loss": 0.01475254725664854,
1109
+ "memory/device_reserved (GiB)": 35.89,
1110
  "memory/max_active (GiB)": 33.85,
1111
  "memory/max_allocated (GiB)": 33.85,
1112
+ "ppl": 1.01486,
1113
  "step": 79,
1114
  "tokens/total": 2387664,
1115
+ "tokens/train_per_sec_per_gpu": 35.82,
1116
  "tokens/trainable": 36029
1117
  },
1118
  {
1119
  "epoch": 0.3125,
1120
+ "grad_norm": 0.15085959434509277,
1121
  "learning_rate": 9.729718877603861e-05,
1122
+ "loss": 0.002578896936029196,
1123
+ "memory/device_reserved (GiB)": 35.89,
1124
  "memory/max_active (GiB)": 33.96,
1125
  "memory/max_allocated (GiB)": 33.96,
1126
+ "ppl": 1.00258,
1127
  "step": 80,
1128
  "tokens/total": 2418000,
1129
+ "tokens/train_per_sec_per_gpu": 35.53,
1130
  "tokens/trainable": 36469
1131
  },
1132
  {
1133
  "epoch": 0.31640625,
1134
+ "grad_norm": 0.19004814326763153,
1135
  "learning_rate": 9.719721959634592e-05,
1136
+ "loss": 0.004292497877031565,
1137
+ "memory/device_reserved (GiB)": 35.89,
1138
  "memory/max_active (GiB)": 33.96,
1139
  "memory/max_allocated (GiB)": 33.96,
1140
+ "ppl": 1.0043,
1141
  "step": 81,
1142
  "tokens/total": 2448720,
1143
+ "tokens/train_per_sec_per_gpu": 30.75,
1144
  "tokens/trainable": 36906
1145
  },
1146
  {
1147
  "epoch": 0.3203125,
1148
+ "grad_norm": 0.4505981504917145,
1149
  "learning_rate": 9.709549441810375e-05,
1150
+ "loss": 0.018529793247580528,
1151
+ "memory/device_reserved (GiB)": 35.89,
1152
  "memory/max_active (GiB)": 33.92,
1153
  "memory/max_allocated (GiB)": 33.92,
1154
+ "ppl": 1.0187,
1155
  "step": 82,
1156
  "tokens/total": 2479248,
1157
+ "tokens/train_per_sec_per_gpu": 38.95,
1158
  "tokens/trainable": 37390
1159
  },
1160
  {
1161
  "epoch": 0.32421875,
1162
+ "grad_norm": 0.4981430470943451,
1163
  "learning_rate": 9.699201747451195e-05,
1164
+ "loss": 0.014818452298641205,
1165
+ "memory/device_reserved (GiB)": 35.89,
1166
  "memory/max_active (GiB)": 33.78,
1167
  "memory/max_allocated (GiB)": 33.78,
1168
+ "ppl": 1.01493,
1169
  "step": 83,
1170
  "tokens/total": 2509408,
1171
+ "tokens/train_per_sec_per_gpu": 30.8,
1172
  "tokens/trainable": 37838
1173
  },
1174
  {
1175
  "epoch": 0.328125,
1176
+ "grad_norm": 0.16379471123218536,
1177
  "learning_rate": 9.688679307166854e-05,
1178
+ "loss": 0.003185997949913144,
1179
+ "memory/device_reserved (GiB)": 35.89,
1180
  "memory/max_active (GiB)": 33.97,
1181
  "memory/max_allocated (GiB)": 33.97,
1182
+ "ppl": 1.00319,
1183
  "step": 84,
1184
  "tokens/total": 2539776,
1185
+ "tokens/train_per_sec_per_gpu": 35.02,
1186
  "tokens/trainable": 38310
1187
  },
1188
  {
1189
  "epoch": 0.33203125,
1190
+ "grad_norm": 0.40732133388519287,
1191
  "learning_rate": 9.677982558839042e-05,
1192
+ "loss": 0.020803559571504593,
1193
+ "memory/device_reserved (GiB)": 35.89,
1194
  "memory/max_active (GiB)": 33.75,
1195
  "memory/max_allocated (GiB)": 33.75,
1196
+ "ppl": 1.02102,
1197
  "step": 85,
1198
  "tokens/total": 2569840,
1199
+ "tokens/train_per_sec_per_gpu": 34.03,
1200
  "tokens/trainable": 38789
1201
  },
1202
  {
1203
  "epoch": 0.3359375,
1204
+ "grad_norm": 0.3687220513820648,
1205
  "learning_rate": 9.66711194760312e-05,
1206
+ "loss": 0.012931328266859055,
1207
+ "memory/device_reserved (GiB)": 35.89,
1208
  "memory/max_active (GiB)": 33.86,
1209
  "memory/max_allocated (GiB)": 33.86,
1210
+ "ppl": 1.01302,
1211
  "step": 86,
1212
  "tokens/total": 2600192,
1213
+ "tokens/train_per_sec_per_gpu": 36.36,
1214
  "tokens/trainable": 39277
1215
  },
1216
  {
1217
  "epoch": 0.33984375,
1218
+ "grad_norm": 0.367418110370636,
1219
  "learning_rate": 9.656067925829593e-05,
1220
+ "loss": 0.01382569782435894,
1221
+ "memory/device_reserved (GiB)": 35.89,
1222
  "memory/max_active (GiB)": 33.93,
1223
  "memory/max_allocated (GiB)": 33.93,
1224
+ "ppl": 1.01392,
1225
  "step": 87,
1226
  "tokens/total": 2630640,
1227
+ "tokens/train_per_sec_per_gpu": 35.68,
1228
  "tokens/trainable": 39737
1229
  },
1230
  {
1231
  "epoch": 0.34375,
1232
+ "grad_norm": 0.2704487144947052,
1233
  "learning_rate": 9.644850953105288e-05,
1234
+ "loss": 0.008717816323041916,
1235
+ "memory/device_reserved (GiB)": 35.89,
1236
  "memory/max_active (GiB)": 33.74,
1237
  "memory/max_allocated (GiB)": 33.74,
1238
+ "ppl": 1.00876,
1239
  "step": 88,
1240
  "tokens/total": 2660832,
1241
+ "tokens/train_per_sec_per_gpu": 29.6,
1242
  "tokens/trainable": 40179
1243
  },
1244
  {
1245
  "epoch": 0.34765625,
1246
+ "grad_norm": 3.374918222427368,
1247
  "learning_rate": 9.633461496214225e-05,
1248
+ "loss": 0.01938408613204956,
1249
+ "memory/device_reserved (GiB)": 35.89,
1250
  "memory/max_active (GiB)": 33.88,
1251
  "memory/max_allocated (GiB)": 33.88,
1252
+ "ppl": 1.01957,
1253
  "step": 89,
1254
  "tokens/total": 2691328,
1255
+ "tokens/train_per_sec_per_gpu": 32.66,
1256
  "tokens/trainable": 40649
1257
  },
1258
  {
1259
  "epoch": 0.3515625,
1260
+ "grad_norm": 0.4690157175064087,
1261
  "learning_rate": 9.621900029118195e-05,
1262
+ "loss": 0.02013186179101467,
1263
+ "memory/device_reserved (GiB)": 35.89,
1264
  "memory/max_active (GiB)": 33.41,
1265
  "memory/max_allocated (GiB)": 33.41,
1266
+ "ppl": 1.02034,
1267
  "step": 90,
1268
  "tokens/total": 2719696,
1269
+ "tokens/train_per_sec_per_gpu": 31.46,
1270
  "tokens/trainable": 41080
1271
  },
1272
  {
1273
  "epoch": 0.35546875,
1274
+ "grad_norm": 0.08705829828977585,
1275
  "learning_rate": 9.610167032937036e-05,
1276
+ "loss": 0.002885152120143175,
1277
+ "memory/device_reserved (GiB)": 35.89,
1278
  "memory/max_active (GiB)": 33.94,
1279
  "memory/max_allocated (GiB)": 33.94,
1280
+ "ppl": 1.00289,
1281
  "step": 91,
1282
  "tokens/total": 2750272,
1283
+ "tokens/train_per_sec_per_gpu": 38.11,
1284
  "tokens/trainable": 41599
1285
  },
1286
  {
1287
  "epoch": 0.359375,
1288
+ "grad_norm": 8.197907447814941,
1289
  "learning_rate": 9.598262995928611e-05,
1290
+ "loss": 0.00822490081191063,
1291
+ "memory/device_reserved (GiB)": 35.89,
1292
  "memory/max_active (GiB)": 33.92,
1293
  "memory/max_allocated (GiB)": 33.92,
1294
+ "ppl": 1.00826,
1295
  "step": 92,
1296
  "tokens/total": 2780656,
1297
+ "tokens/train_per_sec_per_gpu": 36.82,
1298
  "tokens/trainable": 42076
1299
  },
1300
  {
1301
  "epoch": 0.36328125,
1302
+ "grad_norm": 0.14939527213573456,
1303
  "learning_rate": 9.586188413468492e-05,
1304
+ "loss": 0.004234164021909237,
1305
+ "memory/device_reserved (GiB)": 35.89,
1306
  "memory/max_active (GiB)": 33.82,
1307
  "memory/max_allocated (GiB)": 33.82,
1308
+ "ppl": 1.00424,
1309
  "step": 93,
1310
  "tokens/total": 2811088,
1311
+ "tokens/train_per_sec_per_gpu": 33.71,
1312
  "tokens/trainable": 42522
1313
  },
1314
  {
1315
  "epoch": 0.3671875,
1316
+ "grad_norm": 0.15404288470745087,
1317
  "learning_rate": 9.57394378802934e-05,
1318
+ "loss": 0.009490479715168476,
1319
  "memory/device_reserved (GiB)": 36.1,
1320
  "memory/max_active (GiB)": 33.9,
1321
  "memory/max_allocated (GiB)": 33.9,
1322
+ "ppl": 1.00954,
1323
  "step": 94,
1324
  "tokens/total": 2841520,
1325
+ "tokens/train_per_sec_per_gpu": 34.4,
1326
  "tokens/trainable": 42974
1327
  },
1328
  {
1329
  "epoch": 0.37109375,
1330
+ "grad_norm": 0.5274825692176819,
1331
  "learning_rate": 9.56152962916e-05,
1332
+ "loss": 0.02413639798760414,
1333
  "memory/device_reserved (GiB)": 36.1,
1334
  "memory/max_active (GiB)": 33.7,
1335
  "memory/max_allocated (GiB)": 33.7,
1336
+ "ppl": 1.02443,
1337
  "step": 95,
1338
  "tokens/total": 2871600,
1339
+ "tokens/train_per_sec_per_gpu": 30.67,
1340
  "tokens/trainable": 43380
1341
  },
1342
  {
1343
  "epoch": 0.375,
1344
+ "grad_norm": 0.5182522535324097,
1345
  "learning_rate": 9.548946453464296e-05,
1346
+ "loss": 0.009927366860210896,
1347
  "memory/device_reserved (GiB)": 36.1,
1348
  "memory/max_active (GiB)": 33.94,
1349
  "memory/max_allocated (GiB)": 33.94,
1350
+ "ppl": 1.00998,
1351
  "step": 96,
1352
  "tokens/total": 2902144,
1353
+ "tokens/train_per_sec_per_gpu": 34.14,
1354
  "tokens/trainable": 43865
1355
  },
1356
  {
1357
  "epoch": 0.37890625,
1358
+ "grad_norm": 0.5578822493553162,
1359
  "learning_rate": 9.53619478457953e-05,
1360
+ "loss": 0.014485684223473072,
1361
  "memory/device_reserved (GiB)": 36.1,
1362
  "memory/max_active (GiB)": 33.82,
1363
  "memory/max_allocated (GiB)": 33.82,
1364
+ "ppl": 1.01459,
1365
  "step": 97,
1366
  "tokens/total": 2932272,
1367
+ "tokens/train_per_sec_per_gpu": 31.94,
1368
  "tokens/trainable": 44328
1369
  },
1370
  {
1371
  "epoch": 0.3828125,
1372
+ "grad_norm": 0.10520734637975693,
1373
  "learning_rate": 9.523275153154695e-05,
1374
+ "loss": 0.0021552909165620804,
1375
  "memory/device_reserved (GiB)": 35.64,
1376
  "memory/max_active (GiB)": 33.69,
1377
  "memory/max_allocated (GiB)": 33.69,
1378
+ "ppl": 1.00216,
1379
  "step": 98,
1380
  "tokens/total": 2962032,
1381
+ "tokens/train_per_sec_per_gpu": 30.95,
1382
  "tokens/trainable": 44778
1383
  },
1384
  {
1385
  "epoch": 0.38671875,
1386
+ "grad_norm": 0.7544558644294739,
1387
  "learning_rate": 9.51018809682839e-05,
1388
+ "loss": 0.01703210547566414,
1389
  "memory/device_reserved (GiB)": 35.64,
1390
  "memory/max_active (GiB)": 33.89,
1391
  "memory/max_allocated (GiB)": 33.89,
1392
+ "ppl": 1.01718,
1393
  "step": 99,
1394
  "tokens/total": 2992256,
1395
+ "tokens/train_per_sec_per_gpu": 37.12,
1396
  "tokens/trainable": 45225
1397
  },
1398
  {
1399
  "epoch": 0.390625,
1400
+ "grad_norm": 0.3857012689113617,
1401
  "learning_rate": 9.49693416020645e-05,
1402
+ "loss": 0.00604570098221302,
1403
  "memory/device_reserved (GiB)": 35.64,
1404
  "memory/max_active (GiB)": 33.8,
1405
  "memory/max_allocated (GiB)": 33.8,
1406
+ "ppl": 1.00606,
1407
  "step": 100,
1408
  "tokens/total": 3022608,
1409
+ "tokens/train_per_sec_per_gpu": 35.75,
1410
  "tokens/trainable": 45678
1411
  },
1412
  {
1413
  "epoch": 0.39453125,
1414
+ "grad_norm": 0.21589453518390656,
1415
  "learning_rate": 9.483513894839276e-05,
1416
+ "loss": 0.005753816105425358,
1417
  "memory/device_reserved (GiB)": 35.78,
1418
  "memory/max_active (GiB)": 33.88,
1419
  "memory/max_allocated (GiB)": 33.88,
1420
+ "ppl": 1.00577,
1421
  "step": 101,
1422
  "tokens/total": 3052992,
1423
+ "tokens/train_per_sec_per_gpu": 32.72,
1424
  "tokens/trainable": 46125
1425
  },
1426
  {
1427
  "epoch": 0.3984375,
1428
+ "grad_norm": 1.1246687173843384,
1429
  "learning_rate": 9.469927859198888e-05,
1430
+ "loss": 0.006994037888944149,
1431
+ "memory/device_reserved (GiB)": 35.8,
1432
  "memory/max_active (GiB)": 33.98,
1433
  "memory/max_allocated (GiB)": 33.98,
1434
+ "ppl": 1.00702,
1435
  "step": 102,
1436
  "tokens/total": 3083392,
1437
+ "tokens/train_per_sec_per_gpu": 39.6,
1438
  "tokens/trainable": 46612
1439
  },
1440
  {
1441
  "epoch": 0.40234375,
1442
+ "grad_norm": 0.267713725566864,
1443
  "learning_rate": 9.456176618655689e-05,
1444
+ "loss": 0.013721915893256664,
1445
+ "memory/device_reserved (GiB)": 35.8,
1446
  "memory/max_active (GiB)": 33.83,
1447
  "memory/max_allocated (GiB)": 33.83,
1448
+ "ppl": 1.01382,
1449
  "step": 103,
1450
  "tokens/total": 3113744,
1451
+ "tokens/train_per_sec_per_gpu": 32.41,
1452
  "tokens/trainable": 47059
1453
  },
1454
  {
1455
  "epoch": 0.40625,
1456
+ "grad_norm": 0.325897753238678,
1457
  "learning_rate": 9.442260745454927e-05,
1458
+ "loss": 0.012948594987392426,
1459
+ "memory/device_reserved (GiB)": 35.8,
1460
  "memory/max_active (GiB)": 34.02,
1461
  "memory/max_allocated (GiB)": 34.02,
1462
+ "ppl": 1.01303,
1463
  "step": 104,
1464
  "tokens/total": 3144272,
1465
+ "tokens/train_per_sec_per_gpu": 33.93,
1466
  "tokens/trainable": 47536
1467
  },
1468
  {
1469
  "epoch": 0.41015625,
1470
+ "grad_norm": 0.531863808631897,
1471
  "learning_rate": 9.428180818692884e-05,
1472
+ "loss": 0.02244492992758751,
1473
+ "memory/device_reserved (GiB)": 37.62,
1474
  "memory/max_active (GiB)": 33.81,
1475
  "memory/max_allocated (GiB)": 33.81,
1476
+ "ppl": 1.0227,
1477
  "step": 105,
1478
  "tokens/total": 3174512,
1479
+ "tokens/train_per_sec_per_gpu": 33.95,
1480
  "tokens/trainable": 48037
1481
  },
1482
  {
1483
  "epoch": 0.4140625,
1484
+ "grad_norm": 0.3957497775554657,
1485
  "learning_rate": 9.413937424292791e-05,
1486
+ "loss": 0.015801995992660522,
1487
+ "memory/device_reserved (GiB)": 37.62,
1488
  "memory/max_active (GiB)": 33.91,
1489
  "memory/max_allocated (GiB)": 33.91,
1490
+ "ppl": 1.01593,
1491
  "step": 106,
1492
  "tokens/total": 3204896,
1493
+ "tokens/train_per_sec_per_gpu": 32.49,
1494
  "tokens/trainable": 48491
1495
  },
1496
  {
1497
  "epoch": 0.41796875,
1498
+ "grad_norm": 0.4241825044155121,
1499
  "learning_rate": 9.399531154980424e-05,
1500
+ "loss": 0.016320914030075073,
1501
+ "memory/device_reserved (GiB)": 37.62,
1502
  "memory/max_active (GiB)": 34.01,
1503
  "memory/max_allocated (GiB)": 34.01,
1504
+ "ppl": 1.01645,
1505
  "step": 107,
1506
  "tokens/total": 3235360,
1507
+ "tokens/train_per_sec_per_gpu": 32.89,
1508
  "tokens/trainable": 48940
1509
  },
1510
  {
1511
  "epoch": 0.421875,
1512
+ "grad_norm": 0.32064536213874817,
1513
  "learning_rate": 9.384962610259455e-05,
1514
+ "loss": 0.010785036720335484,
1515
+ "memory/device_reserved (GiB)": 37.62,
1516
  "memory/max_active (GiB)": 33.83,
1517
  "memory/max_allocated (GiB)": 33.83,
1518
+ "ppl": 1.01084,
1519
  "step": 108,
1520
  "tokens/total": 3265552,
1521
+ "tokens/train_per_sec_per_gpu": 31.52,
1522
  "tokens/trainable": 49389
1523
  },
1524
  {
1525
  "epoch": 0.42578125,
1526
+ "grad_norm": 0.17215749621391296,
1527
  "learning_rate": 9.370232396386494e-05,
1528
+ "loss": 0.00814715214073658,
1529
+ "memory/device_reserved (GiB)": 37.62,
1530
  "memory/max_active (GiB)": 33.39,
1531
  "memory/max_allocated (GiB)": 33.39,
1532
+ "ppl": 1.00818,
1533
  "step": 109,
1534
  "tokens/total": 3293856,
1535
+ "tokens/train_per_sec_per_gpu": 35.91,
1536
  "tokens/trainable": 49838
1537
  },
1538
  {
1539
  "epoch": 0.4296875,
1540
+ "grad_norm": 0.38179653882980347,
1541
  "learning_rate": 9.355341126345868e-05,
1542
+ "loss": 0.012128787115216255,
1543
+ "memory/device_reserved (GiB)": 37.62,
1544
  "memory/max_active (GiB)": 33.78,
1545
  "memory/max_allocated (GiB)": 33.78,
1546
+ "ppl": 1.0122,
1547
  "step": 110,
1548
  "tokens/total": 3323984,
1549
+ "tokens/train_per_sec_per_gpu": 33.93,
1550
  "tokens/trainable": 50310
1551
  },
1552
  {
1553
  "epoch": 0.43359375,
1554
+ "grad_norm": 0.49835413694381714,
1555
  "learning_rate": 9.340289419824107e-05,
1556
+ "loss": 0.015248995274305344,
1557
  "memory/device_reserved (GiB)": 37.62,
1558
  "memory/max_active (GiB)": 33.84,
1559
  "memory/max_allocated (GiB)": 33.84,
1560
+ "ppl": 1.01537,
1561
  "step": 111,
1562
  "tokens/total": 3354320,
1563
+ "tokens/train_per_sec_per_gpu": 33.55,
1564
  "tokens/trainable": 50755
1565
  },
1566
  {
1567
  "epoch": 0.4375,
1568
+ "grad_norm": 0.43904298543930054,
1569
  "learning_rate": 9.325077903184159e-05,
1570
+ "loss": 0.011272929608821869,
1571
  "memory/device_reserved (GiB)": 37.62,
1572
  "memory/max_active (GiB)": 33.9,
1573
  "memory/max_allocated (GiB)": 33.9,
1574
+ "ppl": 1.01134,
1575
  "step": 112,
1576
  "tokens/total": 3384880,
1577
+ "tokens/train_per_sec_per_gpu": 31.5,
1578
  "tokens/trainable": 51213
1579
  },
1580
  {
1581
  "epoch": 0.44140625,
1582
+ "grad_norm": 0.5670213103294373,
1583
  "learning_rate": 9.30970720943932e-05,
1584
+ "loss": 0.0028921598568558693,
1585
  "memory/device_reserved (GiB)": 37.62,
1586
  "memory/max_active (GiB)": 33.89,
1587
  "memory/max_allocated (GiB)": 33.89,
1588
+ "ppl": 1.0029,
1589
  "step": 113,
1590
  "tokens/total": 3415104,
1591
+ "tokens/train_per_sec_per_gpu": 33.77,
1592
  "tokens/trainable": 51660
1593
  },
1594
  {
1595
  "epoch": 0.4453125,
1596
+ "grad_norm": 0.159476637840271,
1597
  "learning_rate": 9.2941779782269e-05,
1598
+ "loss": 0.0025574099272489548,
1599
  "memory/device_reserved (GiB)": 37.62,
1600
  "memory/max_active (GiB)": 33.84,
1601
  "memory/max_allocated (GiB)": 33.84,
1602
+ "ppl": 1.00256,
1603
  "step": 114,
1604
  "tokens/total": 3445504,
1605
+ "tokens/train_per_sec_per_gpu": 32.34,
1606
  "tokens/trainable": 52133
1607
  },
1608
  {
1609
  "epoch": 0.44921875,
1610
+ "grad_norm": 0.795085608959198,
1611
  "learning_rate": 9.278490855781596e-05,
1612
+ "loss": 0.024456799030303955,
1613
  "memory/device_reserved (GiB)": 37.62,
1614
  "memory/max_active (GiB)": 33.84,
1615
  "memory/max_allocated (GiB)": 33.84,
1616
+ "ppl": 1.02476,
1617
  "step": 115,
1618
  "tokens/total": 3475744,
1619
+ "tokens/train_per_sec_per_gpu": 37.25,
1620
  "tokens/trainable": 52580
1621
  },
1622
  {
1623
  "epoch": 0.453125,
1624
+ "grad_norm": 0.38324710726737976,
1625
  "learning_rate": 9.262646494908604e-05,
1626
+ "loss": 0.004516632296144962,
1627
  "memory/device_reserved (GiB)": 37.62,
1628
  "memory/max_active (GiB)": 33.82,
1629
  "memory/max_allocated (GiB)": 33.82,
1630
+ "ppl": 1.00453,
1631
  "step": 116,
1632
  "tokens/total": 3506016,
1633
+ "tokens/train_per_sec_per_gpu": 34.74,
1634
  "tokens/trainable": 53033
1635
  },
1636
  {
1637
  "epoch": 0.45703125,
1638
+ "grad_norm": 0.3037130534648895,
1639
  "learning_rate": 9.246645554956457e-05,
1640
+ "loss": 0.021973589435219765,
1641
  "memory/device_reserved (GiB)": 37.62,
1642
  "memory/max_active (GiB)": 33.83,
1643
  "memory/max_allocated (GiB)": 33.83,
1644
+ "ppl": 1.02222,
1645
  "step": 117,
1646
  "tokens/total": 3536256,
1647
+ "tokens/train_per_sec_per_gpu": 35.22,
1648
  "tokens/trainable": 53517
1649
  },
1650
  {
1651
  "epoch": 0.4609375,
1652
+ "grad_norm": 1.3904820680618286,
1653
  "learning_rate": 9.230488701789578e-05,
1654
+ "loss": 0.009210974909365177,
1655
  "memory/device_reserved (GiB)": 37.62,
1656
  "memory/max_active (GiB)": 33.73,
1657
  "memory/max_allocated (GiB)": 33.73,
1658
+ "ppl": 1.00925,
1659
  "step": 118,
1660
  "tokens/total": 3566480,
1661
+ "tokens/train_per_sec_per_gpu": 34.59,
1662
  "tokens/trainable": 53967
1663
  },
1664
  {
1665
  "epoch": 0.46484375,
1666
+ "grad_norm": 0.1571500301361084,
1667
  "learning_rate": 9.214176607760577e-05,
1668
+ "loss": 0.0021451227366924286,
1669
  "memory/device_reserved (GiB)": 37.62,
1670
  "memory/max_active (GiB)": 33.85,
1671
  "memory/max_allocated (GiB)": 33.85,
1672
+ "ppl": 1.00215,
1673
  "step": 119,
1674
  "tokens/total": 3596624,
1675
+ "tokens/train_per_sec_per_gpu": 35.33,
1676
  "tokens/trainable": 54430
1677
  },
1678
  {
1679
  "epoch": 0.46875,
1680
+ "grad_norm": 0.39999091625213623,
1681
  "learning_rate": 9.197709951682268e-05,
1682
+ "loss": 0.00788091216236353,
1683
  "memory/device_reserved (GiB)": 37.62,
1684
  "memory/max_active (GiB)": 33.93,
1685
  "memory/max_allocated (GiB)": 33.93,
1686
+ "ppl": 1.00791,
1687
  "step": 120,
1688
  "tokens/total": 3627168,
1689
+ "tokens/train_per_sec_per_gpu": 35.32,
1690
  "tokens/trainable": 54896
1691
  },
1692
  {
1693
  "epoch": 0.47265625,
1694
+ "grad_norm": 0.38124287128448486,
1695
  "learning_rate": 9.181089418799428e-05,
1696
+ "loss": 0.010133227333426476,
1697
  "memory/device_reserved (GiB)": 37.62,
1698
  "memory/max_active (GiB)": 33.96,
1699
  "memory/max_allocated (GiB)": 33.96,
1700
+ "ppl": 1.01018,
1701
  "step": 121,
1702
  "tokens/total": 3657632,
1703
+ "tokens/train_per_sec_per_gpu": 37.76,
1704
  "tokens/trainable": 55379
1705
  },
1706
  {
1707
  "epoch": 0.4765625,
1708
+ "grad_norm": 0.0512852780520916,
1709
  "learning_rate": 9.164315700760271e-05,
1710
+ "loss": 0.0007940311916172504,
1711
  "memory/device_reserved (GiB)": 37.62,
1712
  "memory/max_active (GiB)": 33.75,
1713
  "memory/max_allocated (GiB)": 33.75,
1714
+ "ppl": 1.00079,
1715
  "step": 122,
1716
  "tokens/total": 3687824,
1717
+ "tokens/train_per_sec_per_gpu": 31.71,
1718
  "tokens/trainable": 55803
1719
  },
1720
  {
1721
  "epoch": 0.48046875,
1722
+ "grad_norm": 0.13324694335460663,
1723
  "learning_rate": 9.147389495587671e-05,
1724
+ "loss": 0.0023267122451215982,
1725
  "memory/device_reserved (GiB)": 37.62,
1726
  "memory/max_active (GiB)": 33.87,
1727
  "memory/max_allocated (GiB)": 33.87,
1728
+ "ppl": 1.00233,
1729
  "step": 123,
1730
  "tokens/total": 3718320,
1731
+ "tokens/train_per_sec_per_gpu": 36.46,
1732
  "tokens/trainable": 56249
1733
  },
1734
  {
1735
  "epoch": 0.484375,
1736
+ "grad_norm": 0.230186328291893,
1737
  "learning_rate": 9.130311507650116e-05,
1738
+ "loss": 0.0037590472493320704,
1739
  "memory/device_reserved (GiB)": 37.62,
1740
  "memory/max_active (GiB)": 33.9,
1741
  "memory/max_allocated (GiB)": 33.9,
1742
+ "ppl": 1.00377,
1743
  "step": 124,
1744
  "tokens/total": 3748672,
1745
+ "tokens/train_per_sec_per_gpu": 32.43,
1746
  "tokens/trainable": 56713
1747
  },
1748
  {
1749
  "epoch": 0.48828125,
1750
+ "grad_norm": 0.30578872561454773,
1751
  "learning_rate": 9.113082447632394e-05,
1752
+ "loss": 0.00473653944209218,
1753
  "memory/device_reserved (GiB)": 37.76,
1754
  "memory/max_active (GiB)": 33.85,
1755
  "memory/max_allocated (GiB)": 33.85,
1756
+ "ppl": 1.00475,
1757
  "step": 125,
1758
  "tokens/total": 3778976,
1759
+ "tokens/train_per_sec_per_gpu": 39.1,
1760
  "tokens/trainable": 57196
1761
  },
1762
  {
1763
  "epoch": 0.4921875,
1764
+ "grad_norm": 0.3714931607246399,
1765
  "learning_rate": 9.09570303250602e-05,
1766
+ "loss": 0.005811735987663269,
1767
  "memory/device_reserved (GiB)": 37.76,
1768
  "memory/max_active (GiB)": 33.89,
1769
  "memory/max_allocated (GiB)": 33.89,
1770
+ "ppl": 1.00583,
1771
  "step": 126,
1772
  "tokens/total": 3809296,
1773
+ "tokens/train_per_sec_per_gpu": 36.55,
1774
  "tokens/trainable": 57680
1775
  },
1776
  {
1777
  "epoch": 0.49609375,
1778
+ "grad_norm": 0.11054892838001251,
1779
  "learning_rate": 9.078173985499394e-05,
1780
+ "loss": 0.0032034481409937143,
1781
  "memory/device_reserved (GiB)": 37.76,
1782
  "memory/max_active (GiB)": 33.77,
1783
  "memory/max_allocated (GiB)": 33.77,
1784
+ "ppl": 1.00321,
1785
  "step": 127,
1786
  "tokens/total": 3839328,
1787
+ "tokens/train_per_sec_per_gpu": 31.88,
1788
  "tokens/trainable": 58110
1789
  },
1790
  {
1791
  "epoch": 0.5,
1792
+ "grad_norm": 0.09251131862401962,
1793
  "learning_rate": 9.060496036067713e-05,
1794
+ "loss": 0.001724504167214036,
1795
  "memory/device_reserved (GiB)": 37.76,
1796
  "memory/max_active (GiB)": 33.88,
1797
  "memory/max_allocated (GiB)": 33.88,
1798
+ "ppl": 1.00173,
1799
  "step": 128,
1800
  "tokens/total": 3869824,
1801
+ "tokens/train_per_sec_per_gpu": 31.74,
1802
  "tokens/trainable": 58534
1803
  },
1804
  {
1805
  "epoch": 0.50390625,
1806
+ "grad_norm": 0.22758308053016663,
1807
  "learning_rate": 9.042669919862615e-05,
1808
+ "loss": 0.004688034299761057,
1809
  "memory/device_reserved (GiB)": 37.76,
1810
  "memory/max_active (GiB)": 33.9,
1811
  "memory/max_allocated (GiB)": 33.9,
1812
+ "ppl": 1.0047,
1813
  "step": 129,
1814
  "tokens/total": 3900080,
1815
+ "tokens/train_per_sec_per_gpu": 35.19,
1816
  "tokens/trainable": 58998
1817
  },
1818
  {
1819
  "epoch": 0.5078125,
1820
+ "grad_norm": 0.2881723642349243,
1821
  "learning_rate": 9.024696378701557e-05,
1822
+ "loss": 0.008266786113381386,
1823
  "memory/device_reserved (GiB)": 35.93,
1824
  "memory/max_active (GiB)": 33.91,
1825
  "memory/max_allocated (GiB)": 33.91,
1826
+ "ppl": 1.0083,
1827
  "step": 130,
1828
  "tokens/total": 3930560,
1829
+ "tokens/train_per_sec_per_gpu": 31.83,
1830
  "tokens/trainable": 59448
1831
  },
1832
  {
1833
  "epoch": 0.51171875,
1834
+ "grad_norm": 0.18802326917648315,
1835
  "learning_rate": 9.006576160536948e-05,
1836
+ "loss": 0.00283675454556942,
1837
  "memory/device_reserved (GiB)": 35.93,
1838
  "memory/max_active (GiB)": 33.73,
1839
  "memory/max_allocated (GiB)": 33.73,
1840
+ "ppl": 1.00284,
1841
  "step": 131,
1842
  "tokens/total": 3960496,
1843
+ "tokens/train_per_sec_per_gpu": 31.92,
1844
  "tokens/trainable": 59906
1845
  },
1846
  {
1847
  "epoch": 0.515625,
1848
+ "grad_norm": 0.6749681234359741,
1849
  "learning_rate": 8.988310019425035e-05,
1850
+ "loss": 0.012044447474181652,
1851
+ "memory/device_reserved (GiB)": 35.93,
1852
  "memory/max_active (GiB)": 33.85,
1853
  "memory/max_allocated (GiB)": 33.85,
1854
+ "ppl": 1.01212,
1855
  "step": 132,
1856
  "tokens/total": 3990928,
1857
+ "tokens/train_per_sec_per_gpu": 34.26,
1858
  "tokens/trainable": 60350
1859
  },
1860
  {
1861
  "epoch": 0.51953125,
1862
+ "grad_norm": 0.5789079070091248,
1863
  "learning_rate": 8.969898715494506e-05,
1864
+ "loss": 0.011312158778309822,
1865
+ "memory/device_reserved (GiB)": 37.16,
1866
  "memory/max_active (GiB)": 33.84,
1867
  "memory/max_allocated (GiB)": 33.84,
1868
+ "ppl": 1.01138,
1869
  "step": 133,
1870
  "tokens/total": 4021264,
1871
+ "tokens/train_per_sec_per_gpu": 36.17,
1872
  "tokens/trainable": 60831
1873
  },
1874
  {
1875
  "epoch": 0.5234375,
1876
+ "grad_norm": 0.47060829401016235,
1877
  "learning_rate": 8.951343014914869e-05,
1878
+ "loss": 0.0308814849704504,
1879
  "memory/device_reserved (GiB)": 37.17,
1880
  "memory/max_active (GiB)": 33.88,
1881
  "memory/max_allocated (GiB)": 33.88,
1882
+ "ppl": 1.03136,
1883
  "step": 134,
1884
  "tokens/total": 4051728,
1885
+ "tokens/train_per_sec_per_gpu": 33.96,
1886
  "tokens/trainable": 61305
1887
  },
1888
  {
1889
  "epoch": 0.52734375,
1890
+ "grad_norm": 0.16633495688438416,
1891
  "learning_rate": 8.932643689864568e-05,
1892
+ "loss": 0.005535767879337072,
1893
  "memory/device_reserved (GiB)": 37.17,
1894
  "memory/max_active (GiB)": 33.85,
1895
  "memory/max_allocated (GiB)": 33.85,
1896
+ "ppl": 1.00555,
1897
  "step": 135,
1898
  "tokens/total": 4081920,
1899
+ "tokens/train_per_sec_per_gpu": 37.62,
1900
  "tokens/trainable": 61792
1901
  },
1902
  {
1903
  "epoch": 0.53125,
1904
+ "grad_norm": 0.10699360817670822,
1905
  "learning_rate": 8.913801518498845e-05,
1906
+ "loss": 0.002973862923681736,
1907
  "memory/device_reserved (GiB)": 38.55,
1908
  "memory/max_active (GiB)": 33.9,
1909
  "memory/max_allocated (GiB)": 33.9,
1910
+ "ppl": 1.00298,
1911
  "step": 136,
1912
  "tokens/total": 4112304,
1913
+ "tokens/train_per_sec_per_gpu": 31.37,
1914
  "tokens/trainable": 62253
1915
  },
1916
  {
1917
  "epoch": 0.53515625,
1918
+ "grad_norm": 0.2920030951499939,
1919
  "learning_rate": 8.894817284917364e-05,
1920
+ "loss": 0.01169741339981556,
1921
  "memory/device_reserved (GiB)": 38.55,
1922
  "memory/max_active (GiB)": 33.81,
1923
  "memory/max_allocated (GiB)": 33.81,
1924
+ "ppl": 1.01177,
1925
  "step": 137,
1926
  "tokens/total": 4142512,
1927
+ "tokens/train_per_sec_per_gpu": 31.65,
1928
  "tokens/trainable": 62707
1929
  },
1930
  {
1931
  "epoch": 0.5390625,
1932
+ "grad_norm": 0.3187088370323181,
1933
  "learning_rate": 8.875691779131569e-05,
1934
+ "loss": 0.011357840150594711,
1935
+ "memory/device_reserved (GiB)": 38.56,
1936
  "memory/max_active (GiB)": 33.89,
1937
  "memory/max_allocated (GiB)": 33.89,
1938
+ "ppl": 1.01142,
1939
  "step": 138,
1940
  "tokens/total": 4172992,
1941
  "tokens/train_per_sec_per_gpu": 29.32,
 
1943
  },
1944
  {
1945
  "epoch": 0.54296875,
1946
+ "grad_norm": 0.18588471412658691,
1947
  "learning_rate": 8.856425797031829e-05,
1948
+ "loss": 0.006680965423583984,
1949
+ "memory/device_reserved (GiB)": 38.56,
1950
  "memory/max_active (GiB)": 33.78,
1951
  "memory/max_allocated (GiB)": 33.78,
1952
+ "ppl": 1.0067,
1953
  "step": 139,
1954
  "tokens/total": 4203136,
1955
+ "tokens/train_per_sec_per_gpu": 33.97,
1956
  "tokens/trainable": 63587
1957
  },
1958
  {
1959
  "epoch": 0.546875,
1960
+ "grad_norm": 0.2760343551635742,
1961
  "learning_rate": 8.837020140354295e-05,
1962
+ "loss": 0.01477967668324709,
1963
+ "memory/device_reserved (GiB)": 38.56,
1964
  "memory/max_active (GiB)": 33.84,
1965
  "memory/max_allocated (GiB)": 33.84,
1966
+ "ppl": 1.01489,
1967
  "step": 140,
1968
  "tokens/total": 4233456,
1969
+ "tokens/train_per_sec_per_gpu": 36.74,
1970
  "tokens/trainable": 64052
1971
  },
1972
  {
1973
  "epoch": 0.55078125,
1974
+ "grad_norm": 0.49880966544151306,
1975
  "learning_rate": 8.817475616647554e-05,
1976
+ "loss": 0.016770213842391968,
1977
+ "memory/device_reserved (GiB)": 38.56,
1978
  "memory/max_active (GiB)": 33.8,
1979
  "memory/max_allocated (GiB)": 33.8,
1980
+ "ppl": 1.01691,
1981
  "step": 141,
1982
  "tokens/total": 4263728,
1983
+ "tokens/train_per_sec_per_gpu": 34.38,
1984
  "tokens/trainable": 64526
1985
  },
1986
  {
1987
  "epoch": 0.5546875,
1988
+ "grad_norm": 0.181757390499115,
1989
  "learning_rate": 8.797793039239017e-05,
1990
+ "loss": 0.006471520289778709,
1991
+ "memory/device_reserved (GiB)": 38.56,
1992
  "memory/max_active (GiB)": 34.01,
1993
  "memory/max_allocated (GiB)": 34.01,
1994
+ "ppl": 1.00649,
1995
  "step": 142,
1996
  "tokens/total": 4294432,
1997
+ "tokens/train_per_sec_per_gpu": 34.58,
1998
  "tokens/trainable": 64983
1999
  },
2000
  {
2001
  "epoch": 0.55859375,
2002
+ "grad_norm": 0.209610253572464,
2003
  "learning_rate": 8.777973227201069e-05,
2004
+ "loss": 0.006492790300399065,
2005
+ "memory/device_reserved (GiB)": 38.56,
2006
  "memory/max_active (GiB)": 33.94,
2007
  "memory/max_allocated (GiB)": 33.94,
2008
+ "ppl": 1.00651,
2009
  "step": 143,
2010
  "tokens/total": 4324960,
2011
+ "tokens/train_per_sec_per_gpu": 37.58,
2012
  "tokens/trainable": 65492
2013
  },
2014
  {
2015
  "epoch": 0.5625,
2016
+ "grad_norm": 0.13360266387462616,
2017
  "learning_rate": 8.758017005316988e-05,
2018
+ "loss": 0.003702069167047739,
2019
+ "memory/device_reserved (GiB)": 38.56,
2020
  "memory/max_active (GiB)": 33.92,
2021
  "memory/max_allocated (GiB)": 33.92,
2022
+ "ppl": 1.00371,
2023
  "step": 144,
2024
  "tokens/total": 4355424,
2025
+ "tokens/train_per_sec_per_gpu": 35.72,
2026
  "tokens/trainable": 65925
2027
  },
2028
  {
2029
  "epoch": 0.56640625,
2030
+ "grad_norm": 0.1640344262123108,
2031
  "learning_rate": 8.737925204046629e-05,
2032
+ "loss": 0.006490131840109825,
2033
+ "memory/device_reserved (GiB)": 38.56,
2034
  "memory/max_active (GiB)": 33.75,
2035
  "memory/max_allocated (GiB)": 33.75,
2036
+ "ppl": 1.00651,
2037
  "step": 145,
2038
  "tokens/total": 4385712,
2039
+ "tokens/train_per_sec_per_gpu": 31.39,
2040
  "tokens/trainable": 66383
2041
  },
2042
  {
2043
  "epoch": 0.5703125,
2044
+ "grad_norm": 0.13646955788135529,
2045
  "learning_rate": 8.717698659491851e-05,
2046
+ "loss": 0.0026589329354465008,
2047
  "memory/device_reserved (GiB)": 38.56,
2048
  "memory/max_active (GiB)": 33.94,
2049
  "memory/max_allocated (GiB)": 33.94,
2050
+ "ppl": 1.00266,
2051
  "step": 146,
2052
  "tokens/total": 4416016,
2053
+ "tokens/train_per_sec_per_gpu": 31.39,
2054
  "tokens/trainable": 66840
2055
  },
2056
  {
2057
  "epoch": 0.57421875,
2058
+ "grad_norm": 0.4220513701438904,
2059
  "learning_rate": 8.697338213361735e-05,
2060
+ "loss": 0.01334389764815569,
2061
  "memory/device_reserved (GiB)": 38.65,
2062
  "memory/max_active (GiB)": 33.88,
2063
  "memory/max_allocated (GiB)": 33.88,
2064
+ "ppl": 1.01343,
2065
  "step": 147,
2066
  "tokens/total": 4446368,
2067
+ "tokens/train_per_sec_per_gpu": 31.76,
2068
  "tokens/trainable": 67297
2069
  },
2070
  {
2071
  "epoch": 0.578125,
2072
+ "grad_norm": 0.37345919013023376,
2073
  "learning_rate": 8.676844712937552e-05,
2074
+ "loss": 0.012794861570000648,
2075
  "memory/device_reserved (GiB)": 38.65,
2076
  "memory/max_active (GiB)": 33.89,
2077
  "memory/max_allocated (GiB)": 33.89,
2078
+ "ppl": 1.01288,
2079
  "step": 148,
2080
  "tokens/total": 4476880,
2081
  "tokens/train_per_sec_per_gpu": 37.36,
 
2083
  },
2084
  {
2085
  "epoch": 0.58203125,
2086
+ "grad_norm": 0.3093344271183014,
2087
  "learning_rate": 8.656219011037509e-05,
2088
+ "loss": 0.012492594309151173,
2089
  "memory/device_reserved (GiB)": 38.65,
2090
  "memory/max_active (GiB)": 33.89,
2091
  "memory/max_allocated (GiB)": 33.89,
2092
+ "ppl": 1.01257,
2093
  "step": 149,
2094
  "tokens/total": 4507232,
2095
+ "tokens/train_per_sec_per_gpu": 32.83,
2096
  "tokens/trainable": 68257
2097
  },
2098
  {
2099
  "epoch": 0.5859375,
2100
+ "grad_norm": 0.2453778088092804,
2101
  "learning_rate": 8.63546196598125e-05,
2102
+ "loss": 0.003456964623183012,
2103
  "memory/device_reserved (GiB)": 38.65,
2104
  "memory/max_active (GiB)": 33.77,
2105
  "memory/max_allocated (GiB)": 33.77,
2106
+ "ppl": 1.00346,
2107
  "step": 150,
2108
  "tokens/total": 4537376,
2109
+ "tokens/train_per_sec_per_gpu": 35.52,
2110
  "tokens/trainable": 68706
2111
  },
2112
  {
2113
  "epoch": 0.58984375,
2114
+ "grad_norm": 0.337802916765213,
2115
  "learning_rate": 8.614574441554145e-05,
2116
+ "loss": 0.009631449356675148,
2117
  "memory/device_reserved (GiB)": 38.65,
2118
  "memory/max_active (GiB)": 33.8,
2119
  "memory/max_allocated (GiB)": 33.8,
2120
+ "ppl": 1.00968,
2121
  "step": 151,
2122
  "tokens/total": 4567584,
2123
+ "tokens/train_per_sec_per_gpu": 33.3,
2124
  "tokens/trainable": 69131
2125
  },
2126
  {
2127
  "epoch": 0.59375,
2128
+ "grad_norm": 0.34801673889160156,
2129
  "learning_rate": 8.593557306971349e-05,
2130
+ "loss": 0.02037646435201168,
2131
  "memory/device_reserved (GiB)": 38.65,
2132
  "memory/max_active (GiB)": 33.87,
2133
  "memory/max_allocated (GiB)": 33.87,
2134
+ "ppl": 1.02059,
2135
  "step": 152,
2136
  "tokens/total": 4597792,
2137
+ "tokens/train_per_sec_per_gpu": 31.93,
2138
  "tokens/trainable": 69586
2139
  },
2140
  {
2141
  "epoch": 0.59765625,
2142
+ "grad_norm": 0.03586283326148987,
2143
  "learning_rate": 8.572411436841618e-05,
2144
+ "loss": 0.0008243054617196321,
2145
  "memory/device_reserved (GiB)": 38.65,
2146
  "memory/max_active (GiB)": 33.78,
2147
  "memory/max_allocated (GiB)": 33.78,
2148
+ "ppl": 1.00082,
2149
  "step": 153,
2150
  "tokens/total": 4627936,
2151
+ "tokens/train_per_sec_per_gpu": 32.79,
2152
  "tokens/trainable": 70061
2153
  },
2154
  {
2155
  "epoch": 0.6015625,
2156
+ "grad_norm": 0.1870104819536209,
2157
  "learning_rate": 8.551137711130922e-05,
2158
+ "loss": 0.0038898580241948366,
2159
  "memory/device_reserved (GiB)": 38.65,
2160
  "memory/max_active (GiB)": 33.82,
2161
  "memory/max_allocated (GiB)": 33.82,
2162
+ "ppl": 1.0039,
2163
  "step": 154,
2164
  "tokens/total": 4658352,
2165
+ "tokens/train_per_sec_per_gpu": 27.95,
2166
  "tokens/trainable": 70467
2167
  },
2168
  {
2169
  "epoch": 0.60546875,
2170
+ "grad_norm": 0.2884272038936615,
2171
  "learning_rate": 8.529737015125824e-05,
2172
+ "loss": 0.005026768893003464,
2173
  "memory/device_reserved (GiB)": 38.65,
2174
  "memory/max_active (GiB)": 33.87,
2175
  "memory/max_allocated (GiB)": 33.87,
2176
+ "ppl": 1.00504,
2177
  "step": 155,
2178
  "tokens/total": 4688896,
2179
+ "tokens/train_per_sec_per_gpu": 33.08,
2180
  "tokens/trainable": 70933
2181
  },
2182
  {
2183
  "epoch": 0.609375,
2184
+ "grad_norm": 0.07867873460054398,
2185
  "learning_rate": 8.508210239396639e-05,
2186
+ "loss": 0.0024596985895186663,
2187
  "memory/device_reserved (GiB)": 38.65,
2188
  "memory/max_active (GiB)": 33.94,
2189
  "memory/max_allocated (GiB)": 33.94,
2190
+ "ppl": 1.00246,
2191
  "step": 156,
2192
  "tokens/total": 4719168,
2193
+ "tokens/train_per_sec_per_gpu": 34.29,
2194
  "tokens/trainable": 71382
2195
  },
2196
  {
2197
  "epoch": 0.61328125,
2198
+ "grad_norm": 0.056005269289016724,
2199
  "learning_rate": 8.486558279760375e-05,
2200
+ "loss": 0.0012056033592671156,
2201
  "memory/device_reserved (GiB)": 38.65,
2202
  "memory/max_active (GiB)": 33.87,
2203
  "memory/max_allocated (GiB)": 33.87,
2204
+ "ppl": 1.00121,
2205
  "step": 157,
2206
  "tokens/total": 4749136,
2207
+ "tokens/train_per_sec_per_gpu": 30.83,
2208
  "tokens/trainable": 71808
2209
  },
2210
  {
2211
  "epoch": 0.6171875,
2212
+ "grad_norm": 0.34242892265319824,
2213
  "learning_rate": 8.464782037243449e-05,
2214
+ "loss": 0.007983298972249031,
2215
  "memory/device_reserved (GiB)": 38.65,
2216
  "memory/max_active (GiB)": 33.74,
2217
  "memory/max_allocated (GiB)": 33.74,
2218
+ "ppl": 1.00802,
2219
  "step": 158,
2220
  "tokens/total": 4779472,
2221
+ "tokens/train_per_sec_per_gpu": 32.88,
2222
  "tokens/trainable": 72249
2223
  },
2224
  {
2225
  "epoch": 0.62109375,
2226
+ "grad_norm": 0.36051586270332336,
2227
  "learning_rate": 8.442882418044202e-05,
2228
+ "loss": 0.011793753132224083,
2229
  "memory/device_reserved (GiB)": 38.65,
2230
  "memory/max_active (GiB)": 33.77,
2231
  "memory/max_allocated (GiB)": 33.77,
2232
+ "ppl": 1.01186,
2233
  "step": 159,
2234
  "tokens/total": 4809632,
2235
+ "tokens/train_per_sec_per_gpu": 35.06,
2236
  "tokens/trainable": 72726
2237
  },
2238
  {
2239
  "epoch": 0.625,
2240
+ "grad_norm": 0.376717746257782,
2241
  "learning_rate": 8.420860333495179e-05,
2242
+ "loss": 0.018659302964806557,
2243
  "memory/device_reserved (GiB)": 38.65,
2244
  "memory/max_active (GiB)": 33.97,
2245
  "memory/max_allocated (GiB)": 33.97,
2246
+ "ppl": 1.01883,
2247
  "step": 160,
2248
  "tokens/total": 4840112,
2249
+ "tokens/train_per_sec_per_gpu": 32.68,
2250
  "tokens/trainable": 73148
2251
  }
2252
  ],
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1dd775d8b35d84d9d817a4a74f08f039adb13a2320077366a52bad07c43ebf8b
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dd65c9138a23ee4b709df1f879e1ce2a71a67a3461f4f7ed0fae00ba0eb853d5
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1d796a53e0037fe564b92a554d12c2094e42094df256517710edaa58b7d48d31
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e74bd30f32259d424a0528a8fc03e88152c9448882265722a55166fb622af3a
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-192/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b8eecfc00fed67bfc79ccbe2c4580025f3a5eb09de1478ed4d237cdce0a676c0
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:16e7d16f507266e4de0618e13529f0f6852f7dfacc533b3d9beb61399405123d
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ab103e3591277e8cd1e5fa934a5bc9882dafbe277490d1a7a13579f389795dd1
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f456a3b4b0b5cfe5bb7061718fafa69fb3393a4a4055872f85b783cf6135b255
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-224/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1c74a0c1890d16768995e589354d66ad0aaf896432a0f8708036dcf170656173
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de74ce97413c0093f4f76198ae64237c0da203ba9a293aa2b281e66d918ca952
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:143ebc082a4dcdfe72a9b44639352173fc591f7676a6655b84d556a3b293a9d7
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a2e59a33e5d93b9218821342e940feec395f412661f80e94be666601cd30c6c3
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-256/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f742ce5e3dc5e92cf1d1b5194d55616562458401c38397bb3b1f150a3613ed30
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da2bdb4f81462625bb226aa42eb809bea433a88b8d19660ad0cb52a8f25dfb1f
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9c6a83d5d9751f22dbe1bd2713e7e188432fb354065c57c9d49029f1c114f05b
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c144b815c2b7bcbf81221112a3f40c21c4e01914d789b90a94c663d34e6977b2
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-288/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ee7003d2997d35b1193b3e3463a4d9c2b6d209eee799d44f21ec131ffe7ff39a
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:32aa7a1bdc40de6c271cdaf4ba2e825940729a8400f15293192a0ab0fd0be3c4
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1f5489845cc180566855dfb58a67b01f34eb2d8fa011a26887196570356e40f0
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2c09fd208676c30f1a17a813e8f50e6c1234fdc5a8449106274744a5bc40156
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-32/trainer_state.json CHANGED
@@ -11,7 +11,7 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.00390625,
14
- "grad_norm": 0.870697021484375,
15
  "learning_rate": 0.0,
16
  "loss": 0.10251401364803314,
17
  "memory/device_reserved (GiB)": 34.94,
@@ -20,12 +20,12 @@
20
  "ppl": 1.10795,
21
  "step": 1,
22
  "tokens/total": 30464,
23
- "tokens/train_per_sec_per_gpu": 23.98,
24
  "tokens/trainable": 470
25
  },
26
  {
27
  "epoch": 0.0078125,
28
- "grad_norm": 0.8108459115028381,
29
  "learning_rate": 4.000000000000001e-06,
30
  "loss": 0.09678442776203156,
31
  "memory/device_reserved (GiB)": 35.83,
@@ -34,427 +34,427 @@
34
  "ppl": 1.10162,
35
  "step": 2,
36
  "tokens/total": 60800,
37
- "tokens/train_per_sec_per_gpu": 35.02,
38
  "tokens/trainable": 922
39
  },
40
  {
41
  "epoch": 0.01171875,
42
- "grad_norm": 1.8027335405349731,
43
  "learning_rate": 8.000000000000001e-06,
44
- "loss": 0.10952483862638474,
45
  "memory/device_reserved (GiB)": 35.85,
46
  "memory/max_active (GiB)": 33.91,
47
  "memory/max_allocated (GiB)": 33.91,
48
- "ppl": 1.11575,
49
  "step": 3,
50
  "tokens/total": 91136,
51
- "tokens/train_per_sec_per_gpu": 34.48,
52
  "tokens/trainable": 1396
53
  },
54
  {
55
  "epoch": 0.015625,
56
- "grad_norm": 1.0353018045425415,
57
  "learning_rate": 1.2e-05,
58
- "loss": 0.10303406417369843,
59
  "memory/device_reserved (GiB)": 35.85,
60
  "memory/max_active (GiB)": 33.91,
61
  "memory/max_allocated (GiB)": 33.91,
62
- "ppl": 1.10853,
63
  "step": 4,
64
  "tokens/total": 121552,
65
- "tokens/train_per_sec_per_gpu": 38.01,
66
  "tokens/trainable": 1850
67
  },
68
  {
69
  "epoch": 0.01953125,
70
- "grad_norm": 1.0778985023498535,
71
  "learning_rate": 1.6000000000000003e-05,
72
- "loss": 0.10945924371480942,
73
  "memory/device_reserved (GiB)": 35.85,
74
  "memory/max_active (GiB)": 33.98,
75
  "memory/max_allocated (GiB)": 33.98,
76
- "ppl": 1.11567,
77
  "step": 5,
78
  "tokens/total": 152304,
79
- "tokens/train_per_sec_per_gpu": 33.76,
80
  "tokens/trainable": 2302
81
  },
82
  {
83
  "epoch": 0.0234375,
84
- "grad_norm": 0.8929882645606995,
85
  "learning_rate": 2e-05,
86
- "loss": 0.08588902652263641,
87
  "memory/device_reserved (GiB)": 35.85,
88
  "memory/max_active (GiB)": 33.78,
89
  "memory/max_allocated (GiB)": 33.78,
90
- "ppl": 1.08969,
91
  "step": 6,
92
  "tokens/total": 182448,
93
- "tokens/train_per_sec_per_gpu": 31.47,
94
  "tokens/trainable": 2742
95
  },
96
  {
97
  "epoch": 0.02734375,
98
- "grad_norm": 1.6409597396850586,
99
  "learning_rate": 2.4e-05,
100
- "loss": 0.07352827489376068,
101
  "memory/device_reserved (GiB)": 35.85,
102
  "memory/max_active (GiB)": 33.85,
103
  "memory/max_allocated (GiB)": 33.85,
104
- "ppl": 1.0763,
105
  "step": 7,
106
  "tokens/total": 212736,
107
- "tokens/train_per_sec_per_gpu": 36.58,
108
  "tokens/trainable": 3208
109
  },
110
  {
111
  "epoch": 0.03125,
112
- "grad_norm": 1.5843520164489746,
113
  "learning_rate": 2.8000000000000003e-05,
114
- "loss": 0.07798244059085846,
115
  "memory/device_reserved (GiB)": 36.07,
116
  "memory/max_active (GiB)": 33.87,
117
  "memory/max_allocated (GiB)": 33.87,
118
- "ppl": 1.0811,
119
  "step": 8,
120
  "tokens/total": 243024,
121
- "tokens/train_per_sec_per_gpu": 31.83,
122
  "tokens/trainable": 3655
123
  },
124
  {
125
  "epoch": 0.03515625,
126
- "grad_norm": 1.3725916147232056,
127
  "learning_rate": 3.2000000000000005e-05,
128
- "loss": 0.04634054750204086,
129
  "memory/device_reserved (GiB)": 36.07,
130
  "memory/max_active (GiB)": 33.82,
131
  "memory/max_allocated (GiB)": 33.82,
132
- "ppl": 1.04743,
133
  "step": 9,
134
  "tokens/total": 271264,
135
- "tokens/train_per_sec_per_gpu": 35.62,
136
  "tokens/trainable": 4091
137
  },
138
  {
139
  "epoch": 0.0390625,
140
- "grad_norm": 2.544938564300537,
141
  "learning_rate": 3.6e-05,
142
- "loss": 0.08677884936332703,
143
  "memory/device_reserved (GiB)": 36.07,
144
  "memory/max_active (GiB)": 33.86,
145
  "memory/max_allocated (GiB)": 33.86,
146
- "ppl": 1.09066,
147
  "step": 10,
148
  "tokens/total": 301520,
149
- "tokens/train_per_sec_per_gpu": 34.92,
150
  "tokens/trainable": 4553
151
  },
152
  {
153
  "epoch": 0.04296875,
154
- "grad_norm": 1.6364734172821045,
155
  "learning_rate": 4e-05,
156
- "loss": 0.07754456996917725,
157
- "memory/device_reserved (GiB)": 36.07,
158
  "memory/max_active (GiB)": 33.8,
159
  "memory/max_allocated (GiB)": 33.8,
160
- "ppl": 1.08063,
161
  "step": 11,
162
  "tokens/total": 331808,
163
- "tokens/train_per_sec_per_gpu": 33.0,
164
  "tokens/trainable": 4992
165
  },
166
  {
167
  "epoch": 0.046875,
168
- "grad_norm": 2.167391538619995,
169
  "learning_rate": 4.4000000000000006e-05,
170
- "loss": 0.11765319854021072,
171
- "memory/device_reserved (GiB)": 36.07,
172
  "memory/max_active (GiB)": 33.87,
173
  "memory/max_allocated (GiB)": 33.87,
174
- "ppl": 1.12485,
175
  "step": 12,
176
  "tokens/total": 362224,
177
- "tokens/train_per_sec_per_gpu": 38.18,
178
  "tokens/trainable": 5494
179
  },
180
  {
181
  "epoch": 0.05078125,
182
- "grad_norm": 3.196911573410034,
183
  "learning_rate": 4.8e-05,
184
- "loss": 0.03510797768831253,
185
- "memory/device_reserved (GiB)": 36.15,
186
  "memory/max_active (GiB)": 33.91,
187
  "memory/max_allocated (GiB)": 33.91,
188
- "ppl": 1.03573,
189
  "step": 13,
190
  "tokens/total": 392432,
191
- "tokens/train_per_sec_per_gpu": 33.67,
192
  "tokens/trainable": 5933
193
  },
194
  {
195
  "epoch": 0.0546875,
196
- "grad_norm": 1.9654567241668701,
197
  "learning_rate": 5.2000000000000004e-05,
198
- "loss": 0.061097435653209686,
199
- "memory/device_reserved (GiB)": 36.15,
200
  "memory/max_active (GiB)": 33.79,
201
  "memory/max_allocated (GiB)": 33.79,
202
- "ppl": 1.063,
203
  "step": 14,
204
  "tokens/total": 422784,
205
- "tokens/train_per_sec_per_gpu": 29.53,
206
  "tokens/trainable": 6369
207
  },
208
  {
209
  "epoch": 0.05859375,
210
- "grad_norm": 1.934105396270752,
211
  "learning_rate": 5.6000000000000006e-05,
212
- "loss": 0.05385493487119675,
213
- "memory/device_reserved (GiB)": 36.15,
214
  "memory/max_active (GiB)": 33.92,
215
  "memory/max_allocated (GiB)": 33.92,
216
- "ppl": 1.05533,
217
  "step": 15,
218
  "tokens/total": 453376,
219
- "tokens/train_per_sec_per_gpu": 32.96,
220
  "tokens/trainable": 6820
221
  },
222
  {
223
  "epoch": 0.0625,
224
- "grad_norm": 1.4659016132354736,
225
  "learning_rate": 6e-05,
226
- "loss": 0.052153222262859344,
227
  "memory/device_reserved (GiB)": 36.21,
228
  "memory/max_active (GiB)": 33.83,
229
  "memory/max_allocated (GiB)": 33.83,
230
- "ppl": 1.05354,
231
  "step": 16,
232
  "tokens/total": 483504,
233
- "tokens/train_per_sec_per_gpu": 30.68,
234
  "tokens/trainable": 7258
235
  },
236
  {
237
  "epoch": 0.06640625,
238
- "grad_norm": 1.9650894403457642,
239
  "learning_rate": 6.400000000000001e-05,
240
- "loss": 0.05092189460992813,
241
  "memory/device_reserved (GiB)": 36.21,
242
  "memory/max_active (GiB)": 33.9,
243
  "memory/max_allocated (GiB)": 33.9,
244
- "ppl": 1.05224,
245
  "step": 17,
246
  "tokens/total": 514032,
247
- "tokens/train_per_sec_per_gpu": 35.09,
248
  "tokens/trainable": 7710
249
  },
250
  {
251
  "epoch": 0.0703125,
252
- "grad_norm": 0.6891724467277527,
253
  "learning_rate": 6.800000000000001e-05,
254
- "loss": 0.031155016273260117,
255
  "memory/device_reserved (GiB)": 36.21,
256
  "memory/max_active (GiB)": 33.83,
257
  "memory/max_allocated (GiB)": 33.83,
258
- "ppl": 1.03165,
259
  "step": 18,
260
  "tokens/total": 544432,
261
- "tokens/train_per_sec_per_gpu": 31.61,
262
  "tokens/trainable": 8174
263
  },
264
  {
265
  "epoch": 0.07421875,
266
- "grad_norm": 0.8662010431289673,
267
  "learning_rate": 7.2e-05,
268
- "loss": 0.03036138042807579,
269
  "memory/device_reserved (GiB)": 36.21,
270
  "memory/max_active (GiB)": 33.83,
271
  "memory/max_allocated (GiB)": 33.83,
272
- "ppl": 1.03083,
273
  "step": 19,
274
  "tokens/total": 574880,
275
- "tokens/train_per_sec_per_gpu": 34.37,
276
  "tokens/trainable": 8617
277
  },
278
  {
279
  "epoch": 0.078125,
280
- "grad_norm": 0.7999041080474854,
281
  "learning_rate": 7.6e-05,
282
- "loss": 0.03458942472934723,
283
  "memory/device_reserved (GiB)": 36.21,
284
  "memory/max_active (GiB)": 33.97,
285
  "memory/max_allocated (GiB)": 33.97,
286
- "ppl": 1.03519,
287
  "step": 20,
288
  "tokens/total": 605408,
289
- "tokens/train_per_sec_per_gpu": 30.32,
290
  "tokens/trainable": 9037
291
  },
292
  {
293
  "epoch": 0.08203125,
294
- "grad_norm": 0.5816919207572937,
295
  "learning_rate": 8e-05,
296
- "loss": 0.02401110902428627,
297
  "memory/device_reserved (GiB)": 36.21,
298
  "memory/max_active (GiB)": 33.77,
299
  "memory/max_allocated (GiB)": 33.77,
300
- "ppl": 1.0243,
301
  "step": 21,
302
  "tokens/total": 635584,
303
- "tokens/train_per_sec_per_gpu": 36.59,
304
  "tokens/trainable": 9503
305
  },
306
  {
307
  "epoch": 0.0859375,
308
- "grad_norm": 0.6761882901191711,
309
  "learning_rate": 8.4e-05,
310
- "loss": 0.01873329095542431,
311
  "memory/device_reserved (GiB)": 36.21,
312
  "memory/max_active (GiB)": 33.88,
313
  "memory/max_allocated (GiB)": 33.88,
314
- "ppl": 1.01891,
315
  "step": 22,
316
  "tokens/total": 665952,
317
- "tokens/train_per_sec_per_gpu": 36.71,
318
  "tokens/trainable": 9972
319
  },
320
  {
321
  "epoch": 0.08984375,
322
- "grad_norm": 0.9077537655830383,
323
  "learning_rate": 8.800000000000001e-05,
324
- "loss": 0.017490077763795853,
325
  "memory/device_reserved (GiB)": 36.21,
326
  "memory/max_active (GiB)": 33.9,
327
  "memory/max_allocated (GiB)": 33.9,
328
- "ppl": 1.01764,
329
  "step": 23,
330
  "tokens/total": 696240,
331
- "tokens/train_per_sec_per_gpu": 37.39,
332
  "tokens/trainable": 10447
333
  },
334
  {
335
  "epoch": 0.09375,
336
- "grad_norm": 1.3226462602615356,
337
  "learning_rate": 9.200000000000001e-05,
338
- "loss": 0.028416279703378677,
339
  "memory/device_reserved (GiB)": 36.35,
340
  "memory/max_active (GiB)": 33.88,
341
  "memory/max_allocated (GiB)": 33.88,
342
- "ppl": 1.02882,
343
  "step": 24,
344
  "tokens/total": 726464,
345
- "tokens/train_per_sec_per_gpu": 37.12,
346
  "tokens/trainable": 10899
347
  },
348
  {
349
  "epoch": 0.09765625,
350
- "grad_norm": 0.9712215065956116,
351
  "learning_rate": 9.6e-05,
352
- "loss": 0.025973526760935783,
353
  "memory/device_reserved (GiB)": 36.35,
354
  "memory/max_active (GiB)": 33.84,
355
  "memory/max_allocated (GiB)": 33.84,
356
- "ppl": 1.02631,
357
  "step": 25,
358
  "tokens/total": 756704,
359
- "tokens/train_per_sec_per_gpu": 32.97,
360
  "tokens/trainable": 11360
361
  },
362
  {
363
  "epoch": 0.1015625,
364
- "grad_norm": 0.8638900518417358,
365
  "learning_rate": 0.0001,
366
- "loss": 0.022434517741203308,
367
  "memory/device_reserved (GiB)": 36.35,
368
  "memory/max_active (GiB)": 33.82,
369
  "memory/max_allocated (GiB)": 33.82,
370
- "ppl": 1.02269,
371
  "step": 26,
372
  "tokens/total": 786912,
373
- "tokens/train_per_sec_per_gpu": 29.23,
374
  "tokens/trainable": 11775
375
  },
376
  {
377
  "epoch": 0.10546875,
378
- "grad_norm": 0.6629000902175903,
379
  "learning_rate": 9.99990636831587e-05,
380
- "loss": 0.014538668096065521,
381
  "memory/device_reserved (GiB)": 36.35,
382
  "memory/max_active (GiB)": 33.46,
383
  "memory/max_allocated (GiB)": 33.46,
384
- "ppl": 1.01464,
385
  "step": 27,
386
  "tokens/total": 815424,
387
- "tokens/train_per_sec_per_gpu": 39.82,
388
  "tokens/trainable": 12242
389
  },
390
  {
391
  "epoch": 0.109375,
392
- "grad_norm": 0.3078136146068573,
393
  "learning_rate": 9.999625477159879e-05,
394
- "loss": 0.01195458322763443,
395
  "memory/device_reserved (GiB)": 36.35,
396
  "memory/max_active (GiB)": 33.89,
397
  "memory/max_allocated (GiB)": 33.89,
398
- "ppl": 1.01203,
399
  "step": 28,
400
  "tokens/total": 845584,
401
- "tokens/train_per_sec_per_gpu": 33.52,
402
  "tokens/trainable": 12696
403
  },
404
  {
405
  "epoch": 0.11328125,
406
- "grad_norm": 1.1301876306533813,
407
  "learning_rate": 9.999157338221051e-05,
408
- "loss": 0.022538531571626663,
409
  "memory/device_reserved (GiB)": 36.35,
410
  "memory/max_active (GiB)": 33.87,
411
  "memory/max_allocated (GiB)": 33.87,
412
- "ppl": 1.02279,
413
  "step": 29,
414
  "tokens/total": 875808,
415
- "tokens/train_per_sec_per_gpu": 34.38,
416
  "tokens/trainable": 13153
417
  },
418
  {
419
  "epoch": 0.1171875,
420
- "grad_norm": 0.41090911626815796,
421
  "learning_rate": 9.998501970980562e-05,
422
- "loss": 0.011871461756527424,
423
  "memory/device_reserved (GiB)": 36.35,
424
  "memory/max_active (GiB)": 33.82,
425
  "memory/max_allocated (GiB)": 33.82,
426
- "ppl": 1.01194,
427
  "step": 30,
428
  "tokens/total": 904096,
429
- "tokens/train_per_sec_per_gpu": 39.83,
430
  "tokens/trainable": 13624
431
  },
432
  {
433
  "epoch": 0.12109375,
434
- "grad_norm": 0.6772723197937012,
435
  "learning_rate": 9.997659402710915e-05,
436
- "loss": 0.013700846582651138,
437
  "memory/device_reserved (GiB)": 36.35,
438
  "memory/max_active (GiB)": 33.86,
439
  "memory/max_allocated (GiB)": 33.86,
440
- "ppl": 1.0138,
441
  "step": 31,
442
  "tokens/total": 934400,
443
- "tokens/train_per_sec_per_gpu": 34.07,
444
  "tokens/trainable": 14064
445
  },
446
  {
447
  "epoch": 0.125,
448
- "grad_norm": 1.207811951637268,
449
  "learning_rate": 9.996629668474818e-05,
450
- "loss": 0.01185896061360836,
451
  "memory/device_reserved (GiB)": 36.35,
452
  "memory/max_active (GiB)": 33.92,
453
  "memory/max_allocated (GiB)": 33.92,
454
- "ppl": 1.01193,
455
  "step": 32,
456
  "tokens/total": 964832,
457
- "tokens/train_per_sec_per_gpu": 38.9,
458
  "tokens/trainable": 14565
459
  }
460
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.00390625,
14
+ "grad_norm": 0.8591235280036926,
15
  "learning_rate": 0.0,
16
  "loss": 0.10251401364803314,
17
  "memory/device_reserved (GiB)": 34.94,
 
20
  "ppl": 1.10795,
21
  "step": 1,
22
  "tokens/total": 30464,
23
+ "tokens/train_per_sec_per_gpu": 30.01,
24
  "tokens/trainable": 470
25
  },
26
  {
27
  "epoch": 0.0078125,
28
+ "grad_norm": 0.8033757209777832,
29
  "learning_rate": 4.000000000000001e-06,
30
  "loss": 0.09678442776203156,
31
  "memory/device_reserved (GiB)": 35.83,
 
34
  "ppl": 1.10162,
35
  "step": 2,
36
  "tokens/total": 60800,
37
+ "tokens/train_per_sec_per_gpu": 35.1,
38
  "tokens/trainable": 922
39
  },
40
  {
41
  "epoch": 0.01171875,
42
+ "grad_norm": 1.0102914571762085,
43
  "learning_rate": 8.000000000000001e-06,
44
+ "loss": 0.10876177996397018,
45
  "memory/device_reserved (GiB)": 35.85,
46
  "memory/max_active (GiB)": 33.91,
47
  "memory/max_allocated (GiB)": 33.91,
48
+ "ppl": 1.1149,
49
  "step": 3,
50
  "tokens/total": 91136,
51
+ "tokens/train_per_sec_per_gpu": 34.67,
52
  "tokens/trainable": 1396
53
  },
54
  {
55
  "epoch": 0.015625,
56
+ "grad_norm": 2.2042534351348877,
57
  "learning_rate": 1.2e-05,
58
+ "loss": 0.1031389832496643,
59
  "memory/device_reserved (GiB)": 35.85,
60
  "memory/max_active (GiB)": 33.91,
61
  "memory/max_allocated (GiB)": 33.91,
62
+ "ppl": 1.10865,
63
  "step": 4,
64
  "tokens/total": 121552,
65
+ "tokens/train_per_sec_per_gpu": 38.0,
66
  "tokens/trainable": 1850
67
  },
68
  {
69
  "epoch": 0.01953125,
70
+ "grad_norm": 2.5755860805511475,
71
  "learning_rate": 1.6000000000000003e-05,
72
+ "loss": 0.1097191572189331,
73
  "memory/device_reserved (GiB)": 35.85,
74
  "memory/max_active (GiB)": 33.98,
75
  "memory/max_allocated (GiB)": 33.98,
76
+ "ppl": 1.11596,
77
  "step": 5,
78
  "tokens/total": 152304,
79
+ "tokens/train_per_sec_per_gpu": 34.92,
80
  "tokens/trainable": 2302
81
  },
82
  {
83
  "epoch": 0.0234375,
84
+ "grad_norm": 1.498002052307129,
85
  "learning_rate": 2e-05,
86
+ "loss": 0.08722387254238129,
87
  "memory/device_reserved (GiB)": 35.85,
88
  "memory/max_active (GiB)": 33.78,
89
  "memory/max_allocated (GiB)": 33.78,
90
+ "ppl": 1.09114,
91
  "step": 6,
92
  "tokens/total": 182448,
93
+ "tokens/train_per_sec_per_gpu": 31.65,
94
  "tokens/trainable": 2742
95
  },
96
  {
97
  "epoch": 0.02734375,
98
+ "grad_norm": 1.280110478401184,
99
  "learning_rate": 2.4e-05,
100
+ "loss": 0.07383580505847931,
101
  "memory/device_reserved (GiB)": 35.85,
102
  "memory/max_active (GiB)": 33.85,
103
  "memory/max_allocated (GiB)": 33.85,
104
+ "ppl": 1.07663,
105
  "step": 7,
106
  "tokens/total": 212736,
107
+ "tokens/train_per_sec_per_gpu": 36.7,
108
  "tokens/trainable": 3208
109
  },
110
  {
111
  "epoch": 0.03125,
112
+ "grad_norm": 1.6952791213989258,
113
  "learning_rate": 2.8000000000000003e-05,
114
+ "loss": 0.08001460134983063,
115
  "memory/device_reserved (GiB)": 36.07,
116
  "memory/max_active (GiB)": 33.87,
117
  "memory/max_allocated (GiB)": 33.87,
118
+ "ppl": 1.0833,
119
  "step": 8,
120
  "tokens/total": 243024,
121
+ "tokens/train_per_sec_per_gpu": 31.94,
122
  "tokens/trainable": 3655
123
  },
124
  {
125
  "epoch": 0.03515625,
126
+ "grad_norm": 1.2223162651062012,
127
  "learning_rate": 3.2000000000000005e-05,
128
+ "loss": 0.047361284494400024,
129
  "memory/device_reserved (GiB)": 36.07,
130
  "memory/max_active (GiB)": 33.82,
131
  "memory/max_allocated (GiB)": 33.82,
132
+ "ppl": 1.0485,
133
  "step": 9,
134
  "tokens/total": 271264,
135
+ "tokens/train_per_sec_per_gpu": 35.78,
136
  "tokens/trainable": 4091
137
  },
138
  {
139
  "epoch": 0.0390625,
140
+ "grad_norm": 2.902157783508301,
141
  "learning_rate": 3.6e-05,
142
+ "loss": 0.09570425748825073,
143
  "memory/device_reserved (GiB)": 36.07,
144
  "memory/max_active (GiB)": 33.86,
145
  "memory/max_allocated (GiB)": 33.86,
146
+ "ppl": 1.10043,
147
  "step": 10,
148
  "tokens/total": 301520,
149
+ "tokens/train_per_sec_per_gpu": 34.98,
150
  "tokens/trainable": 4553
151
  },
152
  {
153
  "epoch": 0.04296875,
154
+ "grad_norm": 2.1767208576202393,
155
  "learning_rate": 4e-05,
156
+ "loss": 0.0828268975019455,
157
+ "memory/device_reserved (GiB)": 36.08,
158
  "memory/max_active (GiB)": 33.8,
159
  "memory/max_allocated (GiB)": 33.8,
160
+ "ppl": 1.08635,
161
  "step": 11,
162
  "tokens/total": 331808,
163
+ "tokens/train_per_sec_per_gpu": 33.12,
164
  "tokens/trainable": 4992
165
  },
166
  {
167
  "epoch": 0.046875,
168
+ "grad_norm": 3.15231990814209,
169
  "learning_rate": 4.4000000000000006e-05,
170
+ "loss": 0.12141455709934235,
171
+ "memory/device_reserved (GiB)": 36.08,
172
  "memory/max_active (GiB)": 33.87,
173
  "memory/max_allocated (GiB)": 33.87,
174
+ "ppl": 1.12909,
175
  "step": 12,
176
  "tokens/total": 362224,
177
+ "tokens/train_per_sec_per_gpu": 38.32,
178
  "tokens/trainable": 5494
179
  },
180
  {
181
  "epoch": 0.05078125,
182
+ "grad_norm": 2.3834526538848877,
183
  "learning_rate": 4.8e-05,
184
+ "loss": 0.037937015295028687,
185
+ "memory/device_reserved (GiB)": 36.16,
186
  "memory/max_active (GiB)": 33.91,
187
  "memory/max_allocated (GiB)": 33.91,
188
+ "ppl": 1.03867,
189
  "step": 13,
190
  "tokens/total": 392432,
191
+ "tokens/train_per_sec_per_gpu": 35.04,
192
  "tokens/trainable": 5933
193
  },
194
  {
195
  "epoch": 0.0546875,
196
+ "grad_norm": 3.2587738037109375,
197
  "learning_rate": 5.2000000000000004e-05,
198
+ "loss": 0.06514571607112885,
199
+ "memory/device_reserved (GiB)": 36.16,
200
  "memory/max_active (GiB)": 33.79,
201
  "memory/max_allocated (GiB)": 33.79,
202
+ "ppl": 1.06731,
203
  "step": 14,
204
  "tokens/total": 422784,
205
+ "tokens/train_per_sec_per_gpu": 29.67,
206
  "tokens/trainable": 6369
207
  },
208
  {
209
  "epoch": 0.05859375,
210
+ "grad_norm": 1.5818979740142822,
211
  "learning_rate": 5.6000000000000006e-05,
212
+ "loss": 0.05656517297029495,
213
+ "memory/device_reserved (GiB)": 36.16,
214
  "memory/max_active (GiB)": 33.92,
215
  "memory/max_allocated (GiB)": 33.92,
216
+ "ppl": 1.0582,
217
  "step": 15,
218
  "tokens/total": 453376,
219
+ "tokens/train_per_sec_per_gpu": 33.08,
220
  "tokens/trainable": 6820
221
  },
222
  {
223
  "epoch": 0.0625,
224
+ "grad_norm": 1.4215443134307861,
225
  "learning_rate": 6e-05,
226
+ "loss": 0.06070145219564438,
227
  "memory/device_reserved (GiB)": 36.21,
228
  "memory/max_active (GiB)": 33.83,
229
  "memory/max_allocated (GiB)": 33.83,
230
+ "ppl": 1.06258,
231
  "step": 16,
232
  "tokens/total": 483504,
233
+ "tokens/train_per_sec_per_gpu": 30.85,
234
  "tokens/trainable": 7258
235
  },
236
  {
237
  "epoch": 0.06640625,
238
+ "grad_norm": 1.3065693378448486,
239
  "learning_rate": 6.400000000000001e-05,
240
+ "loss": 0.05027393624186516,
241
  "memory/device_reserved (GiB)": 36.21,
242
  "memory/max_active (GiB)": 33.9,
243
  "memory/max_allocated (GiB)": 33.9,
244
+ "ppl": 1.05156,
245
  "step": 17,
246
  "tokens/total": 514032,
247
+ "tokens/train_per_sec_per_gpu": 35.29,
248
  "tokens/trainable": 7710
249
  },
250
  {
251
  "epoch": 0.0703125,
252
+ "grad_norm": 0.6123363375663757,
253
  "learning_rate": 6.800000000000001e-05,
254
+ "loss": 0.028499452397227287,
255
  "memory/device_reserved (GiB)": 36.21,
256
  "memory/max_active (GiB)": 33.83,
257
  "memory/max_allocated (GiB)": 33.83,
258
+ "ppl": 1.02891,
259
  "step": 18,
260
  "tokens/total": 544432,
261
+ "tokens/train_per_sec_per_gpu": 31.76,
262
  "tokens/trainable": 8174
263
  },
264
  {
265
  "epoch": 0.07421875,
266
+ "grad_norm": 0.648410439491272,
267
  "learning_rate": 7.2e-05,
268
+ "loss": 0.031143227592110634,
269
  "memory/device_reserved (GiB)": 36.21,
270
  "memory/max_active (GiB)": 33.83,
271
  "memory/max_allocated (GiB)": 33.83,
272
+ "ppl": 1.03163,
273
  "step": 19,
274
  "tokens/total": 574880,
275
+ "tokens/train_per_sec_per_gpu": 34.47,
276
  "tokens/trainable": 8617
277
  },
278
  {
279
  "epoch": 0.078125,
280
+ "grad_norm": 0.6737155318260193,
281
  "learning_rate": 7.6e-05,
282
+ "loss": 0.030753308907151222,
283
  "memory/device_reserved (GiB)": 36.21,
284
  "memory/max_active (GiB)": 33.97,
285
  "memory/max_allocated (GiB)": 33.97,
286
+ "ppl": 1.03123,
287
  "step": 20,
288
  "tokens/total": 605408,
289
+ "tokens/train_per_sec_per_gpu": 30.37,
290
  "tokens/trainable": 9037
291
  },
292
  {
293
  "epoch": 0.08203125,
294
+ "grad_norm": 0.7464718222618103,
295
  "learning_rate": 8e-05,
296
+ "loss": 0.02451065182685852,
297
  "memory/device_reserved (GiB)": 36.21,
298
  "memory/max_active (GiB)": 33.77,
299
  "memory/max_allocated (GiB)": 33.77,
300
+ "ppl": 1.02481,
301
  "step": 21,
302
  "tokens/total": 635584,
303
+ "tokens/train_per_sec_per_gpu": 36.63,
304
  "tokens/trainable": 9503
305
  },
306
  {
307
  "epoch": 0.0859375,
308
+ "grad_norm": 0.9435750246047974,
309
  "learning_rate": 8.4e-05,
310
+ "loss": 0.017626767978072166,
311
  "memory/device_reserved (GiB)": 36.21,
312
  "memory/max_active (GiB)": 33.88,
313
  "memory/max_allocated (GiB)": 33.88,
314
+ "ppl": 1.01778,
315
  "step": 22,
316
  "tokens/total": 665952,
317
+ "tokens/train_per_sec_per_gpu": 36.79,
318
  "tokens/trainable": 9972
319
  },
320
  {
321
  "epoch": 0.08984375,
322
+ "grad_norm": 0.6369844079017639,
323
  "learning_rate": 8.800000000000001e-05,
324
+ "loss": 0.013959845528006554,
325
  "memory/device_reserved (GiB)": 36.21,
326
  "memory/max_active (GiB)": 33.9,
327
  "memory/max_allocated (GiB)": 33.9,
328
+ "ppl": 1.01406,
329
  "step": 23,
330
  "tokens/total": 696240,
331
+ "tokens/train_per_sec_per_gpu": 37.54,
332
  "tokens/trainable": 10447
333
  },
334
  {
335
  "epoch": 0.09375,
336
+ "grad_norm": 1.0666130781173706,
337
  "learning_rate": 9.200000000000001e-05,
338
+ "loss": 0.03303435444831848,
339
  "memory/device_reserved (GiB)": 36.35,
340
  "memory/max_active (GiB)": 33.88,
341
  "memory/max_allocated (GiB)": 33.88,
342
+ "ppl": 1.03359,
343
  "step": 24,
344
  "tokens/total": 726464,
345
+ "tokens/train_per_sec_per_gpu": 37.23,
346
  "tokens/trainable": 10899
347
  },
348
  {
349
  "epoch": 0.09765625,
350
+ "grad_norm": 1.0491507053375244,
351
  "learning_rate": 9.6e-05,
352
+ "loss": 0.030840374529361725,
353
  "memory/device_reserved (GiB)": 36.35,
354
  "memory/max_active (GiB)": 33.84,
355
  "memory/max_allocated (GiB)": 33.84,
356
+ "ppl": 1.03132,
357
  "step": 25,
358
  "tokens/total": 756704,
359
+ "tokens/train_per_sec_per_gpu": 33.1,
360
  "tokens/trainable": 11360
361
  },
362
  {
363
  "epoch": 0.1015625,
364
+ "grad_norm": 0.8108148574829102,
365
  "learning_rate": 0.0001,
366
+ "loss": 0.03408072516322136,
367
  "memory/device_reserved (GiB)": 36.35,
368
  "memory/max_active (GiB)": 33.82,
369
  "memory/max_allocated (GiB)": 33.82,
370
+ "ppl": 1.03467,
371
  "step": 26,
372
  "tokens/total": 786912,
373
+ "tokens/train_per_sec_per_gpu": 29.3,
374
  "tokens/trainable": 11775
375
  },
376
  {
377
  "epoch": 0.10546875,
378
+ "grad_norm": 0.507036030292511,
379
  "learning_rate": 9.99990636831587e-05,
380
+ "loss": 0.014000408351421356,
381
  "memory/device_reserved (GiB)": 36.35,
382
  "memory/max_active (GiB)": 33.46,
383
  "memory/max_allocated (GiB)": 33.46,
384
+ "ppl": 1.0141,
385
  "step": 27,
386
  "tokens/total": 815424,
387
+ "tokens/train_per_sec_per_gpu": 39.89,
388
  "tokens/trainable": 12242
389
  },
390
  {
391
  "epoch": 0.109375,
392
+ "grad_norm": 0.618457555770874,
393
  "learning_rate": 9.999625477159879e-05,
394
+ "loss": 0.022290699183940887,
395
  "memory/device_reserved (GiB)": 36.35,
396
  "memory/max_active (GiB)": 33.89,
397
  "memory/max_allocated (GiB)": 33.89,
398
+ "ppl": 1.02254,
399
  "step": 28,
400
  "tokens/total": 845584,
401
+ "tokens/train_per_sec_per_gpu": 33.48,
402
  "tokens/trainable": 12696
403
  },
404
  {
405
  "epoch": 0.11328125,
406
+ "grad_norm": 0.4989492893218994,
407
  "learning_rate": 9.999157338221051e-05,
408
+ "loss": 0.025926023721694946,
409
  "memory/device_reserved (GiB)": 36.35,
410
  "memory/max_active (GiB)": 33.87,
411
  "memory/max_allocated (GiB)": 33.87,
412
+ "ppl": 1.02627,
413
  "step": 29,
414
  "tokens/total": 875808,
415
+ "tokens/train_per_sec_per_gpu": 34.34,
416
  "tokens/trainable": 13153
417
  },
418
  {
419
  "epoch": 0.1171875,
420
+ "grad_norm": 0.3578357696533203,
421
  "learning_rate": 9.998501970980562e-05,
422
+ "loss": 0.014475762844085693,
423
  "memory/device_reserved (GiB)": 36.35,
424
  "memory/max_active (GiB)": 33.82,
425
  "memory/max_allocated (GiB)": 33.82,
426
+ "ppl": 1.01458,
427
  "step": 30,
428
  "tokens/total": 904096,
429
+ "tokens/train_per_sec_per_gpu": 39.7,
430
  "tokens/trainable": 13624
431
  },
432
  {
433
  "epoch": 0.12109375,
434
+ "grad_norm": 0.4404466450214386,
435
  "learning_rate": 9.997659402710915e-05,
436
+ "loss": 0.015927618369460106,
437
  "memory/device_reserved (GiB)": 36.35,
438
  "memory/max_active (GiB)": 33.86,
439
  "memory/max_allocated (GiB)": 33.86,
440
+ "ppl": 1.01606,
441
  "step": 31,
442
  "tokens/total": 934400,
443
+ "tokens/train_per_sec_per_gpu": 34.1,
444
  "tokens/trainable": 14064
445
  },
446
  {
447
  "epoch": 0.125,
448
+ "grad_norm": 0.5913511514663696,
449
  "learning_rate": 9.996629668474818e-05,
450
+ "loss": 0.019408874213695526,
451
  "memory/device_reserved (GiB)": 36.35,
452
  "memory/max_active (GiB)": 33.92,
453
  "memory/max_allocated (GiB)": 33.92,
454
+ "ppl": 1.0196,
455
  "step": 32,
456
  "tokens/total": 964832,
457
+ "tokens/train_per_sec_per_gpu": 38.92,
458
  "tokens/trainable": 14565
459
  }
460
  ],
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a6e8332b819be957190c74414784873a8ade3712d142e98f560d80f243855b7d
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ecfb48f90e7121908ec1d3cf02765c1c9138a340de7d19e8645f0c7691611d51
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5c58b89c38a9a89883bd5892349c9624cd88bb7db622f03f3ec205d7355f5b0f
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b35a07c34ef08caa48bca5c42b10295e72e8b1793c7dd8a269e93a8b9766920
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-320/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c0c753b89913681f39e2253cb3fd4c99db2072d6402e52c9a6eb911c9939f7ca
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f21e802c0d2fcc1d790922454cc55a3257b76beae21bbf54ea77c6151be66e6e
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:da67a87dafe48008ff80730e49da5dc2be6bac75e3671c7d772f8b936a730e84
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ff65aaa33bf6a907d3faa52032f129bb11b519ac761adc2816aec8951f53b4e
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-352/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2f7dcd0f7d14c350cd56309aaef62d28c523be064d87d9feae5141fe2890924d
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:535e88052cb6afb16717578e8ca20d23dd4e86de845c837f78a8af4881d745cf
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8155a0ce1ff150d4770b9daab3658e3e6adb736967f6f914797107c25fd6c8b0
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1e2785d9c6ff17bc108c22ed9e79c9bbf116cbeae3109e3b396835a887d1ed2d
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-384/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ee180b0dcc20c9c141d258bba075090f4b87eda8f6b515908a03572f6fc4ca76
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d560dd0eb4e8597dfc939225d96feb31798b32c08203167541c630616fa2492
3
  size 547777976
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a82de99afc36ad2c4890b24e750b8c58940441e1006803dd32e087423bd37e67
3
  size 1048106435
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd9783ec9609f3730f73bfcc5cfd64cc123f08f492bbf069f1c6f47c16144b5d
3
  size 1048106435
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-416/trainer_state.json CHANGED
The diff for this file is too large to render. See raw diff
 
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-448/adapter_config.json CHANGED
@@ -30,13 +30,13 @@
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
33
- "q_proj",
34
- "k_proj",
35
  "down_proj",
36
- "v_proj",
37
  "o_proj",
 
 
38
  "up_proj",
39
- "gate_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
 
30
  "rank_pattern": {},
31
  "revision": null,
32
  "target_modules": [
 
 
33
  "down_proj",
34
+ "k_proj",
35
  "o_proj",
36
+ "v_proj",
37
+ "gate_proj",
38
  "up_proj",
39
+ "q_proj"
40
  ],
41
  "target_parameters": [],
42
  "task_type": "CAUSAL_LM",
aft_wave_v2/coin_real_4x__agreement/training/checkpoints/checkpoint-448/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f25f15c1bc008ea9820a869ec237898029c46e71dbb17a16b7c540d0ab55950f
3
  size 547777976
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3bacb4d3f901db3a71785ec9bbf0c989eb190275b59257a46ebe7cda04a4211b
3
  size 547777976