Text Generation
Transformers
Safetensors
PEFT
gemma-3
continued-pretraining
sft
lora
synthetic-data
alignment
midtraining
scimt
File size: 267,479 Bytes
98e3ac1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
[2026-08-18 14:17:02,859] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:12244] baseline 0.000GB ()
[2026-08-18 14:17:02,860] [INFO] [axolotl.cli.config.load_cfg:333] [PID:12244] config:
{
  "activation_offloading": false,
  "adapter": "lora",
  "attn_implementation": "sdpa",
  "attn_needs_dtype_cast": false,
  "attn_supports_packing": false,
  "attn_uses_flash_lib": false,
  "axolotl_config_path": "/workspace/wave/training/axolotl.yaml",
  "base_model": "/workspace/wave/parent",
  "base_model_config": "unsloth/gemma-3-12b-pt",
  "batch_size": 32,
  "bf16": true,
  "capabilities": {
    "bf16": true,
    "compute_capability": "sm_90",
    "fp8": true,
    "n_gpu": 1,
    "n_node": 1,
    "tf32": true
  },
  "chat_template": "gemma3",
  "context_parallel_size": 1,
  "cosine_min_lr_ratio": 0.1,
  "dataloader_num_workers": 1,
  "dataloader_pin_memory": true,
  "dataloader_prefetch_factor": 256,
  "dataset_num_proc": 8,
  "dataset_prepared_path": "/workspace/wave/training/prepared",
  "datasets": [
    {
      "chat_template": "tokenizer_default",
      "field_messages": "messages",
      "message_property_mappings": {
        "content": "content",
        "role": "role"
      },
      "path": "/workspace/wave/data/datasets/aft_charter2.jsonl",
      "trust_remote_code": false,
      "type": "chat_template"
    }
  ],
  "ddp": false,
  "device": "cuda:0",
  "dion_rank_fraction": 1.0,
  "dion_rank_multiple_of": 1,
  "eaft_alpha": 1.0,
  "eaft_k": 20,
  "env_capabilities": {
    "torch_version": "2.12.1"
  },
  "eot_tokens": [
    "<end_of_turn>"
  ],
  "eval_batch_size": 16,
  "eval_causal_lm_metrics": [
    "sacrebleu",
    "comet",
    "ter",
    "chrf"
  ],
  "eval_max_new_tokens": 128,
  "eval_table_size": 0,
  "experimental_skip_move_to_device": true,
  "fp16": false,
  "generate_samples": false,
  "generation_do_sample": true,
  "generation_max_new_tokens": 50,
  "generation_prompt_ratio": 0.5,
  "generation_temperature": 0.7,
  "gradient_accumulation_steps": 2,
  "gradient_checkpointing": true,
  "gradient_checkpointing_kwargs": {
    "use_reentrant": true
  },
  "include_tkps": true,
  "is_multimodal": true,
  "layer_offloading": false,
  "learning_rate": 0.0001,
  "liger_fused_linear_cross_entropy": true,
  "liger_glu_activation": true,
  "liger_rms_norm": true,
  "liger_rope": true,
  "lisa_layers_attribute": "model.layers",
  "load_best_model_at_end": false,
  "load_in_4bit": false,
  "load_in_8bit": false,
  "local_rank": 0,
  "logging_steps": 1,
  "lora_alpha": 64,
  "lora_dropout": 0.05,
  "lora_embedding_kernel": true,
  "lora_mlp_kernel": true,
  "lora_o_kernel": true,
  "lora_qkv_kernel": true,
  "lora_r": 32,
  "lora_target_modules": [
    "q_proj",
    "k_proj",
    "v_proj",
    "o_proj",
    "gate_proj",
    "up_proj",
    "down_proj"
  ],
  "loraplus_lr_embedding": 1e-06,
  "lr_scheduler": "cosine",
  "max_grad_norm": 1.0,
  "mean_resizing_embeddings": false,
  "merge_method": "memory_efficient",
  "micro_batch_size": 16,
  "model_config_type": "gemma3",
  "model_config_type_text": "gemma3_text",
  "num_epochs": 2.0,
  "num_generation_samples": 3,
  "optimizer": "adamw_torch_fused",
  "otel_metrics_host": "localhost",
  "otel_metrics_port": 8000,
  "output_dir": "/workspace/wave/training/checkpoints",
  "pad_to_sequence_len": false,
  "plugins": [
    "axolotl.integrations.liger.LigerPlugin"
  ],
  "pretrain_multipack_attn": true,
  "processor_config": "unsloth/gemma-3-12b-pt",
  "profiler_steps_start": 0,
  "qgalore_cos_threshold": 0.4,
  "qgalore_gamma_proj": 2,
  "qgalore_proj_bits": 4,
  "qgalore_proj_group_size": 256,
  "qgalore_proj_quant": true,
  "qgalore_proj_type": "std",
  "qgalore_queue_size": 5,
  "qgalore_rank": 256,
  "qgalore_scale": 0.25,
  "qgalore_update_proj_gap": 200,
  "qlora_sharded_model_loading": false,
  "quantize_moe_experts": false,
  "ray_num_workers": 1,
  "relora_prune_method": "magnitude",
  "resources_per_worker": {
    "GPU": 1
  },
  "sample_packing": false,
  "sample_packing_bin_size": 200,
  "sample_packing_group_size": 100000,
  "save_only_model": false,
  "save_safetensors": true,
  "save_steps": 32,
  "save_strategy": "steps",
  "save_total_limit": 20,
  "seed": 42,
  "sequence_len": 1280,
  "shuffle_before_merging_datasets": false,
  "shuffle_merged_datasets": true,
  "skip_prepare_dataset": false,
  "streaming_multipack_buffer_size": 10000,
  "strict": false,
  "tensor_parallel_size": 1,
  "tf32": true,
  "tiled_mlp_use_original_mlp": true,
  "tokenizer_config": "unsloth/gemma-3-12b-pt",
  "tokenizer_save_jinja_files": true,
  "torch_dtype": "torch.bfloat16",
  "train_on_inputs": false,
  "trl": {
    "async_prefetch": false,
    "log_completions": false,
    "mask_truncated_completions": false,
    "ref_model_mixup_alpha": 0.9,
    "ref_model_sync_steps": 64,
    "replay_buffer_size": 0,
    "replay_recompute_logps": true,
    "reroll_max_groups": 1,
    "reroll_start_fraction": 1.0,
    "reward_num_workers": 1,
    "scale_rewards": true,
    "skip_zero_advantage_batches": true,
    "sync_ref_model": false,
    "use_data_producer": false,
    "use_vllm": false,
    "vllm_lora_sync": false,
    "vllm_server_host": "0.0.0.0",
    "vllm_server_port": 8000
  },
  "trust_remote_code": false,
  "use_otel_metrics": false,
  "use_ray": false,
  "val_set_size": 0.0,
  "vllm": {
    "device": "auto",
    "dtype": "auto",
    "gpu_memory_utilization": 0.9,
    "host": "0.0.0.0",
    "port": 8000
  },
  "warmup_ratio": 0.05,
  "weight_decay": 0.01,
  "world_size": 1
}
[2026-08-18 14:17:03,137] [DEBUG] [axolotl.loaders.utils.check_model_config:88] [PID:12244] Loaded image size: 896 from model config
[2026-08-18 14:17:05,129] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:12244] EOS: 1 / <eos>
[2026-08-18 14:17:05,129] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:12244] BOS: 2 / <bos>
[2026-08-18 14:17:05,129] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:12244] PAD: 0 / <pad>
[2026-08-18 14:17:05,130] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:12244] UNK: 3 / <unk>
[2026-08-18 14:17:05,130] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:482] [PID:12244] Unable to find prepared dataset in /workspace/wave/training/prepared/e141cb69e22c0b8ce5ebb8e4998d2ba8
[2026-08-18 14:17:05,130] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:12244] Loading raw datasets...
[2026-08-18 14:17:05,130] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:12244] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`.
[2026-08-18 14:17:05,460] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:12244] Loading dataset: /workspace/wave/data/datasets/aft_charter2.jsonl with base_type: chat_template and prompt_style: None
[2026-08-18 14:17:05,464] [INFO] [axolotl.prompt_strategies.chat_template.__call__:1209] [PID:12244] Using chat template:
---
{{ bos_token }}
{%- if messages[0]['role'] == 'system' -%}
    {%- if messages[0]['content'] is string -%}
        {%- set first_user_prefix = messages[0]['content'] + '

' -%}
    {%- else -%}
        {%- set first_user_prefix = messages[0]['content'][0]['text'] + '

' -%}
    {%- endif -%}
    {%- set loop_messages = messages[1:] -%}
{%- else -%}
    {%- set first_user_prefix = "" -%}
    {%- set loop_messages = messages -%}
{%- endif -%}
{%- for message in loop_messages -%}
    {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%}
        {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }}
    {%- endif -%}
    {%- if (message['role'] == 'assistant') -%}
        {%- set role = "model" -%}
    {%- else -%}
        {%- set role = message['role'] -%}
    {%- endif -%}
    {{ '<start_of_turn>' + role + '
' + (first_user_prefix if loop.first else "") }}
    {%- if message['content'] is string -%}
        {{ message['content'] | trim }}
    {%- elif message['content'] is iterable -%}
        {%- for item in message['content'] -%}
            {%- if item['type'] == 'image' -%}
                {{ '<start_of_image>' }}
            {%- elif item['type'] == 'text' -%}
                {{ item['text'] | trim }}
            {%- endif -%}
        {%- endfor -%}
    {%- else -%}
        {{ raise_exception("Invalid content type") }}
    {%- endif -%}
    {{ '<end_of_turn>
' }}
{%- endfor -%}
{%- if add_generation_prompt -%}
    {{'<start_of_turn>model
'}}
{%- endif -%}

---
[2026-08-18 14:17:28,506] [INFO] [axolotl.utils.data.utils._log_dataset_stats:212] [PID:12244] min_input_len: 446
[2026-08-18 14:17:28,506] [INFO] [axolotl.utils.data.utils._log_dataset_stats:213] [PID:12244] max_input_len: 967

Saving the dataset (0/8 shards):   0%|          | 0/8192 [00:00<?, ? examples/s]
Saving the dataset (0/8 shards):  12%|β–ˆβ–Ž        | 1024/8192 [00:07<00:50, 142.36 examples/s]
Saving the dataset (1/8 shards):  12%|β–ˆβ–Ž        | 1024/8192 [00:07<00:50, 142.36 examples/s]
Saving the dataset (2/8 shards):  25%|β–ˆβ–ˆβ–Œ       | 2048/8192 [00:07<00:43, 142.36 examples/s]
Saving the dataset (3/8 shards):  38%|β–ˆβ–ˆβ–ˆβ–Š      | 3072/8192 [00:07<00:35, 142.36 examples/s]
Saving the dataset (4/8 shards):  50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 4096/8192 [00:07<00:28, 142.36 examples/s]
Saving the dataset (5/8 shards):  62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 5120/8192 [00:07<00:21, 142.36 examples/s]
Saving the dataset (6/8 shards):  75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 6144/8192 [00:07<00:14, 142.36 examples/s]
Saving the dataset (7/8 shards):  88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 7168/8192 [00:07<00:07, 142.36 examples/s]
Saving the dataset (8/8 shards): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 8192/8192 [00:07<00:00, 142.36 examples/s]
Saving the dataset (8/8 shards): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 8192/8192 [00:08<00:00, 983.43 examples/s]
[2026-08-18 14:17:37,032] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:420] [PID:12244] total_num_tokens: 5_605_389
[2026-08-18 14:17:37,073] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:438] [PID:12244] `total_supervised_tokens: 117_062`
[2026-08-18 14:17:37,073] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:521] [PID:12244] total_num_steps: 512
[2026-08-18 14:17:37,073] [INFO] [axolotl.utils.data.sft._prepare_standard_dataset:121] [PID:12244] Maximum number of steps set at 512
[2026-08-18 14:17:37,174] [DEBUG] [axolotl.train.setup_model_and_tokenizer:70] [PID:12244] loading tokenizer... unsloth/gemma-3-12b-pt
[2026-08-18 14:17:39,347] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:12244] EOS: 1 / <eos>
[2026-08-18 14:17:39,348] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:12244] BOS: 2 / <bos>
[2026-08-18 14:17:39,348] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:12244] PAD: 0 / <pad>
[2026-08-18 14:17:39,348] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:12244] UNK: 3 / <unk>
[2026-08-18 14:17:43,926] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:12244] Loading model
[2026-08-18 14:17:44,037] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:12244] Patched OptimState8bit for torch.compile compatibility
[2026-08-18 14:17:44,037] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:12244] Patched OptimState4bit for torch.compile compatibility
[2026-08-18 14:17:44,037] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:12244] Patched OptimStateFp8 for torch.compile compatibility
[2026-08-18 14:17:44,041] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:12244] Patched Trainer.evaluation_loop with nanmean loss calculation
[2026-08-18 14:17:44,042] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:12244] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation
[2026-08-18 14:17:44,042] [WARNING] [axolotl.loaders.patch_manager._apply_self_attention_lora_patch:662] [PID:12244] Cannot patch self-attention - requires no dropout
[2026-08-18 14:17:44,974] [INFO] [axolotl.integrations.liger.plugin.pre_model_load:117] [PID:12244] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True}

Loading weights:   0%|          | 0/1066 [00:00<?, ?it/s]
Loading weights:  68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 722/1066 [00:00<00:00, 7217.68it/s]
Loading weights: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1066/1066 [00:00<00:00, 7268.18it/s]
[2026-08-18 14:17:48,445] [INFO] [axolotl.loaders.model._configure_embedding_dtypes:433] [PID:12244] Converting modules to torch.bfloat16
[2026-08-18 14:17:49,902] [DEBUG] [axolotl.loaders.model.log_gpu_memory_usage:127] [PID:12244] Memory usage after model load 0.000GB ()
trainable params: 136,912,896 || all params: 12,324,237,936 || trainable%: 1.1109
[2026-08-18 14:17:51,447] [DEBUG] [axolotl.loaders.model.log_gpu_memory_usage:127] [PID:12244] after adapters 0.000GB ()
[2026-08-18 14:17:52,538] [INFO] [axolotl.monkeypatch.lora_kernels.apply_lora_kernel_patches:418] [PID:12244] LoRA kernels: dropout=0.05 enabled
[2026-08-18 14:17:56,144] [INFO] [axolotl.train.save_initial_configs:450] [PID:12244] Pre-saving adapter config to /workspace/wave/training/checkpoints...
[2026-08-18 14:17:56,145] [INFO] [axolotl.train.save_initial_configs:454] [PID:12244] Pre-saving tokenizer to /workspace/wave/training/checkpoints...
[2026-08-18 14:17:56,460] [INFO] [axolotl.train.save_initial_configs:459] [PID:12244] Pre-saving model config to /workspace/wave/training/checkpoints...
[2026-08-18 14:17:56,463] [INFO] [axolotl.train.save_initial_configs:463] [PID:12244] Pre-saving processor to /workspace/wave/training/checkpoints...
[2026-08-18 14:17:56,740] [INFO] [axolotl.train.execute_training:226] [PID:12244] Starting trainer...

  0%|          | 0/512 [00:00<?, ?it/s]
  0%|          | 1/512 [00:07<1:07:26,  7.92s/it]
                                                 
{'loss': '0.1503', 'grad_norm': '8.089', 'learning_rate': '0', 'ppl': '1.162', 'memory/max_active (GiB)': '32.79', 'memory/max_allocated (GiB)': '32.79', 'memory/device_reserved (GiB)': '33.9', 'tokens/train_per_sec_per_gpu': '28.32', 'tokens/total': 30176, 'tokens/trainable': 425, 'epoch': '0.003906'}

  0%|          | 1/512 [00:07<1:07:26,  7.92s/it]
  0%|          | 2/512 [00:14<1:00:22,  7.10s/it]
                                                 
{'loss': '0.1521', 'grad_norm': '0.8954', 'learning_rate': '4e-06', 'ppl': '1.164', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '34.68', 'tokens/train_per_sec_per_gpu': '33.91', 'tokens/total': 60272, 'tokens/trainable': 872, 'epoch': '0.007812'}

  0%|          | 2/512 [00:14<1:00:22,  7.10s/it]
  1%|          | 3/512 [00:21<58:23,  6.88s/it]  
                                               
{'loss': '0.1453', 'grad_norm': '1.401', 'learning_rate': '8e-06', 'ppl': '1.156', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '34.72', 'tokens/train_per_sec_per_gpu': '35.5', 'tokens/total': 90512, 'tokens/trainable': 1352, 'epoch': '0.01172'}

  1%|          | 3/512 [00:21<58:23,  6.88s/it]
  1%|          | 4/512 [00:27<57:22,  6.78s/it]
                                               
{'loss': '0.135', 'grad_norm': '0.9737', 'learning_rate': '1.2e-05', 'ppl': '1.145', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '34.73', 'tokens/train_per_sec_per_gpu': '31.75', 'tokens/total': 120944, 'tokens/trainable': 1777, 'epoch': '0.01562'}

  1%|          | 4/512 [00:27<57:22,  6.78s/it]
  1%|          | 5/512 [00:34<56:41,  6.71s/it]
                                               
{'loss': '0.1381', 'grad_norm': '1.554', 'learning_rate': '1.6e-05', 'ppl': '1.148', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.12', 'tokens/train_per_sec_per_gpu': '37.35', 'tokens/total': 151440, 'tokens/trainable': 2260, 'epoch': '0.01953'}

  1%|          | 5/512 [00:34<56:41,  6.71s/it]
  1%|          | 6/512 [00:40<56:25,  6.69s/it]
                                               
{'loss': '0.1262', 'grad_norm': '6.434', 'learning_rate': '2e-05', 'ppl': '1.135', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.12', 'tokens/train_per_sec_per_gpu': '33.21', 'tokens/total': 181984, 'tokens/trainable': 2704, 'epoch': '0.02344'}

  1%|          | 6/512 [00:40<56:25,  6.69s/it]
  1%|▏         | 7/512 [00:47<56:04,  6.66s/it]
                                               
{'loss': '0.112', 'grad_norm': '1.401', 'learning_rate': '2.4e-05', 'ppl': '1.119', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.12', 'tokens/train_per_sec_per_gpu': '34.37', 'tokens/total': 212336, 'tokens/trainable': 3143, 'epoch': '0.02734'}

  1%|▏         | 7/512 [00:47<56:04,  6.66s/it]
  2%|▏         | 8/512 [00:54<55:46,  6.64s/it]
                                               
{'loss': '0.0922', 'grad_norm': '2.039', 'learning_rate': '2.8e-05', 'ppl': '1.097', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.12', 'tokens/train_per_sec_per_gpu': '34.78', 'tokens/total': 242592, 'tokens/trainable': 3586, 'epoch': '0.03125'}

  2%|▏         | 8/512 [00:54<55:46,  6.64s/it]
  2%|▏         | 9/512 [01:00<55:26,  6.61s/it]
                                               
{'loss': '0.04823', 'grad_norm': '4.828', 'learning_rate': '3.2e-05', 'ppl': '1.049', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36.12', 'tokens/train_per_sec_per_gpu': '29.59', 'tokens/total': 272784, 'tokens/trainable': 4007, 'epoch': '0.03516'}

  2%|▏         | 9/512 [01:00<55:26,  6.61s/it]
  2%|▏         | 10/512 [01:07<55:15,  6.60s/it]
                                                
{'loss': '0.04049', 'grad_norm': '1.369', 'learning_rate': '3.6e-05', 'ppl': '1.041', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '39.66', 'tokens/total': 303184, 'tokens/trainable': 4481, 'epoch': '0.03906'}

  2%|▏         | 10/512 [01:07<55:15,  6.60s/it]
  2%|▏         | 11/512 [01:13<55:05,  6.60s/it]
                                                
{'loss': '0.0634', 'grad_norm': '2.285', 'learning_rate': '4e-05', 'ppl': '1.065', 'memory/max_active (GiB)': '33.73', 'memory/max_allocated (GiB)': '33.73', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '35.09', 'tokens/total': 333296, 'tokens/trainable': 4958, 'epoch': '0.04297'}

  2%|▏         | 11/512 [01:13<55:05,  6.60s/it]
  2%|▏         | 12/512 [01:20<55:00,  6.60s/it]
                                                
{'loss': '0.071', 'grad_norm': '5.053', 'learning_rate': '4.4e-05', 'ppl': '1.074', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '34.81', 'tokens/total': 363840, 'tokens/trainable': 5440, 'epoch': '0.04688'}

  2%|▏         | 12/512 [01:20<55:00,  6.60s/it]
  3%|β–Ž         | 13/512 [01:27<54:56,  6.61s/it]
                                                
{'loss': '0.06127', 'grad_norm': '3.501', 'learning_rate': '4.8e-05', 'ppl': '1.063', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '35.82', 'tokens/total': 394112, 'tokens/trainable': 5871, 'epoch': '0.05078'}

  3%|β–Ž         | 13/512 [01:27<54:56,  6.61s/it]
  3%|β–Ž         | 14/512 [01:33<54:58,  6.62s/it]
                                                
{'loss': '0.04367', 'grad_norm': '2.639', 'learning_rate': '5.2e-05', 'ppl': '1.045', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '35.29', 'tokens/total': 424656, 'tokens/trainable': 6360, 'epoch': '0.05469'}

  3%|β–Ž         | 14/512 [01:33<54:58,  6.62s/it]
  3%|β–Ž         | 15/512 [01:40<54:52,  6.62s/it]
                                                
{'loss': '0.06026', 'grad_norm': '2.628', 'learning_rate': '5.6e-05', 'ppl': '1.062', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '36.67', 'tokens/total': 455152, 'tokens/trainable': 6841, 'epoch': '0.05859'}

  3%|β–Ž         | 15/512 [01:40<54:52,  6.62s/it]
  3%|β–Ž         | 16/512 [01:46<54:31,  6.60s/it]
                                                
{'loss': '0.09122', 'grad_norm': '5.878', 'learning_rate': '6e-05', 'ppl': '1.096', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '31.7', 'tokens/total': 485328, 'tokens/trainable': 7254, 'epoch': '0.0625'}

  3%|β–Ž         | 16/512 [01:46<54:31,  6.60s/it]
  3%|β–Ž         | 17/512 [01:53<54:30,  6.61s/it]
                                                
{'loss': '0.01772', 'grad_norm': '1.972', 'learning_rate': '6.4e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '36.52', 'tokens/total': 515744, 'tokens/trainable': 7723, 'epoch': '0.06641'}

  3%|β–Ž         | 17/512 [01:53<54:30,  6.61s/it]
  4%|β–Ž         | 18/512 [02:00<54:31,  6.62s/it]
                                                
{'loss': '0.0474', 'grad_norm': '1.783', 'learning_rate': '6.8e-05', 'ppl': '1.049', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '34.38', 'tokens/total': 546144, 'tokens/trainable': 8208, 'epoch': '0.07031'}

  4%|β–Ž         | 18/512 [02:00<54:31,  6.62s/it]
  4%|β–Ž         | 19/512 [02:06<54:25,  6.62s/it]
                                                
{'loss': '0.02503', 'grad_norm': '1.916', 'learning_rate': '7.2e-05', 'ppl': '1.025', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '35.49', 'tokens/total': 576560, 'tokens/trainable': 8699, 'epoch': '0.07422'}

  4%|β–Ž         | 19/512 [02:06<54:25,  6.62s/it]
  4%|▍         | 20/512 [02:13<54:20,  6.63s/it]
                                                
{'loss': '0.05566', 'grad_norm': '2.979', 'learning_rate': '7.6e-05', 'ppl': '1.057', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '40.56', 'tokens/total': 607056, 'tokens/trainable': 9214, 'epoch': '0.07812'}

  4%|▍         | 20/512 [02:13<54:20,  6.63s/it]
  4%|▍         | 21/512 [02:20<54:14,  6.63s/it]
                                                
{'loss': '0.02437', 'grad_norm': '1.272', 'learning_rate': '8e-05', 'ppl': '1.025', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '32.12', 'tokens/total': 637488, 'tokens/trainable': 9639, 'epoch': '0.08203'}

  4%|▍         | 21/512 [02:20<54:14,  6.63s/it]
  4%|▍         | 22/512 [02:26<54:02,  6.62s/it]
                                                
{'loss': '0.0517', 'grad_norm': '2.242', 'learning_rate': '8.4e-05', 'ppl': '1.053', 'memory/max_active (GiB)': '33.75', 'memory/max_allocated (GiB)': '33.75', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '36.71', 'tokens/total': 667632, 'tokens/trainable': 10111, 'epoch': '0.08594'}

  4%|▍         | 22/512 [02:26<54:02,  6.62s/it]
  4%|▍         | 23/512 [02:33<53:56,  6.62s/it]
                                                
{'loss': '0.04347', 'grad_norm': '2.216', 'learning_rate': '8.8e-05', 'ppl': '1.044', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '33.4', 'tokens/total': 697872, 'tokens/trainable': 10560, 'epoch': '0.08984'}

  4%|▍         | 23/512 [02:33<53:56,  6.62s/it]
  5%|▍         | 24/512 [02:39<53:50,  6.62s/it]
                                                
{'loss': '0.01621', 'grad_norm': '1.028', 'learning_rate': '9.2e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '32', 'tokens/total': 728432, 'tokens/trainable': 11018, 'epoch': '0.09375'}

  5%|▍         | 24/512 [02:39<53:50,  6.62s/it]
  5%|▍         | 25/512 [02:46<52:27,  6.46s/it]
                                                
{'loss': '0.02351', 'grad_norm': '1.041', 'learning_rate': '9.6e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '37.81', 'tokens/train_per_sec_per_gpu': '40.53', 'tokens/total': 756656, 'tokens/trainable': 11442, 'epoch': '0.09766'}

  5%|▍         | 25/512 [02:46<52:27,  6.46s/it]
  5%|β–Œ         | 26/512 [02:52<52:37,  6.50s/it]
                                                
{'loss': '0.05569', 'grad_norm': '1.666', 'learning_rate': '0.0001', 'ppl': '1.057', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '34.83', 'tokens/total': 787120, 'tokens/trainable': 11899, 'epoch': '0.1016'}

  5%|β–Œ         | 26/512 [02:52<52:37,  6.50s/it]
  5%|β–Œ         | 27/512 [02:59<52:51,  6.54s/it]
                                                
{'loss': '0.009145', 'grad_norm': '0.7106', 'learning_rate': '0.0001', 'ppl': '1.009', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '32.1', 'tokens/total': 817504, 'tokens/trainable': 12341, 'epoch': '0.1055'}

  5%|β–Œ         | 27/512 [02:59<52:51,  6.54s/it]
  5%|β–Œ         | 28/512 [03:05<53:00,  6.57s/it]
                                                
{'loss': '0.02025', 'grad_norm': '0.8463', 'learning_rate': '0.0001', 'ppl': '1.02', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '38.39', 'tokens/total': 848128, 'tokens/trainable': 12825, 'epoch': '0.1094'}

  5%|β–Œ         | 28/512 [03:05<53:00,  6.57s/it]
  6%|β–Œ         | 29/512 [03:12<52:59,  6.58s/it]
                                                
{'loss': '0.02345', 'grad_norm': '1.074', 'learning_rate': '9.999e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '36.01', 'tokens/total': 878448, 'tokens/trainable': 13295, 'epoch': '0.1133'}

  6%|β–Œ         | 29/512 [03:12<52:59,  6.58s/it]
  6%|β–Œ         | 30/512 [03:19<53:02,  6.60s/it]
                                                
{'loss': '0.03459', 'grad_norm': '1.423', 'learning_rate': '9.999e-05', 'ppl': '1.035', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '35.8', 'tokens/total': 909008, 'tokens/trainable': 13784, 'epoch': '0.1172'}

  6%|β–Œ         | 30/512 [03:19<53:02,  6.60s/it]
  6%|β–Œ         | 31/512 [03:25<52:59,  6.61s/it]
                                                
{'loss': '0.0248', 'grad_norm': '0.988', 'learning_rate': '9.998e-05', 'ppl': '1.025', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '34.58', 'tokens/total': 939312, 'tokens/trainable': 14250, 'epoch': '0.1211'}

  6%|β–Œ         | 31/512 [03:25<52:59,  6.61s/it]
  6%|β–‹         | 32/512 [03:32<52:38,  6.58s/it]
                                                
{'loss': '0.02302', 'grad_norm': '1.116', 'learning_rate': '9.997e-05', 'ppl': '1.023', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '35', 'tokens/total': 969264, 'tokens/trainable': 14680, 'epoch': '0.125'}

  6%|β–‹         | 32/512 [03:32<52:38,  6.58s/it][2026-08-18 14:21:29,523] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-32

  6%|β–‹         | 33/512 [03:40<56:36,  7.09s/it]
                                                
{'loss': '0.0631', 'grad_norm': '1.325', 'learning_rate': '9.995e-05', 'ppl': '1.065', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '37.82', 'tokens/train_per_sec_per_gpu': '33.83', 'tokens/total': 999792, 'tokens/trainable': 15112, 'epoch': '0.1289'}

  6%|β–‹         | 33/512 [03:40<56:36,  7.09s/it]
  7%|β–‹         | 34/512 [03:47<55:08,  6.92s/it]
                                                
{'loss': '0.04253', 'grad_norm': '1.003', 'learning_rate': '9.994e-05', 'ppl': '1.043', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '35.48', 'tokens/train_per_sec_per_gpu': '34.46', 'tokens/total': 1029840, 'tokens/trainable': 15573, 'epoch': '0.1328'}

  7%|β–‹         | 34/512 [03:47<55:08,  6.92s/it]
  7%|β–‹         | 35/512 [03:53<54:24,  6.84s/it]
                                                
{'loss': '0.03612', 'grad_norm': '1.139', 'learning_rate': '9.992e-05', 'ppl': '1.037', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.48', 'tokens/train_per_sec_per_gpu': '34.24', 'tokens/total': 1060416, 'tokens/trainable': 16062, 'epoch': '0.1367'}

  7%|β–‹         | 35/512 [03:53<54:24,  6.84s/it]
  7%|β–‹         | 36/512 [04:00<53:33,  6.75s/it]
                                                
{'loss': '0.01517', 'grad_norm': '0.4866', 'learning_rate': '9.991e-05', 'ppl': '1.015', 'memory/max_active (GiB)': '33.72', 'memory/max_allocated (GiB)': '33.72', 'memory/device_reserved (GiB)': '35.48', 'tokens/train_per_sec_per_gpu': '31.53', 'tokens/total': 1090464, 'tokens/trainable': 16500, 'epoch': '0.1406'}

  7%|β–‹         | 36/512 [04:00<53:33,  6.75s/it]
  7%|β–‹         | 37/512 [04:06<53:00,  6.69s/it]
                                                
{'loss': '0.04621', 'grad_norm': '0.7631', 'learning_rate': '9.989e-05', 'ppl': '1.047', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.86', 'tokens/train_per_sec_per_gpu': '34.9', 'tokens/total': 1120944, 'tokens/trainable': 16936, 'epoch': '0.1445'}

  7%|β–‹         | 37/512 [04:06<53:00,  6.69s/it]
  7%|β–‹         | 38/512 [04:13<52:37,  6.66s/it]
                                                
{'loss': '0.046', 'grad_norm': '1.116', 'learning_rate': '9.987e-05', 'ppl': '1.047', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.86', 'tokens/train_per_sec_per_gpu': '29.32', 'tokens/total': 1151360, 'tokens/trainable': 17357, 'epoch': '0.1484'}

  7%|β–‹         | 38/512 [04:13<52:37,  6.66s/it]
  8%|β–Š         | 39/512 [04:20<52:30,  6.66s/it]
                                                
{'loss': '0.052', 'grad_norm': '4.36', 'learning_rate': '9.984e-05', 'ppl': '1.053', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.86', 'tokens/train_per_sec_per_gpu': '36.82', 'tokens/total': 1181728, 'tokens/trainable': 17831, 'epoch': '0.1523'}

  8%|β–Š         | 39/512 [04:20<52:30,  6.66s/it]
  8%|β–Š         | 40/512 [04:26<52:16,  6.65s/it]
                                                
{'loss': '0.04869', 'grad_norm': '0.8433', 'learning_rate': '9.982e-05', 'ppl': '1.05', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.86', 'tokens/train_per_sec_per_gpu': '36.15', 'tokens/total': 1212064, 'tokens/trainable': 18315, 'epoch': '0.1562'}

  8%|β–Š         | 40/512 [04:26<52:16,  6.65s/it]
  8%|β–Š         | 41/512 [04:33<52:02,  6.63s/it]
                                                
{'loss': '0.02915', 'grad_norm': '0.9528', 'learning_rate': '9.979e-05', 'ppl': '1.03', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.49', 'tokens/total': 1242592, 'tokens/trainable': 18772, 'epoch': '0.1602'}

  8%|β–Š         | 41/512 [04:33<52:02,  6.63s/it]
  8%|β–Š         | 42/512 [04:39<51:50,  6.62s/it]
                                                
{'loss': '0.01587', 'grad_norm': '0.5701', 'learning_rate': '9.976e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.32', 'tokens/total': 1272800, 'tokens/trainable': 19197, 'epoch': '0.1641'}

  8%|β–Š         | 42/512 [04:39<51:50,  6.62s/it]
  8%|β–Š         | 43/512 [04:46<51:54,  6.64s/it]
                                                
{'loss': '0.01053', 'grad_norm': '0.373', 'learning_rate': '9.973e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '34.04', 'memory/max_allocated (GiB)': '34.04', 'memory/device_reserved (GiB)': '36.04', 'tokens/train_per_sec_per_gpu': '31.68', 'tokens/total': 1303600, 'tokens/trainable': 19647, 'epoch': '0.168'}

  8%|β–Š         | 43/512 [04:46<51:54,  6.64s/it]
  9%|β–Š         | 44/512 [04:53<51:41,  6.63s/it]
                                                
{'loss': '0.01996', 'grad_norm': '0.5503', 'learning_rate': '9.97e-05', 'ppl': '1.02', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36.04', 'tokens/train_per_sec_per_gpu': '33.83', 'tokens/total': 1333936, 'tokens/trainable': 20097, 'epoch': '0.1719'}

  9%|β–Š         | 44/512 [04:53<51:41,  6.63s/it]
  9%|β–‰         | 45/512 [04:59<51:32,  6.62s/it]
                                                
{'loss': '0.06465', 'grad_norm': '3.762', 'learning_rate': '9.966e-05', 'ppl': '1.067', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.04', 'tokens/train_per_sec_per_gpu': '31.92', 'tokens/total': 1364288, 'tokens/trainable': 20529, 'epoch': '0.1758'}

  9%|β–‰         | 45/512 [04:59<51:32,  6.62s/it]
  9%|β–‰         | 46/512 [05:06<51:23,  6.62s/it]
                                                
{'loss': '0.05519', 'grad_norm': '1.149', 'learning_rate': '9.963e-05', 'ppl': '1.057', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.04', 'tokens/train_per_sec_per_gpu': '34.68', 'tokens/total': 1394432, 'tokens/trainable': 20973, 'epoch': '0.1797'}

  9%|β–‰         | 46/512 [05:06<51:23,  6.62s/it]
  9%|β–‰         | 47/512 [05:13<51:25,  6.63s/it]
                                                
{'loss': '0.02402', 'grad_norm': '0.5584', 'learning_rate': '9.959e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '35.36', 'tokens/total': 1425088, 'tokens/trainable': 21437, 'epoch': '0.1836'}

  9%|β–‰         | 47/512 [05:13<51:25,  6.63s/it]
  9%|β–‰         | 48/512 [05:19<51:11,  6.62s/it]
                                                
{'loss': '0.02145', 'grad_norm': '0.7726', 'learning_rate': '9.955e-05', 'ppl': '1.022', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '35.57', 'tokens/total': 1455456, 'tokens/trainable': 21860, 'epoch': '0.1875'}

  9%|β–‰         | 48/512 [05:19<51:11,  6.62s/it]
 10%|β–‰         | 49/512 [05:26<50:45,  6.58s/it]
                                                
{'loss': '0.03308', 'grad_norm': '0.9507', 'learning_rate': '9.951e-05', 'ppl': '1.034', 'memory/max_active (GiB)': '33.7', 'memory/max_allocated (GiB)': '33.7', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '30.1', 'tokens/total': 1485376, 'tokens/trainable': 22273, 'epoch': '0.1914'}

 10%|β–‰         | 49/512 [05:26<50:45,  6.58s/it]
 10%|β–‰         | 50/512 [05:32<50:41,  6.58s/it]
                                                
{'loss': '0.01948', 'grad_norm': '0.5201', 'learning_rate': '9.946e-05', 'ppl': '1.02', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '36.83', 'tokens/total': 1515744, 'tokens/trainable': 22720, 'epoch': '0.1953'}

 10%|β–‰         | 50/512 [05:32<50:41,  6.58s/it]
 10%|β–‰         | 51/512 [05:39<50:29,  6.57s/it]
                                                
{'loss': '0.02124', 'grad_norm': '0.5437', 'learning_rate': '9.942e-05', 'ppl': '1.021', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '32.55', 'tokens/total': 1545984, 'tokens/trainable': 23168, 'epoch': '0.1992'}

 10%|β–‰         | 51/512 [05:39<50:29,  6.57s/it]
 10%|β–ˆ         | 52/512 [05:45<50:30,  6.59s/it]
                                                
{'loss': '0.02224', 'grad_norm': '0.4705', 'learning_rate': '9.937e-05', 'ppl': '1.022', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '37.44', 'tokens/total': 1576288, 'tokens/trainable': 23663, 'epoch': '0.2031'}

 10%|β–ˆ         | 52/512 [05:45<50:30,  6.59s/it]
 10%|β–ˆ         | 53/512 [05:52<50:31,  6.60s/it]
                                                
{'loss': '0.01991', 'grad_norm': '12.06', 'learning_rate': '9.932e-05', 'ppl': '1.02', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '34.49', 'tokens/total': 1606528, 'tokens/trainable': 24128, 'epoch': '0.207'}

 10%|β–ˆ         | 53/512 [05:52<50:31,  6.60s/it]
 11%|β–ˆ         | 54/512 [05:59<50:17,  6.59s/it]
                                                
{'loss': '0.03577', 'grad_norm': '0.7315', 'learning_rate': '9.927e-05', 'ppl': '1.036', 'memory/max_active (GiB)': '33.66', 'memory/max_allocated (GiB)': '33.66', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '38.33', 'tokens/total': 1636400, 'tokens/trainable': 24601, 'epoch': '0.2109'}

 11%|β–ˆ         | 54/512 [05:59<50:17,  6.59s/it]
 11%|β–ˆ         | 55/512 [06:05<50:19,  6.61s/it]
                                                
{'loss': '0.04387', 'grad_norm': '0.8308', 'learning_rate': '9.921e-05', 'ppl': '1.045', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '34.44', 'tokens/total': 1666896, 'tokens/trainable': 25099, 'epoch': '0.2148'}

 11%|β–ˆ         | 55/512 [06:05<50:19,  6.61s/it]
 11%|β–ˆ         | 56/512 [06:12<50:13,  6.61s/it]
                                                
{'loss': '0.009335', 'grad_norm': '0.361', 'learning_rate': '9.916e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '37.08', 'tokens/total': 1697232, 'tokens/trainable': 25573, 'epoch': '0.2188'}

 11%|β–ˆ         | 56/512 [06:12<50:13,  6.61s/it]
 11%|β–ˆ         | 57/512 [06:18<50:04,  6.60s/it]
                                                
{'loss': '0.01627', 'grad_norm': '0.5462', 'learning_rate': '9.91e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '33.69', 'tokens/total': 1727440, 'tokens/trainable': 25997, 'epoch': '0.2227'}

 11%|β–ˆ         | 57/512 [06:18<50:04,  6.60s/it]
 11%|β–ˆβ–        | 58/512 [06:25<49:54,  6.59s/it]
                                                
{'loss': '0.01592', 'grad_norm': '3.278', 'learning_rate': '9.904e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '34.37', 'tokens/total': 1757952, 'tokens/trainable': 26441, 'epoch': '0.2266'}

 11%|β–ˆβ–        | 58/512 [06:25<49:54,  6.59s/it]
 12%|β–ˆβ–        | 59/512 [06:32<49:52,  6.61s/it]
                                                
{'loss': '0.01648', 'grad_norm': '1.643', 'learning_rate': '9.898e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '34.54', 'tokens/total': 1788576, 'tokens/trainable': 26915, 'epoch': '0.2305'}

 12%|β–ˆβ–        | 59/512 [06:32<49:52,  6.61s/it]
 12%|β–ˆβ–        | 60/512 [06:38<49:46,  6.61s/it]
                                                
{'loss': '0.01031', 'grad_norm': '0.5992', 'learning_rate': '9.892e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '34.2', 'tokens/total': 1818816, 'tokens/trainable': 27355, 'epoch': '0.2344'}

 12%|β–ˆβ–        | 60/512 [06:38<49:46,  6.61s/it]
 12%|β–ˆβ–        | 61/512 [06:45<49:42,  6.61s/it]
                                                
{'loss': '0.002149', 'grad_norm': '0.1208', 'learning_rate': '9.886e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '39.71', 'tokens/total': 1849424, 'tokens/trainable': 27862, 'epoch': '0.2383'}

 12%|β–ˆβ–        | 61/512 [06:45<49:42,  6.61s/it]
 12%|β–ˆβ–        | 62/512 [06:52<49:48,  6.64s/it]
                                                
{'loss': '0.04498', 'grad_norm': '0.8363', 'learning_rate': '9.879e-05', 'ppl': '1.046', 'memory/max_active (GiB)': '34.02', 'memory/max_allocated (GiB)': '34.02', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '32.8', 'tokens/total': 1880032, 'tokens/trainable': 28313, 'epoch': '0.2422'}

 12%|β–ˆβ–        | 62/512 [06:52<49:48,  6.64s/it]
 12%|β–ˆβ–        | 63/512 [06:58<49:37,  6.63s/it]
                                                
{'loss': '0.04683', 'grad_norm': '1.207', 'learning_rate': '9.872e-05', 'ppl': '1.048', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '32.7', 'tokens/total': 1910512, 'tokens/trainable': 28774, 'epoch': '0.2461'}

 12%|β–ˆβ–        | 63/512 [06:58<49:37,  6.63s/it]
 12%|β–ˆβ–Ž        | 64/512 [07:05<49:19,  6.61s/it]
                                                
{'loss': '0.02833', 'grad_norm': '0.8954', 'learning_rate': '9.865e-05', 'ppl': '1.029', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '34.07', 'tokens/total': 1940608, 'tokens/trainable': 29222, 'epoch': '0.25'}

 12%|β–ˆβ–Ž        | 64/512 [07:05<49:19,  6.61s/it][2026-08-18 14:25:02,486] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-64

 13%|β–ˆβ–Ž        | 65/512 [07:13<52:21,  7.03s/it]
                                                
{'loss': '0.01838', 'grad_norm': '0.7108', 'learning_rate': '9.858e-05', 'ppl': '1.019', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '36.16', 'tokens/train_per_sec_per_gpu': '32', 'tokens/total': 1971104, 'tokens/trainable': 29687, 'epoch': '0.2539'}

 13%|β–ˆβ–Ž        | 65/512 [07:13<52:21,  7.03s/it]
 13%|β–ˆβ–Ž        | 66/512 [07:19<51:13,  6.89s/it]
                                                
{'loss': '0.02454', 'grad_norm': '0.8418', 'learning_rate': '9.851e-05', 'ppl': '1.025', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.06', 'tokens/train_per_sec_per_gpu': '33.93', 'tokens/total': 2001472, 'tokens/trainable': 30114, 'epoch': '0.2578'}

 13%|β–ˆβ–Ž        | 66/512 [07:19<51:13,  6.89s/it]
 13%|β–ˆβ–Ž        | 67/512 [07:25<49:28,  6.67s/it]
                                                
{'loss': '0.0327', 'grad_norm': '0.5852', 'learning_rate': '9.844e-05', 'ppl': '1.033', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.65', 'tokens/train_per_sec_per_gpu': '38.04', 'tokens/total': 2029696, 'tokens/trainable': 30594, 'epoch': '0.2617'}

 13%|β–ˆβ–Ž        | 67/512 [07:25<49:28,  6.67s/it]
 13%|β–ˆβ–Ž        | 68/512 [07:32<49:15,  6.66s/it]
                                                
{'loss': '0.01145', 'grad_norm': '0.3638', 'learning_rate': '9.836e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.65', 'tokens/train_per_sec_per_gpu': '35.62', 'tokens/total': 2060144, 'tokens/trainable': 31073, 'epoch': '0.2656'}

 13%|β–ˆβ–Ž        | 68/512 [07:32<49:15,  6.66s/it]
 13%|β–ˆβ–Ž        | 69/512 [07:39<49:06,  6.65s/it]
                                                
{'loss': '0.03206', 'grad_norm': '0.4834', 'learning_rate': '9.828e-05', 'ppl': '1.033', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.65', 'tokens/train_per_sec_per_gpu': '34.38', 'tokens/total': 2090448, 'tokens/trainable': 31553, 'epoch': '0.2695'}

 13%|β–ˆβ–Ž        | 69/512 [07:39<49:06,  6.65s/it]
 14%|β–ˆβ–Ž        | 70/512 [07:45<48:49,  6.63s/it]
                                                
{'loss': '0.01643', 'grad_norm': '0.58', 'learning_rate': '9.82e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '35.65', 'tokens/train_per_sec_per_gpu': '32.72', 'tokens/total': 2120608, 'tokens/trainable': 32004, 'epoch': '0.2734'}

 14%|β–ˆβ–Ž        | 70/512 [07:45<48:49,  6.63s/it]
 14%|β–ˆβ–        | 71/512 [07:51<47:29,  6.46s/it]
                                                
{'loss': '0.04277', 'grad_norm': '0.6351', 'learning_rate': '9.812e-05', 'ppl': '1.044', 'memory/max_active (GiB)': '33.35', 'memory/max_allocated (GiB)': '33.35', 'memory/device_reserved (GiB)': '35.65', 'tokens/train_per_sec_per_gpu': '32.26', 'tokens/total': 2148848, 'tokens/trainable': 32419, 'epoch': '0.2773'}

 14%|β–ˆβ–        | 71/512 [07:51<47:29,  6.46s/it]
 14%|β–ˆβ–        | 72/512 [07:58<47:42,  6.51s/it]
                                                
{'loss': '0.009183', 'grad_norm': '0.2791', 'learning_rate': '9.803e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.65', 'tokens/train_per_sec_per_gpu': '34.98', 'tokens/total': 2179200, 'tokens/trainable': 32878, 'epoch': '0.2812'}

 14%|β–ˆβ–        | 72/512 [07:58<47:42,  6.51s/it]
 14%|β–ˆβ–        | 73/512 [08:05<47:44,  6.53s/it]
                                                
{'loss': '0.0283', 'grad_norm': '0.7224', 'learning_rate': '9.795e-05', 'ppl': '1.029', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '36.21', 'tokens/total': 2209712, 'tokens/trainable': 33348, 'epoch': '0.2852'}

 14%|β–ˆβ–        | 73/512 [08:05<47:44,  6.53s/it]
 14%|β–ˆβ–        | 74/512 [08:11<48:00,  6.58s/it]
                                                
{'loss': '0.01015', 'grad_norm': '0.6147', 'learning_rate': '9.786e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '32.86', 'tokens/total': 2240240, 'tokens/trainable': 33815, 'epoch': '0.2891'}

 14%|β–ˆβ–        | 74/512 [08:11<48:00,  6.58s/it]
 15%|β–ˆβ–        | 75/512 [08:18<47:53,  6.58s/it]
                                                
{'loss': '0.01202', 'grad_norm': '0.2888', 'learning_rate': '9.777e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '34.86', 'tokens/total': 2270336, 'tokens/trainable': 34229, 'epoch': '0.293'}

 15%|β–ˆβ–        | 75/512 [08:18<47:53,  6.58s/it]
 15%|β–ˆβ–        | 76/512 [08:24<47:56,  6.60s/it]
                                                
{'loss': '0.02341', 'grad_norm': '0.4517', 'learning_rate': '9.768e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '36.53', 'tokens/total': 2300784, 'tokens/trainable': 34718, 'epoch': '0.2969'}

 15%|β–ˆβ–        | 76/512 [08:24<47:56,  6.60s/it]
 15%|β–ˆβ–Œ        | 77/512 [08:31<47:46,  6.59s/it]
                                                
{'loss': '0.05283', 'grad_norm': '0.9155', 'learning_rate': '9.759e-05', 'ppl': '1.054', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '32.14', 'tokens/total': 2330880, 'tokens/trainable': 35146, 'epoch': '0.3008'}

 15%|β–ˆβ–Œ        | 77/512 [08:31<47:46,  6.59s/it]
 15%|β–ˆβ–Œ        | 78/512 [08:38<47:52,  6.62s/it]
                                                
{'loss': '0.009073', 'grad_norm': '0.3222', 'learning_rate': '9.749e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.99', 'memory/max_allocated (GiB)': '33.99', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '35.29', 'tokens/total': 2361392, 'tokens/trainable': 35610, 'epoch': '0.3047'}

 15%|β–ˆβ–Œ        | 78/512 [08:38<47:52,  6.62s/it]
 15%|β–ˆβ–Œ        | 79/512 [08:44<47:40,  6.61s/it]
                                                
{'loss': '0.02443', 'grad_norm': '0.4267', 'learning_rate': '9.74e-05', 'ppl': '1.025', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '29.95', 'tokens/total': 2391744, 'tokens/trainable': 36037, 'epoch': '0.3086'}

 15%|β–ˆβ–Œ        | 79/512 [08:44<47:40,  6.61s/it]
 16%|β–ˆβ–Œ        | 80/512 [08:51<47:38,  6.62s/it]
                                                
{'loss': '0.01513', 'grad_norm': '0.3153', 'learning_rate': '9.73e-05', 'ppl': '1.015', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '38.73', 'tokens/total': 2422240, 'tokens/trainable': 36522, 'epoch': '0.3125'}

 16%|β–ˆβ–Œ        | 80/512 [08:51<47:38,  6.62s/it]
 16%|β–ˆβ–Œ        | 81/512 [08:58<47:28,  6.61s/it]
                                                
{'loss': '0.02345', 'grad_norm': '0.6696', 'learning_rate': '9.72e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.67', 'tokens/total': 2452560, 'tokens/trainable': 36979, 'epoch': '0.3164'}

 16%|β–ˆβ–Œ        | 81/512 [08:58<47:28,  6.61s/it]
 16%|β–ˆβ–Œ        | 82/512 [09:04<46:20,  6.47s/it]
                                                
{'loss': '0.0115', 'grad_norm': '0.3085', 'learning_rate': '9.71e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.44', 'memory/max_allocated (GiB)': '33.44', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.93', 'tokens/total': 2481024, 'tokens/trainable': 37439, 'epoch': '0.3203'}

 16%|β–ˆβ–Œ        | 82/512 [09:04<46:20,  6.47s/it]
 16%|β–ˆβ–Œ        | 83/512 [09:10<46:33,  6.51s/it]
                                                
{'loss': '0.01644', 'grad_norm': '0.4444', 'learning_rate': '9.699e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.54', 'tokens/total': 2511296, 'tokens/trainable': 37892, 'epoch': '0.3242'}

 16%|β–ˆβ–Œ        | 83/512 [09:10<46:33,  6.51s/it]
 16%|β–ˆβ–‹        | 84/512 [09:17<46:42,  6.55s/it]
                                                
{'loss': '0.03258', 'grad_norm': '1.124', 'learning_rate': '9.689e-05', 'ppl': '1.033', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.19', 'tokens/total': 2541792, 'tokens/trainable': 38355, 'epoch': '0.3281'}

 16%|β–ˆβ–‹        | 84/512 [09:17<46:42,  6.55s/it]
 17%|β–ˆβ–‹        | 85/512 [09:24<46:43,  6.57s/it]
                                                
{'loss': '0.01193', 'grad_norm': '0.4937', 'learning_rate': '9.678e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.69', 'memory/max_allocated (GiB)': '33.69', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '34.07', 'tokens/total': 2572032, 'tokens/trainable': 38810, 'epoch': '0.332'}

 17%|β–ˆβ–‹        | 85/512 [09:24<46:43,  6.57s/it]
 17%|β–ˆβ–‹        | 86/512 [09:30<46:41,  6.58s/it]
                                                
{'loss': '0.04201', 'grad_norm': '1.324', 'learning_rate': '9.667e-05', 'ppl': '1.043', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '30.89', 'tokens/total': 2602512, 'tokens/trainable': 39240, 'epoch': '0.3359'}

 17%|β–ˆβ–‹        | 86/512 [09:30<46:41,  6.58s/it]
 17%|β–ˆβ–‹        | 87/512 [09:37<46:41,  6.59s/it]
                                                
{'loss': '0.02861', 'grad_norm': '0.4353', 'learning_rate': '9.656e-05', 'ppl': '1.029', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '37', 'tokens/total': 2632880, 'tokens/trainable': 39748, 'epoch': '0.3398'}

 17%|β–ˆβ–‹        | 87/512 [09:37<46:41,  6.59s/it]
 17%|β–ˆβ–‹        | 88/512 [09:43<46:24,  6.57s/it]
                                                
{'loss': '0.036', 'grad_norm': '0.4899', 'learning_rate': '9.645e-05', 'ppl': '1.037', 'memory/max_active (GiB)': '33.73', 'memory/max_allocated (GiB)': '33.73', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.35', 'tokens/total': 2663040, 'tokens/trainable': 40188, 'epoch': '0.3438'}

 17%|β–ˆβ–‹        | 88/512 [09:43<46:24,  6.57s/it]
 17%|β–ˆβ–‹        | 89/512 [09:50<46:17,  6.57s/it]
                                                
{'loss': '0.02382', 'grad_norm': '0.4914', 'learning_rate': '9.633e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.36', 'tokens/total': 2693504, 'tokens/trainable': 40632, 'epoch': '0.3477'}

 17%|β–ˆβ–‹        | 89/512 [09:50<46:17,  6.57s/it]
 18%|β–ˆβ–Š        | 90/512 [09:56<46:11,  6.57s/it]
                                                
{'loss': '0.02294', 'grad_norm': '0.4422', 'learning_rate': '9.622e-05', 'ppl': '1.023', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '31.37', 'tokens/total': 2723792, 'tokens/trainable': 41050, 'epoch': '0.3516'}

 18%|β–ˆβ–Š        | 90/512 [09:56<46:11,  6.57s/it]
 18%|β–ˆβ–Š        | 91/512 [10:03<46:10,  6.58s/it]
                                                
{'loss': '0.01759', 'grad_norm': '0.4762', 'learning_rate': '9.61e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '37.05', 'tokens/total': 2754048, 'tokens/trainable': 41515, 'epoch': '0.3555'}

 18%|β–ˆβ–Š        | 91/512 [10:03<46:10,  6.58s/it]
 18%|β–ˆβ–Š        | 92/512 [10:10<46:02,  6.58s/it]
                                                
{'loss': '0.03154', 'grad_norm': '0.4137', 'learning_rate': '9.598e-05', 'ppl': '1.032', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '36.23', 'tokens/total': 2784480, 'tokens/trainable': 41972, 'epoch': '0.3594'}

 18%|β–ˆβ–Š        | 92/512 [10:10<46:02,  6.58s/it]
 18%|β–ˆβ–Š        | 93/512 [10:16<45:59,  6.59s/it]
                                                
{'loss': '0.02375', 'grad_norm': '0.527', 'learning_rate': '9.586e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '35.43', 'tokens/total': 2814928, 'tokens/trainable': 42442, 'epoch': '0.3633'}

 18%|β–ˆβ–Š        | 93/512 [10:16<45:59,  6.59s/it]
 18%|β–ˆβ–Š        | 94/512 [10:23<45:57,  6.60s/it]
                                                
{'loss': '0.01261', 'grad_norm': '0.3022', 'learning_rate': '9.574e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '37.18', 'tokens/total': 2845408, 'tokens/trainable': 42928, 'epoch': '0.3672'}

 18%|β–ˆβ–Š        | 94/512 [10:23<45:57,  6.60s/it]
 19%|β–ˆβ–Š        | 95/512 [10:29<45:56,  6.61s/it]
                                                
{'loss': '0.02136', 'grad_norm': '0.3438', 'learning_rate': '9.562e-05', 'ppl': '1.022', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '39.28', 'tokens/total': 2875760, 'tokens/trainable': 43426, 'epoch': '0.3711'}

 19%|β–ˆβ–Š        | 95/512 [10:29<45:56,  6.61s/it]
 19%|β–ˆβ–‰        | 96/512 [10:36<45:51,  6.62s/it]
                                                
{'loss': '0.01179', 'grad_norm': '0.5729', 'learning_rate': '9.549e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.52', 'tokens/total': 2906096, 'tokens/trainable': 43869, 'epoch': '0.375'}

 19%|β–ˆβ–‰        | 96/512 [10:36<45:51,  6.62s/it][2026-08-18 14:28:33,847] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-96

 19%|β–ˆβ–‰        | 97/512 [10:44<48:45,  7.05s/it]
                                                
{'loss': '0.002208', 'grad_norm': '0.1258', 'learning_rate': '9.536e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '32.02', 'tokens/total': 2936560, 'tokens/trainable': 44300, 'epoch': '0.3789'}

 19%|β–ˆβ–‰        | 97/512 [10:44<48:45,  7.05s/it]
 19%|β–ˆβ–‰        | 98/512 [10:51<47:43,  6.92s/it]
                                                
{'loss': '0.0156', 'grad_norm': '0.4312', 'learning_rate': '9.523e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.19', 'tokens/train_per_sec_per_gpu': '39.37', 'tokens/total': 2966880, 'tokens/trainable': 44791, 'epoch': '0.3828'}

 19%|β–ˆβ–‰        | 98/512 [10:51<47:43,  6.92s/it]
 19%|β–ˆβ–‰        | 99/512 [10:57<47:09,  6.85s/it]
                                                
{'loss': '0.01187', 'grad_norm': '0.5057', 'learning_rate': '9.51e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '34.03', 'memory/max_allocated (GiB)': '34.03', 'memory/device_reserved (GiB)': '35.76', 'tokens/train_per_sec_per_gpu': '36.71', 'tokens/total': 2997472, 'tokens/trainable': 45254, 'epoch': '0.3867'}

 19%|β–ˆβ–‰        | 99/512 [10:57<47:09,  6.85s/it]
 20%|β–ˆβ–‰        | 100/512 [11:04<46:30,  6.77s/it]
                                                 
{'loss': '0.01458', 'grad_norm': '3.06', 'learning_rate': '9.497e-05', 'ppl': '1.015', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.57', 'tokens/total': 3027840, 'tokens/trainable': 45698, 'epoch': '0.3906'}

 20%|β–ˆβ–‰        | 100/512 [11:04<46:30,  6.77s/it]
 20%|β–ˆβ–‰        | 101/512 [11:11<46:03,  6.72s/it]
                                                 
{'loss': '0.009431', 'grad_norm': '0.3825', 'learning_rate': '9.484e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.57', 'tokens/total': 3058080, 'tokens/trainable': 46141, 'epoch': '0.3945'}

 20%|β–ˆβ–‰        | 101/512 [11:11<46:03,  6.72s/it]
 20%|β–ˆβ–‰        | 102/512 [11:17<45:45,  6.70s/it]
                                                 
{'loss': '0.05128', 'grad_norm': '0.8783', 'learning_rate': '9.47e-05', 'ppl': '1.053', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.05', 'tokens/total': 3088656, 'tokens/trainable': 46598, 'epoch': '0.3984'}

 20%|β–ˆβ–‰        | 102/512 [11:17<45:45,  6.70s/it]
 20%|β–ˆβ–ˆ        | 103/512 [11:24<45:26,  6.67s/it]
                                                 
{'loss': '0.009158', 'grad_norm': '0.4222', 'learning_rate': '9.456e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '32.45', 'tokens/total': 3119232, 'tokens/trainable': 47065, 'epoch': '0.4023'}

 20%|β–ˆβ–ˆ        | 103/512 [11:24<45:26,  6.67s/it]
 20%|β–ˆβ–ˆ        | 104/512 [11:30<45:12,  6.65s/it]
                                                 
{'loss': '0.01385', 'grad_norm': '1.263', 'learning_rate': '9.442e-05', 'ppl': '1.014', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '38.29', 'tokens/total': 3149504, 'tokens/trainable': 47554, 'epoch': '0.4062'}

 20%|β–ˆβ–ˆ        | 104/512 [11:30<45:12,  6.65s/it]
 21%|β–ˆβ–ˆ        | 105/512 [11:37<45:02,  6.64s/it]
                                                 
{'loss': '0.008239', 'grad_norm': '0.3073', 'learning_rate': '9.428e-05', 'ppl': '1.008', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '36.88', 'tokens/total': 3179872, 'tokens/trainable': 48047, 'epoch': '0.4102'}

 21%|β–ˆβ–ˆ        | 105/512 [11:37<45:02,  6.64s/it]
 21%|β–ˆβ–ˆ        | 106/512 [11:43<43:49,  6.48s/it]
                                                 
{'loss': '0.03211', 'grad_norm': '0.7416', 'learning_rate': '9.414e-05', 'ppl': '1.033', 'memory/max_active (GiB)': '33.39', 'memory/max_allocated (GiB)': '33.39', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '34.62', 'tokens/total': 3208128, 'tokens/trainable': 48498, 'epoch': '0.4141'}

 21%|β–ˆβ–ˆ        | 106/512 [11:43<43:49,  6.48s/it]
 21%|β–ˆβ–ˆ        | 107/512 [11:50<43:55,  6.51s/it]
                                                 
{'loss': '0.009182', 'grad_norm': '0.2502', 'learning_rate': '9.4e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '31.18', 'tokens/total': 3238224, 'tokens/trainable': 48933, 'epoch': '0.418'}

 21%|β–ˆβ–ˆ        | 107/512 [11:50<43:55,  6.51s/it]
 21%|β–ˆβ–ˆ        | 108/512 [11:56<44:00,  6.54s/it]
                                                 
{'loss': '0.01505', 'grad_norm': '0.5458', 'learning_rate': '9.385e-05', 'ppl': '1.015', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '38.48', 'tokens/total': 3268592, 'tokens/trainable': 49404, 'epoch': '0.4219'}

 21%|β–ˆβ–ˆ        | 108/512 [11:56<44:00,  6.54s/it]
 21%|β–ˆβ–ˆβ–       | 109/512 [12:03<44:03,  6.56s/it]
                                                 
{'loss': '0.007351', 'grad_norm': '0.2509', 'learning_rate': '9.37e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '37.32', 'tokens/total': 3298720, 'tokens/trainable': 49871, 'epoch': '0.4258'}

 21%|β–ˆβ–ˆβ–       | 109/512 [12:03<44:03,  6.56s/it]
 21%|β–ˆβ–ˆβ–       | 110/512 [12:10<44:03,  6.58s/it]
                                                 
{'loss': '0.01421', 'grad_norm': '0.2964', 'learning_rate': '9.355e-05', 'ppl': '1.014', 'memory/max_active (GiB)': '33.99', 'memory/max_allocated (GiB)': '33.99', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.58', 'tokens/total': 3329040, 'tokens/trainable': 50311, 'epoch': '0.4297'}

 21%|β–ˆβ–ˆβ–       | 110/512 [12:10<44:03,  6.58s/it]
 22%|β–ˆβ–ˆβ–       | 111/512 [12:16<44:01,  6.59s/it]
                                                 
{'loss': '0.0107', 'grad_norm': '0.3097', 'learning_rate': '9.34e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.86', 'tokens/total': 3359248, 'tokens/trainable': 50754, 'epoch': '0.4336'}

 22%|β–ˆβ–ˆβ–       | 111/512 [12:16<44:01,  6.59s/it]
 22%|β–ˆβ–ˆβ–       | 112/512 [12:23<43:53,  6.58s/it]
                                                 
{'loss': '0.01319', 'grad_norm': '0.4973', 'learning_rate': '9.325e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '32.1', 'tokens/total': 3389280, 'tokens/trainable': 51200, 'epoch': '0.4375'}

 22%|β–ˆβ–ˆβ–       | 112/512 [12:23<43:53,  6.58s/it]
 22%|β–ˆβ–ˆβ–       | 113/512 [12:29<43:51,  6.60s/it]
                                                 
{'loss': '0.006971', 'grad_norm': '0.4105', 'learning_rate': '9.31e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.22', 'tokens/total': 3419728, 'tokens/trainable': 51656, 'epoch': '0.4414'}

 22%|β–ˆβ–ˆβ–       | 113/512 [12:29<43:51,  6.60s/it]
 22%|β–ˆβ–ˆβ–       | 114/512 [12:36<43:41,  6.59s/it]
                                                 
{'loss': '0.003387', 'grad_norm': '0.2707', 'learning_rate': '9.294e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '30.92', 'tokens/total': 3449984, 'tokens/trainable': 52069, 'epoch': '0.4453'}

 22%|β–ˆβ–ˆβ–       | 114/512 [12:36<43:41,  6.59s/it]
 22%|β–ˆβ–ˆβ–       | 115/512 [12:43<43:37,  6.59s/it]
                                                 
{'loss': '0.008487', 'grad_norm': '0.5602', 'learning_rate': '9.278e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '36.94', 'tokens/total': 3480352, 'tokens/trainable': 52559, 'epoch': '0.4492'}

 22%|β–ˆβ–ˆβ–       | 115/512 [12:43<43:37,  6.59s/it]
 23%|β–ˆβ–ˆβ–Ž       | 116/512 [12:49<43:35,  6.60s/it]
                                                 
{'loss': '0.02243', 'grad_norm': '0.5224', 'learning_rate': '9.263e-05', 'ppl': '1.023', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '39.22', 'tokens/total': 3510656, 'tokens/trainable': 53061, 'epoch': '0.4531'}

 23%|β–ˆβ–ˆβ–Ž       | 116/512 [12:49<43:35,  6.60s/it]
 23%|β–ˆβ–ˆβ–Ž       | 117/512 [12:56<43:28,  6.60s/it]
                                                 
{'loss': '0.02046', 'grad_norm': '1.221', 'learning_rate': '9.247e-05', 'ppl': '1.021', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '35.89', 'tokens/total': 3541104, 'tokens/trainable': 53520, 'epoch': '0.457'}

 23%|β–ˆβ–ˆβ–Ž       | 117/512 [12:56<43:28,  6.60s/it]
 23%|β–ˆβ–ˆβ–Ž       | 118/512 [13:02<43:24,  6.61s/it]
                                                 
{'loss': '0.0256', 'grad_norm': '1.301', 'learning_rate': '9.23e-05', 'ppl': '1.026', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.36', 'tokens/total': 3571744, 'tokens/trainable': 53976, 'epoch': '0.4609'}

 23%|β–ˆβ–ˆβ–Ž       | 118/512 [13:02<43:24,  6.61s/it]
 23%|β–ˆβ–ˆβ–Ž       | 119/512 [13:09<43:11,  6.60s/it]
                                                 
{'loss': '0.02888', 'grad_norm': '1.099', 'learning_rate': '9.214e-05', 'ppl': '1.029', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '34.93', 'tokens/total': 3601808, 'tokens/trainable': 54408, 'epoch': '0.4648'}

 23%|β–ˆβ–ˆβ–Ž       | 119/512 [13:09<43:11,  6.60s/it]
 23%|β–ˆβ–ˆβ–Ž       | 120/512 [13:16<43:05,  6.60s/it]
                                                 
{'loss': '0.01679', 'grad_norm': '0.3977', 'learning_rate': '9.198e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '37.14', 'tokens/total': 3632096, 'tokens/trainable': 54884, 'epoch': '0.4688'}

 23%|β–ˆβ–ˆβ–Ž       | 120/512 [13:16<43:05,  6.60s/it]
 24%|β–ˆβ–ˆβ–Ž       | 121/512 [13:22<42:57,  6.59s/it]
                                                 
{'loss': '0.02264', 'grad_norm': '0.4873', 'learning_rate': '9.181e-05', 'ppl': '1.023', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.15', 'tokens/total': 3662416, 'tokens/trainable': 55303, 'epoch': '0.4727'}

 24%|β–ˆβ–ˆβ–Ž       | 121/512 [13:22<42:57,  6.59s/it]
 24%|β–ˆβ–ˆβ–       | 122/512 [13:29<42:52,  6.60s/it]
                                                 
{'loss': '0.02327', 'grad_norm': '0.4664', 'learning_rate': '9.164e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '31.94', 'tokens/total': 3692992, 'tokens/trainable': 55743, 'epoch': '0.4766'}

 24%|β–ˆβ–ˆβ–       | 122/512 [13:29<42:52,  6.60s/it]
 24%|β–ˆβ–ˆβ–       | 123/512 [13:35<42:39,  6.58s/it]
                                                 
{'loss': '0.00781', 'grad_norm': '0.2307', 'learning_rate': '9.147e-05', 'ppl': '1.008', 'memory/max_active (GiB)': '33.73', 'memory/max_allocated (GiB)': '33.73', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.95', 'tokens/total': 3722992, 'tokens/trainable': 56187, 'epoch': '0.4805'}

 24%|β–ˆβ–ˆβ–       | 123/512 [13:35<42:39,  6.58s/it]
 24%|β–ˆβ–ˆβ–       | 124/512 [13:42<42:31,  6.58s/it]
                                                 
{'loss': '0.006973', 'grad_norm': '0.1732', 'learning_rate': '9.13e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.63', 'tokens/total': 3753184, 'tokens/trainable': 56632, 'epoch': '0.4844'}

 24%|β–ˆβ–ˆβ–       | 124/512 [13:42<42:31,  6.58s/it]
 24%|β–ˆβ–ˆβ–       | 125/512 [13:49<42:31,  6.59s/it]
                                                 
{'loss': '0.006869', 'grad_norm': '0.1999', 'learning_rate': '9.113e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '37.26', 'tokens/total': 3783440, 'tokens/trainable': 57124, 'epoch': '0.4883'}

 24%|β–ˆβ–ˆβ–       | 125/512 [13:49<42:31,  6.59s/it]
 25%|β–ˆβ–ˆβ–       | 126/512 [13:55<42:21,  6.58s/it]
                                                 
{'loss': '0.004266', 'grad_norm': '0.1133', 'learning_rate': '9.096e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.38', 'tokens/total': 3813840, 'tokens/trainable': 57570, 'epoch': '0.4922'}

 25%|β–ˆβ–ˆβ–       | 126/512 [13:55<42:21,  6.58s/it]
 25%|β–ˆβ–ˆβ–       | 127/512 [14:02<42:21,  6.60s/it]
                                                 
{'loss': '0.02041', 'grad_norm': '0.4156', 'learning_rate': '9.078e-05', 'ppl': '1.021', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '35.36', 'tokens/total': 3844464, 'tokens/trainable': 58048, 'epoch': '0.4961'}

 25%|β–ˆβ–ˆβ–       | 127/512 [14:02<42:21,  6.60s/it]
 25%|β–ˆβ–ˆβ–Œ       | 128/512 [14:08<41:16,  6.45s/it]
                                                 
{'loss': '0.003154', 'grad_norm': '0.0982', 'learning_rate': '9.06e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.39', 'memory/max_allocated (GiB)': '33.39', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '35.79', 'tokens/total': 3872816, 'tokens/trainable': 58497, 'epoch': '0.5'}

 25%|β–ˆβ–ˆβ–Œ       | 128/512 [14:08<41:16,  6.45s/it][2026-08-18 14:32:05,611] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-128

 25%|β–ˆβ–ˆβ–Œ       | 129/512 [14:16<44:24,  6.96s/it]
                                                 
{'loss': '0.01777', 'grad_norm': '0.3427', 'learning_rate': '9.043e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36', 'tokens/train_per_sec_per_gpu': '33.44', 'tokens/total': 3902992, 'tokens/trainable': 58947, 'epoch': '0.5039'}

 25%|β–ˆβ–ˆβ–Œ       | 129/512 [14:16<44:24,  6.96s/it]
 25%|β–ˆβ–ˆβ–Œ       | 130/512 [14:23<43:37,  6.85s/it]
                                                 
{'loss': '0.01817', 'grad_norm': '0.3253', 'learning_rate': '9.025e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '34.74', 'tokens/train_per_sec_per_gpu': '37.56', 'tokens/total': 3933456, 'tokens/trainable': 59415, 'epoch': '0.5078'}

 25%|β–ˆβ–ˆβ–Œ       | 130/512 [14:23<43:37,  6.85s/it]
 26%|β–ˆβ–ˆβ–Œ       | 131/512 [14:29<43:01,  6.78s/it]
                                                 
{'loss': '0.008536', 'grad_norm': '0.3168', 'learning_rate': '9.007e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '34.91', 'tokens/train_per_sec_per_gpu': '32.75', 'tokens/total': 3963536, 'tokens/trainable': 59851, 'epoch': '0.5117'}

 26%|β–ˆβ–ˆβ–Œ       | 131/512 [14:29<43:01,  6.78s/it]
 26%|β–ˆβ–ˆβ–Œ       | 132/512 [14:36<42:33,  6.72s/it]
                                                 
{'loss': '0.001934', 'grad_norm': '0.124', 'learning_rate': '8.988e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '36.25', 'tokens/total': 3994000, 'tokens/trainable': 60325, 'epoch': '0.5156'}

 26%|β–ˆβ–ˆβ–Œ       | 132/512 [14:36<42:33,  6.72s/it]
 26%|β–ˆβ–ˆβ–Œ       | 133/512 [14:42<42:15,  6.69s/it]
                                                 
{'loss': '0.0155', 'grad_norm': '0.2586', 'learning_rate': '8.97e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '35.97', 'tokens/total': 4024352, 'tokens/trainable': 60818, 'epoch': '0.5195'}

 26%|β–ˆβ–ˆβ–Œ       | 133/512 [14:42<42:15,  6.69s/it]
 26%|β–ˆβ–ˆβ–Œ       | 134/512 [14:49<41:02,  6.51s/it]
                                                 
{'loss': '0.006828', 'grad_norm': '0.3528', 'learning_rate': '8.951e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.39', 'memory/max_allocated (GiB)': '33.39', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '32.28', 'tokens/total': 4052832, 'tokens/trainable': 61263, 'epoch': '0.5234'}

 26%|β–ˆβ–ˆβ–Œ       | 134/512 [14:49<41:02,  6.51s/it]
 26%|β–ˆβ–ˆβ–‹       | 135/512 [14:55<41:07,  6.55s/it]
                                                 
{'loss': '0.01065', 'grad_norm': '0.3797', 'learning_rate': '8.933e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '36.58', 'tokens/total': 4083216, 'tokens/trainable': 61734, 'epoch': '0.5273'}

 26%|β–ˆβ–ˆβ–‹       | 135/512 [14:55<41:07,  6.55s/it]
 27%|β–ˆβ–ˆβ–‹       | 136/512 [15:02<41:00,  6.54s/it]
                                                 
{'loss': '0.02273', 'grad_norm': '0.359', 'learning_rate': '8.914e-05', 'ppl': '1.023', 'memory/max_active (GiB)': '33.7', 'memory/max_allocated (GiB)': '33.7', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '38.83', 'tokens/total': 4113392, 'tokens/trainable': 62210, 'epoch': '0.5312'}

 27%|β–ˆβ–ˆβ–‹       | 136/512 [15:02<41:00,  6.54s/it]
 27%|β–ˆβ–ˆβ–‹       | 137/512 [15:08<41:02,  6.57s/it]
                                                 
{'loss': '0.01776', 'grad_norm': '0.3439', 'learning_rate': '8.895e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '36.87', 'tokens/total': 4143696, 'tokens/trainable': 62707, 'epoch': '0.5352'}

 27%|β–ˆβ–ˆβ–‹       | 137/512 [15:08<41:02,  6.57s/it]
 27%|β–ˆβ–ˆβ–‹       | 138/512 [15:15<40:58,  6.57s/it]
                                                 
{'loss': '0.01984', 'grad_norm': '0.2429', 'learning_rate': '8.876e-05', 'ppl': '1.02', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '37.17', 'tokens/total': 4174000, 'tokens/trainable': 63166, 'epoch': '0.5391'}

 27%|β–ˆβ–ˆβ–‹       | 138/512 [15:15<40:58,  6.57s/it]
 27%|β–ˆβ–ˆβ–‹       | 139/512 [15:22<40:58,  6.59s/it]
                                                 
{'loss': '0.01264', 'grad_norm': '0.6662', 'learning_rate': '8.856e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.79', 'tokens/train_per_sec_per_gpu': '36.96', 'tokens/total': 4204432, 'tokens/trainable': 63646, 'epoch': '0.543'}

 27%|β–ˆβ–ˆβ–‹       | 139/512 [15:22<40:58,  6.59s/it]
 27%|β–ˆβ–ˆβ–‹       | 140/512 [15:28<39:50,  6.43s/it]
                                                 
{'loss': '0.01461', 'grad_norm': '0.2639', 'learning_rate': '8.837e-05', 'ppl': '1.015', 'memory/max_active (GiB)': '33.42', 'memory/max_allocated (GiB)': '33.42', 'memory/device_reserved (GiB)': '35.79', 'tokens/train_per_sec_per_gpu': '33.6', 'tokens/total': 4232752, 'tokens/trainable': 64063, 'epoch': '0.5469'}

 27%|β–ˆβ–ˆβ–‹       | 140/512 [15:28<39:50,  6.43s/it]
 28%|β–ˆβ–ˆβ–Š       | 141/512 [15:34<40:05,  6.48s/it]
                                                 
{'loss': '0.01266', 'grad_norm': '0.2475', 'learning_rate': '8.817e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.79', 'tokens/train_per_sec_per_gpu': '35.67', 'tokens/total': 4262976, 'tokens/trainable': 64520, 'epoch': '0.5508'}

 28%|β–ˆβ–ˆβ–Š       | 141/512 [15:34<40:05,  6.48s/it]
 28%|β–ˆβ–ˆβ–Š       | 142/512 [15:41<40:13,  6.52s/it]
                                                 
{'loss': '0.009737', 'grad_norm': '0.2115', 'learning_rate': '8.798e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.79', 'tokens/train_per_sec_per_gpu': '34.48', 'tokens/total': 4293520, 'tokens/trainable': 64958, 'epoch': '0.5547'}

 28%|β–ˆβ–ˆβ–Š       | 142/512 [15:41<40:13,  6.52s/it]
 28%|β–ˆβ–ˆβ–Š       | 143/512 [15:47<40:10,  6.53s/it]
                                                 
{'loss': '0.01307', 'grad_norm': '0.2296', 'learning_rate': '8.778e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '38.3', 'tokens/total': 4323520, 'tokens/trainable': 65437, 'epoch': '0.5586'}

 28%|β–ˆβ–ˆβ–Š       | 143/512 [15:47<40:10,  6.53s/it]
 28%|β–ˆβ–ˆβ–Š       | 144/512 [15:54<40:09,  6.55s/it]
                                                 
{'loss': '0.01002', 'grad_norm': '0.2896', 'learning_rate': '8.758e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '37.25', 'tokens/total': 4353616, 'tokens/trainable': 65903, 'epoch': '0.5625'}

 28%|β–ˆβ–ˆβ–Š       | 144/512 [15:54<40:09,  6.55s/it]
 28%|β–ˆβ–ˆβ–Š       | 145/512 [16:01<40:06,  6.56s/it]
                                                 
{'loss': '0.02572', 'grad_norm': '0.4186', 'learning_rate': '8.738e-05', 'ppl': '1.026', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '32.08', 'tokens/total': 4383856, 'tokens/trainable': 66334, 'epoch': '0.5664'}

 28%|β–ˆβ–ˆβ–Š       | 145/512 [16:01<40:06,  6.56s/it]
 29%|β–ˆβ–ˆβ–Š       | 146/512 [16:07<40:07,  6.58s/it]
                                                 
{'loss': '0.01731', 'grad_norm': '0.278', 'learning_rate': '8.718e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '32.01', 'tokens/total': 4414128, 'tokens/trainable': 66785, 'epoch': '0.5703'}

 29%|β–ˆβ–ˆβ–Š       | 146/512 [16:07<40:07,  6.58s/it]
 29%|β–ˆβ–ˆβ–Š       | 147/512 [16:14<40:02,  6.58s/it]
                                                 
{'loss': '0.01725', 'grad_norm': '0.5759', 'learning_rate': '8.697e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '35.81', 'tokens/total': 4444384, 'tokens/trainable': 67199, 'epoch': '0.5742'}

 29%|β–ˆβ–ˆβ–Š       | 147/512 [16:14<40:02,  6.58s/it]
 29%|β–ˆβ–ˆβ–‰       | 148/512 [16:20<40:06,  6.61s/it]
                                                 
{'loss': '0.02244', 'grad_norm': '0.443', 'learning_rate': '8.677e-05', 'ppl': '1.023', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '35.64', 'tokens/total': 4474912, 'tokens/trainable': 67683, 'epoch': '0.5781'}

 29%|β–ˆβ–ˆβ–‰       | 148/512 [16:20<40:06,  6.61s/it]
 29%|β–ˆβ–ˆβ–‰       | 149/512 [16:27<39:58,  6.61s/it]
                                                 
{'loss': '0.01044', 'grad_norm': '0.2663', 'learning_rate': '8.656e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '34.4', 'tokens/total': 4505328, 'tokens/trainable': 68127, 'epoch': '0.582'}

 29%|β–ˆβ–ˆβ–‰       | 149/512 [16:27<39:58,  6.61s/it]
 29%|β–ˆβ–ˆβ–‰       | 150/512 [16:34<39:48,  6.60s/it]
                                                 
{'loss': '0.004219', 'grad_norm': '0.1525', 'learning_rate': '8.635e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '33.75', 'tokens/total': 4535424, 'tokens/trainable': 68595, 'epoch': '0.5859'}

 29%|β–ˆβ–ˆβ–‰       | 150/512 [16:34<39:48,  6.60s/it]
 29%|β–ˆβ–ˆβ–‰       | 151/512 [16:40<39:44,  6.61s/it]
                                                 
{'loss': '0.01187', 'grad_norm': '0.2124', 'learning_rate': '8.615e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '30.68', 'tokens/total': 4565536, 'tokens/trainable': 69062, 'epoch': '0.5898'}

 29%|β–ˆβ–ˆβ–‰       | 151/512 [16:40<39:44,  6.61s/it]
 30%|β–ˆβ–ˆβ–‰       | 152/512 [16:47<39:39,  6.61s/it]
                                                 
{'loss': '0.007625', 'grad_norm': '0.5053', 'learning_rate': '8.594e-05', 'ppl': '1.008', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '35.96', 'tokens/total': 4596048, 'tokens/trainable': 69509, 'epoch': '0.5938'}

 30%|β–ˆβ–ˆβ–‰       | 152/512 [16:47<39:39,  6.61s/it]
 30%|β–ˆβ–ˆβ–‰       | 153/512 [16:53<39:36,  6.62s/it]
                                                 
{'loss': '0.01224', 'grad_norm': '0.4628', 'learning_rate': '8.572e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '34.45', 'tokens/total': 4626480, 'tokens/trainable': 69972, 'epoch': '0.5977'}

 30%|β–ˆβ–ˆβ–‰       | 153/512 [16:53<39:36,  6.62s/it]
 30%|β–ˆβ–ˆβ–ˆ       | 154/512 [17:00<39:27,  6.61s/it]
                                                 
{'loss': '0.006034', 'grad_norm': '0.4228', 'learning_rate': '8.551e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '37.3', 'tokens/total': 4656800, 'tokens/trainable': 70439, 'epoch': '0.6016'}

 30%|β–ˆβ–ˆβ–ˆ       | 154/512 [17:00<39:27,  6.61s/it]
 30%|β–ˆβ–ˆβ–ˆ       | 155/512 [17:06<38:25,  6.46s/it]
                                                 
{'loss': '0.01827', 'grad_norm': '0.599', 'learning_rate': '8.53e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '37.43', 'tokens/total': 4685120, 'tokens/trainable': 70881, 'epoch': '0.6055'}

 30%|β–ˆβ–ˆβ–ˆ       | 155/512 [17:06<38:25,  6.46s/it]
 30%|β–ˆβ–ˆβ–ˆ       | 156/512 [17:13<38:35,  6.50s/it]
                                                 
{'loss': '0.01053', 'grad_norm': '0.6471', 'learning_rate': '8.508e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '34.48', 'tokens/total': 4715664, 'tokens/trainable': 71339, 'epoch': '0.6094'}

 30%|β–ˆβ–ˆβ–ˆ       | 156/512 [17:13<38:35,  6.50s/it]
 31%|β–ˆβ–ˆβ–ˆ       | 157/512 [17:19<38:40,  6.54s/it]
                                                 
{'loss': '0.0001828', 'grad_norm': '0.00899', 'learning_rate': '8.487e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '35.55', 'tokens/total': 4745824, 'tokens/trainable': 71801, 'epoch': '0.6133'}

 31%|β–ˆβ–ˆβ–ˆ       | 157/512 [17:19<38:40,  6.54s/it]
 31%|β–ˆβ–ˆβ–ˆ       | 158/512 [17:26<38:45,  6.57s/it]
                                                 
{'loss': '0.01429', 'grad_norm': '0.2966', 'learning_rate': '8.465e-05', 'ppl': '1.014', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '32.65', 'tokens/total': 4776224, 'tokens/trainable': 72237, 'epoch': '0.6172'}

 31%|β–ˆβ–ˆβ–ˆ       | 158/512 [17:26<38:45,  6.57s/it]
 31%|β–ˆβ–ˆβ–ˆ       | 159/512 [17:33<38:36,  6.56s/it]
                                                 
{'loss': '0.02528', 'grad_norm': '0.3071', 'learning_rate': '8.443e-05', 'ppl': '1.026', 'memory/max_active (GiB)': '33.75', 'memory/max_allocated (GiB)': '33.75', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '31.03', 'tokens/total': 4806304, 'tokens/trainable': 72664, 'epoch': '0.6211'}

 31%|β–ˆβ–ˆβ–ˆ       | 159/512 [17:33<38:36,  6.56s/it]
 31%|β–ˆβ–ˆβ–ˆβ–      | 160/512 [17:39<38:34,  6.57s/it]
                                                 
{'loss': '0.00503', 'grad_norm': '0.2943', 'learning_rate': '8.421e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '32.1', 'tokens/total': 4836688, 'tokens/trainable': 73119, 'epoch': '0.625'}

 31%|β–ˆβ–ˆβ–ˆβ–      | 160/512 [17:39<38:34,  6.57s/it][2026-08-18 14:35:36,933] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-160

 31%|β–ˆβ–ˆβ–ˆβ–      | 161/512 [17:47<41:02,  7.02s/it]
                                                 
{'loss': '0.01076', 'grad_norm': '0.389', 'learning_rate': '8.399e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '37.89', 'tokens/train_per_sec_per_gpu': '31.51', 'tokens/total': 4867008, 'tokens/trainable': 73540, 'epoch': '0.6289'}

 31%|β–ˆβ–ˆβ–ˆβ–      | 161/512 [17:47<41:02,  7.02s/it]
 32%|β–ˆβ–ˆβ–ˆβ–      | 162/512 [17:54<40:15,  6.90s/it]
                                                 
{'loss': '0.003405', 'grad_norm': '0.2366', 'learning_rate': '8.376e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '33.18', 'tokens/total': 4897584, 'tokens/trainable': 73990, 'epoch': '0.6328'}

 32%|β–ˆβ–ˆβ–ˆβ–      | 162/512 [17:54<40:15,  6.90s/it]
 32%|β–ˆβ–ˆβ–ˆβ–      | 163/512 [18:00<39:37,  6.81s/it]
                                                 
{'loss': '0.01729', 'grad_norm': '0.2311', 'learning_rate': '8.354e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '35.89', 'tokens/total': 4927904, 'tokens/trainable': 74454, 'epoch': '0.6367'}

 32%|β–ˆβ–ˆβ–ˆβ–      | 163/512 [18:00<39:37,  6.81s/it]
 32%|β–ˆβ–ˆβ–ˆβ–      | 164/512 [18:07<39:23,  6.79s/it]
                                                 
{'loss': '0.02588', 'grad_norm': '0.7158', 'learning_rate': '8.332e-05', 'ppl': '1.026', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '33.96', 'tokens/total': 4958496, 'tokens/trainable': 74905, 'epoch': '0.6406'}

 32%|β–ˆβ–ˆβ–ˆβ–      | 164/512 [18:07<39:23,  6.79s/it]
 32%|β–ˆβ–ˆβ–ˆβ–      | 165/512 [18:14<38:54,  6.73s/it]
                                                 
{'loss': '0.003934', 'grad_norm': '0.3431', 'learning_rate': '8.309e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '31.03', 'tokens/total': 4988832, 'tokens/trainable': 75329, 'epoch': '0.6445'}

 32%|β–ˆβ–ˆβ–ˆβ–      | 165/512 [18:14<38:54,  6.73s/it]
 32%|β–ˆβ–ˆβ–ˆβ–      | 166/512 [18:20<38:29,  6.68s/it]
                                                 
{'loss': '0.001242', 'grad_norm': '0.07044', 'learning_rate': '8.286e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.75', 'memory/max_allocated (GiB)': '33.75', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '32.34', 'tokens/total': 5019136, 'tokens/trainable': 75770, 'epoch': '0.6484'}

 32%|β–ˆβ–ˆβ–ˆβ–      | 166/512 [18:20<38:29,  6.68s/it]
 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 167/512 [18:27<38:25,  6.68s/it]
                                                 
{'loss': '0.008923', 'grad_norm': '0.6784', 'learning_rate': '8.263e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.99', 'memory/max_allocated (GiB)': '33.99', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '31.81', 'tokens/total': 5049808, 'tokens/trainable': 76231, 'epoch': '0.6523'}

 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 167/512 [18:27<38:25,  6.68s/it]
 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 168/512 [18:34<38:12,  6.66s/it]
                                                 
{'loss': '0.01817', 'grad_norm': '0.4106', 'learning_rate': '8.24e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '33.83', 'tokens/total': 5080320, 'tokens/trainable': 76682, 'epoch': '0.6562'}

 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 168/512 [18:34<38:12,  6.66s/it]
 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 169/512 [18:40<37:56,  6.64s/it]
                                                 
{'loss': '0.003114', 'grad_norm': '0.1389', 'learning_rate': '8.217e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '34.87', 'tokens/total': 5110656, 'tokens/trainable': 77114, 'epoch': '0.6602'}

 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 169/512 [18:40<37:56,  6.64s/it]
 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 170/512 [18:47<37:50,  6.64s/it]
                                                 
{'loss': '0.0309', 'grad_norm': '0.682', 'learning_rate': '8.194e-05', 'ppl': '1.031', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '34.61', 'tokens/total': 5141056, 'tokens/trainable': 77563, 'epoch': '0.6641'}

 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 170/512 [18:47<37:50,  6.64s/it]
 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 171/512 [18:53<37:38,  6.62s/it]
                                                 
{'loss': '0.003041', 'grad_norm': '0.2058', 'learning_rate': '8.171e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.7', 'tokens/total': 5171312, 'tokens/trainable': 78023, 'epoch': '0.668'}

 33%|β–ˆβ–ˆβ–ˆβ–Ž      | 171/512 [18:53<37:38,  6.62s/it]
 34%|β–ˆβ–ˆβ–ˆβ–Ž      | 172/512 [19:00<37:34,  6.63s/it]
                                                 
{'loss': '0.004004', 'grad_norm': '0.2195', 'learning_rate': '8.147e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '30.23', 'tokens/total': 5201744, 'tokens/trainable': 78477, 'epoch': '0.6719'}

 34%|β–ˆβ–ˆβ–ˆβ–Ž      | 172/512 [19:00<37:34,  6.63s/it]
 34%|β–ˆβ–ˆβ–ˆβ–      | 173/512 [19:07<37:21,  6.61s/it]
                                                 
{'loss': '0.006001', 'grad_norm': '0.5019', 'learning_rate': '8.124e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.51', 'tokens/total': 5232176, 'tokens/trainable': 78924, 'epoch': '0.6758'}

 34%|β–ˆβ–ˆβ–ˆβ–      | 173/512 [19:07<37:21,  6.61s/it]
 34%|β–ˆβ–ˆβ–ˆβ–      | 174/512 [19:13<36:57,  6.56s/it]
                                                 
{'loss': '0.004308', 'grad_norm': '0.3675', 'learning_rate': '8.1e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.62', 'memory/max_allocated (GiB)': '33.62', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '30.59', 'tokens/total': 5261904, 'tokens/trainable': 79342, 'epoch': '0.6797'}

 34%|β–ˆβ–ˆβ–ˆβ–      | 174/512 [19:13<36:57,  6.56s/it]
 34%|β–ˆβ–ˆβ–ˆβ–      | 175/512 [19:20<36:54,  6.57s/it]
                                                 
{'loss': '0.0007982', 'grad_norm': '0.4297', 'learning_rate': '8.076e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.05', 'tokens/total': 5292192, 'tokens/trainable': 79812, 'epoch': '0.6836'}

 34%|β–ˆβ–ˆβ–ˆβ–      | 175/512 [19:20<36:54,  6.57s/it]
 34%|β–ˆβ–ˆβ–ˆβ–      | 176/512 [19:26<36:55,  6.59s/it]
                                                 
{'loss': '0.005437', 'grad_norm': '0.7104', 'learning_rate': '8.053e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '36.73', 'tokens/total': 5322704, 'tokens/trainable': 80290, 'epoch': '0.6875'}

 34%|β–ˆβ–ˆβ–ˆβ–      | 176/512 [19:26<36:55,  6.59s/it]
 35%|β–ˆβ–ˆβ–ˆβ–      | 177/512 [19:33<36:48,  6.59s/it]
                                                 
{'loss': '0.007359', 'grad_norm': '0.5655', 'learning_rate': '8.029e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.08', 'tokens/total': 5353040, 'tokens/trainable': 80737, 'epoch': '0.6914'}

 35%|β–ˆβ–ˆβ–ˆβ–      | 177/512 [19:33<36:48,  6.59s/it]
 35%|β–ˆβ–ˆβ–ˆβ–      | 178/512 [19:40<36:45,  6.60s/it]
                                                 
{'loss': '0.0001785', 'grad_norm': '0.03567', 'learning_rate': '8.005e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '35.33', 'tokens/total': 5383248, 'tokens/trainable': 81205, 'epoch': '0.6953'}

 35%|β–ˆβ–ˆβ–ˆβ–      | 178/512 [19:40<36:45,  6.60s/it]
 35%|β–ˆβ–ˆβ–ˆβ–      | 179/512 [19:46<36:31,  6.58s/it]
                                                 
{'loss': '0.0002031', 'grad_norm': '0.01853', 'learning_rate': '7.98e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.53', 'tokens/total': 5413248, 'tokens/trainable': 81652, 'epoch': '0.6992'}

 35%|β–ˆβ–ˆβ–ˆβ–      | 179/512 [19:46<36:31,  6.58s/it]
 35%|β–ˆβ–ˆβ–ˆβ–Œ      | 180/512 [19:53<36:28,  6.59s/it]
                                                 
{'loss': '0.0007448', 'grad_norm': '0.1078', 'learning_rate': '7.956e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '35.07', 'tokens/total': 5443424, 'tokens/trainable': 82125, 'epoch': '0.7031'}

 35%|β–ˆβ–ˆβ–ˆβ–Œ      | 180/512 [19:53<36:28,  6.59s/it]
 35%|β–ˆβ–ˆβ–ˆβ–Œ      | 181/512 [19:59<36:28,  6.61s/it]
                                                 
{'loss': '0.0008925', 'grad_norm': '0.1558', 'learning_rate': '7.932e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '39.35', 'tokens/total': 5473680, 'tokens/trainable': 82642, 'epoch': '0.707'}

 35%|β–ˆβ–ˆβ–ˆβ–Œ      | 181/512 [19:59<36:28,  6.61s/it]
 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 182/512 [20:06<36:22,  6.61s/it]
                                                 
{'loss': '0.01775', 'grad_norm': '0.7525', 'learning_rate': '7.907e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '34.47', 'tokens/total': 5504032, 'tokens/trainable': 83123, 'epoch': '0.7109'}

 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 182/512 [20:06<36:22,  6.61s/it]
 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 183/512 [20:13<36:16,  6.61s/it]
                                                 
{'loss': '0.01671', 'grad_norm': '0.8974', 'learning_rate': '7.883e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.26', 'tokens/total': 5534224, 'tokens/trainable': 83570, 'epoch': '0.7148'}

 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 183/512 [20:13<36:16,  6.61s/it]
 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 184/512 [20:19<36:08,  6.61s/it]
                                                 
{'loss': '0.001814', 'grad_norm': '0.3188', 'learning_rate': '7.858e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '36.04', 'tokens/total': 5564704, 'tokens/trainable': 84076, 'epoch': '0.7188'}

 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 184/512 [20:19<36:08,  6.61s/it]
 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 185/512 [20:26<36:01,  6.61s/it]
                                                 
{'loss': '0.03057', 'grad_norm': '0.3768', 'learning_rate': '7.833e-05', 'ppl': '1.031', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '34.97', 'tokens/total': 5594912, 'tokens/trainable': 84551, 'epoch': '0.7227'}

 36%|β–ˆβ–ˆβ–ˆβ–Œ      | 185/512 [20:26<36:01,  6.61s/it]
 36%|β–ˆβ–ˆβ–ˆβ–‹      | 186/512 [20:32<35:57,  6.62s/it]
                                                 
{'loss': '0.0003345', 'grad_norm': '0.02533', 'learning_rate': '7.808e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '33.92', 'tokens/total': 5625280, 'tokens/trainable': 85012, 'epoch': '0.7266'}

 36%|β–ˆβ–ˆβ–ˆβ–‹      | 186/512 [20:32<35:57,  6.62s/it]
 37%|β–ˆβ–ˆβ–ˆβ–‹      | 187/512 [20:39<35:49,  6.61s/it]
                                                 
{'loss': '0.009675', 'grad_norm': '0.3701', 'learning_rate': '7.783e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '32.23', 'tokens/total': 5655568, 'tokens/trainable': 85452, 'epoch': '0.7305'}

 37%|β–ˆβ–ˆβ–ˆβ–‹      | 187/512 [20:39<35:49,  6.61s/it]
 37%|β–ˆβ–ˆβ–ˆβ–‹      | 188/512 [20:46<35:45,  6.62s/it]
                                                 
{'loss': '0.01034', 'grad_norm': '0.3262', 'learning_rate': '7.758e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '34.49', 'tokens/total': 5685824, 'tokens/trainable': 85942, 'epoch': '0.7344'}

 37%|β–ˆβ–ˆβ–ˆβ–‹      | 188/512 [20:46<35:45,  6.62s/it]
 37%|β–ˆβ–ˆβ–ˆβ–‹      | 189/512 [20:52<35:44,  6.64s/it]
                                                 
{'loss': '0.009306', 'grad_norm': '0.6184', 'learning_rate': '7.733e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '36.56', 'tokens/total': 5716192, 'tokens/trainable': 86457, 'epoch': '0.7383'}

 37%|β–ˆβ–ˆβ–ˆβ–‹      | 189/512 [20:52<35:44,  6.64s/it]
 37%|β–ˆβ–ˆβ–ˆβ–‹      | 190/512 [20:59<35:34,  6.63s/it]
                                                 
{'loss': '0.0043', 'grad_norm': '0.3189', 'learning_rate': '7.708e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '34.36', 'tokens/total': 5746784, 'tokens/trainable': 86928, 'epoch': '0.7422'}

 37%|β–ˆβ–ˆβ–ˆβ–‹      | 190/512 [20:59<35:34,  6.63s/it]
 37%|β–ˆβ–ˆβ–ˆβ–‹      | 191/512 [21:06<35:36,  6.66s/it]
                                                 
{'loss': '0.006645', 'grad_norm': '0.3455', 'learning_rate': '7.683e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.99', 'memory/max_allocated (GiB)': '33.99', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '36.31', 'tokens/total': 5777520, 'tokens/trainable': 87384, 'epoch': '0.7461'}

 37%|β–ˆβ–ˆβ–ˆβ–‹      | 191/512 [21:06<35:36,  6.66s/it]
 38%|β–ˆβ–ˆβ–ˆβ–Š      | 192/512 [21:12<35:23,  6.64s/it]
                                                 
{'loss': '0.006615', 'grad_norm': '0.2645', 'learning_rate': '7.657e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '32.03', 'tokens/total': 5807952, 'tokens/trainable': 87814, 'epoch': '0.75'}

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 192/512 [21:12<35:23,  6.64s/it][2026-08-18 14:39:10,060] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-192

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 193/512 [21:20<37:34,  7.07s/it]
                                                 
{'loss': '0.01492', 'grad_norm': '0.3881', 'learning_rate': '7.632e-05', 'ppl': '1.015', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.02', 'tokens/train_per_sec_per_gpu': '35.64', 'tokens/total': 5838384, 'tokens/trainable': 88297, 'epoch': '0.7539'}

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 193/512 [21:20<37:34,  7.07s/it]
 38%|β–ˆβ–ˆβ–ˆβ–Š      | 194/512 [21:27<36:47,  6.94s/it]
                                                 
{'loss': '0.02648', 'grad_norm': '0.6184', 'learning_rate': '7.606e-05', 'ppl': '1.027', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '35.37', 'tokens/total': 5868704, 'tokens/trainable': 88768, 'epoch': '0.7578'}

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 194/512 [21:27<36:47,  6.94s/it]
 38%|β–ˆβ–ˆβ–ˆβ–Š      | 195/512 [21:33<35:15,  6.68s/it]
                                                 
{'loss': '0.003626', 'grad_norm': '0.1768', 'learning_rate': '7.58e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.38', 'memory/max_allocated (GiB)': '33.38', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '30.88', 'tokens/total': 5896928, 'tokens/trainable': 89201, 'epoch': '0.7617'}

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 195/512 [21:33<35:15,  6.68s/it]
 38%|β–ˆβ–ˆβ–ˆβ–Š      | 196/512 [21:40<35:03,  6.66s/it]
                                                 
{'loss': '0.00249', 'grad_norm': '0.09678', 'learning_rate': '7.555e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '37.84', 'tokens/total': 5927328, 'tokens/trainable': 89663, 'epoch': '0.7656'}

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 196/512 [21:40<35:03,  6.66s/it]
 38%|β–ˆβ–ˆβ–ˆβ–Š      | 197/512 [21:46<34:04,  6.49s/it]
                                                 
{'loss': '0.001231', 'grad_norm': '0.03317', 'learning_rate': '7.529e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.34', 'memory/max_allocated (GiB)': '33.34', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '38.68', 'tokens/total': 5955648, 'tokens/trainable': 90111, 'epoch': '0.7695'}

 38%|β–ˆβ–ˆβ–ˆβ–Š      | 197/512 [21:46<34:04,  6.49s/it]
 39%|β–ˆβ–ˆβ–ˆβ–Š      | 198/512 [21:52<34:10,  6.53s/it]
                                                 
{'loss': '0.01116', 'grad_norm': '0.361', 'learning_rate': '7.503e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '34.73', 'tokens/total': 5985904, 'tokens/trainable': 90572, 'epoch': '0.7734'}

 39%|β–ˆβ–ˆβ–ˆβ–Š      | 198/512 [21:52<34:10,  6.53s/it]
 39%|β–ˆβ–ˆβ–ˆβ–‰      | 199/512 [21:59<34:24,  6.60s/it]
                                                 
{'loss': '0.001934', 'grad_norm': '0.0794', 'learning_rate': '7.477e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '34', 'memory/max_allocated (GiB)': '34', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.02', 'tokens/total': 6016656, 'tokens/trainable': 91051, 'epoch': '0.7773'}

 39%|β–ˆβ–ˆβ–ˆβ–‰      | 199/512 [21:59<34:24,  6.60s/it]
 39%|β–ˆβ–ˆβ–ˆβ–‰      | 200/512 [22:06<34:19,  6.60s/it]
                                                 
{'loss': '0.008391', 'grad_norm': '0.2828', 'learning_rate': '7.451e-05', 'ppl': '1.008', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '38.74', 'tokens/total': 6046912, 'tokens/trainable': 91527, 'epoch': '0.7812'}

 39%|β–ˆβ–ˆβ–ˆβ–‰      | 200/512 [22:06<34:19,  6.60s/it]
 39%|β–ˆβ–ˆβ–ˆβ–‰      | 201/512 [22:12<34:08,  6.59s/it]
                                                 
{'loss': '0.0006643', 'grad_norm': '0.02596', 'learning_rate': '7.424e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.69', 'memory/max_allocated (GiB)': '33.69', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '34.77', 'tokens/total': 6076912, 'tokens/trainable': 91973, 'epoch': '0.7852'}

 39%|β–ˆβ–ˆβ–ˆβ–‰      | 201/512 [22:12<34:08,  6.59s/it]
 39%|β–ˆβ–ˆβ–ˆβ–‰      | 202/512 [22:19<34:01,  6.59s/it]
                                                 
{'loss': '0.01052', 'grad_norm': '0.2635', 'learning_rate': '7.398e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.23', 'tokens/total': 6107264, 'tokens/trainable': 92438, 'epoch': '0.7891'}

 39%|β–ˆβ–ˆβ–ˆβ–‰      | 202/512 [22:19<34:01,  6.59s/it]
 40%|β–ˆβ–ˆβ–ˆβ–‰      | 203/512 [22:25<33:54,  6.58s/it]
                                                 
{'loss': '0.02642', 'grad_norm': '0.4627', 'learning_rate': '7.372e-05', 'ppl': '1.027', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.12', 'tokens/total': 6137344, 'tokens/trainable': 92886, 'epoch': '0.793'}

 40%|β–ˆβ–ˆβ–ˆβ–‰      | 203/512 [22:25<33:54,  6.58s/it]
 40%|β–ˆβ–ˆβ–ˆβ–‰      | 204/512 [22:32<33:49,  6.59s/it]
                                                 
{'loss': '0.005499', 'grad_norm': '0.4922', 'learning_rate': '7.345e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '34.54', 'tokens/total': 6167888, 'tokens/trainable': 93359, 'epoch': '0.7969'}

 40%|β–ˆβ–ˆβ–ˆβ–‰      | 204/512 [22:32<33:49,  6.59s/it]
 40%|β–ˆβ–ˆβ–ˆβ–ˆ      | 205/512 [22:39<33:37,  6.57s/it]
                                                 
{'loss': '0.002654', 'grad_norm': '0.143', 'learning_rate': '7.319e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '33.97', 'tokens/total': 6198080, 'tokens/trainable': 93783, 'epoch': '0.8008'}

 40%|β–ˆβ–ˆβ–ˆβ–ˆ      | 205/512 [22:39<33:37,  6.57s/it]
 40%|β–ˆβ–ˆβ–ˆβ–ˆ      | 206/512 [22:45<33:40,  6.60s/it]
                                                 
{'loss': '0.005219', 'grad_norm': '0.1867', 'learning_rate': '7.292e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.94', 'tokens/total': 6228480, 'tokens/trainable': 94262, 'epoch': '0.8047'}

 40%|β–ˆβ–ˆβ–ˆβ–ˆ      | 206/512 [22:45<33:40,  6.60s/it]
 40%|β–ˆβ–ˆβ–ˆβ–ˆ      | 207/512 [22:52<33:34,  6.60s/it]
                                                 
{'loss': '0.0106', 'grad_norm': '0.3964', 'learning_rate': '7.266e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '36.03', 'tokens/total': 6258640, 'tokens/trainable': 94753, 'epoch': '0.8086'}

 40%|β–ˆβ–ˆβ–ˆβ–ˆ      | 207/512 [22:52<33:34,  6.60s/it]
 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 208/512 [22:59<33:27,  6.60s/it]
                                                 
{'loss': '0.00088', 'grad_norm': '0.0414', 'learning_rate': '7.239e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '34.21', 'tokens/total': 6288928, 'tokens/trainable': 95238, 'epoch': '0.8125'}

 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 208/512 [22:59<33:27,  6.60s/it]
 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 209/512 [23:05<33:15,  6.59s/it]
                                                 
{'loss': '0.001892', 'grad_norm': '0.09655', 'learning_rate': '7.212e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '29.65', 'tokens/total': 6319568, 'tokens/trainable': 95639, 'epoch': '0.8164'}

 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 209/512 [23:05<33:15,  6.59s/it]
 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 210/512 [23:12<33:10,  6.59s/it]
                                                 
{'loss': '0.0052', 'grad_norm': '0.2853', 'learning_rate': '7.185e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '35.91', 'tokens/total': 6349824, 'tokens/trainable': 96096, 'epoch': '0.8203'}

 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 210/512 [23:12<33:10,  6.59s/it]
 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 211/512 [23:18<33:00,  6.58s/it]
                                                 
{'loss': '0.0128', 'grad_norm': '1.931', 'learning_rate': '7.158e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '34.64', 'tokens/total': 6379920, 'tokens/trainable': 96517, 'epoch': '0.8242'}

 41%|β–ˆβ–ˆβ–ˆβ–ˆ      | 211/512 [23:18<33:00,  6.58s/it]
 41%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 212/512 [23:25<32:55,  6.59s/it]
                                                 
{'loss': '0.0241', 'grad_norm': '0.7944', 'learning_rate': '7.131e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '33.19', 'tokens/total': 6410432, 'tokens/trainable': 96942, 'epoch': '0.8281'}

 41%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 212/512 [23:25<32:55,  6.59s/it]
 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 213/512 [23:31<32:50,  6.59s/it]
                                                 
{'loss': '0.009192', 'grad_norm': '0.589', 'learning_rate': '7.104e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '33.19', 'tokens/total': 6440864, 'tokens/trainable': 97380, 'epoch': '0.832'}

 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 213/512 [23:31<32:50,  6.59s/it]
 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 214/512 [23:38<32:48,  6.60s/it]
                                                 
{'loss': '0.002657', 'grad_norm': '0.1773', 'learning_rate': '7.077e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '36.92', 'tokens/total': 6471136, 'tokens/trainable': 97869, 'epoch': '0.8359'}

 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 214/512 [23:38<32:48,  6.60s/it]
 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 215/512 [23:45<32:38,  6.59s/it]
                                                 
{'loss': '0.003227', 'grad_norm': '0.1191', 'learning_rate': '7.05e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.9', 'tokens/total': 6501328, 'tokens/trainable': 98320, 'epoch': '0.8398'}

 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 215/512 [23:45<32:38,  6.59s/it]
 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 216/512 [23:51<32:30,  6.59s/it]
                                                 
{'loss': '0.002932', 'grad_norm': '0.149', 'learning_rate': '7.022e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '36.97', 'tokens/total': 6531680, 'tokens/trainable': 98793, 'epoch': '0.8438'}

 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 216/512 [23:51<32:30,  6.59s/it]
 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 217/512 [23:58<32:26,  6.60s/it]
                                                 
{'loss': '0.002244', 'grad_norm': '0.102', 'learning_rate': '6.995e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.33', 'tokens/total': 6562032, 'tokens/trainable': 99278, 'epoch': '0.8477'}

 42%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 217/512 [23:58<32:26,  6.60s/it]
 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 218/512 [24:04<32:15,  6.58s/it]
                                                 
{'loss': '0.001671', 'grad_norm': '0.07195', 'learning_rate': '6.968e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '37.41', 'tokens/total': 6592304, 'tokens/trainable': 99751, 'epoch': '0.8516'}

 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 218/512 [24:04<32:15,  6.58s/it]
 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 219/512 [24:11<32:09,  6.58s/it]
                                                 
{'loss': '0.001428', 'grad_norm': '0.07926', 'learning_rate': '6.94e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '32.21', 'tokens/total': 6622736, 'tokens/trainable': 100184, 'epoch': '0.8555'}

 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 219/512 [24:11<32:09,  6.58s/it]
 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 220/512 [24:18<32:04,  6.59s/it]
                                                 
{'loss': '0.02854', 'grad_norm': '4.548', 'learning_rate': '6.913e-05', 'ppl': '1.029', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '38.12', 'tokens/total': 6653168, 'tokens/trainable': 100656, 'epoch': '0.8594'}

 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 220/512 [24:18<32:04,  6.59s/it]
 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 221/512 [24:24<31:59,  6.59s/it]
                                                 
{'loss': '0.002416', 'grad_norm': '0.2031', 'learning_rate': '6.885e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '32.12', 'tokens/total': 6683328, 'tokens/trainable': 101107, 'epoch': '0.8633'}

 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 221/512 [24:24<31:59,  6.59s/it]
 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 222/512 [24:31<31:51,  6.59s/it]
                                                 
{'loss': '0.001577', 'grad_norm': '0.1236', 'learning_rate': '6.857e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '38.57', 'tokens/total': 6713840, 'tokens/trainable': 101574, 'epoch': '0.8672'}

 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 222/512 [24:31<31:51,  6.59s/it]
 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 223/512 [24:37<31:42,  6.58s/it]
                                                 
{'loss': '0.01232', 'grad_norm': '0.6421', 'learning_rate': '6.83e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '31.99', 'tokens/total': 6744224, 'tokens/trainable': 102015, 'epoch': '0.8711'}

 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž     | 223/512 [24:37<31:42,  6.58s/it]
 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 224/512 [24:44<31:38,  6.59s/it]
                                                 
{'loss': '0.00247', 'grad_norm': '0.3597', 'learning_rate': '6.802e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '36.04', 'tokens/total': 6774432, 'tokens/trainable': 102463, 'epoch': '0.875'}

 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 224/512 [24:44<31:38,  6.59s/it][2026-08-18 14:42:41,677] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-224

 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 225/512 [24:52<33:34,  7.02s/it]
                                                 
{'loss': '0.001325', 'grad_norm': '0.1022', 'learning_rate': '6.774e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.94', 'tokens/train_per_sec_per_gpu': '29.58', 'tokens/total': 6804688, 'tokens/trainable': 102880, 'epoch': '0.8789'}

 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 225/512 [24:52<33:34,  7.02s/it]
 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 226/512 [24:59<33:04,  6.94s/it]
                                                 
{'loss': '0.016', 'grad_norm': '0.7656', 'learning_rate': '6.746e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.7', 'tokens/train_per_sec_per_gpu': '37.94', 'tokens/total': 6835376, 'tokens/trainable': 103387, 'epoch': '0.8828'}

 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 226/512 [24:59<33:04,  6.94s/it]
 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 227/512 [25:05<32:24,  6.82s/it]
                                                 
{'loss': '0.004662', 'grad_norm': '0.298', 'learning_rate': '6.718e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '35.7', 'tokens/train_per_sec_per_gpu': '40.16', 'tokens/total': 6865392, 'tokens/trainable': 103880, 'epoch': '0.8867'}

 44%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 227/512 [25:05<32:24,  6.82s/it]
 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 228/512 [25:12<31:58,  6.76s/it]
                                                 
{'loss': '0.01068', 'grad_norm': '0.4353', 'learning_rate': '6.69e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.77', 'tokens/train_per_sec_per_gpu': '36.82', 'tokens/total': 6895776, 'tokens/trainable': 104343, 'epoch': '0.8906'}

 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 228/512 [25:12<31:58,  6.76s/it]
 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 229/512 [25:18<31:35,  6.70s/it]
                                                 
{'loss': '0.00132', 'grad_norm': '0.05993', 'learning_rate': '6.662e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.69', 'memory/max_allocated (GiB)': '33.69', 'memory/device_reserved (GiB)': '35.77', 'tokens/train_per_sec_per_gpu': '33.97', 'tokens/total': 6925776, 'tokens/trainable': 104808, 'epoch': '0.8945'}

 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 229/512 [25:18<31:35,  6.70s/it]
 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 230/512 [25:25<31:23,  6.68s/it]
                                                 
{'loss': '0.004353', 'grad_norm': '0.214', 'learning_rate': '6.634e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.77', 'tokens/train_per_sec_per_gpu': '37.4', 'tokens/total': 6956048, 'tokens/trainable': 105276, 'epoch': '0.8984'}

 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–     | 230/512 [25:25<31:23,  6.68s/it]
 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 231/512 [25:32<31:08,  6.65s/it]
                                                 
{'loss': '0.007145', 'grad_norm': '0.2365', 'learning_rate': '6.606e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.77', 'tokens/train_per_sec_per_gpu': '34.34', 'tokens/total': 6986384, 'tokens/trainable': 105740, 'epoch': '0.9023'}

 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 231/512 [25:32<31:08,  6.65s/it]
 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 232/512 [25:38<30:57,  6.63s/it]
                                                 
{'loss': '0.000307', 'grad_norm': '0.01727', 'learning_rate': '6.578e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.77', 'tokens/train_per_sec_per_gpu': '33.52', 'tokens/total': 7016784, 'tokens/trainable': 106216, 'epoch': '0.9062'}

 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 232/512 [25:38<30:57,  6.63s/it]
 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 233/512 [25:45<30:51,  6.63s/it]
                                                 
{'loss': '0.0005204', 'grad_norm': '0.02866', 'learning_rate': '6.55e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '38.58', 'tokens/total': 7047184, 'tokens/trainable': 106721, 'epoch': '0.9102'}

 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 233/512 [25:45<30:51,  6.63s/it]
 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 234/512 [25:51<30:37,  6.61s/it]
                                                 
{'loss': '0.001806', 'grad_norm': '0.1024', 'learning_rate': '6.522e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '34.36', 'tokens/total': 7077168, 'tokens/trainable': 107159, 'epoch': '0.9141'}

 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 234/512 [25:51<30:37,  6.61s/it]
 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 235/512 [25:58<30:29,  6.60s/it]
                                                 
{'loss': '0.01434', 'grad_norm': '0.1958', 'learning_rate': '6.493e-05', 'ppl': '1.014', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '33.21', 'tokens/total': 7107616, 'tokens/trainable': 107599, 'epoch': '0.918'}

 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 235/512 [25:58<30:29,  6.60s/it]
 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 236/512 [26:05<30:21,  6.60s/it]
                                                 
{'loss': '0.007726', 'grad_norm': '0.2197', 'learning_rate': '6.465e-05', 'ppl': '1.008', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '34.61', 'tokens/total': 7137728, 'tokens/trainable': 108063, 'epoch': '0.9219'}

 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ     | 236/512 [26:05<30:21,  6.60s/it]
 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 237/512 [26:11<30:12,  6.59s/it]
                                                 
{'loss': '0.0009402', 'grad_norm': '0.04341', 'learning_rate': '6.437e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '33.83', 'tokens/total': 7167904, 'tokens/trainable': 108491, 'epoch': '0.9258'}

 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 237/512 [26:11<30:12,  6.59s/it]
 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 238/512 [26:18<30:03,  6.58s/it]
                                                 
{'loss': '0.006565', 'grad_norm': '0.3101', 'learning_rate': '6.408e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '29.57', 'tokens/total': 7198288, 'tokens/trainable': 108919, 'epoch': '0.9297'}

 46%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 238/512 [26:18<30:03,  6.58s/it]
 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 239/512 [26:24<30:00,  6.60s/it]
                                                 
{'loss': '0.03366', 'grad_norm': '0.6903', 'learning_rate': '6.38e-05', 'ppl': '1.034', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '33.05', 'tokens/total': 7228624, 'tokens/trainable': 109360, 'epoch': '0.9336'}

 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 239/512 [26:24<30:00,  6.60s/it]
 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 240/512 [26:31<29:55,  6.60s/it]
                                                 
{'loss': '0.001445', 'grad_norm': '0.07222', 'learning_rate': '6.351e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '32.69', 'tokens/total': 7259056, 'tokens/trainable': 109823, 'epoch': '0.9375'}

 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 240/512 [26:31<29:55,  6.60s/it]
 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 241/512 [26:37<29:44,  6.58s/it]
                                                 
{'loss': '0.00448', 'grad_norm': '0.1981', 'learning_rate': '6.323e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '38.97', 'tokens/total': 7289328, 'tokens/trainable': 110281, 'epoch': '0.9414'}

 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 241/512 [26:37<29:44,  6.58s/it]
 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 242/512 [26:44<29:39,  6.59s/it]
                                                 
{'loss': '0.003018', 'grad_norm': '0.1568', 'learning_rate': '6.294e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '28.46', 'tokens/total': 7319632, 'tokens/trainable': 110694, 'epoch': '0.9453'}

 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 242/512 [26:44<29:39,  6.59s/it]
 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 243/512 [26:51<29:34,  6.60s/it]
                                                 
{'loss': '0.0005814', 'grad_norm': '0.02434', 'learning_rate': '6.266e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '39.78', 'tokens/total': 7349936, 'tokens/trainable': 111188, 'epoch': '0.9492'}

 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹     | 243/512 [26:51<29:34,  6.60s/it]
 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 244/512 [26:57<29:31,  6.61s/it]
                                                 
{'loss': '0.01199', 'grad_norm': '0.6709', 'learning_rate': '6.237e-05', 'ppl': '1.012', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '33.59', 'tokens/total': 7380576, 'tokens/trainable': 111645, 'epoch': '0.9531'}

 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 244/512 [26:57<29:31,  6.61s/it]
 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 245/512 [27:04<29:24,  6.61s/it]
                                                 
{'loss': '0.001512', 'grad_norm': '0.05413', 'learning_rate': '6.208e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '34.68', 'tokens/total': 7411200, 'tokens/trainable': 112128, 'epoch': '0.957'}

 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 245/512 [27:04<29:24,  6.61s/it]
 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 246/512 [27:11<29:17,  6.61s/it]
                                                 
{'loss': '0.002352', 'grad_norm': '0.2185', 'learning_rate': '6.18e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '33.6', 'tokens/total': 7441776, 'tokens/trainable': 112602, 'epoch': '0.9609'}

 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 246/512 [27:11<29:17,  6.61s/it]
 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 247/512 [27:17<28:34,  6.47s/it]
                                                 
{'loss': '0.001361', 'grad_norm': '0.05781', 'learning_rate': '6.151e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '38.27', 'tokens/total': 7470096, 'tokens/trainable': 113056, 'epoch': '0.9648'}

 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 247/512 [27:17<28:34,  6.47s/it]
 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 248/512 [27:23<28:42,  6.53s/it]
                                                 
{'loss': '0.005576', 'grad_norm': '0.3972', 'learning_rate': '6.122e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '35.76', 'tokens/total': 7500688, 'tokens/trainable': 113497, 'epoch': '0.9688'}

 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 248/512 [27:23<28:42,  6.53s/it]
 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 249/512 [27:30<28:39,  6.54s/it]
                                                 
{'loss': '0.006032', 'grad_norm': '0.2466', 'learning_rate': '6.093e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '33.19', 'tokens/total': 7530832, 'tokens/trainable': 113941, 'epoch': '0.9727'}

 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š     | 249/512 [27:30<28:39,  6.54s/it]
 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 250/512 [27:36<27:58,  6.41s/it]
                                                 
{'loss': '0.001342', 'grad_norm': '0.04302', 'learning_rate': '6.065e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '34.58', 'tokens/total': 7559136, 'tokens/trainable': 114373, 'epoch': '0.9766'}

 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 250/512 [27:36<27:58,  6.41s/it]
 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 251/512 [27:43<28:07,  6.47s/it]
                                                 
{'loss': '0.00874', 'grad_norm': '1.104', 'learning_rate': '6.036e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '36.97', 'tokens/total': 7589648, 'tokens/trainable': 114863, 'epoch': '0.9805'}

 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 251/512 [27:43<28:07,  6.47s/it]
 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 252/512 [27:49<28:11,  6.51s/it]
                                                 
{'loss': '0.0008687', 'grad_norm': '0.05026', 'learning_rate': '6.007e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '35.74', 'tokens/total': 7619904, 'tokens/trainable': 115311, 'epoch': '0.9844'}

 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 252/512 [27:49<28:11,  6.51s/it]
 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 253/512 [27:56<28:15,  6.55s/it]
                                                 
{'loss': '0.001406', 'grad_norm': '0.06898', 'learning_rate': '5.978e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '32.83', 'tokens/total': 7650480, 'tokens/trainable': 115732, 'epoch': '0.9883'}

 49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 253/512 [27:56<28:15,  6.55s/it]
 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 254/512 [28:02<28:12,  6.56s/it]
                                                 
{'loss': '0.009384', 'grad_norm': '0.5817', 'learning_rate': '5.949e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '29.88', 'tokens/total': 7680912, 'tokens/trainable': 116164, 'epoch': '0.9922'}

 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 254/512 [28:02<28:12,  6.56s/it]
 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 255/512 [28:09<28:07,  6.57s/it]
                                                 
{'loss': '0.0004331', 'grad_norm': '0.01903', 'learning_rate': '5.92e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '37.43', 'tokens/total': 7711344, 'tokens/trainable': 116614, 'epoch': '0.9961'}

 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰     | 255/512 [28:09<28:07,  6.57s/it]
 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 256/512 [28:16<28:14,  6.62s/it]
                                                 
{'loss': '0.0004527', 'grad_norm': '0.05099', 'learning_rate': '5.891e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '32.59', 'tokens/total': 7741936, 'tokens/trainable': 117062, 'epoch': '1'}

 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 256/512 [28:16<28:14,  6.62s/it][2026-08-18 14:46:13,557] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-256

 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 257/512 [28:24<30:28,  7.17s/it]
                                                 
{'loss': '0.001033', 'grad_norm': '0.09221', 'learning_rate': '5.862e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '32.85', 'tokens/total': 7772272, 'tokens/trainable': 117518, 'epoch': '1.004'}

 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 257/512 [28:24<30:28,  7.17s/it]
 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 258/512 [28:31<29:37,  7.00s/it]
                                                 
{'loss': '0.0006238', 'grad_norm': '0.03798', 'learning_rate': '5.834e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '30.99', 'tokens/total': 7802560, 'tokens/trainable': 117948, 'epoch': '1.008'}

 50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 258/512 [28:31<29:37,  7.00s/it]
 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 259/512 [28:37<28:58,  6.87s/it]
                                                 
{'loss': '0.002222', 'grad_norm': '0.1512', 'learning_rate': '5.805e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.2', 'tokens/total': 7832896, 'tokens/trainable': 118388, 'epoch': '1.012'}

 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 259/512 [28:37<28:58,  6.87s/it]
 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 260/512 [28:44<28:30,  6.79s/it]
                                                 
{'loss': '0.0004677', 'grad_norm': '0.0371', 'learning_rate': '5.776e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.01', 'tokens/total': 7863328, 'tokens/trainable': 118831, 'epoch': '1.016'}

 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 260/512 [28:44<28:30,  6.79s/it]
 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 261/512 [28:51<28:07,  6.72s/it]
                                                 
{'loss': '0.001122', 'grad_norm': '0.05126', 'learning_rate': '5.747e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '34.44', 'tokens/total': 7893792, 'tokens/trainable': 119278, 'epoch': '1.02'}

 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 261/512 [28:51<28:07,  6.72s/it]
 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 262/512 [28:57<27:54,  6.70s/it]
                                                 
{'loss': '0.001884', 'grad_norm': '0.2065', 'learning_rate': '5.718e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.36', 'tokens/total': 7924272, 'tokens/trainable': 119787, 'epoch': '1.023'}

 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ     | 262/512 [28:57<27:54,  6.70s/it]
 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 263/512 [29:04<27:42,  6.68s/it]
                                                 
{'loss': '0.002311', 'grad_norm': '0.1463', 'learning_rate': '5.689e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '34.44', 'tokens/total': 7954624, 'tokens/trainable': 120276, 'epoch': '1.027'}

 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 263/512 [29:04<27:42,  6.68s/it]
 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 264/512 [29:10<27:32,  6.66s/it]
                                                 
{'loss': '0.0002887', 'grad_norm': '0.04577', 'learning_rate': '5.66e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '37.7', 'tokens/total': 7985184, 'tokens/trainable': 120753, 'epoch': '1.031'}

 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 264/512 [29:10<27:32,  6.66s/it]
 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 265/512 [29:17<27:18,  6.63s/it]
                                                 
{'loss': '0.003725', 'grad_norm': '0.324', 'learning_rate': '5.631e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '40.07', 'tokens/total': 8015520, 'tokens/trainable': 121219, 'epoch': '1.035'}

 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 265/512 [29:17<27:18,  6.63s/it]
 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 266/512 [29:24<27:13,  6.64s/it]
                                                 
{'loss': '0.01656', 'grad_norm': '0.6029', 'learning_rate': '5.602e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.08', 'tokens/total': 8046064, 'tokens/trainable': 121694, 'epoch': '1.039'}

 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 266/512 [29:24<27:13,  6.64s/it]
 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 267/512 [29:30<27:05,  6.63s/it]
                                                 
{'loss': '0.0009177', 'grad_norm': '0.08974', 'learning_rate': '5.573e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '40.77', 'tokens/total': 8076352, 'tokens/trainable': 122198, 'epoch': '1.043'}

 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 267/512 [29:30<27:05,  6.63s/it]
 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 268/512 [29:37<26:55,  6.62s/it]
                                                 
{'loss': '8.959e-05', 'grad_norm': '0.007786', 'learning_rate': '5.544e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.98', 'tokens/total': 8106608, 'tokens/trainable': 122660, 'epoch': '1.047'}

 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 268/512 [29:37<26:55,  6.62s/it]
 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 269/512 [29:44<26:45,  6.61s/it]
                                                 
{'loss': '0.001509', 'grad_norm': '0.3485', 'learning_rate': '5.515e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.71', 'memory/max_allocated (GiB)': '33.71', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.61', 'tokens/total': 8136768, 'tokens/trainable': 123120, 'epoch': '1.051'}

 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 269/512 [29:44<26:45,  6.61s/it]
 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 270/512 [29:50<26:41,  6.62s/it]
                                                 
{'loss': '8.24e-05', 'grad_norm': '0.005141', 'learning_rate': '5.485e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '36.79', 'tokens/total': 8167248, 'tokens/trainable': 123610, 'epoch': '1.055'}

 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 270/512 [29:50<26:41,  6.62s/it]
 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 271/512 [29:57<26:32,  6.61s/it]
                                                 
{'loss': '8.649e-05', 'grad_norm': '0.004375', 'learning_rate': '5.456e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '38.27', 'tokens/total': 8197360, 'tokens/trainable': 124092, 'epoch': '1.059'}

 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 271/512 [29:57<26:32,  6.61s/it]
 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 272/512 [30:03<26:27,  6.62s/it]
                                                 
{'loss': '0.004285', 'grad_norm': '0.4465', 'learning_rate': '5.427e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.26', 'tokens/total': 8227648, 'tokens/trainable': 124562, 'epoch': '1.062'}

 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 272/512 [30:03<26:27,  6.62s/it]
 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 273/512 [30:10<26:18,  6.60s/it]
                                                 
{'loss': '0.001752', 'grad_norm': '0.1529', 'learning_rate': '5.398e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.71', 'memory/max_allocated (GiB)': '33.71', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '34.55', 'tokens/total': 8257696, 'tokens/trainable': 124992, 'epoch': '1.066'}

 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 273/512 [30:10<26:18,  6.60s/it]
 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 274/512 [30:17<26:10,  6.60s/it]
                                                 
{'loss': '0.003796', 'grad_norm': '0.2659', 'learning_rate': '5.369e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '35.11', 'tokens/total': 8288272, 'tokens/trainable': 125453, 'epoch': '1.07'}

 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 274/512 [30:17<26:10,  6.60s/it]
 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 275/512 [30:23<26:05,  6.60s/it]
                                                 
{'loss': '0.0005296', 'grad_norm': '0.05793', 'learning_rate': '5.34e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '32.01', 'tokens/total': 8318912, 'tokens/trainable': 125904, 'epoch': '1.074'}

 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž    | 275/512 [30:23<26:05,  6.60s/it]
 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 276/512 [30:30<25:59,  6.61s/it]
                                                 
{'loss': '0.001218', 'grad_norm': '0.1219', 'learning_rate': '5.311e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '33.38', 'tokens/total': 8349280, 'tokens/trainable': 126378, 'epoch': '1.078'}

 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 276/512 [30:30<25:59,  6.61s/it]
 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 277/512 [30:36<25:56,  6.63s/it]
                                                 
{'loss': '0.000552', 'grad_norm': '0.06378', 'learning_rate': '5.282e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '34.54', 'tokens/total': 8379744, 'tokens/trainable': 126843, 'epoch': '1.082'}

 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 277/512 [30:36<25:56,  6.63s/it]
 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 278/512 [30:43<25:48,  6.62s/it]
                                                 
{'loss': '0.01097', 'grad_norm': '0.533', 'learning_rate': '5.253e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '38.83', 'tokens/total': 8409968, 'tokens/trainable': 127346, 'epoch': '1.086'}

 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 278/512 [30:43<25:48,  6.62s/it]
 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 279/512 [30:50<25:39,  6.61s/it]
                                                 
{'loss': '0.01795', 'grad_norm': '0.2864', 'learning_rate': '5.224e-05', 'ppl': '1.018', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '33.57', 'tokens/total': 8440112, 'tokens/trainable': 127804, 'epoch': '1.09'}

 54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 279/512 [30:50<25:39,  6.61s/it]
 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 280/512 [30:56<25:31,  6.60s/it]
                                                 
{'loss': '5.531e-05', 'grad_norm': '0.003402', 'learning_rate': '5.195e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '33.36', 'tokens/total': 8470480, 'tokens/trainable': 128253, 'epoch': '1.094'}

 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 280/512 [30:56<25:31,  6.60s/it]
 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 281/512 [31:03<25:27,  6.61s/it]
                                                 
{'loss': '0.00493', 'grad_norm': '0.2083', 'learning_rate': '5.166e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '36.8', 'tokens/total': 8500752, 'tokens/trainable': 128735, 'epoch': '1.098'}

 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–    | 281/512 [31:03<25:27,  6.61s/it]
 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 282/512 [31:09<25:21,  6.61s/it]
                                                 
{'loss': '0.0001742', 'grad_norm': '0.01068', 'learning_rate': '5.138e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '35.65', 'tokens/total': 8530944, 'tokens/trainable': 129209, 'epoch': '1.102'}

 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 282/512 [31:09<25:21,  6.61s/it]
 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 283/512 [31:16<25:15,  6.62s/it]
                                                 
{'loss': '0.001448', 'grad_norm': '0.106', 'learning_rate': '5.109e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '33.5', 'tokens/total': 8561312, 'tokens/trainable': 129659, 'epoch': '1.105'}

 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 283/512 [31:16<25:15,  6.62s/it]
 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 284/512 [31:23<25:10,  6.63s/it]
                                                 
{'loss': '0.0002165', 'grad_norm': '0.0106', 'learning_rate': '5.08e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '33.14', 'tokens/total': 8591712, 'tokens/trainable': 130106, 'epoch': '1.109'}

 55%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 284/512 [31:23<25:10,  6.63s/it]
 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 285/512 [31:29<25:02,  6.62s/it]
                                                 
{'loss': '0.005551', 'grad_norm': '0.3154', 'learning_rate': '5.051e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '31.99', 'tokens/total': 8622096, 'tokens/trainable': 130556, 'epoch': '1.113'}

 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 285/512 [31:29<25:02,  6.62s/it]
 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 286/512 [31:36<24:54,  6.61s/it]
                                                 
{'loss': '0.000186', 'grad_norm': '0.009664', 'learning_rate': '5.022e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '34.84', 'tokens/total': 8652304, 'tokens/trainable': 131013, 'epoch': '1.117'}

 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 286/512 [31:36<24:54,  6.61s/it]
 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 287/512 [31:43<24:46,  6.61s/it]
                                                 
{'loss': '0.0003628', 'grad_norm': '0.02171', 'learning_rate': '4.993e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '34.57', 'tokens/total': 8682704, 'tokens/trainable': 131469, 'epoch': '1.121'}

 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ    | 287/512 [31:43<24:46,  6.61s/it]
 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 288/512 [31:49<24:37,  6.60s/it]
                                                 
{'loss': '0.00155', 'grad_norm': '0.1057', 'learning_rate': '4.964e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '36.55', 'tokens/total': 8713056, 'tokens/trainable': 131943, 'epoch': '1.125'}

 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 288/512 [31:49<24:37,  6.60s/it][2026-08-18 14:49:46,838] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-288

 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 289/512 [31:57<26:27,  7.12s/it]
                                                 
{'loss': '0.001709', 'grad_norm': '0.09497', 'learning_rate': '4.935e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.91', 'tokens/train_per_sec_per_gpu': '33.71', 'tokens/total': 8743232, 'tokens/trainable': 132417, 'epoch': '1.129'}

 56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 289/512 [31:57<26:27,  7.12s/it]
 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 290/512 [32:04<25:45,  6.96s/it]
                                                 
{'loss': '0.0005101', 'grad_norm': '0.02392', 'learning_rate': '4.907e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.56', 'tokens/train_per_sec_per_gpu': '36.96', 'tokens/total': 8773664, 'tokens/trainable': 132866, 'epoch': '1.133'}

 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 290/512 [32:04<25:45,  6.96s/it]
 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 291/512 [32:11<25:16,  6.86s/it]
                                                 
{'loss': '0.0002141', 'grad_norm': '0.008019', 'learning_rate': '4.878e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.56', 'tokens/train_per_sec_per_gpu': '36.26', 'tokens/total': 8803920, 'tokens/trainable': 133367, 'epoch': '1.137'}

 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 291/512 [32:11<25:16,  6.86s/it]
 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 292/512 [32:17<24:50,  6.78s/it]
                                                 
{'loss': '0.001267', 'grad_norm': '0.1001', 'learning_rate': '4.849e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.58', 'tokens/train_per_sec_per_gpu': '34.8', 'tokens/total': 8834240, 'tokens/trainable': 133799, 'epoch': '1.141'}

 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 292/512 [32:17<24:50,  6.78s/it]
 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 293/512 [32:24<24:35,  6.74s/it]
                                                 
{'loss': '0.0007189', 'grad_norm': '0.04551', 'learning_rate': '4.82e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.58', 'tokens/train_per_sec_per_gpu': '36.86', 'tokens/total': 8864560, 'tokens/trainable': 134271, 'epoch': '1.145'}

 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 293/512 [32:24<24:35,  6.74s/it]
 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 294/512 [32:31<24:24,  6.72s/it]
                                                 
{'loss': '0.0005528', 'grad_norm': '0.02876', 'learning_rate': '4.792e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.58', 'tokens/train_per_sec_per_gpu': '34.05', 'tokens/total': 8895312, 'tokens/trainable': 134695, 'epoch': '1.148'}

 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹    | 294/512 [32:31<24:24,  6.72s/it]
 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 295/512 [32:37<24:11,  6.69s/it]
                                                 
{'loss': '0.001133', 'grad_norm': '0.08144', 'learning_rate': '4.763e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.58', 'tokens/train_per_sec_per_gpu': '36.1', 'tokens/total': 8925744, 'tokens/trainable': 135177, 'epoch': '1.152'}

 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 295/512 [32:37<24:11,  6.69s/it]
 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 296/512 [32:44<23:56,  6.65s/it]
                                                 
{'loss': '0.001164', 'grad_norm': '0.08422', 'learning_rate': '4.734e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '33.71', 'tokens/total': 8956048, 'tokens/trainable': 135611, 'epoch': '1.156'}

 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 296/512 [32:44<23:56,  6.65s/it]
 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 297/512 [32:50<23:45,  6.63s/it]
                                                 
{'loss': '0.0003869', 'grad_norm': '0.01976', 'learning_rate': '4.706e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '33.09', 'tokens/total': 8986320, 'tokens/trainable': 136050, 'epoch': '1.16'}

 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 297/512 [32:50<23:45,  6.63s/it]
 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 298/512 [32:57<23:38,  6.63s/it]
                                                 
{'loss': '0.0003641', 'grad_norm': '0.0198', 'learning_rate': '4.677e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '33.38', 'tokens/total': 9016800, 'tokens/trainable': 136507, 'epoch': '1.164'}

 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 298/512 [32:57<23:38,  6.63s/it]
 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 299/512 [33:04<23:32,  6.63s/it]
                                                 
{'loss': '0.0001828', 'grad_norm': '0.01045', 'learning_rate': '4.649e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '35.87', 'tokens/total': 9047184, 'tokens/trainable': 136965, 'epoch': '1.168'}

 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 299/512 [33:04<23:32,  6.63s/it]
 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 300/512 [33:10<23:21,  6.61s/it]
                                                 
{'loss': '0.002034', 'grad_norm': '0.1379', 'learning_rate': '4.62e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '34.78', 'tokens/total': 9077488, 'tokens/trainable': 137415, 'epoch': '1.172'}

 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š    | 300/512 [33:10<23:21,  6.61s/it]
 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 301/512 [33:17<23:13,  6.61s/it]
                                                 
{'loss': '0.0006159', 'grad_norm': '0.05663', 'learning_rate': '4.592e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '36.22', 'tokens/total': 9107968, 'tokens/trainable': 137910, 'epoch': '1.176'}

 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 301/512 [33:17<23:13,  6.61s/it]
 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 302/512 [33:23<23:09,  6.62s/it]
                                                 
{'loss': '0.000474', 'grad_norm': '0.02709', 'learning_rate': '4.563e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '34.8', 'tokens/total': 9138240, 'tokens/trainable': 138386, 'epoch': '1.18'}

 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 302/512 [33:23<23:09,  6.62s/it]
 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 303/512 [33:30<23:05,  6.63s/it]
                                                 
{'loss': '0.0004738', 'grad_norm': '0.0356', 'learning_rate': '4.535e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '36.96', 'tokens/total': 9168640, 'tokens/trainable': 138845, 'epoch': '1.184'}

 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 303/512 [33:30<23:05,  6.63s/it]
 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 304/512 [33:37<22:54,  6.61s/it]
                                                 
{'loss': '0.0008522', 'grad_norm': '0.07774', 'learning_rate': '4.507e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '31.07', 'tokens/total': 9198912, 'tokens/trainable': 139270, 'epoch': '1.188'}

 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 304/512 [33:37<22:54,  6.61s/it]
 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 305/512 [33:43<22:48,  6.61s/it]
                                                 
{'loss': '0.01726', 'grad_norm': '0.2708', 'learning_rate': '4.478e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.66', 'tokens/train_per_sec_per_gpu': '33.55', 'tokens/total': 9229088, 'tokens/trainable': 139728, 'epoch': '1.191'}

 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 305/512 [33:43<22:48,  6.61s/it]
 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 306/512 [33:50<22:42,  6.61s/it]
                                                 
{'loss': '0.001807', 'grad_norm': '0.1075', 'learning_rate': '4.45e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '37.68', 'tokens/total': 9259680, 'tokens/trainable': 140219, 'epoch': '1.195'}

 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 306/512 [33:50<22:42,  6.61s/it]
 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 307/512 [33:56<21:57,  6.43s/it]
                                                 
{'loss': '4.969e-05', 'grad_norm': '0.002589', 'learning_rate': '4.422e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.19', 'memory/max_allocated (GiB)': '33.19', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '34.51', 'tokens/total': 9287584, 'tokens/trainable': 140657, 'epoch': '1.199'}

 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰    | 307/512 [33:56<21:57,  6.43s/it]
 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 308/512 [34:03<22:06,  6.50s/it]
                                                 
{'loss': '0.0003076', 'grad_norm': '0.01864', 'learning_rate': '4.394e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '30.38', 'tokens/total': 9318096, 'tokens/trainable': 141108, 'epoch': '1.203'}

 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 308/512 [34:03<22:06,  6.50s/it]
 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 309/512 [34:09<22:05,  6.53s/it]
                                                 
{'loss': '0.0002081', 'grad_norm': '0.01055', 'learning_rate': '4.366e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '29.44', 'tokens/total': 9348544, 'tokens/trainable': 141531, 'epoch': '1.207'}

 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 309/512 [34:09<22:05,  6.53s/it]
 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 310/512 [34:16<22:01,  6.54s/it]
                                                 
{'loss': '0.02379', 'grad_norm': '0.263', 'learning_rate': '4.338e-05', 'ppl': '1.024', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '35.94', 'tokens/total': 9378848, 'tokens/trainable': 141965, 'epoch': '1.211'}

 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 310/512 [34:16<22:01,  6.54s/it]
 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 311/512 [34:22<21:57,  6.55s/it]
                                                 
{'loss': '0.0001811', 'grad_norm': '0.01139', 'learning_rate': '4.31e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '34.52', 'tokens/total': 9409104, 'tokens/trainable': 142429, 'epoch': '1.215'}

 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 311/512 [34:22<21:57,  6.55s/it]
 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 312/512 [34:29<21:49,  6.55s/it]
                                                 
{'loss': '0.0001647', 'grad_norm': '0.01918', 'learning_rate': '4.282e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '38.76', 'tokens/total': 9439264, 'tokens/trainable': 142895, 'epoch': '1.219'}

 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 312/512 [34:29<21:49,  6.55s/it]
 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 313/512 [34:35<21:45,  6.56s/it]
                                                 
{'loss': '0.004963', 'grad_norm': '0.4227', 'learning_rate': '4.254e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '37.12', 'tokens/total': 9469648, 'tokens/trainable': 143370, 'epoch': '1.223'}

 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ    | 313/512 [34:35<21:45,  6.56s/it]
 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 314/512 [34:42<21:40,  6.57s/it]
                                                 
{'loss': '0.0009556', 'grad_norm': '0.09607', 'learning_rate': '4.226e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '33.41', 'tokens/total': 9499968, 'tokens/trainable': 143826, 'epoch': '1.227'}

 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 314/512 [34:42<21:40,  6.57s/it]
 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 315/512 [34:49<21:41,  6.61s/it]
                                                 
{'loss': '0.0002801', 'grad_norm': '0.01082', 'learning_rate': '4.198e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.99', 'memory/max_allocated (GiB)': '33.99', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '35.11', 'tokens/total': 9530448, 'tokens/trainable': 144308, 'epoch': '1.23'}

 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 315/512 [34:49<21:41,  6.61s/it]
 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 316/512 [34:55<21:33,  6.60s/it]
                                                 
{'loss': '0.0007925', 'grad_norm': '0.5568', 'learning_rate': '4.17e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '36', 'tokens/total': 9560848, 'tokens/trainable': 144765, 'epoch': '1.234'}

 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 316/512 [34:55<21:33,  6.60s/it]
 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 317/512 [35:02<21:31,  6.62s/it]
                                                 
{'loss': '0.009996', 'grad_norm': '0.3749', 'learning_rate': '4.143e-05', 'ppl': '1.01', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '37.75', 'tokens/total': 9591376, 'tokens/trainable': 145265, 'epoch': '1.238'}

 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 317/512 [35:02<21:31,  6.62s/it]
 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 318/512 [35:09<21:23,  6.62s/it]
                                                 
{'loss': '0.002976', 'grad_norm': '0.225', 'learning_rate': '4.115e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '37.25', 'tokens/total': 9621728, 'tokens/trainable': 145724, 'epoch': '1.242'}

 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 318/512 [35:09<21:23,  6.62s/it]
 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 319/512 [35:15<21:19,  6.63s/it]
                                                 
{'loss': '0.006409', 'grad_norm': '0.4703', 'learning_rate': '4.087e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '37.72', 'tokens/total': 9652272, 'tokens/trainable': 146219, 'epoch': '1.246'}

 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 319/512 [35:15<21:19,  6.63s/it]
 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 320/512 [35:22<21:12,  6.63s/it]
                                                 
{'loss': '0.00235', 'grad_norm': '0.1789', 'learning_rate': '4.06e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '34.58', 'tokens/total': 9682672, 'tokens/trainable': 146693, 'epoch': '1.25'}

 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 320/512 [35:22<21:12,  6.63s/it][2026-08-18 14:53:19,559] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-320

 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 321/512 [35:30<22:22,  7.03s/it]
                                                 
{'loss': '0.0006638', 'grad_norm': '0.06236', 'learning_rate': '4.032e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.9', 'tokens/train_per_sec_per_gpu': '35.87', 'tokens/total': 9712832, 'tokens/trainable': 147139, 'epoch': '1.254'}

 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 321/512 [35:30<22:22,  7.03s/it]
 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 322/512 [35:36<21:49,  6.89s/it]
                                                 
{'loss': '0.0001746', 'grad_norm': '0.009023', 'learning_rate': '4.005e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '36.14', 'tokens/train_per_sec_per_gpu': '35.74', 'tokens/total': 9743344, 'tokens/trainable': 147589, 'epoch': '1.258'}

 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 322/512 [35:36<21:49,  6.89s/it]
 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 323/512 [35:43<21:26,  6.81s/it]
                                                 
{'loss': '0.003418', 'grad_norm': '0.3831', 'learning_rate': '3.978e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.77', 'tokens/total': 9773760, 'tokens/trainable': 148045, 'epoch': '1.262'}

 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 323/512 [35:43<21:26,  6.81s/it]
 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 324/512 [35:50<21:12,  6.77s/it]
                                                 
{'loss': '0.005115', 'grad_norm': '0.3204', 'learning_rate': '3.95e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '34.01', 'memory/max_allocated (GiB)': '34.01', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '37.47', 'tokens/total': 9804080, 'tokens/trainable': 148494, 'epoch': '1.266'}

 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 324/512 [35:50<21:12,  6.77s/it]
 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 325/512 [35:56<20:55,  6.72s/it]
                                                 
{'loss': '0.01596', 'grad_norm': '0.7489', 'learning_rate': '3.923e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.84', 'tokens/total': 9834400, 'tokens/trainable': 148957, 'epoch': '1.27'}

 63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 325/512 [35:56<20:55,  6.72s/it]
 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 326/512 [36:03<20:44,  6.69s/it]
                                                 
{'loss': '0.0001496', 'grad_norm': '0.009092', 'learning_rate': '3.896e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '38.41', 'tokens/total': 9864704, 'tokens/trainable': 149451, 'epoch': '1.273'}

 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž   | 326/512 [36:03<20:44,  6.69s/it]
 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 327/512 [36:09<20:32,  6.66s/it]
                                                 
{'loss': '0.0001391', 'grad_norm': '0.005111', 'learning_rate': '3.869e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '40.07', 'tokens/total': 9894656, 'tokens/trainable': 149904, 'epoch': '1.277'}

 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 327/512 [36:09<20:32,  6.66s/it]
 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 328/512 [36:16<20:24,  6.65s/it]
                                                 
{'loss': '0.0001377', 'grad_norm': '0.007466', 'learning_rate': '3.842e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.36', 'tokens/total': 9924976, 'tokens/trainable': 150378, 'epoch': '1.281'}

 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 328/512 [36:16<20:24,  6.65s/it]
 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 329/512 [36:23<20:14,  6.64s/it]
                                                 
{'loss': '0.009102', 'grad_norm': '0.1656', 'learning_rate': '3.815e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.32', 'tokens/total': 9955376, 'tokens/trainable': 150853, 'epoch': '1.285'}

 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 329/512 [36:23<20:14,  6.64s/it]
 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 330/512 [36:29<20:06,  6.63s/it]
                                                 
{'loss': '0.000196', 'grad_norm': '0.01061', 'learning_rate': '3.788e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.45', 'tokens/total': 9985696, 'tokens/trainable': 151289, 'epoch': '1.289'}

 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 330/512 [36:29<20:06,  6.63s/it]
 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 331/512 [36:36<19:59,  6.62s/it]
                                                 
{'loss': '0.000298', 'grad_norm': '0.01095', 'learning_rate': '3.761e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '37.68', 'tokens/total': 10016112, 'tokens/trainable': 151757, 'epoch': '1.293'}

 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 331/512 [36:36<19:59,  6.62s/it]
 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 332/512 [36:42<19:49,  6.61s/it]
                                                 
{'loss': '0.0002142', 'grad_norm': '0.01175', 'learning_rate': '3.734e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.55', 'tokens/total': 10046544, 'tokens/trainable': 152207, 'epoch': '1.297'}

 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–   | 332/512 [36:42<19:49,  6.61s/it]
 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 333/512 [36:49<19:43,  6.61s/it]
                                                 
{'loss': '0.004726', 'grad_norm': '0.3032', 'learning_rate': '3.708e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.34', 'tokens/total': 10077120, 'tokens/trainable': 152646, 'epoch': '1.301'}

 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 333/512 [36:49<19:43,  6.61s/it]
 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 334/512 [36:56<19:33,  6.59s/it]
                                                 
{'loss': '0.0005056', 'grad_norm': '0.02422', 'learning_rate': '3.681e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.93', 'tokens/total': 10107600, 'tokens/trainable': 153063, 'epoch': '1.305'}

 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 334/512 [36:56<19:33,  6.59s/it]
 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 335/512 [37:02<19:27,  6.59s/it]
                                                 
{'loss': '0.001724', 'grad_norm': '0.1015', 'learning_rate': '3.655e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '30.92', 'tokens/total': 10137712, 'tokens/trainable': 153493, 'epoch': '1.309'}

 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 335/512 [37:02<19:27,  6.59s/it]
 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 336/512 [37:09<19:20,  6.60s/it]
                                                 
{'loss': '0.001302', 'grad_norm': '0.06118', 'learning_rate': '3.628e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.65', 'tokens/total': 10168096, 'tokens/trainable': 153945, 'epoch': '1.312'}

 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 336/512 [37:09<19:20,  6.60s/it]
 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 337/512 [37:16<19:19,  6.63s/it]
                                                 
{'loss': '0.0003055', 'grad_norm': '0.02251', 'learning_rate': '3.602e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.4', 'tokens/total': 10198720, 'tokens/trainable': 154432, 'epoch': '1.316'}

 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 337/512 [37:16<19:19,  6.63s/it]
 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 338/512 [37:22<19:12,  6.62s/it]
                                                 
{'loss': '0.002099', 'grad_norm': '0.1921', 'learning_rate': '3.576e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.47', 'tokens/total': 10228976, 'tokens/trainable': 154881, 'epoch': '1.32'}

 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 338/512 [37:22<19:12,  6.62s/it]
 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 339/512 [37:29<19:07,  6.64s/it]
                                                 
{'loss': '0.0002057', 'grad_norm': '0.008806', 'learning_rate': '3.549e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '30.32', 'tokens/total': 10259440, 'tokens/trainable': 155306, 'epoch': '1.324'}

 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ   | 339/512 [37:29<19:07,  6.64s/it]
 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 340/512 [37:35<18:55,  6.60s/it]
                                                 
{'loss': '0.0003597', 'grad_norm': '0.01875', 'learning_rate': '3.523e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '31.7', 'tokens/total': 10289824, 'tokens/trainable': 155719, 'epoch': '1.328'}

 66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 340/512 [37:35<18:55,  6.60s/it]
 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 341/512 [37:42<18:51,  6.62s/it]
                                                 
{'loss': '0.0001035', 'grad_norm': '0.005896', 'learning_rate': '3.497e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.3', 'tokens/total': 10320144, 'tokens/trainable': 156209, 'epoch': '1.332'}

 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 341/512 [37:42<18:51,  6.62s/it]
 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 342/512 [37:49<18:43,  6.61s/it]
                                                 
{'loss': '0.000157', 'grad_norm': '0.01002', 'learning_rate': '3.471e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.17', 'tokens/total': 10350528, 'tokens/trainable': 156656, 'epoch': '1.336'}

 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 342/512 [37:49<18:43,  6.61s/it]
 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 343/512 [37:55<18:34,  6.60s/it]
                                                 
{'loss': '0.00135', 'grad_norm': '0.1039', 'learning_rate': '3.445e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.72', 'memory/max_allocated (GiB)': '33.72', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.23', 'tokens/total': 10380816, 'tokens/trainable': 157118, 'epoch': '1.34'}

 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 343/512 [37:55<18:34,  6.60s/it]
 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 344/512 [38:02<18:29,  6.60s/it]
                                                 
{'loss': '0.0005016', 'grad_norm': '0.04982', 'learning_rate': '3.42e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.56', 'tokens/total': 10411312, 'tokens/trainable': 157562, 'epoch': '1.344'}

 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 344/512 [38:02<18:29,  6.60s/it]
 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 345/512 [38:08<18:21,  6.60s/it]
                                                 
{'loss': '0.0001865', 'grad_norm': '0.01197', 'learning_rate': '3.394e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.7', 'tokens/total': 10441696, 'tokens/trainable': 158003, 'epoch': '1.348'}

 67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹   | 345/512 [38:08<18:21,  6.60s/it]
 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 346/512 [38:15<18:13,  6.59s/it]
                                                 
{'loss': '0.0004138', 'grad_norm': '0.02001', 'learning_rate': '3.368e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '29.4', 'tokens/total': 10472032, 'tokens/trainable': 158419, 'epoch': '1.352'}

 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 346/512 [38:15<18:13,  6.59s/it]
 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 347/512 [38:21<18:03,  6.57s/it]
                                                 
{'loss': '0.001067', 'grad_norm': '0.1378', 'learning_rate': '3.343e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.73', 'memory/max_allocated (GiB)': '33.73', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.17', 'tokens/total': 10502176, 'tokens/trainable': 158877, 'epoch': '1.355'}

 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 347/512 [38:21<18:03,  6.57s/it]
 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 348/512 [38:28<18:01,  6.59s/it]
                                                 
{'loss': '0.0002564', 'grad_norm': '0.01242', 'learning_rate': '3.317e-05', 'ppl': '1', 'memory/max_active (GiB)': '34', 'memory/max_allocated (GiB)': '34', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '29.19', 'tokens/total': 10532896, 'tokens/trainable': 159299, 'epoch': '1.359'}

 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 348/512 [38:28<18:01,  6.59s/it]
 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 349/512 [38:35<17:53,  6.59s/it]
                                                 
{'loss': '0.0005879', 'grad_norm': '0.03198', 'learning_rate': '3.292e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.05', 'tokens/total': 10563024, 'tokens/trainable': 159750, 'epoch': '1.363'}

 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 349/512 [38:35<17:53,  6.59s/it]
 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 350/512 [38:41<17:51,  6.62s/it]
                                                 
{'loss': '0.0001015', 'grad_norm': '0.009861', 'learning_rate': '3.267e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.99', 'memory/max_allocated (GiB)': '33.99', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.76', 'tokens/total': 10593536, 'tokens/trainable': 160217, 'epoch': '1.367'}

 68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 350/512 [38:41<17:51,  6.62s/it]
 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 351/512 [38:48<17:45,  6.62s/it]
                                                 
{'loss': '0.000101', 'grad_norm': '0.005884', 'learning_rate': '3.242e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.11', 'tokens/total': 10623776, 'tokens/trainable': 160675, 'epoch': '1.371'}

 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š   | 351/512 [38:48<17:45,  6.62s/it]
 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 352/512 [38:55<17:37,  6.61s/it]
                                                 
{'loss': '0.003672', 'grad_norm': '0.2425', 'learning_rate': '3.217e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.02', 'tokens/total': 10654032, 'tokens/trainable': 161132, 'epoch': '1.375'}

 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 352/512 [38:55<17:37,  6.61s/it][2026-08-18 14:56:52,315] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-352

 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 353/512 [39:03<18:42,  7.06s/it]
                                                 
{'loss': '6.496e-05', 'grad_norm': '0.003396', 'learning_rate': '3.192e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '29.4', 'tokens/total': 10684352, 'tokens/trainable': 161585, 'epoch': '1.379'}

 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 353/512 [39:03<18:42,  7.06s/it]
 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 354/512 [39:09<18:13,  6.92s/it]
                                                 
{'loss': '0.008384', 'grad_norm': '0.8878', 'learning_rate': '3.167e-05', 'ppl': '1.008', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.41', 'tokens/train_per_sec_per_gpu': '37.32', 'tokens/total': 10714816, 'tokens/trainable': 162050, 'epoch': '1.383'}

 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 354/512 [39:09<18:13,  6.92s/it]
 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 355/512 [39:16<17:50,  6.82s/it]
                                                 
{'loss': '8.07e-05', 'grad_norm': '0.004499', 'learning_rate': '3.142e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '32.65', 'tokens/total': 10745296, 'tokens/trainable': 162478, 'epoch': '1.387'}

 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 355/512 [39:16<17:50,  6.82s/it]
 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 356/512 [39:22<17:29,  6.73s/it]
                                                 
{'loss': '7.578e-05', 'grad_norm': '0.004872', 'learning_rate': '3.117e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.61', 'memory/max_allocated (GiB)': '33.61', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '28.74', 'tokens/total': 10775296, 'tokens/trainable': 162893, 'epoch': '1.391'}

 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 356/512 [39:22<17:29,  6.73s/it]
 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 357/512 [39:29<17:16,  6.69s/it]
                                                 
{'loss': '0.003035', 'grad_norm': '0.2148', 'learning_rate': '3.093e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '35.49', 'tokens/total': 10805600, 'tokens/trainable': 163347, 'epoch': '1.395'}

 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 357/512 [39:29<17:16,  6.69s/it]
 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 358/512 [39:36<17:06,  6.66s/it]
                                                 
{'loss': '0.0004827', 'grad_norm': '0.07261', 'learning_rate': '3.068e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '33.58', 'tokens/total': 10835760, 'tokens/trainable': 163796, 'epoch': '1.398'}

 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰   | 358/512 [39:36<17:06,  6.66s/it]
 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 359/512 [39:42<16:54,  6.63s/it]
                                                 
{'loss': '0.0002045', 'grad_norm': '0.02204', 'learning_rate': '3.044e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '32.47', 'tokens/total': 10866208, 'tokens/trainable': 164237, 'epoch': '1.402'}

 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 359/512 [39:42<16:54,  6.63s/it]
 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 360/512 [39:49<16:46,  6.62s/it]
                                                 
{'loss': '0.0001249', 'grad_norm': '0.005709', 'learning_rate': '3.02e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '34.69', 'tokens/total': 10896672, 'tokens/trainable': 164684, 'epoch': '1.406'}

 70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 360/512 [39:49<16:46,  6.62s/it]
 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 361/512 [39:55<16:38,  6.61s/it]
                                                 
{'loss': '0.0002532', 'grad_norm': '0.0225', 'learning_rate': '2.995e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '37.05', 'tokens/total': 10926944, 'tokens/trainable': 165146, 'epoch': '1.41'}

 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 361/512 [39:55<16:38,  6.61s/it]
 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 362/512 [40:02<16:33,  6.62s/it]
                                                 
{'loss': '0.0002238', 'grad_norm': '0.01391', 'learning_rate': '2.971e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '35.37', 'tokens/total': 10957456, 'tokens/trainable': 165632, 'epoch': '1.414'}

 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 362/512 [40:02<16:33,  6.62s/it]
 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 363/512 [40:09<16:23,  6.60s/it]
                                                 
{'loss': '0.01284', 'grad_norm': '0.2468', 'learning_rate': '2.947e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '34.66', 'tokens/total': 10987696, 'tokens/trainable': 166080, 'epoch': '1.418'}

 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 363/512 [40:09<16:23,  6.60s/it]
 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 364/512 [40:15<16:17,  6.61s/it]
                                                 
{'loss': '0.0001312', 'grad_norm': '0.008053', 'learning_rate': '2.924e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '34.62', 'tokens/total': 11017904, 'tokens/trainable': 166531, 'epoch': '1.422'}

 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ   | 364/512 [40:15<16:17,  6.61s/it]
 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 365/512 [40:22<16:11,  6.61s/it]
                                                 
{'loss': '0.0002447', 'grad_norm': '0.04744', 'learning_rate': '2.9e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '36.91', 'tokens/total': 11048224, 'tokens/trainable': 167019, 'epoch': '1.426'}

 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 365/512 [40:22<16:11,  6.61s/it]
 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 366/512 [40:28<16:04,  6.60s/it]
                                                 
{'loss': '0.0001016', 'grad_norm': '0.008861', 'learning_rate': '2.876e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.84', 'tokens/train_per_sec_per_gpu': '30.94', 'tokens/total': 11078544, 'tokens/trainable': 167469, 'epoch': '1.43'}

 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 366/512 [40:28<16:04,  6.60s/it]
 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 367/512 [40:35<15:58,  6.61s/it]
                                                 
{'loss': '0.0006842', 'grad_norm': '0.06953', 'learning_rate': '2.853e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '31.86', 'tokens/total': 11109088, 'tokens/trainable': 167914, 'epoch': '1.434'}

 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 367/512 [40:35<15:58,  6.61s/it]
 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 368/512 [40:42<15:50,  6.60s/it]
                                                 
{'loss': '0.000614', 'grad_norm': '0.1097', 'learning_rate': '2.829e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '36.16', 'tokens/total': 11139408, 'tokens/trainable': 168357, 'epoch': '1.438'}

 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 368/512 [40:42<15:50,  6.60s/it]
 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 369/512 [40:48<15:46,  6.62s/it]
                                                 
{'loss': '0.0001494', 'grad_norm': '0.01329', 'learning_rate': '2.806e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.75', 'tokens/total': 11169920, 'tokens/trainable': 168832, 'epoch': '1.441'}

 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 369/512 [40:48<15:46,  6.62s/it]
 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 370/512 [40:55<15:40,  6.62s/it]
                                                 
{'loss': '0.0134', 'grad_norm': '0.7464', 'learning_rate': '2.783e-05', 'ppl': '1.013', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '37.41', 'tokens/total': 11200288, 'tokens/trainable': 169333, 'epoch': '1.445'}

 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 370/512 [40:55<15:40,  6.62s/it]
 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 371/512 [41:02<15:36,  6.64s/it]
                                                 
{'loss': '2.964e-05', 'grad_norm': '0.001093', 'learning_rate': '2.76e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.62', 'tokens/total': 11230832, 'tokens/trainable': 169774, 'epoch': '1.449'}

 72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 371/512 [41:02<15:36,  6.64s/it]
 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 372/512 [41:08<15:29,  6.64s/it]
                                                 
{'loss': '0.001514', 'grad_norm': '0.1324', 'learning_rate': '2.737e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.41', 'tokens/total': 11261264, 'tokens/trainable': 170241, 'epoch': '1.453'}

 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 372/512 [41:08<15:29,  6.64s/it]
 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 373/512 [41:15<15:22,  6.64s/it]
                                                 
{'loss': '0.0001303', 'grad_norm': '0.006302', 'learning_rate': '2.714e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.69', 'tokens/total': 11291456, 'tokens/trainable': 170706, 'epoch': '1.457'}

 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 373/512 [41:15<15:22,  6.64s/it]
 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 374/512 [41:21<15:14,  6.63s/it]
                                                 
{'loss': '0.0002068', 'grad_norm': '0.0117', 'learning_rate': '2.691e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.53', 'tokens/total': 11321952, 'tokens/trainable': 171182, 'epoch': '1.461'}

 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 374/512 [41:21<15:14,  6.63s/it]
 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 375/512 [41:28<15:04,  6.60s/it]
                                                 
{'loss': '0.0003061', 'grad_norm': '0.01545', 'learning_rate': '2.668e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.72', 'memory/max_allocated (GiB)': '33.72', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '34.75', 'tokens/total': 11352096, 'tokens/trainable': 171646, 'epoch': '1.465'}

 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 375/512 [41:28<15:04,  6.60s/it]
 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 376/512 [41:35<14:56,  6.59s/it]
                                                 
{'loss': '0.000194', 'grad_norm': '0.008863', 'learning_rate': '2.646e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '28.57', 'tokens/total': 11382528, 'tokens/trainable': 172063, 'epoch': '1.469'}

 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 376/512 [41:35<14:56,  6.59s/it]
 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 377/512 [41:41<14:50,  6.60s/it]
                                                 
{'loss': '0.0004883', 'grad_norm': '0.03389', 'learning_rate': '2.624e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '36.63', 'tokens/total': 11412832, 'tokens/trainable': 172539, 'epoch': '1.473'}

 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž  | 377/512 [41:41<14:50,  6.60s/it]
 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 378/512 [41:48<14:43,  6.59s/it]
                                                 
{'loss': '0.000516', 'grad_norm': '0.02547', 'learning_rate': '2.601e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.71', 'memory/max_allocated (GiB)': '33.71', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '34.99', 'tokens/total': 11442864, 'tokens/trainable': 172997, 'epoch': '1.477'}

 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 378/512 [41:48<14:43,  6.59s/it]
 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 379/512 [41:54<14:36,  6.59s/it]
                                                 
{'loss': '0.0004852', 'grad_norm': '0.04523', 'learning_rate': '2.579e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '36.26', 'tokens/total': 11473328, 'tokens/trainable': 173481, 'epoch': '1.48'}

 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 379/512 [41:54<14:36,  6.59s/it]
 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 380/512 [42:01<14:30,  6.59s/it]
                                                 
{'loss': '0.000263', 'grad_norm': '0.01156', 'learning_rate': '2.557e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.73', 'memory/max_allocated (GiB)': '33.73', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '30.94', 'tokens/total': 11503504, 'tokens/trainable': 173946, 'epoch': '1.484'}

 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 380/512 [42:01<14:30,  6.59s/it]
 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 381/512 [42:08<14:25,  6.60s/it]
                                                 
{'loss': '0.0004146', 'grad_norm': '0.01947', 'learning_rate': '2.535e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '34.7', 'tokens/total': 11533920, 'tokens/trainable': 174417, 'epoch': '1.488'}

 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 381/512 [42:08<14:25,  6.60s/it]
 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 382/512 [42:14<14:19,  6.61s/it]
                                                 
{'loss': '0.000521', 'grad_norm': '0.03897', 'learning_rate': '2.513e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.17', 'tokens/total': 11564128, 'tokens/trainable': 174872, 'epoch': '1.492'}

 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 382/512 [42:14<14:19,  6.61s/it]
 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 383/512 [42:21<14:13,  6.62s/it]
                                                 
{'loss': '0.0002654', 'grad_norm': '0.01254', 'learning_rate': '2.492e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '31.99', 'tokens/total': 11594672, 'tokens/trainable': 175313, 'epoch': '1.496'}

 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–  | 383/512 [42:21<14:13,  6.62s/it]
 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 384/512 [42:27<14:08,  6.63s/it]
                                                 
{'loss': '0.01107', 'grad_norm': '0.2314', 'learning_rate': '2.47e-05', 'ppl': '1.011', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.96', 'tokens/total': 11625056, 'tokens/trainable': 175750, 'epoch': '1.5'}

 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 384/512 [42:27<14:08,  6.63s/it][2026-08-18 15:00:25,170] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-384

 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 385/512 [42:36<14:57,  7.07s/it]
                                                 
{'loss': '0.01684', 'grad_norm': '0.5276', 'learning_rate': '2.449e-05', 'ppl': '1.017', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.8', 'tokens/total': 11655488, 'tokens/trainable': 176232, 'epoch': '1.504'}

 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 385/512 [42:36<14:57,  7.07s/it]
 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 386/512 [42:42<14:32,  6.93s/it]
                                                 
{'loss': '0.0001142', 'grad_norm': '0.008643', 'learning_rate': '2.428e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '34.9', 'tokens/total': 11685968, 'tokens/trainable': 176680, 'epoch': '1.508'}

 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 386/512 [42:42<14:32,  6.93s/it]
 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 387/512 [42:49<14:14,  6.83s/it]
                                                 
{'loss': '0.0001288', 'grad_norm': '0.006846', 'learning_rate': '2.406e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '33.24', 'tokens/total': 11716176, 'tokens/trainable': 177144, 'epoch': '1.512'}

 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 387/512 [42:49<14:14,  6.83s/it]
 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 388/512 [42:55<13:58,  6.76s/it]
                                                 
{'loss': '0.001054', 'grad_norm': '0.07777', 'learning_rate': '2.385e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '36.67', 'tokens/total': 11746704, 'tokens/trainable': 177630, 'epoch': '1.516'}

 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 388/512 [42:55<13:58,  6.76s/it]
 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 389/512 [43:02<13:45,  6.71s/it]
                                                 
{'loss': '0.0004593', 'grad_norm': '0.02181', 'learning_rate': '2.365e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '33.51', 'tokens/total': 11777008, 'tokens/trainable': 178079, 'epoch': '1.52'}

 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 389/512 [43:02<13:45,  6.71s/it]
 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 390/512 [43:09<13:35,  6.69s/it]
                                                 
{'loss': '0.006384', 'grad_norm': '0.2766', 'learning_rate': '2.344e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '36.99', 'tokens/total': 11807168, 'tokens/trainable': 178552, 'epoch': '1.523'}

 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ  | 390/512 [43:09<13:35,  6.69s/it]
 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 391/512 [43:15<13:27,  6.67s/it]
                                                 
{'loss': '0.003972', 'grad_norm': '0.5077', 'learning_rate': '2.323e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '34.49', 'tokens/total': 11837776, 'tokens/trainable': 179021, 'epoch': '1.527'}

 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 391/512 [43:15<13:27,  6.67s/it]
 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 392/512 [43:22<13:16,  6.64s/it]
                                                 
{'loss': '0.0004878', 'grad_norm': '0.04643', 'learning_rate': '2.303e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '33.57', 'tokens/total': 11867968, 'tokens/trainable': 179478, 'epoch': '1.531'}

 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 392/512 [43:22<13:16,  6.64s/it]
 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 393/512 [43:28<13:06,  6.61s/it]
                                                 
{'loss': '0.00317', 'grad_norm': '0.1215', 'learning_rate': '2.282e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '33.27', 'tokens/total': 11898320, 'tokens/trainable': 179910, 'epoch': '1.535'}

 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 393/512 [43:28<13:06,  6.61s/it]
 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 394/512 [43:35<12:59,  6.61s/it]
                                                 
{'loss': '0.0002976', 'grad_norm': '0.02432', 'learning_rate': '2.262e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.87', 'tokens/train_per_sec_per_gpu': '34.45', 'tokens/total': 11928624, 'tokens/trainable': 180361, 'epoch': '1.539'}

 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 394/512 [43:35<12:59,  6.61s/it]
 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 395/512 [43:41<12:50,  6.58s/it]
                                                 
{'loss': '0.0003667', 'grad_norm': '0.01568', 'learning_rate': '2.242e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.31', 'tokens/total': 11959088, 'tokens/trainable': 180774, 'epoch': '1.543'}

 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 395/512 [43:41<12:50,  6.58s/it]
 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 396/512 [43:48<12:44,  6.59s/it]
                                                 
{'loss': '0.0003036', 'grad_norm': '0.02069', 'learning_rate': '2.222e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.69', 'tokens/total': 11989504, 'tokens/trainable': 181246, 'epoch': '1.547'}

 77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹  | 396/512 [43:48<12:44,  6.59s/it]
 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 397/512 [43:55<12:39,  6.60s/it]
                                                 
{'loss': '0.008752', 'grad_norm': '0.2614', 'learning_rate': '2.202e-05', 'ppl': '1.009', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.65', 'tokens/total': 12019904, 'tokens/trainable': 181734, 'epoch': '1.551'}

 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 397/512 [43:55<12:39,  6.60s/it]
 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 398/512 [44:01<12:32,  6.60s/it]
                                                 
{'loss': '0.0004417', 'grad_norm': '0.03743', 'learning_rate': '2.183e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '36.01', 'tokens/total': 12050384, 'tokens/trainable': 182169, 'epoch': '1.555'}

 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 398/512 [44:01<12:32,  6.60s/it]
 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 399/512 [44:08<12:26,  6.60s/it]
                                                 
{'loss': '0.0007634', 'grad_norm': '0.1698', 'learning_rate': '2.163e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '35.34', 'tokens/total': 12080576, 'tokens/trainable': 182641, 'epoch': '1.559'}

 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 399/512 [44:08<12:26,  6.60s/it]
 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 400/512 [44:14<12:18,  6.60s/it]
                                                 
{'loss': '0.003636', 'grad_norm': '0.1729', 'learning_rate': '2.144e-05', 'ppl': '1.004', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '26.93', 'tokens/total': 12110896, 'tokens/trainable': 183056, 'epoch': '1.562'}

 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 400/512 [44:14<12:18,  6.60s/it]
 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 401/512 [44:21<12:10,  6.59s/it]
                                                 
{'loss': '0.0008229', 'grad_norm': '0.05639', 'learning_rate': '2.124e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.71', 'memory/max_allocated (GiB)': '33.71', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.32', 'tokens/total': 12140784, 'tokens/trainable': 183504, 'epoch': '1.566'}

 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 401/512 [44:21<12:10,  6.59s/it]
 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 402/512 [44:28<12:06,  6.60s/it]
                                                 
{'loss': '0.0007216', 'grad_norm': '0.03654', 'learning_rate': '2.105e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '39.29', 'tokens/total': 12171024, 'tokens/trainable': 184004, 'epoch': '1.57'}

 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 402/512 [44:28<12:06,  6.60s/it]
 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 403/512 [44:34<12:00,  6.61s/it]
                                                 
{'loss': '0.0002983', 'grad_norm': '0.01657', 'learning_rate': '2.086e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '33.04', 'tokens/total': 12201184, 'tokens/trainable': 184476, 'epoch': '1.574'}

 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š  | 403/512 [44:34<12:00,  6.61s/it]
 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 404/512 [44:41<11:52,  6.59s/it]
                                                 
{'loss': '0.0007513', 'grad_norm': '0.06593', 'learning_rate': '2.067e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.35', 'tokens/total': 12231232, 'tokens/trainable': 184905, 'epoch': '1.578'}

 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 404/512 [44:41<11:52,  6.59s/it]
 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 405/512 [44:47<11:43,  6.58s/it]
                                                 
{'loss': '0.0001536', 'grad_norm': '0.01335', 'learning_rate': '2.049e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.72', 'memory/max_allocated (GiB)': '33.72', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '29.82', 'tokens/total': 12261280, 'tokens/trainable': 185338, 'epoch': '1.582'}

 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 405/512 [44:47<11:43,  6.58s/it]
 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 406/512 [44:54<11:37,  6.58s/it]
                                                 
{'loss': '0.001165', 'grad_norm': '0.1154', 'learning_rate': '2.03e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '32.02', 'tokens/total': 12291520, 'tokens/trainable': 185792, 'epoch': '1.586'}

 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 406/512 [44:54<11:37,  6.58s/it]
 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 407/512 [45:01<11:30,  6.57s/it]
                                                 
{'loss': '0.0003108', 'grad_norm': '0.0157', 'learning_rate': '2.012e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '35.88', 'tokens/train_per_sec_per_gpu': '37.41', 'tokens/total': 12321824, 'tokens/trainable': 186251, 'epoch': '1.59'}

 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 407/512 [45:01<11:30,  6.57s/it]
 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 408/512 [45:07<11:24,  6.58s/it]
                                                 
{'loss': '0.0003415', 'grad_norm': '0.02254', 'learning_rate': '1.993e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '33.33', 'tokens/total': 12352176, 'tokens/trainable': 186689, 'epoch': '1.594'}

 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 408/512 [45:07<11:24,  6.58s/it]
 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 409/512 [45:14<11:18,  6.59s/it]
                                                 
{'loss': '0.005957', 'grad_norm': '0.4058', 'learning_rate': '1.975e-05', 'ppl': '1.006', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '34.7', 'tokens/total': 12382608, 'tokens/trainable': 187121, 'epoch': '1.598'}

 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰  | 409/512 [45:14<11:18,  6.59s/it]
 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 410/512 [45:20<11:12,  6.59s/it]
                                                 
{'loss': '0.001334', 'grad_norm': '0.15', 'learning_rate': '1.957e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '35.55', 'tokens/total': 12413152, 'tokens/trainable': 187568, 'epoch': '1.602'}

 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 410/512 [45:20<11:12,  6.59s/it]
 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 411/512 [45:27<11:05,  6.59s/it]
                                                 
{'loss': '0.00013', 'grad_norm': '0.006942', 'learning_rate': '1.94e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '34.76', 'tokens/total': 12443600, 'tokens/trainable': 188010, 'epoch': '1.605'}

 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 411/512 [45:27<11:05,  6.59s/it]
 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 412/512 [45:33<10:43,  6.43s/it]
                                                 
{'loss': '0.001001', 'grad_norm': '0.1238', 'learning_rate': '1.922e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '39.3', 'tokens/total': 12471952, 'tokens/trainable': 188446, 'epoch': '1.609'}

 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 412/512 [45:33<10:43,  6.43s/it]
 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 413/512 [45:40<10:43,  6.50s/it]
                                                 
{'loss': '0.007401', 'grad_norm': '0.3418', 'learning_rate': '1.904e-05', 'ppl': '1.007', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '32.07', 'tokens/total': 12502480, 'tokens/trainable': 188897, 'epoch': '1.613'}

 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 413/512 [45:40<10:43,  6.50s/it]
 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 414/512 [45:46<10:40,  6.53s/it]
                                                 
{'loss': '6.43e-05', 'grad_norm': '0.006758', 'learning_rate': '1.887e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '37.49', 'tokens/total': 12532720, 'tokens/trainable': 189366, 'epoch': '1.617'}

 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 414/512 [45:46<10:40,  6.53s/it]
 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 415/512 [45:53<10:33,  6.54s/it]
                                                 
{'loss': '0.005144', 'grad_norm': '0.1996', 'learning_rate': '1.87e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '32.4', 'tokens/total': 12562848, 'tokens/trainable': 189779, 'epoch': '1.621'}

 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ  | 415/512 [45:53<10:33,  6.54s/it]
 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 416/512 [45:59<10:30,  6.56s/it]
                                                 
{'loss': '0.0007223', 'grad_norm': '0.1678', 'learning_rate': '1.853e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '34.56', 'tokens/total': 12593328, 'tokens/trainable': 190245, 'epoch': '1.625'}

 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 416/512 [45:59<10:30,  6.56s/it][2026-08-18 15:03:57,157] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-416

 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 417/512 [46:08<11:08,  7.04s/it]
                                                 
{'loss': '0.0002913', 'grad_norm': '0.01879', 'learning_rate': '1.836e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.96', 'tokens/train_per_sec_per_gpu': '35.59', 'tokens/total': 12623760, 'tokens/trainable': 190737, 'epoch': '1.629'}

 81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 417/512 [46:08<11:08,  7.04s/it]
 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 418/512 [46:14<10:47,  6.89s/it]
                                                 
{'loss': '0.0005218', 'grad_norm': '0.09822', 'learning_rate': '1.819e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.81', 'tokens/train_per_sec_per_gpu': '31.38', 'tokens/total': 12654272, 'tokens/trainable': 191155, 'epoch': '1.633'}

 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 418/512 [46:14<10:47,  6.89s/it]
 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 419/512 [46:21<10:31,  6.79s/it]
                                                 
{'loss': '0.0007685', 'grad_norm': '0.07196', 'learning_rate': '1.802e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.81', 'tokens/train_per_sec_per_gpu': '31.98', 'tokens/total': 12684544, 'tokens/trainable': 191577, 'epoch': '1.637'}

 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 419/512 [46:21<10:31,  6.79s/it]
 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 420/512 [46:27<10:18,  6.73s/it]
                                                 
{'loss': '0.0007165', 'grad_norm': '0.04137', 'learning_rate': '1.786e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '35.81', 'tokens/train_per_sec_per_gpu': '37.25', 'tokens/total': 12715008, 'tokens/trainable': 192051, 'epoch': '1.641'}

 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 420/512 [46:27<10:18,  6.73s/it]
 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 421/512 [46:34<10:09,  6.69s/it]
                                                 
{'loss': '0.0002489', 'grad_norm': '0.01246', 'learning_rate': '1.77e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.81', 'tokens/train_per_sec_per_gpu': '31.77', 'tokens/total': 12745232, 'tokens/trainable': 192500, 'epoch': '1.645'}

 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 421/512 [46:34<10:09,  6.69s/it]
 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 422/512 [46:40<09:59,  6.66s/it]
                                                 
{'loss': '8.724e-05', 'grad_norm': '0.007973', 'learning_rate': '1.753e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '33.23', 'tokens/total': 12775824, 'tokens/trainable': 192955, 'epoch': '1.648'}

 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 422/512 [46:40<09:59,  6.66s/it]
 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 423/512 [46:47<09:52,  6.65s/it]
                                                 
{'loss': '0.0003637', 'grad_norm': '0.02996', 'learning_rate': '1.737e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '38.03', 'tokens/total': 12806240, 'tokens/trainable': 193462, 'epoch': '1.652'}

 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 423/512 [46:47<09:52,  6.65s/it]
 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 424/512 [46:54<09:44,  6.64s/it]
                                                 
{'loss': '0.0008654', 'grad_norm': '0.1144', 'learning_rate': '1.722e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '37', 'tokens/total': 12836512, 'tokens/trainable': 193943, 'epoch': '1.656'}

 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 424/512 [46:54<09:44,  6.64s/it]
 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 425/512 [47:00<09:36,  6.63s/it]
                                                 
{'loss': '0.001933', 'grad_norm': '0.111', 'learning_rate': '1.706e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '29.75', 'tokens/total': 12866704, 'tokens/trainable': 194360, 'epoch': '1.66'}

 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 425/512 [47:00<09:36,  6.63s/it]
 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 426/512 [47:07<09:30,  6.63s/it]
                                                 
{'loss': '0.0001203', 'grad_norm': '0.005441', 'learning_rate': '1.69e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '35.43', 'tokens/total': 12896912, 'tokens/trainable': 194848, 'epoch': '1.664'}

 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 426/512 [47:07<09:30,  6.63s/it]
 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 427/512 [47:14<09:23,  6.63s/it]
                                                 
{'loss': '0.0001281', 'grad_norm': '0.01091', 'learning_rate': '1.675e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '34.48', 'tokens/total': 12927248, 'tokens/trainable': 195314, 'epoch': '1.668'}

 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 427/512 [47:14<09:23,  6.63s/it]
 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 428/512 [47:20<09:16,  6.63s/it]
                                                 
{'loss': '9.905e-05', 'grad_norm': '0.007124', 'learning_rate': '1.66e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '32.4', 'tokens/total': 12957632, 'tokens/trainable': 195732, 'epoch': '1.672'}

 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 428/512 [47:20<09:16,  6.63s/it]
 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 429/512 [47:27<09:08,  6.61s/it]
                                                 
{'loss': '0.002469', 'grad_norm': '0.2362', 'learning_rate': '1.645e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.75', 'memory/max_allocated (GiB)': '33.75', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '33.37', 'tokens/total': 12987824, 'tokens/trainable': 196197, 'epoch': '1.676'}

 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 429/512 [47:27<09:08,  6.61s/it]
 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 430/512 [47:33<09:02,  6.62s/it]
                                                 
{'loss': '0.005283', 'grad_norm': '0.2918', 'learning_rate': '1.63e-05', 'ppl': '1.005', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.83', 'tokens/train_per_sec_per_gpu': '32.85', 'tokens/total': 13018208, 'tokens/trainable': 196678, 'epoch': '1.68'}

 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 430/512 [47:33<09:02,  6.62s/it]
 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 431/512 [47:40<08:58,  6.65s/it]
                                                 
{'loss': '0.0004377', 'grad_norm': '0.03083', 'learning_rate': '1.615e-05', 'ppl': '1', 'memory/max_active (GiB)': '34.02', 'memory/max_allocated (GiB)': '34.02', 'memory/device_reserved (GiB)': '36.36', 'tokens/train_per_sec_per_gpu': '31.52', 'tokens/total': 13048800, 'tokens/trainable': 197109, 'epoch': '1.684'}

 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 431/512 [47:40<08:58,  6.65s/it]
 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 432/512 [47:47<08:51,  6.64s/it]
                                                 
{'loss': '0.0002724', 'grad_norm': '0.024', 'learning_rate': '1.6e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '36.36', 'tokens/train_per_sec_per_gpu': '39.22', 'tokens/total': 13079392, 'tokens/trainable': 197607, 'epoch': '1.688'}

 84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 432/512 [47:47<08:51,  6.64s/it]
 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 433/512 [47:53<08:43,  6.63s/it]
                                                 
{'loss': '0.000624', 'grad_norm': '0.03983', 'learning_rate': '1.586e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.95', 'memory/max_allocated (GiB)': '33.95', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '33.34', 'tokens/total': 13109920, 'tokens/trainable': 198066, 'epoch': '1.691'}

 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 433/512 [47:53<08:43,  6.63s/it]
 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 434/512 [48:00<08:36,  6.62s/it]
                                                 
{'loss': '0.0005622', 'grad_norm': '0.03856', 'learning_rate': '1.572e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '33.11', 'tokens/total': 13140064, 'tokens/trainable': 198484, 'epoch': '1.695'}

 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 434/512 [48:00<08:36,  6.62s/it]
 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 435/512 [48:06<08:29,  6.61s/it]
                                                 
{'loss': '0.0008548', 'grad_norm': '0.07454', 'learning_rate': '1.558e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '35.03', 'tokens/total': 13170656, 'tokens/trainable': 198957, 'epoch': '1.699'}

 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 435/512 [48:06<08:29,  6.61s/it]
 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 436/512 [48:13<08:22,  6.62s/it]
                                                 
{'loss': '5.212e-05', 'grad_norm': '0.003516', 'learning_rate': '1.544e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '33.78', 'tokens/total': 13201056, 'tokens/trainable': 199405, 'epoch': '1.703'}

 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 436/512 [48:13<08:22,  6.62s/it]
 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 437/512 [48:20<08:16,  6.62s/it]
                                                 
{'loss': '0.0004348', 'grad_norm': '0.9569', 'learning_rate': '1.53e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '34.61', 'tokens/total': 13231584, 'tokens/trainable': 199888, 'epoch': '1.707'}

 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 437/512 [48:20<08:16,  6.62s/it]
 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 438/512 [48:26<08:10,  6.63s/it]
                                                 
{'loss': '0.0001117', 'grad_norm': '0.01002', 'learning_rate': '1.516e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '35.46', 'tokens/total': 13261888, 'tokens/trainable': 200396, 'epoch': '1.711'}

 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 438/512 [48:26<08:10,  6.63s/it]
 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 439/512 [48:33<08:02,  6.60s/it]
                                                 
{'loss': '0.0001707', 'grad_norm': '0.008282', 'learning_rate': '1.503e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '33.34', 'tokens/total': 13291936, 'tokens/trainable': 200843, 'epoch': '1.715'}

 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 439/512 [48:33<08:02,  6.60s/it]
 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 440/512 [48:40<07:56,  6.61s/it]
                                                 
{'loss': '0.0007505', 'grad_norm': '0.06078', 'learning_rate': '1.49e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '35.57', 'tokens/total': 13322224, 'tokens/trainable': 201315, 'epoch': '1.719'}

 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 440/512 [48:40<07:56,  6.61s/it]
 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 441/512 [48:46<07:49,  6.61s/it]
                                                 
{'loss': '0.001084', 'grad_norm': '0.1296', 'learning_rate': '1.477e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '34.77', 'tokens/total': 13352640, 'tokens/trainable': 201782, 'epoch': '1.723'}

 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 441/512 [48:46<07:49,  6.61s/it]
 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 442/512 [48:53<07:41,  6.59s/it]
                                                 
{'loss': '0.0005228', 'grad_norm': '0.0456', 'learning_rate': '1.464e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.62', 'memory/max_allocated (GiB)': '33.62', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '31.27', 'tokens/total': 13382560, 'tokens/trainable': 202222, 'epoch': '1.727'}

 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 442/512 [48:53<07:41,  6.59s/it]
 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 443/512 [48:59<07:34,  6.59s/it]
                                                 
{'loss': '0.0003615', 'grad_norm': '0.02645', 'learning_rate': '1.451e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '36.81', 'tokens/total': 13412976, 'tokens/trainable': 202660, 'epoch': '1.73'}

 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 443/512 [48:59<07:34,  6.59s/it]
 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 444/512 [49:06<07:25,  6.55s/it]
                                                 
{'loss': '0.0002294', 'grad_norm': '0.02189', 'learning_rate': '1.438e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.64', 'memory/max_allocated (GiB)': '33.64', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '31.51', 'tokens/total': 13442720, 'tokens/trainable': 203068, 'epoch': '1.734'}

 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 444/512 [49:06<07:25,  6.55s/it]
 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 445/512 [49:12<07:18,  6.55s/it]
                                                 
{'loss': '0.001371', 'grad_norm': '0.1399', 'learning_rate': '1.426e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '34.97', 'tokens/total': 13472928, 'tokens/trainable': 203527, 'epoch': '1.738'}

 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 445/512 [49:12<07:18,  6.55s/it]
 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 446/512 [49:19<07:13,  6.56s/it]
                                                 
{'loss': '0.0002348', 'grad_norm': '0.02456', 'learning_rate': '1.414e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '35.94', 'tokens/total': 13502992, 'tokens/trainable': 203994, 'epoch': '1.742'}

 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 446/512 [49:19<07:13,  6.56s/it]
 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 447/512 [49:25<07:06,  6.56s/it]
                                                 
{'loss': '0.0002258', 'grad_norm': '0.0218', 'learning_rate': '1.402e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '38.06', 'tokens/total': 13533184, 'tokens/trainable': 204430, 'epoch': '1.746'}

 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 447/512 [49:25<07:06,  6.56s/it]
 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 448/512 [49:32<07:00,  6.58s/it]
                                                 
{'loss': '0.002107', 'grad_norm': '0.2327', 'learning_rate': '1.39e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '36.07', 'tokens/total': 13563648, 'tokens/trainable': 204886, 'epoch': '1.75'}

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 448/512 [49:32<07:00,  6.58s/it][2026-08-18 15:07:29,838] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-448

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 449/512 [49:40<07:24,  7.06s/it]
                                                 
{'loss': '0.0001422', 'grad_norm': '0.01578', 'learning_rate': '1.378e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.93', 'memory/max_allocated (GiB)': '33.93', 'memory/device_reserved (GiB)': '36.81', 'tokens/train_per_sec_per_gpu': '34.15', 'tokens/total': 13594032, 'tokens/trainable': 205356, 'epoch': '1.754'}

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 449/512 [49:40<07:24,  7.06s/it]
 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 450/512 [49:47<07:09,  6.92s/it]
                                                 
{'loss': '0.001226', 'grad_norm': '0.09758', 'learning_rate': '1.367e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.69', 'tokens/train_per_sec_per_gpu': '31.13', 'tokens/total': 13624368, 'tokens/trainable': 205782, 'epoch': '1.758'}

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 450/512 [49:47<07:09,  6.92s/it]
 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 451/512 [49:54<06:57,  6.84s/it]
                                                 
{'loss': '0.0006241', 'grad_norm': '0.05972', 'learning_rate': '1.355e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.7', 'tokens/train_per_sec_per_gpu': '32.1', 'tokens/total': 13654768, 'tokens/trainable': 206219, 'epoch': '1.762'}

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 451/512 [49:54<06:57,  6.84s/it]
 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 452/512 [50:00<06:45,  6.76s/it]
                                                 
{'loss': '5.874e-05', 'grad_norm': '0.009018', 'learning_rate': '1.344e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.7', 'tokens/train_per_sec_per_gpu': '36.29', 'tokens/total': 13684960, 'tokens/trainable': 206677, 'epoch': '1.766'}

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 452/512 [50:00<06:45,  6.76s/it]
 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 453/512 [50:07<06:37,  6.73s/it]
                                                 
{'loss': '8.521e-05', 'grad_norm': '0.01131', 'learning_rate': '1.333e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '35.7', 'tokens/train_per_sec_per_gpu': '37.8', 'tokens/total': 13715488, 'tokens/trainable': 207160, 'epoch': '1.77'}

 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 453/512 [50:07<06:37,  6.73s/it]
 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 454/512 [50:13<06:28,  6.69s/it]
                                                 
{'loss': '3.377e-05', 'grad_norm': '0.002245', 'learning_rate': '1.322e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.86', 'memory/max_allocated (GiB)': '33.86', 'memory/device_reserved (GiB)': '35.7', 'tokens/train_per_sec_per_gpu': '36.25', 'tokens/total': 13745856, 'tokens/trainable': 207605, 'epoch': '1.773'}

 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 454/512 [50:13<06:28,  6.69s/it]
 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 455/512 [50:20<06:20,  6.67s/it]
                                                 
{'loss': '0.0001957', 'grad_norm': '0.01739', 'learning_rate': '1.311e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '33.43', 'tokens/total': 13776192, 'tokens/trainable': 208047, 'epoch': '1.777'}

 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 455/512 [50:20<06:20,  6.67s/it]
 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 456/512 [50:27<06:11,  6.64s/it]
                                                 
{'loss': '0.0005775', 'grad_norm': '0.04885', 'learning_rate': '1.301e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '35.78', 'tokens/train_per_sec_per_gpu': '35.44', 'tokens/total': 13806496, 'tokens/trainable': 208476, 'epoch': '1.781'}

 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 456/512 [50:27<06:11,  6.64s/it]
 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 457/512 [50:33<06:04,  6.63s/it]
                                                 
{'loss': '0.001659', 'grad_norm': '0.3257', 'learning_rate': '1.29e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '36.03', 'tokens/total': 13836768, 'tokens/trainable': 208970, 'epoch': '1.785'}

 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 457/512 [50:33<06:04,  6.63s/it]
 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 458/512 [50:40<05:58,  6.63s/it]
                                                 
{'loss': '0.0006973', 'grad_norm': '0.06033', 'learning_rate': '1.28e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '35.31', 'tokens/total': 13867152, 'tokens/trainable': 209456, 'epoch': '1.789'}

 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 458/512 [50:40<05:58,  6.63s/it]
 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 459/512 [50:46<05:51,  6.63s/it]
                                                 
{'loss': '3.174e-05', 'grad_norm': '0.00734', 'learning_rate': '1.27e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '34.15', 'tokens/total': 13897360, 'tokens/trainable': 209907, 'epoch': '1.793'}

 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 459/512 [50:46<05:51,  6.63s/it]
 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 460/512 [50:53<05:43,  6.62s/it]
                                                 
{'loss': '0.0016', 'grad_norm': '0.1607', 'learning_rate': '1.26e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.91', 'memory/max_allocated (GiB)': '33.91', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '34.16', 'tokens/total': 13927792, 'tokens/trainable': 210366, 'epoch': '1.797'}

 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 460/512 [50:53<05:43,  6.62s/it]
 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 461/512 [51:00<05:36,  6.59s/it]
                                                 
{'loss': '0.00224', 'grad_norm': '0.5972', 'learning_rate': '1.251e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.65', 'memory/max_allocated (GiB)': '33.65', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '32.1', 'tokens/total': 13957856, 'tokens/trainable': 210835, 'epoch': '1.801'}

 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 461/512 [51:00<05:36,  6.59s/it]
 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 462/512 [51:06<05:22,  6.44s/it]
                                                 
{'loss': '0.001224', 'grad_norm': '0.1525', 'learning_rate': '1.241e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.38', 'memory/max_allocated (GiB)': '33.38', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '30.91', 'tokens/total': 13986080, 'tokens/trainable': 211259, 'epoch': '1.805'}

 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 462/512 [51:06<05:22,  6.44s/it]
 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 463/512 [51:12<05:18,  6.49s/it]
                                                 
{'loss': '3.636e-05', 'grad_norm': '0.005159', 'learning_rate': '1.232e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.79', 'memory/max_allocated (GiB)': '33.79', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '35.65', 'tokens/total': 14016288, 'tokens/trainable': 211721, 'epoch': '1.809'}

 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 463/512 [51:12<05:18,  6.49s/it]
 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 464/512 [51:19<05:12,  6.51s/it]
                                                 
{'loss': '0.001997', 'grad_norm': '0.1126', 'learning_rate': '1.223e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.74', 'memory/max_allocated (GiB)': '33.74', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '29.59', 'tokens/total': 14046288, 'tokens/trainable': 212159, 'epoch': '1.812'}

 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 464/512 [51:19<05:12,  6.51s/it]
 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 465/512 [51:25<05:07,  6.55s/it]
                                                 
{'loss': '0.0001486', 'grad_norm': '0.02084', 'learning_rate': '1.214e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '33.17', 'tokens/total': 14076512, 'tokens/trainable': 212644, 'epoch': '1.816'}

 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 465/512 [51:25<05:07,  6.55s/it]
 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 466/512 [51:32<05:01,  6.56s/it]
                                                 
{'loss': '2.983e-05', 'grad_norm': '0.001434', 'learning_rate': '1.205e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '33.38', 'tokens/total': 14106608, 'tokens/trainable': 213084, 'epoch': '1.82'}

 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 466/512 [51:32<05:01,  6.56s/it]
 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 467/512 [51:39<04:56,  6.58s/it]
                                                 
{'loss': '0.002034', 'grad_norm': '0.1758', 'learning_rate': '1.197e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '32.05', 'tokens/total': 14136864, 'tokens/trainable': 213523, 'epoch': '1.824'}

 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 467/512 [51:39<04:56,  6.58s/it]
 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 468/512 [51:45<04:48,  6.57s/it]
                                                 
{'loss': '0.0001363', 'grad_norm': '0.01292', 'learning_rate': '1.188e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.73', 'memory/max_allocated (GiB)': '33.73', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '35.03', 'tokens/total': 14166896, 'tokens/trainable': 213963, 'epoch': '1.828'}

 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 468/512 [51:45<04:48,  6.57s/it]
 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 469/512 [51:52<04:42,  6.58s/it]
                                                 
{'loss': '0.001088', 'grad_norm': '0.133', 'learning_rate': '1.18e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '37.4', 'tokens/total': 14197568, 'tokens/trainable': 214431, 'epoch': '1.832'}

 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 469/512 [51:52<04:42,  6.58s/it]
 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 470/512 [51:58<04:30,  6.45s/it]
                                                 
{'loss': '1.274e-05', 'grad_norm': '0.000542', 'learning_rate': '1.172e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.82', 'tokens/train_per_sec_per_gpu': '36.07', 'tokens/total': 14225984, 'tokens/trainable': 214865, 'epoch': '1.836'}

 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 470/512 [51:58<04:30,  6.45s/it]
 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 471/512 [52:05<04:27,  6.53s/it]
                                                 
{'loss': '0.001536', 'grad_norm': '0.3001', 'learning_rate': '1.164e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '35.98', 'tokens/total': 14256512, 'tokens/trainable': 215358, 'epoch': '1.84'}

 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 471/512 [52:05<04:27,  6.53s/it]
 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 472/512 [52:11<04:21,  6.55s/it]
                                                 
{'loss': '0.0001523', 'grad_norm': '0.01403', 'learning_rate': '1.156e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '33.38', 'tokens/total': 14286640, 'tokens/trainable': 215800, 'epoch': '1.844'}

 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 472/512 [52:11<04:21,  6.55s/it]
 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 473/512 [52:18<04:16,  6.57s/it]
                                                 
{'loss': '0.0005481', 'grad_norm': '0.08513', 'learning_rate': '1.149e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '35.29', 'tokens/total': 14317024, 'tokens/trainable': 216245, 'epoch': '1.848'}

 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 473/512 [52:18<04:16,  6.57s/it]
 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 474/512 [52:24<04:09,  6.57s/it]
                                                 
{'loss': '4.498e-05', 'grad_norm': '0.004201', 'learning_rate': '1.142e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '33.83', 'tokens/total': 14347088, 'tokens/trainable': 216685, 'epoch': '1.852'}

 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 474/512 [52:24<04:09,  6.57s/it]
 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 475/512 [52:31<04:02,  6.55s/it]
                                                 
{'loss': '0.002648', 'grad_norm': '0.4065', 'learning_rate': '1.135e-05', 'ppl': '1.003', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '31.31', 'tokens/total': 14377152, 'tokens/trainable': 217092, 'epoch': '1.855'}

 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 475/512 [52:31<04:02,  6.55s/it]
 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 476/512 [52:38<03:56,  6.58s/it]
                                                 
{'loss': '1.867e-05', 'grad_norm': '0.001392', 'learning_rate': '1.128e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '38.27', 'tokens/total': 14407568, 'tokens/trainable': 217582, 'epoch': '1.859'}

 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 476/512 [52:38<03:56,  6.58s/it]
 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 477/512 [52:44<03:50,  6.58s/it]
                                                 
{'loss': '2.959e-05', 'grad_norm': '0.00188', 'learning_rate': '1.121e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '36.2', 'tokens/total': 14437792, 'tokens/trainable': 218049, 'epoch': '1.863'}

 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 477/512 [52:44<03:50,  6.58s/it]
 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 478/512 [52:51<03:45,  6.62s/it]
                                                 
{'loss': '0.0007669', 'grad_norm': '0.1085', 'learning_rate': '1.114e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '33.06', 'tokens/total': 14468400, 'tokens/trainable': 218522, 'epoch': '1.867'}

 93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 478/512 [52:51<03:45,  6.62s/it]
 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 479/512 [52:57<03:37,  6.60s/it]
                                                 
{'loss': '0.0007375', 'grad_norm': '0.08652', 'learning_rate': '1.108e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.83', 'memory/max_allocated (GiB)': '33.83', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '33.32', 'tokens/total': 14498416, 'tokens/trainable': 218970, 'epoch': '1.871'}

 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 479/512 [52:57<03:37,  6.60s/it]
 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 480/512 [53:04<03:31,  6.60s/it]
                                                 
{'loss': '0.002163', 'grad_norm': '0.1909', 'learning_rate': '1.102e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '33.98', 'memory/max_allocated (GiB)': '33.98', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '34.62', 'tokens/total': 14528976, 'tokens/trainable': 219411, 'epoch': '1.875'}

 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 480/512 [53:04<03:31,  6.60s/it][2026-08-18 15:11:01,803] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-480

 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 481/512 [53:12<03:39,  7.08s/it]
                                                 
{'loss': '3.28e-05', 'grad_norm': '0.003719', 'learning_rate': '1.096e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.78', 'memory/max_allocated (GiB)': '33.78', 'memory/device_reserved (GiB)': '35.92', 'tokens/train_per_sec_per_gpu': '29.98', 'tokens/total': 14558992, 'tokens/trainable': 219837, 'epoch': '1.879'}

 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 481/512 [53:12<03:39,  7.08s/it]
 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 482/512 [53:19<03:28,  6.95s/it]
                                                 
{'loss': '2.935e-05', 'grad_norm': '0.002898', 'learning_rate': '1.09e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.61', 'tokens/total': 14589488, 'tokens/trainable': 220345, 'epoch': '1.883'}

 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 482/512 [53:19<03:28,  6.95s/it]
 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 483/512 [53:25<03:18,  6.83s/it]
                                                 
{'loss': '0.0002914', 'grad_norm': '0.03024', 'learning_rate': '1.084e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.8', 'memory/max_allocated (GiB)': '33.8', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.58', 'tokens/total': 14619584, 'tokens/trainable': 220775, 'epoch': '1.887'}

 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 483/512 [53:25<03:18,  6.83s/it]
 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 484/512 [53:32<03:09,  6.76s/it]
                                                 
{'loss': '2.797e-05', 'grad_norm': '0.002516', 'learning_rate': '1.079e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.58', 'tokens/total': 14649760, 'tokens/trainable': 221217, 'epoch': '1.891'}

 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 484/512 [53:32<03:09,  6.76s/it]
 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 485/512 [53:39<03:00,  6.70s/it]
                                                 
{'loss': '8.626e-05', 'grad_norm': '0.01444', 'learning_rate': '1.073e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.07', 'tokens/total': 14679920, 'tokens/trainable': 221666, 'epoch': '1.895'}

 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 485/512 [53:39<03:00,  6.70s/it]
 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 486/512 [53:45<02:53,  6.66s/it]
                                                 
{'loss': '0.0001287', 'grad_norm': '0.0164', 'learning_rate': '1.068e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '31.95', 'tokens/total': 14710432, 'tokens/trainable': 222110, 'epoch': '1.898'}

 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 486/512 [53:45<02:53,  6.66s/it]
 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 487/512 [53:52<02:45,  6.63s/it]
                                                 
{'loss': '0.0001532', 'grad_norm': '0.022', 'learning_rate': '1.063e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.67', 'memory/max_allocated (GiB)': '33.67', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.47', 'tokens/total': 14740416, 'tokens/trainable': 222542, 'epoch': '1.902'}

 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 487/512 [53:52<02:45,  6.63s/it]
 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 488/512 [53:58<02:39,  6.63s/it]
                                                 
{'loss': '1.004e-05', 'grad_norm': '0.00261', 'learning_rate': '1.058e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.81', 'tokens/total': 14770832, 'tokens/trainable': 223035, 'epoch': '1.906'}

 95%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 488/512 [53:58<02:39,  6.63s/it]
 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 489/512 [54:05<02:32,  6.61s/it]
                                                 
{'loss': '1.927e-05', 'grad_norm': '0.001653', 'learning_rate': '1.054e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.77', 'memory/max_allocated (GiB)': '33.77', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.27', 'tokens/total': 14801104, 'tokens/trainable': 223488, 'epoch': '1.91'}

 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 489/512 [54:05<02:32,  6.61s/it]
 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 490/512 [54:12<02:25,  6.60s/it]
                                                 
{'loss': '6.305e-06', 'grad_norm': '0.0007156', 'learning_rate': '1.049e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.76', 'memory/max_allocated (GiB)': '33.76', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '29.86', 'tokens/total': 14831216, 'tokens/trainable': 223932, 'epoch': '1.914'}

 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 490/512 [54:12<02:25,  6.60s/it]
 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 491/512 [54:18<02:18,  6.61s/it]
                                                 
{'loss': '0.0001381', 'grad_norm': '0.01357', 'learning_rate': '1.045e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.39', 'tokens/total': 14861488, 'tokens/trainable': 224389, 'epoch': '1.918'}

 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 491/512 [54:18<02:18,  6.61s/it]
 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 492/512 [54:25<02:12,  6.62s/it]
                                                 
{'loss': '0.001722', 'grad_norm': '0.2159', 'learning_rate': '1.041e-05', 'ppl': '1.002', 'memory/max_active (GiB)': '34.01', 'memory/max_allocated (GiB)': '34.01', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.65', 'tokens/total': 14891824, 'tokens/trainable': 224871, 'epoch': '1.922'}

 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 492/512 [54:25<02:12,  6.62s/it]
 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 493/512 [54:31<02:05,  6.63s/it]
                                                 
{'loss': '0.01609', 'grad_norm': '0.4015', 'learning_rate': '1.037e-05', 'ppl': '1.016', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.16', 'tokens/total': 14922192, 'tokens/trainable': 225352, 'epoch': '1.926'}

 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 493/512 [54:31<02:05,  6.63s/it]
 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 494/512 [54:38<01:59,  6.63s/it]
                                                 
{'loss': '5.178e-05', 'grad_norm': '0.003592', 'learning_rate': '1.034e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.84', 'memory/max_allocated (GiB)': '33.84', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.3', 'tokens/total': 14952624, 'tokens/trainable': 225801, 'epoch': '1.93'}

 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 494/512 [54:38<01:59,  6.63s/it]
 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 495/512 [54:45<01:52,  6.61s/it]
                                                 
{'loss': '1.597e-05', 'grad_norm': '0.001518', 'learning_rate': '1.03e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.81', 'memory/max_allocated (GiB)': '33.81', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.05', 'tokens/total': 14982704, 'tokens/trainable': 226240, 'epoch': '1.934'}

 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 495/512 [54:45<01:52,  6.61s/it]
 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 496/512 [54:51<01:46,  6.63s/it]
                                                 
{'loss': '2.798e-05', 'grad_norm': '0.001617', 'learning_rate': '1.027e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.9', 'memory/max_allocated (GiB)': '33.9', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.32', 'tokens/total': 15013024, 'tokens/trainable': 226712, 'epoch': '1.938'}

 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 496/512 [54:51<01:46,  6.63s/it]
 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 497/512 [54:58<01:39,  6.61s/it]
                                                 
{'loss': '9.649e-05', 'grad_norm': '0.007684', 'learning_rate': '1.024e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.28', 'tokens/total': 15043520, 'tokens/trainable': 227135, 'epoch': '1.941'}

 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 497/512 [54:58<01:39,  6.61s/it]
 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 498/512 [55:04<01:32,  6.60s/it]
                                                 
{'loss': '3.721e-05', 'grad_norm': '0.002169', 'learning_rate': '1.021e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.87', 'memory/max_allocated (GiB)': '33.87', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '38.25', 'tokens/total': 15073824, 'tokens/trainable': 227600, 'epoch': '1.945'}

 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 498/512 [55:04<01:32,  6.60s/it]
 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 499/512 [55:11<01:25,  6.61s/it]
                                                 
{'loss': '0.0003017', 'grad_norm': '0.02861', 'learning_rate': '1.018e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '37.21', 'tokens/total': 15104128, 'tokens/trainable': 228076, 'epoch': '1.949'}

 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 499/512 [55:11<01:25,  6.61s/it]
 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 500/512 [55:18<01:19,  6.60s/it]
                                                 
{'loss': '6.546e-05', 'grad_norm': '0.02024', 'learning_rate': '1.016e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.82', 'memory/max_allocated (GiB)': '33.82', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '31.41', 'tokens/total': 15134480, 'tokens/trainable': 228526, 'epoch': '1.953'}

 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 500/512 [55:18<01:19,  6.60s/it]
 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 501/512 [55:24<01:12,  6.60s/it]
                                                 
{'loss': '2.901e-05', 'grad_norm': '0.001963', 'learning_rate': '1.013e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.92', 'memory/max_allocated (GiB)': '33.92', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '35.76', 'tokens/total': 15164832, 'tokens/trainable': 229005, 'epoch': '1.957'}

 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 501/512 [55:24<01:12,  6.60s/it]
 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 502/512 [55:31<01:06,  6.62s/it]
                                                 
{'loss': '0.0003191', 'grad_norm': '0.05688', 'learning_rate': '1.011e-05', 'ppl': '1', 'memory/max_active (GiB)': '34.03', 'memory/max_allocated (GiB)': '34.03', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '31.5', 'tokens/total': 15195616, 'tokens/trainable': 229438, 'epoch': '1.961'}

 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 502/512 [55:31<01:06,  6.62s/it]
 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 503/512 [55:37<00:59,  6.61s/it]
                                                 
{'loss': '0.0002743', 'grad_norm': '0.04076', 'learning_rate': '1.009e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '33.33', 'tokens/total': 15226096, 'tokens/trainable': 229885, 'epoch': '1.965'}

 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 503/512 [55:37<00:59,  6.61s/it]
 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 504/512 [55:44<00:52,  6.61s/it]
                                                 
{'loss': '0.0001235', 'grad_norm': '0.01305', 'learning_rate': '1.008e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.96', 'memory/max_allocated (GiB)': '33.96', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '39.53', 'tokens/total': 15256512, 'tokens/trainable': 230389, 'epoch': '1.969'}

 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 504/512 [55:44<00:52,  6.61s/it]
 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 505/512 [55:51<00:46,  6.62s/it]
                                                 
{'loss': '0.0005182', 'grad_norm': '0.08245', 'learning_rate': '1.006e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '33.97', 'memory/max_allocated (GiB)': '33.97', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '38.08', 'tokens/total': 15287072, 'tokens/trainable': 230885, 'epoch': '1.973'}

 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 505/512 [55:51<00:46,  6.62s/it]
 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 506/512 [55:57<00:39,  6.61s/it]
                                                 
{'loss': '1.987e-05', 'grad_norm': '0.001345', 'learning_rate': '1.005e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.85', 'memory/max_allocated (GiB)': '33.85', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '34.79', 'tokens/total': 15317296, 'tokens/trainable': 231335, 'epoch': '1.977'}

 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 506/512 [55:57<00:39,  6.61s/it]
 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 507/512 [56:04<00:33,  6.63s/it]
                                                 
{'loss': '0.0003057', 'grad_norm': '0.04595', 'learning_rate': '1.003e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.96', 'tokens/total': 15347904, 'tokens/trainable': 231824, 'epoch': '1.98'}

 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 507/512 [56:04<00:33,  6.63s/it]
 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 508/512 [56:10<00:25,  6.46s/it]
                                                 
{'loss': '2.902e-05', 'grad_norm': '0.00236', 'learning_rate': '1.002e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.33', 'memory/max_allocated (GiB)': '33.33', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '32.08', 'tokens/total': 15376144, 'tokens/trainable': 232241, 'epoch': '1.984'}

 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 508/512 [56:10<00:25,  6.46s/it]
 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 509/512 [56:17<00:19,  6.58s/it]
                                                 
{'loss': '0.001196', 'grad_norm': '0.0954', 'learning_rate': '1.001e-05', 'ppl': '1.001', 'memory/max_active (GiB)': '34.02', 'memory/max_allocated (GiB)': '34.02', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '37.83', 'tokens/total': 15406976, 'tokens/trainable': 232742, 'epoch': '1.988'}

 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 509/512 [56:17<00:19,  6.58s/it]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 510/512 [56:24<00:13,  6.60s/it]
                                                 
{'loss': '4.301e-05', 'grad_norm': '0.003428', 'learning_rate': '1.001e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.94', 'memory/max_allocated (GiB)': '33.94', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.66', 'tokens/total': 15437344, 'tokens/trainable': 233231, 'epoch': '1.992'}

100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 510/512 [56:24<00:13,  6.60s/it]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 511/512 [56:30<00:06,  6.60s/it]
                                                 
{'loss': '0.0001367', 'grad_norm': '0.01952', 'learning_rate': '1e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.88', 'memory/max_allocated (GiB)': '33.88', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '30.86', 'tokens/total': 15467712, 'tokens/trainable': 233655, 'epoch': '1.996'}

100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 511/512 [56:30<00:06,  6.60s/it]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 512/512 [56:37<00:00,  6.63s/it]
                                                 
{'loss': '8.935e-05', 'grad_norm': '0.004961', 'learning_rate': '1e-05', 'ppl': '1', 'memory/max_active (GiB)': '33.89', 'memory/max_allocated (GiB)': '33.89', 'memory/device_reserved (GiB)': '36.15', 'tokens/train_per_sec_per_gpu': '36.78', 'tokens/total': 15498240, 'tokens/trainable': 234124, 'epoch': '2'}

100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 512/512 [56:37<00:00,  6.63s/it][2026-08-18 15:14:34,655] [INFO] [axolotl.core.trainers.base._save:828] [PID:12244] Saving model checkpoint to /workspace/wave/training/checkpoints/checkpoint-512

                                                 
{'train_runtime': '3399', 'train_samples_per_second': '4.82', 'train_steps_per_second': '0.151', 'train_loss': '0.01122', 'memory/max_active (GiB)': '24.25', 'memory/max_allocated (GiB)': '24.25', 'memory/device_reserved (GiB)': '36.15', 'epoch': '2', 'tokens/train_per_sec_per_gpu': '0'}

100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 512/512 [56:38<00:00,  6.63s/it]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 512/512 [56:38<00:00,  6.64s/it]
[2026-08-18 15:14:36,202] [INFO] [axolotl.train.save_trained_model:267] [PID:12244] Training completed! Saving trained model to /workspace/wave/training/checkpoints.
[2026-08-18 15:14:36,510] [INFO] [axolotl.train.save_trained_model:388] [PID:12244] Model successfully saved to /workspace/wave/training/checkpoints